diff --git a/.gitea/workflows/conformance.yml b/.gitea/workflows/conformance.yml index f0b51e3..3317d21 100644 --- a/.gitea/workflows/conformance.yml +++ b/.gitea/workflows/conformance.yml @@ -40,6 +40,8 @@ jobs: run: cargo test --release --manifest-path conformance/probe/Cargo.toml env: CARGO_TARGET_DIR: conformance/.cache/target + - name: Reference-side tests + run: /opt/conformance/bin/python conformance/test_ref.py - name: Sweep # The corpora come from GitHub (pinned commits, conformance/corpus.txt), # so this job needs a runner that reaches github.com. diff --git a/conformance/README.md b/conformance/README.md index 5f08558..919d1fe 100644 --- a/conformance/README.md +++ b/conformance/README.md @@ -19,9 +19,11 @@ probe's `szip` feature; `libaec-dev`), and a Python with the packages in | `fetch-corpus.sh` | shallow, sparse, blob-filtered checkout of each pinned commit into `.cache/src/` (gitignored); no-op when already there | | `list_files.py` | which files are probed (HDF5/netCDF-4 extensions minus netCDF classic, plus the CVE reproducers) | | `probe/` | the clawhdf5 side: a standalone crate (outside the workspace, so `cargo test --workspace` never builds it) that walks a file with `clawhdf5-format` and prints canonical JSON | -| `ref.py` | the h5py side: the same JSON from h5py | +| `ref.py` | the h5py side: the same JSON from h5py (values corrected for a known h5py bug are marked `ref_fix`) | +| `ref_bugs.py` | re-reads the objects h5py reads only through a libhdf5 bug in six differently-set-up processes; an object whose values change is confirmed as a libhdf5 over-read | +| `test_ref.py` | tests of `ref.py`'s correction and `ref_bugs.py`'s confirmation (`python conformance/test_ref.py`) | | `run_one.sh` | runs both sides on one file (and `h5dump` on the CVE corpus) under a timeout and an address-space limit | -| `compare.py` | classifies each file (ok / our-error / mismatch / h5py-cannot-read / panic / hang / crash / oom) and groups root causes | +| `compare.py` | classifies each file (ok / our-error / mismatch / h5py-cannot-read / ref-bug / panic / hang / crash / oom) and groups root causes | | `report.py` | writes `CONFORMANCE.md` | | `check.py` | the gate: fails on any panic/hang/crash/oom, on an ok count below `baseline.json`, or on a baseline-ok file that is no longer ok | | `baseline.json` | the ok files the gate holds the line on | diff --git a/conformance/compare.py b/conformance/compare.py index c1d65b2..b3c944a 100755 --- a/conformance/compare.py +++ b/conformance/compare.py @@ -5,6 +5,9 @@ Writes /results.csv, results.json and summary.md. File classes (first match wins): hang, oom, crash, panic ours: timeout / allocation failure / signal / any panic (caught or not) h5py-cannot-read libhdf5/h5py failed to open the file (or crashed/hung) + ref-bug every issue is an object we refuse that h5py reads only through a + libhdf5 bug, confirmed in this run by ref_bugs.py (its values + change with the reading process's heap) our-error we fail to open, list, or read something h5py reads mismatch we read something with different shape/values, or a different object set ok @@ -19,6 +22,20 @@ import sys R = sys.argv[1] RUNS = os.path.join(R, "runs") +# Objects ref_bugs.py confirmed in this run: h5py's values for them come from +# libhdf5 reading memory the file does not determine. +try: + REF_BUGS = {(b["file"], b["object"]) + for b in json.load(open(os.path.join(R, "ref_bugs.json")))["read_bugs"] if b.get("confirmed")} +except (OSError, ValueError, KeyError): + REF_BUGS = set() + + +def is_ref_bug(rel, issue): + """An our-error on reading an object that ref_bugs.py confirmed.""" + kind, detail = issue[0], issue[1] + return kind == "our-error" and any(f == rel and detail.startswith(obj + ": error: ") for f, obj in REF_BUGS) + def load(d, name): rc_p = os.path.join(d, name + ".rc") @@ -86,6 +103,8 @@ mismatch_causes = collections.defaultdict(lambda: {"files": set(), "count": 0, " panics = [] ref_only_errors = collections.Counter() incomparable = collections.Counter() +# Values ref.py corrected for a known h5py bug: (file, object, fixes, same as ours) +ref_fixes = [] def add(bucket, key, file, example): @@ -169,6 +188,8 @@ for rel in files: elif a.get("hash") != b.get("hash"): issues.append(("mismatch", f"{p}: values differ (h5py {a.get('dtype')} vs ours {b.get('dtype')})", "values", b | {"ref_head": a.get("head"), "ref_dtype": a.get("dtype")})) ok = False + if a.get("ref_fix") and "hash" in b: + ref_fixes.append((rel, p, a["ref_fix"], a.get("hash") == b.get("hash"))) ra, oa = a.get("attrs") or {}, b.get("attrs") or {} if "attrs_error" not in b and "attrs_error" not in a and not ref_unopened: for an in sorted(set(ra) | set(oa)): @@ -187,6 +208,8 @@ for rel in files: issues.append(("mismatch", f"{p}@{an}: attr shape {x.get('shape')} vs ours {y.get('shape')}", "attr-shape", y | {"ref_dtype": x.get("dtype")})) elif x.get("hash") != y.get("hash"): issues.append(("mismatch", f"{p}@{an}: attr values differ (h5py {x.get('dtype')} vs ours {y.get('dtype')})", "attr-values", y | {"ref_head": x.get("head"), "ref_dtype": x.get("dtype")})) + if x.get("ref_fix") and "hash" in y: + ref_fixes.append((rel, f"{p}@{an}", x["ref_fix"], x.get("hash") == y.get("hash"))) if ok: n_ok += 1 @@ -197,6 +220,8 @@ for rel in files: cls = "panic" elif ref_open_fail: cls = "h5py-cannot-read" + elif issues and all(is_ref_bug(rel, i) for i in issues): + cls = "ref-bug" elif ours_open_err: cls = "our-error" issues.append(("our-error", f"open: {ours_open_err}", ours_open_err, {})) @@ -259,10 +284,11 @@ def ser(b): json.dump({"rows": rows, "issues": issues_by_file, "root_causes": ser(root_causes), "mismatch_causes": ser(mismatch_causes), - "panics": panics, "incomparable": incomparable.most_common(), "ref_only_errors": ref_only_errors.most_common()}, + "panics": panics, "incomparable": incomparable.most_common(), "ref_only_errors": ref_only_errors.most_common(), + "ref_fixes": ref_fixes, "ref_bugs_confirmed": sorted(REF_BUGS)}, open(os.path.join(R, "results.json"), "w"), indent=1) -classes = ["ok", "our-error", "mismatch", "h5py-cannot-read", "hang", "panic", "crash", "oom"] +classes = ["ok", "our-error", "mismatch", "h5py-cannot-read", "ref-bug", "hang", "panic", "crash", "oom"] by_corpus = collections.defaultdict(collections.Counter) for r in rows: by_corpus[r["corpus"]][r["class"]] += 1 diff --git a/conformance/ref.py b/conformance/ref.py index 428b6a5..499fc3a 100755 --- a/conformance/ref.py +++ b/conformance/ref.py @@ -53,6 +53,49 @@ def packed(dt): return dt +# --- reference corrections --------------------------------------------------- +# Where h5py is known to return values the file does not hold, and the right +# values follow from what it returned, ref.py corrects them and records the +# correction on the object ("ref_fix"), so the comparison is still a real +# comparison and CONFORMANCE.md lists every corrected object. Each correction +# first checks that the installed h5py still has the bug. + +# Corrections applied while encoding the current object. +FIXES = set() +_BE_VLEN_BUG = None + + +def be_vlen_bug(): + """h5py (3.16 / HDF5 2.0 at least) returns the elements of a + variable-length sequence whose base type is big-endian with the file's + big-endian bytes under a native (little-endian) dtype: a + `vlen_dtype('>f4')` dataset holding [1.0, 2.0] reads back as + [4.6e-41, 9.0e-44]. `h5dump` prints the file's values. Checked once per + process by writing and reading exactly that dataset in memory.""" + global _BE_VLEN_BUG + if _BE_VLEN_BUG is None: + import io + try: + bio = io.BytesIO() + with h5py.File(bio, "w") as f: + d = f.create_dataset("v", (1,), dtype=h5py.vlen_dtype(np.dtype(">f4"))) + d[0] = np.array([1.0, 2.0], dtype=">f4") + with h5py.File(bio, "r") as f: + got = np.asarray(f["v"][0]) + _BE_VLEN_BUG = (got.dtype == np.dtype("f4").tolist() == [1.0, 2.0] + and got.tolist() != [1.0, 2.0]) + except Exception: # noqa: BLE001 + _BE_VLEN_BUG = False + return _BE_VLEN_BUG + + +def unswapped(got, base): + """`got` is `base` (big-endian somewhere) with every field in native + little-endian order instead: the shape of h5py's big-endian VL bug.""" + return base.newbyteorder("<") == got and base != got + + def canon_el(dt, val, out): if dt.fields: for n in dt.names: @@ -79,7 +122,13 @@ def canon_el(dt, val, out): base = h5py.check_vlen_dtype(dt) if base is None: raise TypeError(f"unhandled object dtype {dt!r}") - arr = np.asarray(val if val is not None else [], dtype=base).reshape(-1) + arr = np.asarray(val if val is not None else []) + if arr.dtype != base and be_vlen_bug() and unswapped(arr.dtype, base): + # h5py's big-endian VL bug (see be_vlen_bug): the bytes are + # the file's, the dtype label is wrong. Relabel, don't convert. + arr = arr.view(base) + FIXES.add("h5py-be-vlen") + arr = np.asarray(arr, dtype=base).reshape(-1) out += b"V" + struct.pack(": re-check the objects h5py reads only through a +libhdf5 bug. + +For each object of READ_BUGS (below), h5py reads it in several fresh +processes whose heaps differ: h5py imported before numpy (three runs, plus +two with glibc's MALLOC_PERTURB_, which fills newly allocated and freed heap +blocks with a byte pattern) and numpy imported first. Values the file +determines come out the same every time. An object whose values differ +between those runs is read from memory the file does not determine — an +over-read or an uninitialised buffer in libhdf5 — so the values h5py reports +for it are not the file's, and clawhdf5 refusing the object is not a +clawhdf5 error. compare.py classifies a file as `ref-bug` only on objects +confirmed that way in the same run (`$OUT/ref_bugs.json`); an object whose +reading turns out stable stays an our-error. + +Run by conformance/run.sh; on its own it is the reproducer (JSON on stdout). +""" +import concurrent.futures +import json +import os +import subprocess +import sys + +# (file, object) -> what goes wrong. Checked 2026-09-27 against HDF5 2.0.0 +# (h5py 3.16), h5dump 1.14.6 and the HDFGroup/hdf5 sources (tag hdf5_1_14_6 +# and develop); see docs/known-issues.md, "Conformance: the last non-ok files". +READ_BUGS = { + ("cve_hdf5/cvefiles/cve-2025-2308.h5", "/Scale_offset_long_long_data_le"): + "the first chunk records minbits 11: its 12 values need 17 bytes of codes, and the " + "26-byte chunk holds 5 after its 21-byte header; libhdf5's scale-offset decoder reads " + "past its buffer, and develop refuses the chunk (\"Buffer too short\")", + ("cve_hdf5/cvefiles/cve-2025-44904.h5", "/Scale_offset_float_data_le"): + "unfiltered chunks stored as 38 and 37 bytes for 48-byte chunks: 1.14/2.0 read the " + "stored bytes into a buffer of that size and use it as the whole chunk " + "(H5D__chunk_lock), so the rest is heap memory; develop refuses them (\"incorrect chunk " + "size returned from index for unfiltered chunk\")", + ("hdf5/test/testfiles/bad_nbit_parms_walk.h5", "/Nbit_int_data_le"): + "the N-Bit parameter list holds 7 values (cd_values[0] = 7) where an integer needs 8: " + "the decoder takes the bit offset from cd_values[7], past the list; libhdf5's own test " + "(`test_filter_bad_params`, test/dsets.c on develop) requires the read to fail", +} + +# (which module is imported first, MALLOC_PERTURB_) +RUNS = [("h5py", None), ("h5py", None), ("h5py", None), ("h5py", "170"), ("h5py", "255"), + ("numpy", None)] + +READ = r""" +import hashlib, sys +if sys.argv[3] == "h5py": + import h5py, numpy as np +else: + import numpy as np, h5py +try: + import hdf5plugin # noqa: F401 +except Exception: + pass +try: + with h5py.File(sys.argv[1], "r") as f: + a = np.ascontiguousarray(f[sys.argv[2]][()]) + print("values " + hashlib.sha256(a.tobytes()).hexdigest()[:16]) +except Exception as e: + print("error " + (str(e).splitlines() or [type(e).__name__])[0][:120]) +""" + + +def read_once(path, obj, first, perturb): + env = dict(os.environ) + env.pop("MALLOC_PERTURB_", None) + if perturb: + env["MALLOC_PERTURB_"] = perturb + try: + p = subprocess.run([sys.executable, "-c", READ, path, obj, first], env=env, + capture_output=True, text=True, timeout=60) + out = p.stdout.strip().splitlines() + return out[-1] if out else f"exit {p.returncode}" + except subprocess.TimeoutExpired: + return "timeout" + + +def check(corpus, key): + f, obj = key + path = os.path.join(corpus, f) + rec = {"file": f, "object": obj, "why": READ_BUGS[key]} + if not os.path.exists(path): + return rec | {"missing": True, "confirmed": False} + runs = [{"first": a, "malloc_perturb": p, "outcome": read_once(path, obj, a, p)} for a, p in RUNS] + distinct = sorted({r["outcome"] for r in runs}) + return rec | { + "runs": runs, + "distinct": len(distinct), + "confirmed": len(distinct) > 1 and any(o.startswith("values ") for o in distinct), + } + + +def main(): + corpus = sys.argv[1] + keys = list(READ_BUGS) + with concurrent.futures.ThreadPoolExecutor(max_workers=len(keys)) as ex: + out = list(ex.map(lambda k: check(corpus, k), keys)) + print(json.dumps({"read_bugs": out}, indent=1)) + + +if __name__ == "__main__": + main() diff --git a/conformance/report.py b/conformance/report.py index 85ad0aa..6c4cb13 100644 --- a/conformance/report.py +++ b/conformance/report.py @@ -26,7 +26,7 @@ except Exception: # noqa: BLE001 R, OUT_MD, CORPUS = sys.argv[1], sys.argv[2], sys.argv[3] HERE = os.path.dirname(os.path.abspath(__file__)) ROOT = os.path.dirname(HERE) -CLASSES = ["ok", "our-error", "mismatch", "h5py-cannot-read", "panic", "hang", "crash", "oom"] +CLASSES = ["ok", "our-error", "mismatch", "h5py-cannot-read", "ref-bug", "panic", "hang", "crash", "oom"] def sh(*cmd, cwd=ROOT): @@ -89,46 +89,15 @@ def ex_list(files, n=3): return s + (f" (+{len(files) - n} more)" if len(files) > n else "") -# --- known causes that are not clawhdf5 bugs -------------------------------- -def is_h5py_be_vlen(i): - """h5py returns the elements of a VL sequence of a big-endian base type - with their file (big-endian) bytes but a native-endian dtype.""" - return (i["kind"] == "mismatch" and i["key"] in ("values", "attr-values") - and (i.get("ref_dtype") == "object") and (i.get("ours_dtype") or "").startswith("vlen(") - and ">" in (i.get("ours_dtype") or "")) - - -# Objects the reference (h5py 3.16 / HDF5 2.0) reads only because of an -# HDF5 2.0 bug, and that clawhdf5 refuses: each one reads past a buffer or -# returns bytes the file does not hold, and libhdf5's develop branch refuses all -# three. (file, object) -> why. Checked 2026-09-26 against HDF5 2.0.0 -# and HDFGroup/hdf5 develop sources; see docs/known-issues.md. -LIBHDF5_BUGS = { - ("cve_hdf5/cvefiles/cve-2025-2308.h5", "/Scale_offset_long_long_data_le"): - "scale-offset codes run past the end of the chunk: HDF5 2.0 reads past its buffer; " - "libhdf5's develop branch refuses the chunk (\"Buffer too short\")", - ("cve_hdf5/cvefiles/cve-2025-44904.h5", "/Scale_offset_float_data_le"): - "unfiltered chunks of 38 and 37 bytes for 48-byte chunks: HDF5 2.0 fills the rest with " - "whatever its buffer held; libhdf5's develop branch refuses them (\"incorrect chunk size returned " - "from index for unfiltered chunk\")", - ("hdf5/test/testfiles/bad_nbit_parms_walk.h5", "/Nbit_int_data_le"): - "an N-Bit parameter list one value short: HDF5 2.0 reads past the list; libhdf5's own " - "test (`test_filter_bad_params`, test/dsets.c) now requires the read to fail", -} - - -def is_libhdf5_bug(rel, i): - return i["kind"] == "our-error" and any( - f == rel and i["detail"].startswith(obj + ":") for (f, obj) in LIBHDF5_BUGS) - - -known = collections.defaultdict(list) -for r in rows: - iss = issues.get(r["file"], []) - if r["class"] == "mismatch" and iss and all(is_h5py_be_vlen(i) for i in iss): - known["h5py-be-vlen"].append(r["file"]) - if r["class"] == "our-error" and iss and all(is_libhdf5_bug(r["file"], i) for i in iss): - known["libhdf5-2.0"].append(r["file"]) +# --- reference bugs --------------------------------------------------------- +# ref_bugs.py's re-check of the objects h5py reads only through a libhdf5 bug +# (compare.py classifies on the confirmed ones), and the objects whose h5py +# values ref.py corrected (compare.py's ref_fixes). +try: + ref_bugs = json.load(open(os.path.join(R, "ref_bugs.json")))["read_bugs"] +except (OSError, ValueError, KeyError): + ref_bugs = [] +ref_fixes = res.get("ref_fixes", []) # --- the CVE corpus: clawhdf5 vs h5dump vs h5py ------------------------------ @@ -243,6 +212,7 @@ w("A file's class is the first that applies:") w("") w("- **panic / hang / crash / oom** — clawhdf5 panicked (caught per object or not), hit the timeout, died on a signal, or failed an allocation. The CI gate fails on any of these.") w("- **h5py-cannot-read** — libhdf5 could not open the file (or itself crashed or hung). Nothing to compare against; most are the deliberately malformed CVE reproducers.") +w("- **ref-bug** — every difference is an object clawhdf5 refuses that h5py reads only through a libhdf5 bug: the values h5py returns for it change with the reading process's heap, re-checked in every run (see *Reference bugs*).") w("- **our-error** — clawhdf5 returned an error for something h5py reads.") w("- **mismatch** — both read it, but the shapes, values, object set or attribute set differ.") w("- **ok** — every object h5py reads, clawhdf5 reads identically.") @@ -254,14 +224,13 @@ for c in sorted(by_corpus): w(f"| {c} | {sum(cnt.values())} | " + " | ".join(str(cnt.get(k, 0)) for k in CLASSES) + " |") w(f"| **all** | **{len(rows)}** | " + " | ".join(f"**{total.get(k, 0)}**" for k in CLASSES) + " |") w("") -if known["h5py-be-vlen"]: - w(f"{len(known['h5py-be-vlen'])} of the {total.get('mismatch', 0)} mismatches are a known h5py bug, " - "not ours (see *Known not-our-bug*).") - w("") -if known["libhdf5-2.0"]: - w(f"{len(known['libhdf5-2.0'])} of the {total.get('our-error', 0)} our-errors are corrupt data that " - "HDF5 2.0 reads only through a bug and clawhdf5 refuses (see *Known not-our-bug*).") - w("") +nonok = total.get("our-error", 0) + total.get("mismatch", 0) +w(f"**Our errors and mismatches: {nonok}.** " + + ("Every file clawhdf5 does not read like h5py is either unreadable by h5py or a confirmed libhdf5 " + "bug (*ref-bug*)." if nonok == 0 else "See the root causes below.") + + (f" {len(ref_fixes)} object(s) were compared against h5py's values corrected for a known h5py bug" + f" ({sum(1 for x in ref_fixes if x[3])} identical to clawhdf5's; see *Reference bugs*)." if ref_fixes else "")) +w("") w("Corpora (fetched by `conformance/fetch-corpus.sh` into the gitignored `conformance/.cache/`):") w("") w("| corpus | source | commit |") @@ -321,14 +290,38 @@ w("") w("") w("") -w("## Known not-our-bug") +w("## Reference bugs") +w("") +w("### Objects h5py reads only through a libhdf5 bug (*ref-bug*)") +w("") +w("clawhdf5 refuses these objects; h5py 3.16 / HDF5 2.0 returns values for them. `conformance/ref_bugs.py`") +w("re-reads each with h5py in six fresh processes whose heaps differ (h5py imported before numpy, three") +w("times and twice more with `MALLOC_PERTURB_`, and numpy imported first). Values the file determines") +w("come out the same every time; these do not, so they are memory libhdf5 over-reads, not the file's") +w("data. A file is *ref-bug* only while every one of its differences is such an object confirmed in") +w("the same run; an object that reads the same every time goes back to *our-error*. Reproducer:") +w("`python conformance/ref_bugs.py conformance/.cache/corpus` (prints every read's outcome).") +w("") +w("| file | object | distinct results in 6 reads | confirmed | what goes wrong |") +w("|---|---|---:|---|---|") +for b in ref_bugs: + n = "missing" if b.get("missing") else b.get("distinct", "?") + w(f"| `{b['file']}` | `{b['object']}` | {n} | {'yes' if b.get('confirmed') else '**no**'} | {b['why']} |") +w("") +w("### Values corrected for a known h5py bug") w("") w("- **h5py big-endian variable-length sequences.** h5py returns the elements of a VL sequence") w(" whose base type is big-endian with the file's big-endian bytes but a native (little-endian)") -w(" numpy dtype, so the values it reports are byte-swapped garbage; `h5dump` prints the values") -w(" clawhdf5 reads. Reproducer: `h5py.vlen_dtype(np.dtype('>f4'))` dataset holding `[1.0, 2.0]`") -w(" reads back in h5py as `[4.6e-41, 9.0e-44]`. Affected here: " - + (ex_list(sorted(known["h5py-be-vlen"]), 10) if known["h5py-be-vlen"] else "none") + ".") +w(" numpy dtype: a `h5py.vlen_dtype(np.dtype('>f4'))` dataset holding `[1.0, 2.0]` reads back as") +w(" `[4.6e-41, 9.0e-44]`; `h5dump` prints the file's values. `ref.py` checks that the installed") +w(" h5py still does this (by writing and reading exactly that dataset in memory) and, if so,") +w(" relabels such elements with the file's byte order before hashing, so the values are still") +w(" compared. Corrected objects: " + + (", ".join(f"`{f}` `{p}` ({'same as clawhdf5' if same else '**differs from clawhdf5**'})" + for f, p, _, same in ref_fixes) if ref_fixes else "none") + ".") +w("") +w("## Other comparison rules") +w("") w("- **Non-IEEE floats and partial-precision integers (N-Bit).** libhdf5 converts a float whose") w(" bit layout is not IEEE (e.g. `H5Tset_precision` for the N-Bit filter) or an integer with a") w(" bit offset / reduced precision into the plain numpy type of the same size. The probe") @@ -339,11 +332,6 @@ if res["incomparable"]: w(" (FP8 -> float16, bfloat16 -> float32, x87 long double -> float128) the values are not") w(" compared (shape and presence still are): " + ", ".join(f"{k} ({n}x)" for k, n in res["incomparable"]) + ".") -w("- **Corrupt data HDF5 2.0 reads through a bug.** clawhdf5 refuses these objects; h5py 3.16 /") -w(" HDF5 2.0 returns values for them that the file does not hold:") -for (f, obj), why in sorted(LIBHDF5_BUGS.items()): - here = "" if f in known["libhdf5-2.0"] else " (not an our-error in this run)" - w(f" - `{f}` `{obj}`: {why}{here}.") w("- **References** are compared by presence only (`R`), not by target.") w("") if res.get("ref_only_errors"): diff --git a/conformance/run.sh b/conformance/run.sh index 5ceb635..1c4bd85 100755 --- a/conformance/run.sh +++ b/conformance/run.sh @@ -71,6 +71,8 @@ xargs -a "$OUT/files.txt" -d '\n' -P "$JOBS" -I{} bash -c ' f="$1"; d="$OUT/runs/${f//\//__}" case "$f" in cve_hdf5/*) export WITH_H5DUMP=1 ;; esac "$HERE/run_one.sh" "$C/$f" "$d"' _ {} 2>"$OUT/probe.log" +echo "== re-checking the objects h5py reads only through a libhdf5 bug" +"$PY" "$HERE/ref_bugs.py" "$C" > "$OUT/ref_bugs.json" 2> "$OUT/ref_bugs.err" || true echo "== comparing" "$PY" "$HERE/compare.py" "$OUT" >/dev/null t2=$(date +%s) diff --git a/conformance/test_ref.py b/conformance/test_ref.py new file mode 100644 index 0000000..ed2d68f --- /dev/null +++ b/conformance/test_ref.py @@ -0,0 +1,92 @@ +#!/usr/bin/env python3 +"""Tests of the reference side's corrections: `python conformance/test_ref.py`. + +- ref.py compares a big-endian VL sequence by the file's values even though + h5py returns them byte-swapped (and records that it corrected them); +- ref_bugs.py confirms an object only when its reads disagree. +""" +import json +import os +import subprocess +import sys +import tempfile +import unittest + +import h5py +import numpy as np + +HERE = os.path.dirname(os.path.abspath(__file__)) +sys.path.insert(0, HERE) +import ref_bugs # noqa: E402 + + +def ref_objects(path): + out = subprocess.run([sys.executable, os.path.join(HERE, "ref.py"), path], + capture_output=True, text=True, check=True).stdout + return {o["path"]: o for o in json.loads(out)["objects"]} + + +class BigEndianVlen(unittest.TestCase): + def test_be_vlen_compared_by_file_values(self): + with tempfile.TemporaryDirectory() as d: + path = os.path.join(d, "v.h5") + with h5py.File(path, "w") as f: + for name, order in (("be", ">"), ("le", "<")): + t = np.dtype(order + "f4") + ds = f.create_dataset(name, (2,), dtype=h5py.vlen_dtype(t)) + ds[0] = np.array([1.0, 2.0], dtype=t) + ds[1] = np.array([3.0], dtype=t) + u = np.dtype(order + "u8") + f.attrs.create(name, [np.array([1, 2], dtype=u), np.array([42], dtype=u)], + dtype=h5py.vlen_dtype(u)) + objs = ref_objects(path) + be, le = objs["/be"], objs["/le"] + # Same values, so the same canonical hash whatever the file's byte order. + self.assertEqual(be["hash"], le["hash"]) + self.assertEqual(objs["/"]["attrs"]["be"]["hash"], objs["/"]["attrs"]["le"]["hash"]) + self.assertNotIn("ref_fix", le) + # And the correction is recorded wherever h5py needed it. + import ref + if ref.be_vlen_bug(): + self.assertEqual(be.get("ref_fix"), ["h5py-be-vlen"]) + self.assertEqual(objs["/"]["attrs"]["be"].get("ref_fix"), ["h5py-be-vlen"]) + + +class RefBugsConfirmation(unittest.TestCase): + def run_check(self, outcomes): + seq = iter(outcomes) + saved = ref_bugs.read_once + ref_bugs.read_once = lambda *a: next(seq) + try: + key = next(iter(ref_bugs.READ_BUGS)) + with tempfile.TemporaryDirectory() as d: + p = os.path.join(d, key[0]) + os.makedirs(os.path.dirname(p)) + open(p, "wb").close() + return ref_bugs.check(d, key) + finally: + ref_bugs.read_once = saved + + def test_stable_values_are_not_confirmed(self): + r = self.run_check(["values a"] * len(ref_bugs.RUNS)) + self.assertFalse(r["confirmed"]) + + def test_changing_values_are_confirmed(self): + r = self.run_check(["values a"] * (len(ref_bugs.RUNS) - 1) + ["values b"]) + self.assertTrue(r["confirmed"]) + r = self.run_check(["values a"] * (len(ref_bugs.RUNS) - 1) + ["error filter failed"]) + self.assertTrue(r["confirmed"]) + + def test_errors_only_are_not_confirmed(self): + # h5py cannot read it at all: nothing it reads, nothing to excuse. + r = self.run_check(["error x"] * (len(ref_bugs.RUNS) - 1) + ["error y"]) + self.assertFalse(r["confirmed"]) + + def test_missing_file_is_not_confirmed(self): + key = next(iter(ref_bugs.READ_BUGS)) + with tempfile.TemporaryDirectory() as d: + self.assertFalse(ref_bugs.check(d, key)["confirmed"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/docs/known-issues.md b/docs/known-issues.md index 98e1fac..9e572a6 100644 --- a/docs/known-issues.md +++ b/docs/known-issues.md @@ -25,6 +25,49 @@ An earlier run the same day under load also listed single-thread contiguous hyperslab reads as 5.6% slower; the idle rerun puts them at −1.7% with overlapping ranges (noise), so that item is withdrawn. +## Conformance: the last non-ok files (checked 2026-09-27) + +**Status:** classified with evidence — none is a clawhdf5 bug. The +conformance report had 3 our-errors and 2 mismatches left, each documented +as "not ours" but still counted against us. Re-checked on tank +(h5py 3.16 / HDF5 2.0.0, h5dump 1.14.6): + +- **2 mismatches, h5py's big-endian VL bug:** + `NCAS-CMS_pyfive/tests/data/attr_datatypes.hdf5` `/@vlen_uint64` and + `hdf5/tools/test/testfiles/tcomplex_be.h5` + `/VariableLengthDatasetFloatComplex`. h5py returns the elements with the + file's big-endian bytes under a little-endian dtype (a `vlen('>f4')` + holding `[1.0, 2.0]` reads back as `[4.6e-41, 9.0e-44]`). h5dump 1.14.6 + prints `(1, 2), (3, 4, 5), (42)` for `vlen_uint64`, which is what we read; + it cannot print `tcomplex_be.h5` (complex types are HDF5 2.0). The + reference side (`conformance/ref.py`) now checks that the installed h5py + has the bug and relabels such elements with the file's byte order, so the + values are compared rather than excused: both objects are identical to + ours, and both files are now *ok*. +- **3 our-errors, objects HDF5 2.0 reads only by over-reading memory:** + `cve-2025-2308.h5` `/Scale_offset_long_long_data_le` (first chunk records + `minbits` 11; its 12 values need 17 bytes of codes and the 26-byte chunk + holds 5 after its 21-byte header), `cve-2025-44904.h5` + `/Scale_offset_float_data_le` (unfiltered chunks stored as 38 and 37 bytes + for 48-byte chunks; 1.14.6's `H5D__chunk_lock` reads the stored bytes + into a buffer of that size and uses it as the whole chunk) and + `bad_nbit_parms_walk.h5` `/Nbit_int_data_le` (N-Bit parameters + `(7, 0, 40, 1, 4, 0, 20)`: an integer needs 8, and the decoder takes the + bit offset from `cd_values[7]`, past the list). **Could we match + libhdf5?** No: its output is not determined by the file. h5py's values for + the first two change from one run to the next (three plain runs gave three + different results), and all three change with `MALLOC_PERTURB_` or with + whether numpy is imported before h5py (`bad_nbit_parms_walk` then fails + with "filter returned failure during read" or reads all zeros); h5dump + 1.14.6 prints yet other values for the over-read parts (`1280, 0, 0` where + h5py gave e.g. `1283, 749, 1713`). libhdf5's develop branch refuses all + three. We keep + refusing them. `conformance/ref_bugs.py` repeats the check in every + conformance run (six reads per object in differently set-up processes) and + such a file is classified *ref-bug* only while its values keep changing. + +Result (`conformance/run.sh --no-fetch`, 2026-09-27): see CHANGELOG. + ## Files a SWMR writer had open could not be read past a stale end of file **Status:** fixed 2026-09-27 (branch `feat/p3-m5-swmr-reader`), before any @@ -505,7 +548,9 @@ fill-value item that did is fixed). short"); `bad_nbit_parms_walk.h5` has an N-Bit parameter list one value short (libhdf5's own `test_filter_bad_params` in `test/dsets.c` now requires that read to fail). We refuse both; `CONFORMANCE.md` lists them - under *Known not-our-bug*. Scale-offset did decode three cases + under *Known not-our-bug* (since 2026-09-27: the *ref-bug* class, with + the evidence re-checked every run — see *Conformance: the last non-ok + files* above). Scale-offset did decode three cases differently from libhdf5 (codes after a `minval` of recorded size other than 8, `minbits` 0 with a fill value, full-width `minbits`): fixed 2026-09-26, and the full-width case was silent wrong data on ordinary