conformance: compare h5py's big-endian VL values corrected, confirm libhdf5 over-reads per run
The last 3 our-errors and 2 mismatches were documented as not ours but
still counted against us, on a heuristic (any big-endian VL mismatch) and
a fixed list.
- ref.py checks that the installed h5py returns big-endian VL elements
with the file's bytes under a little-endian dtype (writing and reading
a vlen('>f4') in memory) and, if so, relabels them with the file's byte
order before hashing, marking the object `ref_fix`. The values are now
compared: attr_datatypes.hdf5 /@vlen_uint64 and tcomplex_be.h5
/VariableLengthDatasetFloatComplex are identical to ours (h5dump 1.14.6
prints the same (1, 2), (3, 4, 5), (42)).
- ref_bugs.py re-reads each object h5py reads only through a libhdf5 bug
in six processes with different heaps (import order, MALLOC_PERTURB_).
Values the file determines are the same every time; these three change
(6, 6 and 3 distinct results), so they are over-read memory, not data
clawhdf5 could match. compare.py classifies a file `ref-bug` only when
every difference is such an object confirmed in the same run.
- report.py: the ref-bug class, the evidence table, the corrected
objects; test_ref.py covers both (run in the nightly job).
Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
+47
-59
@@ -26,7 +26,7 @@ except Exception: # noqa: BLE001
|
||||
R, OUT_MD, CORPUS = sys.argv[1], sys.argv[2], sys.argv[3]
|
||||
HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
ROOT = os.path.dirname(HERE)
|
||||
CLASSES = ["ok", "our-error", "mismatch", "h5py-cannot-read", "panic", "hang", "crash", "oom"]
|
||||
CLASSES = ["ok", "our-error", "mismatch", "h5py-cannot-read", "ref-bug", "panic", "hang", "crash", "oom"]
|
||||
|
||||
|
||||
def sh(*cmd, cwd=ROOT):
|
||||
@@ -89,46 +89,15 @@ def ex_list(files, n=3):
|
||||
return s + (f" (+{len(files) - n} more)" if len(files) > n else "")
|
||||
|
||||
|
||||
# --- known causes that are not clawhdf5 bugs --------------------------------
|
||||
def is_h5py_be_vlen(i):
|
||||
"""h5py returns the elements of a VL sequence of a big-endian base type
|
||||
with their file (big-endian) bytes but a native-endian dtype."""
|
||||
return (i["kind"] == "mismatch" and i["key"] in ("values", "attr-values")
|
||||
and (i.get("ref_dtype") == "object") and (i.get("ours_dtype") or "").startswith("vlen(")
|
||||
and ">" in (i.get("ours_dtype") or ""))
|
||||
|
||||
|
||||
# Objects the reference (h5py 3.16 / HDF5 2.0) reads only because of an
|
||||
# HDF5 2.0 bug, and that clawhdf5 refuses: each one reads past a buffer or
|
||||
# returns bytes the file does not hold, and libhdf5's develop branch refuses all
|
||||
# three. (file, object) -> why. Checked 2026-09-26 against HDF5 2.0.0
|
||||
# and HDFGroup/hdf5 develop sources; see docs/known-issues.md.
|
||||
LIBHDF5_BUGS = {
|
||||
("cve_hdf5/cvefiles/cve-2025-2308.h5", "/Scale_offset_long_long_data_le"):
|
||||
"scale-offset codes run past the end of the chunk: HDF5 2.0 reads past its buffer; "
|
||||
"libhdf5's develop branch refuses the chunk (\"Buffer too short\")",
|
||||
("cve_hdf5/cvefiles/cve-2025-44904.h5", "/Scale_offset_float_data_le"):
|
||||
"unfiltered chunks of 38 and 37 bytes for 48-byte chunks: HDF5 2.0 fills the rest with "
|
||||
"whatever its buffer held; libhdf5's develop branch refuses them (\"incorrect chunk size returned "
|
||||
"from index for unfiltered chunk\")",
|
||||
("hdf5/test/testfiles/bad_nbit_parms_walk.h5", "/Nbit_int_data_le"):
|
||||
"an N-Bit parameter list one value short: HDF5 2.0 reads past the list; libhdf5's own "
|
||||
"test (`test_filter_bad_params`, test/dsets.c) now requires the read to fail",
|
||||
}
|
||||
|
||||
|
||||
def is_libhdf5_bug(rel, i):
|
||||
return i["kind"] == "our-error" and any(
|
||||
f == rel and i["detail"].startswith(obj + ":") for (f, obj) in LIBHDF5_BUGS)
|
||||
|
||||
|
||||
known = collections.defaultdict(list)
|
||||
for r in rows:
|
||||
iss = issues.get(r["file"], [])
|
||||
if r["class"] == "mismatch" and iss and all(is_h5py_be_vlen(i) for i in iss):
|
||||
known["h5py-be-vlen"].append(r["file"])
|
||||
if r["class"] == "our-error" and iss and all(is_libhdf5_bug(r["file"], i) for i in iss):
|
||||
known["libhdf5-2.0"].append(r["file"])
|
||||
# --- reference bugs ---------------------------------------------------------
|
||||
# ref_bugs.py's re-check of the objects h5py reads only through a libhdf5 bug
|
||||
# (compare.py classifies on the confirmed ones), and the objects whose h5py
|
||||
# values ref.py corrected (compare.py's ref_fixes).
|
||||
try:
|
||||
ref_bugs = json.load(open(os.path.join(R, "ref_bugs.json")))["read_bugs"]
|
||||
except (OSError, ValueError, KeyError):
|
||||
ref_bugs = []
|
||||
ref_fixes = res.get("ref_fixes", [])
|
||||
|
||||
|
||||
# --- the CVE corpus: clawhdf5 vs h5dump vs h5py ------------------------------
|
||||
@@ -243,6 +212,7 @@ w("A file's class is the first that applies:")
|
||||
w("")
|
||||
w("- **panic / hang / crash / oom** — clawhdf5 panicked (caught per object or not), hit the timeout, died on a signal, or failed an allocation. The CI gate fails on any of these.")
|
||||
w("- **h5py-cannot-read** — libhdf5 could not open the file (or itself crashed or hung). Nothing to compare against; most are the deliberately malformed CVE reproducers.")
|
||||
w("- **ref-bug** — every difference is an object clawhdf5 refuses that h5py reads only through a libhdf5 bug: the values h5py returns for it change with the reading process's heap, re-checked in every run (see *Reference bugs*).")
|
||||
w("- **our-error** — clawhdf5 returned an error for something h5py reads.")
|
||||
w("- **mismatch** — both read it, but the shapes, values, object set or attribute set differ.")
|
||||
w("- **ok** — every object h5py reads, clawhdf5 reads identically.")
|
||||
@@ -254,14 +224,13 @@ for c in sorted(by_corpus):
|
||||
w(f"| {c} | {sum(cnt.values())} | " + " | ".join(str(cnt.get(k, 0)) for k in CLASSES) + " |")
|
||||
w(f"| **all** | **{len(rows)}** | " + " | ".join(f"**{total.get(k, 0)}**" for k in CLASSES) + " |")
|
||||
w("")
|
||||
if known["h5py-be-vlen"]:
|
||||
w(f"{len(known['h5py-be-vlen'])} of the {total.get('mismatch', 0)} mismatches are a known h5py bug, "
|
||||
"not ours (see *Known not-our-bug*).")
|
||||
w("")
|
||||
if known["libhdf5-2.0"]:
|
||||
w(f"{len(known['libhdf5-2.0'])} of the {total.get('our-error', 0)} our-errors are corrupt data that "
|
||||
"HDF5 2.0 reads only through a bug and clawhdf5 refuses (see *Known not-our-bug*).")
|
||||
w("")
|
||||
nonok = total.get("our-error", 0) + total.get("mismatch", 0)
|
||||
w(f"**Our errors and mismatches: {nonok}.** "
|
||||
+ ("Every file clawhdf5 does not read like h5py is either unreadable by h5py or a confirmed libhdf5 "
|
||||
"bug (*ref-bug*)." if nonok == 0 else "See the root causes below.")
|
||||
+ (f" {len(ref_fixes)} object(s) were compared against h5py's values corrected for a known h5py bug"
|
||||
f" ({sum(1 for x in ref_fixes if x[3])} identical to clawhdf5's; see *Reference bugs*)." if ref_fixes else ""))
|
||||
w("")
|
||||
w("Corpora (fetched by `conformance/fetch-corpus.sh` into the gitignored `conformance/.cache/`):")
|
||||
w("")
|
||||
w("| corpus | source | commit |")
|
||||
@@ -321,14 +290,38 @@ w("")
|
||||
w("</details>")
|
||||
w("")
|
||||
|
||||
w("## Known not-our-bug")
|
||||
w("## Reference bugs")
|
||||
w("")
|
||||
w("### Objects h5py reads only through a libhdf5 bug (*ref-bug*)")
|
||||
w("")
|
||||
w("clawhdf5 refuses these objects; h5py 3.16 / HDF5 2.0 returns values for them. `conformance/ref_bugs.py`")
|
||||
w("re-reads each with h5py in six fresh processes whose heaps differ (h5py imported before numpy, three")
|
||||
w("times and twice more with `MALLOC_PERTURB_`, and numpy imported first). Values the file determines")
|
||||
w("come out the same every time; these do not, so they are memory libhdf5 over-reads, not the file's")
|
||||
w("data. A file is *ref-bug* only while every one of its differences is such an object confirmed in")
|
||||
w("the same run; an object that reads the same every time goes back to *our-error*. Reproducer:")
|
||||
w("`python conformance/ref_bugs.py conformance/.cache/corpus` (prints every read's outcome).")
|
||||
w("")
|
||||
w("| file | object | distinct results in 6 reads | confirmed | what goes wrong |")
|
||||
w("|---|---|---:|---|---|")
|
||||
for b in ref_bugs:
|
||||
n = "missing" if b.get("missing") else b.get("distinct", "?")
|
||||
w(f"| `{b['file']}` | `{b['object']}` | {n} | {'yes' if b.get('confirmed') else '**no**'} | {b['why']} |")
|
||||
w("")
|
||||
w("### Values corrected for a known h5py bug")
|
||||
w("")
|
||||
w("- **h5py big-endian variable-length sequences.** h5py returns the elements of a VL sequence")
|
||||
w(" whose base type is big-endian with the file's big-endian bytes but a native (little-endian)")
|
||||
w(" numpy dtype, so the values it reports are byte-swapped garbage; `h5dump` prints the values")
|
||||
w(" clawhdf5 reads. Reproducer: `h5py.vlen_dtype(np.dtype('>f4'))` dataset holding `[1.0, 2.0]`")
|
||||
w(" reads back in h5py as `[4.6e-41, 9.0e-44]`. Affected here: "
|
||||
+ (ex_list(sorted(known["h5py-be-vlen"]), 10) if known["h5py-be-vlen"] else "none") + ".")
|
||||
w(" numpy dtype: a `h5py.vlen_dtype(np.dtype('>f4'))` dataset holding `[1.0, 2.0]` reads back as")
|
||||
w(" `[4.6e-41, 9.0e-44]`; `h5dump` prints the file's values. `ref.py` checks that the installed")
|
||||
w(" h5py still does this (by writing and reading exactly that dataset in memory) and, if so,")
|
||||
w(" relabels such elements with the file's byte order before hashing, so the values are still")
|
||||
w(" compared. Corrected objects: "
|
||||
+ (", ".join(f"`{f}` `{p}` ({'same as clawhdf5' if same else '**differs from clawhdf5**'})"
|
||||
for f, p, _, same in ref_fixes) if ref_fixes else "none") + ".")
|
||||
w("")
|
||||
w("## Other comparison rules")
|
||||
w("")
|
||||
w("- **Non-IEEE floats and partial-precision integers (N-Bit).** libhdf5 converts a float whose")
|
||||
w(" bit layout is not IEEE (e.g. `H5Tset_precision` for the N-Bit filter) or an integer with a")
|
||||
w(" bit offset / reduced precision into the plain numpy type of the same size. The probe")
|
||||
@@ -339,11 +332,6 @@ if res["incomparable"]:
|
||||
w(" (FP8 -> float16, bfloat16 -> float32, x87 long double -> float128) the values are not")
|
||||
w(" compared (shape and presence still are): "
|
||||
+ ", ".join(f"{k} ({n}x)" for k, n in res["incomparable"]) + ".")
|
||||
w("- **Corrupt data HDF5 2.0 reads through a bug.** clawhdf5 refuses these objects; h5py 3.16 /")
|
||||
w(" HDF5 2.0 returns values for them that the file does not hold:")
|
||||
for (f, obj), why in sorted(LIBHDF5_BUGS.items()):
|
||||
here = "" if f in known["libhdf5-2.0"] else " (not an our-error in this run)"
|
||||
w(f" - `{f}` `{obj}`: {why}{here}.")
|
||||
w("- **References** are compared by presence only (`R`), not by target.")
|
||||
w("")
|
||||
if res.get("ref_only_errors"):
|
||||
|
||||
Reference in New Issue
Block a user