From 5c44630ea22a708ae3dc46dcb2e6faba9fc693f2 Mon Sep 17 00:00:00 2001 From: osobh Date: Sun, 27 Sep 2026 22:26:32 -0500 Subject: [PATCH 01/10] conformance: compare h5py's big-endian VL values corrected, confirm libhdf5 over-reads per run The last 3 our-errors and 2 mismatches were documented as not ours but still counted against us, on a heuristic (any big-endian VL mismatch) and a fixed list. - ref.py checks that the installed h5py returns big-endian VL elements with the file's bytes under a little-endian dtype (writing and reading a vlen('>f4') in memory) and, if so, relabels them with the file's byte order before hashing, marking the object `ref_fix`. The values are now compared: attr_datatypes.hdf5 /@vlen_uint64 and tcomplex_be.h5 /VariableLengthDatasetFloatComplex are identical to ours (h5dump 1.14.6 prints the same (1, 2), (3, 4, 5), (42)). - ref_bugs.py re-reads each object h5py reads only through a libhdf5 bug in six processes with different heaps (import order, MALLOC_PERTURB_). Values the file determines are the same every time; these three change (6, 6 and 3 distinct results), so they are over-read memory, not data clawhdf5 could match. compare.py classifies a file `ref-bug` only when every difference is such an object confirmed in the same run. - report.py: the ref-bug class, the evidence table, the corrected objects; test_ref.py covers both (run in the nightly job). Co-Authored-By: Claude Opus 5.5 (1M context) --- .gitea/workflows/conformance.yml | 2 + conformance/README.md | 6 +- conformance/compare.py | 30 ++++++++- conformance/ref.py | 54 +++++++++++++++- conformance/ref_bugs.py | 105 ++++++++++++++++++++++++++++++ conformance/report.py | 106 ++++++++++++++----------------- conformance/run.sh | 2 + conformance/test_ref.py | 92 +++++++++++++++++++++++++++ docs/known-issues.md | 47 +++++++++++++- 9 files changed, 379 insertions(+), 65 deletions(-) create mode 100644 conformance/ref_bugs.py create mode 100644 conformance/test_ref.py diff --git a/.gitea/workflows/conformance.yml b/.gitea/workflows/conformance.yml index f0b51e3..3317d21 100644 --- a/.gitea/workflows/conformance.yml +++ b/.gitea/workflows/conformance.yml @@ -40,6 +40,8 @@ jobs: run: cargo test --release --manifest-path conformance/probe/Cargo.toml env: CARGO_TARGET_DIR: conformance/.cache/target + - name: Reference-side tests + run: /opt/conformance/bin/python conformance/test_ref.py - name: Sweep # The corpora come from GitHub (pinned commits, conformance/corpus.txt), # so this job needs a runner that reaches github.com. diff --git a/conformance/README.md b/conformance/README.md index 5f08558..919d1fe 100644 --- a/conformance/README.md +++ b/conformance/README.md @@ -19,9 +19,11 @@ probe's `szip` feature; `libaec-dev`), and a Python with the packages in | `fetch-corpus.sh` | shallow, sparse, blob-filtered checkout of each pinned commit into `.cache/src/` (gitignored); no-op when already there | | `list_files.py` | which files are probed (HDF5/netCDF-4 extensions minus netCDF classic, plus the CVE reproducers) | | `probe/` | the clawhdf5 side: a standalone crate (outside the workspace, so `cargo test --workspace` never builds it) that walks a file with `clawhdf5-format` and prints canonical JSON | -| `ref.py` | the h5py side: the same JSON from h5py | +| `ref.py` | the h5py side: the same JSON from h5py (values corrected for a known h5py bug are marked `ref_fix`) | +| `ref_bugs.py` | re-reads the objects h5py reads only through a libhdf5 bug in six differently-set-up processes; an object whose values change is confirmed as a libhdf5 over-read | +| `test_ref.py` | tests of `ref.py`'s correction and `ref_bugs.py`'s confirmation (`python conformance/test_ref.py`) | | `run_one.sh` | runs both sides on one file (and `h5dump` on the CVE corpus) under a timeout and an address-space limit | -| `compare.py` | classifies each file (ok / our-error / mismatch / h5py-cannot-read / panic / hang / crash / oom) and groups root causes | +| `compare.py` | classifies each file (ok / our-error / mismatch / h5py-cannot-read / ref-bug / panic / hang / crash / oom) and groups root causes | | `report.py` | writes `CONFORMANCE.md` | | `check.py` | the gate: fails on any panic/hang/crash/oom, on an ok count below `baseline.json`, or on a baseline-ok file that is no longer ok | | `baseline.json` | the ok files the gate holds the line on | diff --git a/conformance/compare.py b/conformance/compare.py index c1d65b2..b3c944a 100755 --- a/conformance/compare.py +++ b/conformance/compare.py @@ -5,6 +5,9 @@ Writes /results.csv, results.json and summary.md. File classes (first match wins): hang, oom, crash, panic ours: timeout / allocation failure / signal / any panic (caught or not) h5py-cannot-read libhdf5/h5py failed to open the file (or crashed/hung) + ref-bug every issue is an object we refuse that h5py reads only through a + libhdf5 bug, confirmed in this run by ref_bugs.py (its values + change with the reading process's heap) our-error we fail to open, list, or read something h5py reads mismatch we read something with different shape/values, or a different object set ok @@ -19,6 +22,20 @@ import sys R = sys.argv[1] RUNS = os.path.join(R, "runs") +# Objects ref_bugs.py confirmed in this run: h5py's values for them come from +# libhdf5 reading memory the file does not determine. +try: + REF_BUGS = {(b["file"], b["object"]) + for b in json.load(open(os.path.join(R, "ref_bugs.json")))["read_bugs"] if b.get("confirmed")} +except (OSError, ValueError, KeyError): + REF_BUGS = set() + + +def is_ref_bug(rel, issue): + """An our-error on reading an object that ref_bugs.py confirmed.""" + kind, detail = issue[0], issue[1] + return kind == "our-error" and any(f == rel and detail.startswith(obj + ": error: ") for f, obj in REF_BUGS) + def load(d, name): rc_p = os.path.join(d, name + ".rc") @@ -86,6 +103,8 @@ mismatch_causes = collections.defaultdict(lambda: {"files": set(), "count": 0, " panics = [] ref_only_errors = collections.Counter() incomparable = collections.Counter() +# Values ref.py corrected for a known h5py bug: (file, object, fixes, same as ours) +ref_fixes = [] def add(bucket, key, file, example): @@ -169,6 +188,8 @@ for rel in files: elif a.get("hash") != b.get("hash"): issues.append(("mismatch", f"{p}: values differ (h5py {a.get('dtype')} vs ours {b.get('dtype')})", "values", b | {"ref_head": a.get("head"), "ref_dtype": a.get("dtype")})) ok = False + if a.get("ref_fix") and "hash" in b: + ref_fixes.append((rel, p, a["ref_fix"], a.get("hash") == b.get("hash"))) ra, oa = a.get("attrs") or {}, b.get("attrs") or {} if "attrs_error" not in b and "attrs_error" not in a and not ref_unopened: for an in sorted(set(ra) | set(oa)): @@ -187,6 +208,8 @@ for rel in files: issues.append(("mismatch", f"{p}@{an}: attr shape {x.get('shape')} vs ours {y.get('shape')}", "attr-shape", y | {"ref_dtype": x.get("dtype")})) elif x.get("hash") != y.get("hash"): issues.append(("mismatch", f"{p}@{an}: attr values differ (h5py {x.get('dtype')} vs ours {y.get('dtype')})", "attr-values", y | {"ref_head": x.get("head"), "ref_dtype": x.get("dtype")})) + if x.get("ref_fix") and "hash" in y: + ref_fixes.append((rel, f"{p}@{an}", x["ref_fix"], x.get("hash") == y.get("hash"))) if ok: n_ok += 1 @@ -197,6 +220,8 @@ for rel in files: cls = "panic" elif ref_open_fail: cls = "h5py-cannot-read" + elif issues and all(is_ref_bug(rel, i) for i in issues): + cls = "ref-bug" elif ours_open_err: cls = "our-error" issues.append(("our-error", f"open: {ours_open_err}", ours_open_err, {})) @@ -259,10 +284,11 @@ def ser(b): json.dump({"rows": rows, "issues": issues_by_file, "root_causes": ser(root_causes), "mismatch_causes": ser(mismatch_causes), - "panics": panics, "incomparable": incomparable.most_common(), "ref_only_errors": ref_only_errors.most_common()}, + "panics": panics, "incomparable": incomparable.most_common(), "ref_only_errors": ref_only_errors.most_common(), + "ref_fixes": ref_fixes, "ref_bugs_confirmed": sorted(REF_BUGS)}, open(os.path.join(R, "results.json"), "w"), indent=1) -classes = ["ok", "our-error", "mismatch", "h5py-cannot-read", "hang", "panic", "crash", "oom"] +classes = ["ok", "our-error", "mismatch", "h5py-cannot-read", "ref-bug", "hang", "panic", "crash", "oom"] by_corpus = collections.defaultdict(collections.Counter) for r in rows: by_corpus[r["corpus"]][r["class"]] += 1 diff --git a/conformance/ref.py b/conformance/ref.py index 428b6a5..499fc3a 100755 --- a/conformance/ref.py +++ b/conformance/ref.py @@ -53,6 +53,49 @@ def packed(dt): return dt +# --- reference corrections --------------------------------------------------- +# Where h5py is known to return values the file does not hold, and the right +# values follow from what it returned, ref.py corrects them and records the +# correction on the object ("ref_fix"), so the comparison is still a real +# comparison and CONFORMANCE.md lists every corrected object. Each correction +# first checks that the installed h5py still has the bug. + +# Corrections applied while encoding the current object. +FIXES = set() +_BE_VLEN_BUG = None + + +def be_vlen_bug(): + """h5py (3.16 / HDF5 2.0 at least) returns the elements of a + variable-length sequence whose base type is big-endian with the file's + big-endian bytes under a native (little-endian) dtype: a + `vlen_dtype('>f4')` dataset holding [1.0, 2.0] reads back as + [4.6e-41, 9.0e-44]. `h5dump` prints the file's values. Checked once per + process by writing and reading exactly that dataset in memory.""" + global _BE_VLEN_BUG + if _BE_VLEN_BUG is None: + import io + try: + bio = io.BytesIO() + with h5py.File(bio, "w") as f: + d = f.create_dataset("v", (1,), dtype=h5py.vlen_dtype(np.dtype(">f4"))) + d[0] = np.array([1.0, 2.0], dtype=">f4") + with h5py.File(bio, "r") as f: + got = np.asarray(f["v"][0]) + _BE_VLEN_BUG = (got.dtype == np.dtype("f4").tolist() == [1.0, 2.0] + and got.tolist() != [1.0, 2.0]) + except Exception: # noqa: BLE001 + _BE_VLEN_BUG = False + return _BE_VLEN_BUG + + +def unswapped(got, base): + """`got` is `base` (big-endian somewhere) with every field in native + little-endian order instead: the shape of h5py's big-endian VL bug.""" + return base.newbyteorder("<") == got and base != got + + def canon_el(dt, val, out): if dt.fields: for n in dt.names: @@ -79,7 +122,13 @@ def canon_el(dt, val, out): base = h5py.check_vlen_dtype(dt) if base is None: raise TypeError(f"unhandled object dtype {dt!r}") - arr = np.asarray(val if val is not None else [], dtype=base).reshape(-1) + arr = np.asarray(val if val is not None else []) + if arr.dtype != base and be_vlen_bug() and unswapped(arr.dtype, base): + # h5py's big-endian VL bug (see be_vlen_bug): the bytes are + # the file's, the dtype label is wrong. Relabel, don't convert. + arr = arr.view(base) + FIXES.add("h5py-be-vlen") + arr = np.asarray(arr, dtype=base).reshape(-1) out += b"V" + struct.pack(": re-check the objects h5py reads only through a +libhdf5 bug. + +For each object of READ_BUGS (below), h5py reads it in several fresh +processes whose heaps differ: h5py imported before numpy (three runs, plus +two with glibc's MALLOC_PERTURB_, which fills newly allocated and freed heap +blocks with a byte pattern) and numpy imported first. Values the file +determines come out the same every time. An object whose values differ +between those runs is read from memory the file does not determine — an +over-read or an uninitialised buffer in libhdf5 — so the values h5py reports +for it are not the file's, and clawhdf5 refusing the object is not a +clawhdf5 error. compare.py classifies a file as `ref-bug` only on objects +confirmed that way in the same run (`$OUT/ref_bugs.json`); an object whose +reading turns out stable stays an our-error. + +Run by conformance/run.sh; on its own it is the reproducer (JSON on stdout). +""" +import concurrent.futures +import json +import os +import subprocess +import sys + +# (file, object) -> what goes wrong. Checked 2026-09-27 against HDF5 2.0.0 +# (h5py 3.16), h5dump 1.14.6 and the HDFGroup/hdf5 sources (tag hdf5_1_14_6 +# and develop); see docs/known-issues.md, "Conformance: the last non-ok files". +READ_BUGS = { + ("cve_hdf5/cvefiles/cve-2025-2308.h5", "/Scale_offset_long_long_data_le"): + "the first chunk records minbits 11: its 12 values need 17 bytes of codes, and the " + "26-byte chunk holds 5 after its 21-byte header; libhdf5's scale-offset decoder reads " + "past its buffer, and develop refuses the chunk (\"Buffer too short\")", + ("cve_hdf5/cvefiles/cve-2025-44904.h5", "/Scale_offset_float_data_le"): + "unfiltered chunks stored as 38 and 37 bytes for 48-byte chunks: 1.14/2.0 read the " + "stored bytes into a buffer of that size and use it as the whole chunk " + "(H5D__chunk_lock), so the rest is heap memory; develop refuses them (\"incorrect chunk " + "size returned from index for unfiltered chunk\")", + ("hdf5/test/testfiles/bad_nbit_parms_walk.h5", "/Nbit_int_data_le"): + "the N-Bit parameter list holds 7 values (cd_values[0] = 7) where an integer needs 8: " + "the decoder takes the bit offset from cd_values[7], past the list; libhdf5's own test " + "(`test_filter_bad_params`, test/dsets.c on develop) requires the read to fail", +} + +# (which module is imported first, MALLOC_PERTURB_) +RUNS = [("h5py", None), ("h5py", None), ("h5py", None), ("h5py", "170"), ("h5py", "255"), + ("numpy", None)] + +READ = r""" +import hashlib, sys +if sys.argv[3] == "h5py": + import h5py, numpy as np +else: + import numpy as np, h5py +try: + import hdf5plugin # noqa: F401 +except Exception: + pass +try: + with h5py.File(sys.argv[1], "r") as f: + a = np.ascontiguousarray(f[sys.argv[2]][()]) + print("values " + hashlib.sha256(a.tobytes()).hexdigest()[:16]) +except Exception as e: + print("error " + (str(e).splitlines() or [type(e).__name__])[0][:120]) +""" + + +def read_once(path, obj, first, perturb): + env = dict(os.environ) + env.pop("MALLOC_PERTURB_", None) + if perturb: + env["MALLOC_PERTURB_"] = perturb + try: + p = subprocess.run([sys.executable, "-c", READ, path, obj, first], env=env, + capture_output=True, text=True, timeout=60) + out = p.stdout.strip().splitlines() + return out[-1] if out else f"exit {p.returncode}" + except subprocess.TimeoutExpired: + return "timeout" + + +def check(corpus, key): + f, obj = key + path = os.path.join(corpus, f) + rec = {"file": f, "object": obj, "why": READ_BUGS[key]} + if not os.path.exists(path): + return rec | {"missing": True, "confirmed": False} + runs = [{"first": a, "malloc_perturb": p, "outcome": read_once(path, obj, a, p)} for a, p in RUNS] + distinct = sorted({r["outcome"] for r in runs}) + return rec | { + "runs": runs, + "distinct": len(distinct), + "confirmed": len(distinct) > 1 and any(o.startswith("values ") for o in distinct), + } + + +def main(): + corpus = sys.argv[1] + keys = list(READ_BUGS) + with concurrent.futures.ThreadPoolExecutor(max_workers=len(keys)) as ex: + out = list(ex.map(lambda k: check(corpus, k), keys)) + print(json.dumps({"read_bugs": out}, indent=1)) + + +if __name__ == "__main__": + main() diff --git a/conformance/report.py b/conformance/report.py index 85ad0aa..6c4cb13 100644 --- a/conformance/report.py +++ b/conformance/report.py @@ -26,7 +26,7 @@ except Exception: # noqa: BLE001 R, OUT_MD, CORPUS = sys.argv[1], sys.argv[2], sys.argv[3] HERE = os.path.dirname(os.path.abspath(__file__)) ROOT = os.path.dirname(HERE) -CLASSES = ["ok", "our-error", "mismatch", "h5py-cannot-read", "panic", "hang", "crash", "oom"] +CLASSES = ["ok", "our-error", "mismatch", "h5py-cannot-read", "ref-bug", "panic", "hang", "crash", "oom"] def sh(*cmd, cwd=ROOT): @@ -89,46 +89,15 @@ def ex_list(files, n=3): return s + (f" (+{len(files) - n} more)" if len(files) > n else "") -# --- known causes that are not clawhdf5 bugs -------------------------------- -def is_h5py_be_vlen(i): - """h5py returns the elements of a VL sequence of a big-endian base type - with their file (big-endian) bytes but a native-endian dtype.""" - return (i["kind"] == "mismatch" and i["key"] in ("values", "attr-values") - and (i.get("ref_dtype") == "object") and (i.get("ours_dtype") or "").startswith("vlen(") - and ">" in (i.get("ours_dtype") or "")) - - -# Objects the reference (h5py 3.16 / HDF5 2.0) reads only because of an -# HDF5 2.0 bug, and that clawhdf5 refuses: each one reads past a buffer or -# returns bytes the file does not hold, and libhdf5's develop branch refuses all -# three. (file, object) -> why. Checked 2026-09-26 against HDF5 2.0.0 -# and HDFGroup/hdf5 develop sources; see docs/known-issues.md. -LIBHDF5_BUGS = { - ("cve_hdf5/cvefiles/cve-2025-2308.h5", "/Scale_offset_long_long_data_le"): - "scale-offset codes run past the end of the chunk: HDF5 2.0 reads past its buffer; " - "libhdf5's develop branch refuses the chunk (\"Buffer too short\")", - ("cve_hdf5/cvefiles/cve-2025-44904.h5", "/Scale_offset_float_data_le"): - "unfiltered chunks of 38 and 37 bytes for 48-byte chunks: HDF5 2.0 fills the rest with " - "whatever its buffer held; libhdf5's develop branch refuses them (\"incorrect chunk size returned " - "from index for unfiltered chunk\")", - ("hdf5/test/testfiles/bad_nbit_parms_walk.h5", "/Nbit_int_data_le"): - "an N-Bit parameter list one value short: HDF5 2.0 reads past the list; libhdf5's own " - "test (`test_filter_bad_params`, test/dsets.c) now requires the read to fail", -} - - -def is_libhdf5_bug(rel, i): - return i["kind"] == "our-error" and any( - f == rel and i["detail"].startswith(obj + ":") for (f, obj) in LIBHDF5_BUGS) - - -known = collections.defaultdict(list) -for r in rows: - iss = issues.get(r["file"], []) - if r["class"] == "mismatch" and iss and all(is_h5py_be_vlen(i) for i in iss): - known["h5py-be-vlen"].append(r["file"]) - if r["class"] == "our-error" and iss and all(is_libhdf5_bug(r["file"], i) for i in iss): - known["libhdf5-2.0"].append(r["file"]) +# --- reference bugs --------------------------------------------------------- +# ref_bugs.py's re-check of the objects h5py reads only through a libhdf5 bug +# (compare.py classifies on the confirmed ones), and the objects whose h5py +# values ref.py corrected (compare.py's ref_fixes). +try: + ref_bugs = json.load(open(os.path.join(R, "ref_bugs.json")))["read_bugs"] +except (OSError, ValueError, KeyError): + ref_bugs = [] +ref_fixes = res.get("ref_fixes", []) # --- the CVE corpus: clawhdf5 vs h5dump vs h5py ------------------------------ @@ -243,6 +212,7 @@ w("A file's class is the first that applies:") w("") w("- **panic / hang / crash / oom** — clawhdf5 panicked (caught per object or not), hit the timeout, died on a signal, or failed an allocation. The CI gate fails on any of these.") w("- **h5py-cannot-read** — libhdf5 could not open the file (or itself crashed or hung). Nothing to compare against; most are the deliberately malformed CVE reproducers.") +w("- **ref-bug** — every difference is an object clawhdf5 refuses that h5py reads only through a libhdf5 bug: the values h5py returns for it change with the reading process's heap, re-checked in every run (see *Reference bugs*).") w("- **our-error** — clawhdf5 returned an error for something h5py reads.") w("- **mismatch** — both read it, but the shapes, values, object set or attribute set differ.") w("- **ok** — every object h5py reads, clawhdf5 reads identically.") @@ -254,14 +224,13 @@ for c in sorted(by_corpus): w(f"| {c} | {sum(cnt.values())} | " + " | ".join(str(cnt.get(k, 0)) for k in CLASSES) + " |") w(f"| **all** | **{len(rows)}** | " + " | ".join(f"**{total.get(k, 0)}**" for k in CLASSES) + " |") w("") -if known["h5py-be-vlen"]: - w(f"{len(known['h5py-be-vlen'])} of the {total.get('mismatch', 0)} mismatches are a known h5py bug, " - "not ours (see *Known not-our-bug*).") - w("") -if known["libhdf5-2.0"]: - w(f"{len(known['libhdf5-2.0'])} of the {total.get('our-error', 0)} our-errors are corrupt data that " - "HDF5 2.0 reads only through a bug and clawhdf5 refuses (see *Known not-our-bug*).") - w("") +nonok = total.get("our-error", 0) + total.get("mismatch", 0) +w(f"**Our errors and mismatches: {nonok}.** " + + ("Every file clawhdf5 does not read like h5py is either unreadable by h5py or a confirmed libhdf5 " + "bug (*ref-bug*)." if nonok == 0 else "See the root causes below.") + + (f" {len(ref_fixes)} object(s) were compared against h5py's values corrected for a known h5py bug" + f" ({sum(1 for x in ref_fixes if x[3])} identical to clawhdf5's; see *Reference bugs*)." if ref_fixes else "")) +w("") w("Corpora (fetched by `conformance/fetch-corpus.sh` into the gitignored `conformance/.cache/`):") w("") w("| corpus | source | commit |") @@ -321,14 +290,38 @@ w("") w("") w("") -w("## Known not-our-bug") +w("## Reference bugs") +w("") +w("### Objects h5py reads only through a libhdf5 bug (*ref-bug*)") +w("") +w("clawhdf5 refuses these objects; h5py 3.16 / HDF5 2.0 returns values for them. `conformance/ref_bugs.py`") +w("re-reads each with h5py in six fresh processes whose heaps differ (h5py imported before numpy, three") +w("times and twice more with `MALLOC_PERTURB_`, and numpy imported first). Values the file determines") +w("come out the same every time; these do not, so they are memory libhdf5 over-reads, not the file's") +w("data. A file is *ref-bug* only while every one of its differences is such an object confirmed in") +w("the same run; an object that reads the same every time goes back to *our-error*. Reproducer:") +w("`python conformance/ref_bugs.py conformance/.cache/corpus` (prints every read's outcome).") +w("") +w("| file | object | distinct results in 6 reads | confirmed | what goes wrong |") +w("|---|---|---:|---|---|") +for b in ref_bugs: + n = "missing" if b.get("missing") else b.get("distinct", "?") + w(f"| `{b['file']}` | `{b['object']}` | {n} | {'yes' if b.get('confirmed') else '**no**'} | {b['why']} |") +w("") +w("### Values corrected for a known h5py bug") w("") w("- **h5py big-endian variable-length sequences.** h5py returns the elements of a VL sequence") w(" whose base type is big-endian with the file's big-endian bytes but a native (little-endian)") -w(" numpy dtype, so the values it reports are byte-swapped garbage; `h5dump` prints the values") -w(" clawhdf5 reads. Reproducer: `h5py.vlen_dtype(np.dtype('>f4'))` dataset holding `[1.0, 2.0]`") -w(" reads back in h5py as `[4.6e-41, 9.0e-44]`. Affected here: " - + (ex_list(sorted(known["h5py-be-vlen"]), 10) if known["h5py-be-vlen"] else "none") + ".") +w(" numpy dtype: a `h5py.vlen_dtype(np.dtype('>f4'))` dataset holding `[1.0, 2.0]` reads back as") +w(" `[4.6e-41, 9.0e-44]`; `h5dump` prints the file's values. `ref.py` checks that the installed") +w(" h5py still does this (by writing and reading exactly that dataset in memory) and, if so,") +w(" relabels such elements with the file's byte order before hashing, so the values are still") +w(" compared. Corrected objects: " + + (", ".join(f"`{f}` `{p}` ({'same as clawhdf5' if same else '**differs from clawhdf5**'})" + for f, p, _, same in ref_fixes) if ref_fixes else "none") + ".") +w("") +w("## Other comparison rules") +w("") w("- **Non-IEEE floats and partial-precision integers (N-Bit).** libhdf5 converts a float whose") w(" bit layout is not IEEE (e.g. `H5Tset_precision` for the N-Bit filter) or an integer with a") w(" bit offset / reduced precision into the plain numpy type of the same size. The probe") @@ -339,11 +332,6 @@ if res["incomparable"]: w(" (FP8 -> float16, bfloat16 -> float32, x87 long double -> float128) the values are not") w(" compared (shape and presence still are): " + ", ".join(f"{k} ({n}x)" for k, n in res["incomparable"]) + ".") -w("- **Corrupt data HDF5 2.0 reads through a bug.** clawhdf5 refuses these objects; h5py 3.16 /") -w(" HDF5 2.0 returns values for them that the file does not hold:") -for (f, obj), why in sorted(LIBHDF5_BUGS.items()): - here = "" if f in known["libhdf5-2.0"] else " (not an our-error in this run)" - w(f" - `{f}` `{obj}`: {why}{here}.") w("- **References** are compared by presence only (`R`), not by target.") w("") if res.get("ref_only_errors"): diff --git a/conformance/run.sh b/conformance/run.sh index 5ceb635..1c4bd85 100755 --- a/conformance/run.sh +++ b/conformance/run.sh @@ -71,6 +71,8 @@ xargs -a "$OUT/files.txt" -d '\n' -P "$JOBS" -I{} bash -c ' f="$1"; d="$OUT/runs/${f//\//__}" case "$f" in cve_hdf5/*) export WITH_H5DUMP=1 ;; esac "$HERE/run_one.sh" "$C/$f" "$d"' _ {} 2>"$OUT/probe.log" +echo "== re-checking the objects h5py reads only through a libhdf5 bug" +"$PY" "$HERE/ref_bugs.py" "$C" > "$OUT/ref_bugs.json" 2> "$OUT/ref_bugs.err" || true echo "== comparing" "$PY" "$HERE/compare.py" "$OUT" >/dev/null t2=$(date +%s) diff --git a/conformance/test_ref.py b/conformance/test_ref.py new file mode 100644 index 0000000..ed2d68f --- /dev/null +++ b/conformance/test_ref.py @@ -0,0 +1,92 @@ +#!/usr/bin/env python3 +"""Tests of the reference side's corrections: `python conformance/test_ref.py`. + +- ref.py compares a big-endian VL sequence by the file's values even though + h5py returns them byte-swapped (and records that it corrected them); +- ref_bugs.py confirms an object only when its reads disagree. +""" +import json +import os +import subprocess +import sys +import tempfile +import unittest + +import h5py +import numpy as np + +HERE = os.path.dirname(os.path.abspath(__file__)) +sys.path.insert(0, HERE) +import ref_bugs # noqa: E402 + + +def ref_objects(path): + out = subprocess.run([sys.executable, os.path.join(HERE, "ref.py"), path], + capture_output=True, text=True, check=True).stdout + return {o["path"]: o for o in json.loads(out)["objects"]} + + +class BigEndianVlen(unittest.TestCase): + def test_be_vlen_compared_by_file_values(self): + with tempfile.TemporaryDirectory() as d: + path = os.path.join(d, "v.h5") + with h5py.File(path, "w") as f: + for name, order in (("be", ">"), ("le", "<")): + t = np.dtype(order + "f4") + ds = f.create_dataset(name, (2,), dtype=h5py.vlen_dtype(t)) + ds[0] = np.array([1.0, 2.0], dtype=t) + ds[1] = np.array([3.0], dtype=t) + u = np.dtype(order + "u8") + f.attrs.create(name, [np.array([1, 2], dtype=u), np.array([42], dtype=u)], + dtype=h5py.vlen_dtype(u)) + objs = ref_objects(path) + be, le = objs["/be"], objs["/le"] + # Same values, so the same canonical hash whatever the file's byte order. + self.assertEqual(be["hash"], le["hash"]) + self.assertEqual(objs["/"]["attrs"]["be"]["hash"], objs["/"]["attrs"]["le"]["hash"]) + self.assertNotIn("ref_fix", le) + # And the correction is recorded wherever h5py needed it. + import ref + if ref.be_vlen_bug(): + self.assertEqual(be.get("ref_fix"), ["h5py-be-vlen"]) + self.assertEqual(objs["/"]["attrs"]["be"].get("ref_fix"), ["h5py-be-vlen"]) + + +class RefBugsConfirmation(unittest.TestCase): + def run_check(self, outcomes): + seq = iter(outcomes) + saved = ref_bugs.read_once + ref_bugs.read_once = lambda *a: next(seq) + try: + key = next(iter(ref_bugs.READ_BUGS)) + with tempfile.TemporaryDirectory() as d: + p = os.path.join(d, key[0]) + os.makedirs(os.path.dirname(p)) + open(p, "wb").close() + return ref_bugs.check(d, key) + finally: + ref_bugs.read_once = saved + + def test_stable_values_are_not_confirmed(self): + r = self.run_check(["values a"] * len(ref_bugs.RUNS)) + self.assertFalse(r["confirmed"]) + + def test_changing_values_are_confirmed(self): + r = self.run_check(["values a"] * (len(ref_bugs.RUNS) - 1) + ["values b"]) + self.assertTrue(r["confirmed"]) + r = self.run_check(["values a"] * (len(ref_bugs.RUNS) - 1) + ["error filter failed"]) + self.assertTrue(r["confirmed"]) + + def test_errors_only_are_not_confirmed(self): + # h5py cannot read it at all: nothing it reads, nothing to excuse. + r = self.run_check(["error x"] * (len(ref_bugs.RUNS) - 1) + ["error y"]) + self.assertFalse(r["confirmed"]) + + def test_missing_file_is_not_confirmed(self): + key = next(iter(ref_bugs.READ_BUGS)) + with tempfile.TemporaryDirectory() as d: + self.assertFalse(ref_bugs.check(d, key)["confirmed"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/docs/known-issues.md b/docs/known-issues.md index 98e1fac..9e572a6 100644 --- a/docs/known-issues.md +++ b/docs/known-issues.md @@ -25,6 +25,49 @@ An earlier run the same day under load also listed single-thread contiguous hyperslab reads as 5.6% slower; the idle rerun puts them at −1.7% with overlapping ranges (noise), so that item is withdrawn. +## Conformance: the last non-ok files (checked 2026-09-27) + +**Status:** classified with evidence — none is a clawhdf5 bug. The +conformance report had 3 our-errors and 2 mismatches left, each documented +as "not ours" but still counted against us. Re-checked on tank +(h5py 3.16 / HDF5 2.0.0, h5dump 1.14.6): + +- **2 mismatches, h5py's big-endian VL bug:** + `NCAS-CMS_pyfive/tests/data/attr_datatypes.hdf5` `/@vlen_uint64` and + `hdf5/tools/test/testfiles/tcomplex_be.h5` + `/VariableLengthDatasetFloatComplex`. h5py returns the elements with the + file's big-endian bytes under a little-endian dtype (a `vlen('>f4')` + holding `[1.0, 2.0]` reads back as `[4.6e-41, 9.0e-44]`). h5dump 1.14.6 + prints `(1, 2), (3, 4, 5), (42)` for `vlen_uint64`, which is what we read; + it cannot print `tcomplex_be.h5` (complex types are HDF5 2.0). The + reference side (`conformance/ref.py`) now checks that the installed h5py + has the bug and relabels such elements with the file's byte order, so the + values are compared rather than excused: both objects are identical to + ours, and both files are now *ok*. +- **3 our-errors, objects HDF5 2.0 reads only by over-reading memory:** + `cve-2025-2308.h5` `/Scale_offset_long_long_data_le` (first chunk records + `minbits` 11; its 12 values need 17 bytes of codes and the 26-byte chunk + holds 5 after its 21-byte header), `cve-2025-44904.h5` + `/Scale_offset_float_data_le` (unfiltered chunks stored as 38 and 37 bytes + for 48-byte chunks; 1.14.6's `H5D__chunk_lock` reads the stored bytes + into a buffer of that size and uses it as the whole chunk) and + `bad_nbit_parms_walk.h5` `/Nbit_int_data_le` (N-Bit parameters + `(7, 0, 40, 1, 4, 0, 20)`: an integer needs 8, and the decoder takes the + bit offset from `cd_values[7]`, past the list). **Could we match + libhdf5?** No: its output is not determined by the file. h5py's values for + the first two change from one run to the next (three plain runs gave three + different results), and all three change with `MALLOC_PERTURB_` or with + whether numpy is imported before h5py (`bad_nbit_parms_walk` then fails + with "filter returned failure during read" or reads all zeros); h5dump + 1.14.6 prints yet other values for the over-read parts (`1280, 0, 0` where + h5py gave e.g. `1283, 749, 1713`). libhdf5's develop branch refuses all + three. We keep + refusing them. `conformance/ref_bugs.py` repeats the check in every + conformance run (six reads per object in differently set-up processes) and + such a file is classified *ref-bug* only while its values keep changing. + +Result (`conformance/run.sh --no-fetch`, 2026-09-27): see CHANGELOG. + ## Files a SWMR writer had open could not be read past a stale end of file **Status:** fixed 2026-09-27 (branch `feat/p3-m5-swmr-reader`), before any @@ -505,7 +548,9 @@ fill-value item that did is fixed). short"); `bad_nbit_parms_walk.h5` has an N-Bit parameter list one value short (libhdf5's own `test_filter_bad_params` in `test/dsets.c` now requires that read to fail). We refuse both; `CONFORMANCE.md` lists them - under *Known not-our-bug*. Scale-offset did decode three cases + under *Known not-our-bug* (since 2026-09-27: the *ref-bug* class, with + the evidence re-checked every run — see *Conformance: the last non-ok + files* above). Scale-offset did decode three cases differently from libhdf5 (codes after a `minval` of recorded size other than 8, `minbits` 0 with a fill value, full-width `minbits`): fixed 2026-09-26, and the full-width case was silent wrong data on ordinary From 96086add99eff43ba2389797cce1d39af7244328 Mon Sep 17 00:00:00 2001 From: osobh Date: Sun, 27 Sep 2026 22:32:11 -0500 Subject: [PATCH 02/10] format: inline the version-1 message loop again (ObjectHeader::parse back to 8f59b2e's speed) 4313917 kept parse_v1_messages out of line (#[inline(never)]) so the generic parser would not carry the loop; since ef428d7 the slice path is compiled once in this crate, and the call itself was the remaining cost of the chunk queue: A/B builds of object_header_parse_x401 with only this attribute changed put #[inline(never)] and no attribute at 24.5-24.9 us and #[inline] at 23.6-24.0 us, with 8f59b2e at 23.6-24.1 us. Lazily creating the chunk list only when a continuation is found (tried too) measured no faster and was not kept. Same code otherwise: every chunk-queue check (65,536 chunks, cycles, file size budget, one chunk buffer at a time, libhdf5 order, overlap allowed) is unchanged. Co-Authored-By: Claude Opus 5.5 (1M context) --- crates/clawhdf5-format/src/object_header.rs | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/crates/clawhdf5-format/src/object_header.rs b/crates/clawhdf5-format/src/object_header.rs index afcb9c5..2045a4c 100644 --- a/crates/clawhdf5-format/src/object_header.rs +++ b/crates/clawhdf5-format/src/object_header.rs @@ -295,7 +295,13 @@ impl ObjectHeader { /// The messages of one version-1 chunk: each checked and appended to /// `messages` (NIL ones dropped), each continuation added to `spans`. /// Returns how many messages (NIL ones included) the chunk holds. - #[inline(never)] + /// + /// Inlined into the chunk loop: kept out of line (`#[inline(never)]`, + /// 4313917), the call cost `ObjectHeader::parse` about 2.5 ns per + /// header, 4% on small version-1 headers (A/B builds, 2026-09-27; see + /// `BENCHMARKS.md`). Without an attribute the compiler keeps it out of + /// line too. + #[inline] fn parse_v1_messages( data: &[u8], offset_size: u8, From ff0b2f8a4e4ef105d7ff4694349ca475abf3f611 Mon Sep 17 00:00:00 2001 From: osobh Date: Sun, 27 Sep 2026 22:33:42 -0500 Subject: [PATCH 03/10] conformance: keep ref-bug files out of the root-cause tables, name the non-ok classes A ref-bug file's refused objects were still listed as our-error root causes. The summary line now names every non-ok class and its count, and an empty root-cause table says None. Co-Authored-By: Claude Opus 5.5 (1M context) --- conformance/compare.py | 3 ++- conformance/report.py | 31 ++++++++++++++++++------------- 2 files changed, 20 insertions(+), 14 deletions(-) diff --git a/conformance/compare.py b/conformance/compare.py index b3c944a..ec568b5 100755 --- a/conformance/compare.py +++ b/conformance/compare.py @@ -239,7 +239,8 @@ for rel in files: "caught": [(p, w, m[:2500]) for p, w, m in caught_panics[:3]], "n_caught": len(caught_panics), }) - for kind, detail, key, rec in issues: + # A ref-bug file's differences are listed with the evidence instead. + for kind, detail, key, rec in (issues if cls != "ref-bug" else []): if kind == "our-error": add(root_causes, norm(key), rel, detail[:300]) else: diff --git a/conformance/report.py b/conformance/report.py index 6c4cb13..0770936 100644 --- a/conformance/report.py +++ b/conformance/report.py @@ -225,9 +225,8 @@ for c in sorted(by_corpus): w(f"| **all** | **{len(rows)}** | " + " | ".join(f"**{total.get(k, 0)}**" for k in CLASSES) + " |") w("") nonok = total.get("our-error", 0) + total.get("mismatch", 0) -w(f"**Our errors and mismatches: {nonok}.** " - + ("Every file clawhdf5 does not read like h5py is either unreadable by h5py or a confirmed libhdf5 " - "bug (*ref-bug*)." if nonok == 0 else "See the root causes below.") +w(f"**Our errors and mismatches: {nonok}.** Files not ok: " + + (", ".join(f"{total[c]} {c}" for c in CLASSES if c != "ok" and total.get(c)) or "none") + "." + (f" {len(ref_fixes)} object(s) were compared against h5py's values corrected for a known h5py bug" f" ({sum(1 for x in ref_fixes if x[3])} identical to clawhdf5's; see *Reference bugs*)." if ref_fixes else "")) w("") @@ -250,19 +249,25 @@ w("") w("## Our-error root causes") w("") -w("Grouped by normalised error message. *files* counts files whose class this cause affects.") -w("") -w("| files | objects | error | examples |") -w("|---:|---:|---|---|") -for k, v in res["root_causes"].items(): - w(f"| {v['files']} | {v['count']} | `{k.replace('|', '/')}` | {ex_list(v['file_list'])} |") +if res["root_causes"]: + w("Grouped by normalised error message. *files* counts files whose class this cause affects.") + w("") + w("| files | objects | error | examples |") + w("|---:|---:|---|---|") + for k, v in res["root_causes"].items(): + w(f"| {v['files']} | {v['count']} | `{k.replace('|', '/')}` | {ex_list(v['file_list'])} |") +else: + w("None.") w("") w("## Mismatch root causes") w("") -w("| files | objects | cause | examples |") -w("|---:|---:|---|---|") -for k, v in res["mismatch_causes"].items(): - w(f"| {v['files']} | {v['count']} | `{k.replace('|', '/')}` | {ex_list(v['file_list'])} |") +if res["mismatch_causes"]: + w("| files | objects | cause | examples |") + w("|---:|---:|---|---|") + for k, v in res["mismatch_causes"].items(): + w(f"| {v['files']} | {v['count']} | `{k.replace('|', '/')}` | {ex_list(v['file_list'])} |") +else: + w("None.") w("") w("## CVE corpus: clawhdf5 vs h5dump vs h5py") From 05e11360273b7608224d3f0612f52a7b2d5798b4 Mon Sep 17 00:00:00 2001 From: osobh Date: Sun, 27 Sep 2026 22:40:03 -0500 Subject: [PATCH 04/10] docs: conformance report with the last non-ok files classified (602 of 697 ok, 0 our-error, 0 mismatch) Co-Authored-By: Claude Opus 5.5 (1M context) --- CHANGELOG.md | 21 +++++++++++ CONFORMANCE.md | 88 ++++++++++++++++++++++++-------------------- docs/known-issues.md | 3 +- 3 files changed, 72 insertions(+), 40 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 306dd13..1518e19 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,27 @@ ## Unreleased +### Conformance: no our-errors or mismatches left (2026-09-27) +- The last 3 our-errors and 2 mismatches were documented as not ours but + still counted against us. Re-checked with evidence + (`docs/known-issues.md`, "Conformance: the last non-ok files"): + - The 2 mismatches were h5py's big-endian VL bug (elements returned with + the file's bytes under a little-endian dtype). `conformance/ref.py` now + checks the installed h5py has the bug and relabels such elements before + hashing, so their values are compared: `attr_datatypes.hdf5` and + `tcomplex_be.h5` are identical to clawhdf5's and now **ok**. + - The 3 our-errors are objects HDF5 2.0 reads by over-reading memory + (scale-offset codes past a chunk, unfiltered chunks shorter than a + chunk, an N-Bit parameter list one value short). h5py's values for them + change between runs, with `MALLOC_PERTURB_` and with import order, and + h5dump 1.14.6 prints others, so they are not the file's data and + clawhdf5 keeps refusing them. `conformance/ref_bugs.py` repeats that + check in every run, and such a file is the new class **ref-bug** only + while its values keep changing. +- Result (`conformance/run.sh --no-fetch`, tank, 2026-09-27): 602 of 697 + ok (baseline 600), 0 our-error, 0 mismatch, 3 ref-bug, 92 + h5py-cannot-read, no panic/hang/crash/oom. + ### Deterministic errors on damaged chunked datasets (2026-09-27) - A read through the file's chunk cache listed a damaged dataset's chunks in hash-map order, seeded per `File`, so two opens of the same file could diff --git a/CONFORMANCE.md b/CONFORMANCE.md index ff5b581..a900e17 100644 --- a/CONFORMANCE.md +++ b/CONFORMANCE.md @@ -13,15 +13,15 @@ fatal. This file is generated by `conformance/run.sh`; do not edit it by hand. | | | |---|---| -| date | 2026-09-27 00:34 UTC | -| clawhdf5 commit | `f37e7ae3263277319dba4bc39be5397194eb00c3` | +| date | 2026-09-28 03:33 UTC | +| clawhdf5 commit | `ff0b2f8a4e4ef105d7ff4694349ca475abf3f611` | | machine | `tank`: AMD Ryzen 7 7800X3D 8-Core Processor, 16 CPUs, 61 GiB, Linux 7.0.0-34-generic x86_64 | -| command | `conformance/run.sh --no-fetch --update-baseline` | -| rustc | rustc 1.98.1 (48a229cea 2026-09-01) | +| command | `conformance/run.sh --no-fetch` | +| rustc | | | reference | h5py 3.16.0, HDF5 2.0.0, numpy 2.5.3, hdf5plugin 7.1.0, Python 3.14.4 | | h5dump | Version 1.14.6 (CVE corpus only) | | limits | 20 s timeout (SIGKILL), 4096 MiB address space, per process; 16 files in parallel | -| runtime | 21 s probing + comparing (0 s fetch/build before it) | +| runtime | 21 s probing + comparing (18 s fetch/build before it) | ## Results @@ -29,25 +29,24 @@ A file's class is the first that applies: - **panic / hang / crash / oom** — clawhdf5 panicked (caught per object or not), hit the timeout, died on a signal, or failed an allocation. The CI gate fails on any of these. - **h5py-cannot-read** — libhdf5 could not open the file (or itself crashed or hung). Nothing to compare against; most are the deliberately malformed CVE reproducers. +- **ref-bug** — every difference is an object clawhdf5 refuses that h5py reads only through a libhdf5 bug: the values h5py returns for it change with the reading process's heap, re-checked in every run (see *Reference bugs*). - **our-error** — clawhdf5 returned an error for something h5py reads. - **mismatch** — both read it, but the shapes, values, object set or attribute set differ. - **ok** — every object h5py reads, clawhdf5 reads identically. -| corpus | files | ok | our-error | mismatch | h5py-cannot-read | panic | hang | crash | oom | -|---|---|---|---|---|---|---|---|---|---| -| NCAS-CMS_pyfive | 33 | 32 | 0 | 1 | 0 | 0 | 0 | 0 | 0 | -| cve_hdf5 | 147 | 113 | 2 | 0 | 32 | 0 | 0 | 0 | 0 | -| h5py_data | 4 | 4 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | -| hdf5 | 466 | 404 | 1 | 1 | 60 | 0 | 0 | 0 | 0 | -| netcdf-c | 20 | 20 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | -| netcdf4-python | 18 | 18 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | -| usnistgov_h5wasm | 5 | 5 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | -| xarray-data | 4 | 4 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | -| **all** | **697** | **600** | **3** | **2** | **92** | **0** | **0** | **0** | **0** | +| corpus | files | ok | our-error | mismatch | h5py-cannot-read | ref-bug | panic | hang | crash | oom | +|---|---|---|---|---|---|---|---|---|---|---| +| NCAS-CMS_pyfive | 33 | 33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | +| cve_hdf5 | 147 | 113 | 0 | 0 | 32 | 2 | 0 | 0 | 0 | 0 | +| h5py_data | 4 | 4 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | +| hdf5 | 466 | 405 | 0 | 0 | 60 | 1 | 0 | 0 | 0 | 0 | +| netcdf-c | 20 | 20 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | +| netcdf4-python | 18 | 18 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | +| usnistgov_h5wasm | 5 | 5 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | +| xarray-data | 4 | 4 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | +| **all** | **697** | **602** | **0** | **0** | **92** | **3** | **0** | **0** | **0** | **0** | -2 of the 2 mismatches are a known h5py bug, not ours (see *Known not-our-bug*). - -3 of the 3 our-errors are corrupt data that HDF5 2.0 reads only through a bug and clawhdf5 refuses (see *Known not-our-bug*). +**Our errors and mismatches: 0.** Files not ok: 92 h5py-cannot-read, 3 ref-bug. 2 object(s) were compared against h5py's values corrected for a known h5py bug (2 identical to clawhdf5's; see *Reference bugs*). Corpora (fetched by `conformance/fetch-corpus.sh` into the gitignored `conformance/.cache/`): @@ -68,18 +67,11 @@ None. ## Our-error root causes -Grouped by normalised error message. *files* counts files whose class this cause affects. - -| files | objects | error | examples | -|---:|---:|---|---| -| 3 | 3 | `ChunkedReadError("…")` | `cve_hdf5/cvefiles/cve-2025-2308.h5`, `cve_hdf5/cvefiles/cve-2025-44904.h5`, `hdf5/test/testfiles/bad_nbit_parms_walk.h5` | +None. ## Mismatch root causes -| files | objects | cause | examples | -|---:|---:|---|---| -| 1 | 1 | `attr-values: ours=vlen(>u8) h5py=object layout=- filters=-` | `NCAS-CMS_pyfive/tests/data/attr_datatypes.hdf5` | -| 1 | 1 | `values: ours=vlen({r:>f4,i:>f4}8) h5py=object layout=contiguous filters=-` | `hdf5/tools/test/testfiles/tcomplex_be.h5` | +None. ## CVE corpus: clawhdf5 vs h5dump vs h5py @@ -203,7 +195,7 @@ columns are. | cvefiles/cve-2024-33876.h5 | ok | read 3 obj, 1 errors | read 3 obj, 1 errors | ok | | cvefiles/cve-2024-33877.h5 | error exit | read 8 obj, 1 errors | read 8 obj, 1 errors | ok | | cvefiles/cve-2025-2153.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read | -| cvefiles/cve-2025-2308.h5 | error exit | read 25 obj, 1 errors | read 25 obj, 2 errors | our-error | +| cvefiles/cve-2025-2308.h5 | error exit | read 25 obj, 1 errors | read 25 obj, 2 errors | ref-bug | | cvefiles/cve-2025-2309.h5 | ok | read 6 obj, 1 errors | read 6 obj | ok | | cvefiles/cve-2025-2310.h5 | error exit | read 24 obj, 8 errors | read 24 obj, 8 errors | ok | | cvefiles/cve-2025-2912.h5 | error exit | open error | open error | h5py-cannot-read | @@ -214,7 +206,7 @@ columns are. | cvefiles/cve-2025-2924.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok | | cvefiles/cve-2025-2925.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok | | cvefiles/cve-2025-2926.h5 | error exit | open error | open error | h5py-cannot-read | -| cvefiles/cve-2025-44904.h5 | error exit | read 25 obj, 1 errors | read 25 obj, 2 errors | our-error | +| cvefiles/cve-2025-44904.h5 | error exit | read 25 obj, 1 errors | read 25 obj, 2 errors | ref-bug | | cvefiles/cve-2025-44905.h5 | error exit | read 25 obj, 3 errors | read 25 obj, 3 errors | ok | | cvefiles/cve-2025-6269-1.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok | | cvefiles/cve-2025-6269-2.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok | @@ -249,13 +241,36 @@ columns are. -## Known not-our-bug +## Reference bugs + +### Objects h5py reads only through a libhdf5 bug (*ref-bug*) + +clawhdf5 refuses these objects; h5py 3.16 / HDF5 2.0 returns values for them. `conformance/ref_bugs.py` +re-reads each with h5py in six fresh processes whose heaps differ (h5py imported before numpy, three +times and twice more with `MALLOC_PERTURB_`, and numpy imported first). Values the file determines +come out the same every time; these do not, so they are memory libhdf5 over-reads, not the file's +data. A file is *ref-bug* only while every one of its differences is such an object confirmed in +the same run; an object that reads the same every time goes back to *our-error*. Reproducer: +`python conformance/ref_bugs.py conformance/.cache/corpus` (prints every read's outcome). + +| file | object | distinct results in 6 reads | confirmed | what goes wrong | +|---|---|---:|---|---| +| `cve_hdf5/cvefiles/cve-2025-2308.h5` | `/Scale_offset_long_long_data_le` | 6 | yes | the first chunk records minbits 11: its 12 values need 17 bytes of codes, and the 26-byte chunk holds 5 after its 21-byte header; libhdf5's scale-offset decoder reads past its buffer, and develop refuses the chunk ("Buffer too short") | +| `cve_hdf5/cvefiles/cve-2025-44904.h5` | `/Scale_offset_float_data_le` | 6 | yes | unfiltered chunks stored as 38 and 37 bytes for 48-byte chunks: 1.14/2.0 read the stored bytes into a buffer of that size and use it as the whole chunk (H5D__chunk_lock), so the rest is heap memory; develop refuses them ("incorrect chunk size returned from index for unfiltered chunk") | +| `hdf5/test/testfiles/bad_nbit_parms_walk.h5` | `/Nbit_int_data_le` | 3 | yes | the N-Bit parameter list holds 7 values (cd_values[0] = 7) where an integer needs 8: the decoder takes the bit offset from cd_values[7], past the list; libhdf5's own test (`test_filter_bad_params`, test/dsets.c on develop) requires the read to fail | + +### Values corrected for a known h5py bug - **h5py big-endian variable-length sequences.** h5py returns the elements of a VL sequence whose base type is big-endian with the file's big-endian bytes but a native (little-endian) - numpy dtype, so the values it reports are byte-swapped garbage; `h5dump` prints the values - clawhdf5 reads. Reproducer: `h5py.vlen_dtype(np.dtype('>f4'))` dataset holding `[1.0, 2.0]` - reads back in h5py as `[4.6e-41, 9.0e-44]`. Affected here: `NCAS-CMS_pyfive/tests/data/attr_datatypes.hdf5`, `hdf5/tools/test/testfiles/tcomplex_be.h5`. + numpy dtype: a `h5py.vlen_dtype(np.dtype('>f4'))` dataset holding `[1.0, 2.0]` reads back as + `[4.6e-41, 9.0e-44]`; `h5dump` prints the file's values. `ref.py` checks that the installed + h5py still does this (by writing and reading exactly that dataset in memory) and, if so, + relabels such elements with the file's byte order before hashing, so the values are still + compared. Corrected objects: `NCAS-CMS_pyfive/tests/data/attr_datatypes.hdf5` `/@vlen_uint64` (same as clawhdf5), `hdf5/tools/test/testfiles/tcomplex_be.h5` `/VariableLengthDatasetFloatComplex` (same as clawhdf5). + +## Other comparison rules + - **Non-IEEE floats and partial-precision integers (N-Bit).** libhdf5 converts a float whose bit layout is not IEEE (e.g. `H5Tset_precision` for the N-Bit filter) or an integer with a bit offset / reduced precision into the plain numpy type of the same size. The probe @@ -264,11 +279,6 @@ columns are. - **Types h5py widens.** Where h5py reads a type into a numpy type of a different size (FP8 -> float16, bfloat16 -> float32, x87 long double -> float128) the values are not compared (shape and presence still are): dataset file type size 1 -> numpy float16 (2) (15x), attr file type size 1 -> numpy float16 (2) (15x), dataset file type size 2 -> numpy float32 (4) (2x), dataset file type size 8 -> numpy float128 (16) (1x), dataset file type size 12 -> numpy float128 (16) (1x), attr file type size 2 -> numpy float32 (4) (1x), dataset file type size 2 -> numpy >f4 (4) (1x), attr file type size 2 -> numpy >f4 (4) (1x). -- **Corrupt data HDF5 2.0 reads through a bug.** clawhdf5 refuses these objects; h5py 3.16 / - HDF5 2.0 returns values for them that the file does not hold: - - `cve_hdf5/cvefiles/cve-2025-2308.h5` `/Scale_offset_long_long_data_le`: scale-offset codes run past the end of the chunk: HDF5 2.0 reads past its buffer; libhdf5's develop branch refuses the chunk ("Buffer too short"). - - `cve_hdf5/cvefiles/cve-2025-44904.h5` `/Scale_offset_float_data_le`: unfiltered chunks of 38 and 37 bytes for 48-byte chunks: HDF5 2.0 fills the rest with whatever its buffer held; libhdf5's develop branch refuses them ("incorrect chunk size returned from index for unfiltered chunk"). - - `hdf5/test/testfiles/bad_nbit_parms_walk.h5` `/Nbit_int_data_le`: an N-Bit parameter list one value short: HDF5 2.0 reads past the list; libhdf5's own test (`test_filter_bad_params`, test/dsets.c) now requires the read to fail. - **References** are compared by presence only (`R`), not by target. ## Objects h5py fails on but clawhdf5 reads diff --git a/docs/known-issues.md b/docs/known-issues.md index 9e572a6..d87d371 100644 --- a/docs/known-issues.md +++ b/docs/known-issues.md @@ -66,7 +66,8 @@ as "not ours" but still counted against us. Re-checked on tank conformance run (six reads per object in differently set-up processes) and such a file is classified *ref-bug* only while its values keep changing. -Result (`conformance/run.sh --no-fetch`, 2026-09-27): see CHANGELOG. +Result (`conformance/run.sh --no-fetch`, tank, 2026-09-27): 602 of 697 ok +(was 600), 0 our-error, 0 mismatch, 3 ref-bug, 92 h5py-cannot-read. ## Files a SWMR writer had open could not be read past a stale end of file From ca81c3ebfa78e1090c824743f367c22913de58ae Mon Sep 17 00:00:00 2001 From: osobh Date: Sun, 27 Sep 2026 22:41:09 -0500 Subject: [PATCH 05/10] format, wasm: Storage::hint, fetched by the lazy reader with a pass's misses A parser often learns where the next structures are (a node's children, a structure's body once its prefix gives its length) before it reads them one at a time. Over openUrl's restartable reader a structure only reached after a miss costs a pass, and a round trip, of its own. - clawhdf5-format: `Storage::hint(offset, len)`, "about to be read by this operation". Default: nothing (every backend that reads when asked); `&T`, `Box`, `Arc` and the facade's `FileData` forward it (shifted past a user block, clamped to the file). - clawhdf5-wasm `LazyStorage` records hinted blocks it lacks. A pass that misses nothing ignores them (a hint never adds a round trip); a pass that misses also asks for them, in file order, while the pass stays within what is left of the operation's `maxFetch` budget (a hint never makes a call fail). At most 1 MiB (or a block) per hint and 65536 blocks per pass are recorded, whatever a file makes a parser hint. `Operation::attempt` follows hints; the plain `LazyStorage::attempt` does not. `run_blocking` and the browser's driver use the former. `LazyStats::hinted_blocks` counts them. No parser hints yet: results, passes and requests are unchanged. Tests: hinted blocks come with a miss and never alone (cached ones skipped, a plain attempt ignores them), and stay within the fetch budget, a hint past the end or longer than the file harmless. Co-Authored-By: Claude Opus 5.5 (1M context) --- crates/clawhdf5-format/src/storage.rs | 30 ++++ crates/clawhdf5-wasm/src/lazy.rs | 199 +++++++++++++++++++++++++- crates/clawhdf5-wasm/src/lib.rs | 2 +- crates/clawhdf5/src/reader.rs | 15 ++ 4 files changed, 239 insertions(+), 7 deletions(-) diff --git a/crates/clawhdf5-format/src/storage.rs b/crates/clawhdf5-format/src/storage.rs index 5fe9497..6b74378 100644 --- a/crates/clawhdf5-format/src/storage.rs +++ b/crates/clawhdf5-format/src/storage.rs @@ -89,6 +89,21 @@ pub trait Storage { fn as_contiguous(&self) -> Option<&[u8]> { None } + + /// A hint that `[offset, offset + len)` is about to be read by the + /// same operation: a parser that has just learnt where the next + /// structures are (a node's children, a structure's body) says so + /// before it reads them one at a time. Nothing is read and nothing + /// fails. The default ignores it, as does every backend that reads + /// when asked; the browser's restartable reader, which fetches over + /// the network between attempts, fetches hinted bytes along with the + /// bytes an attempt actually missed, so structures a parser only + /// reaches after a miss arrive in the same round trip. Results never + /// depend on hints. + #[inline] + fn hint(&self, offset: u64, len: usize) { + let _ = (offset, len); + } } impl Storage for [u8] { @@ -148,6 +163,11 @@ impl Storage for &T { fn as_contiguous(&self) -> Option<&[u8]> { (**self).as_contiguous() } + + #[inline] + fn hint(&self, offset: u64, len: usize) { + (**self).hint(offset, len) + } } impl Storage for Box { @@ -170,6 +190,11 @@ impl Storage for Box { fn as_contiguous(&self) -> Option<&[u8]> { (**self).as_contiguous() } + + #[inline] + fn hint(&self, offset: u64, len: usize) { + (**self).hint(offset, len) + } } #[cfg(feature = "std")] @@ -193,6 +218,11 @@ impl Storage for std::sync::Arc { fn as_contiguous(&self) -> Option<&[u8]> { (**self).as_contiguous() } + + #[inline] + fn hint(&self, offset: u64, len: usize) { + (**self).hint(offset, len) + } } /// `storage.len()` as the `usize` the parsers' end-of-file errors report diff --git a/crates/clawhdf5-wasm/src/lazy.rs b/crates/clawhdf5-wasm/src/lazy.rs index dadb939..e45dce4 100644 --- a/crates/clawhdf5-wasm/src/lazy.rs +++ b/crates/clawhdf5-wasm/src/lazy.rs @@ -25,6 +25,14 @@ //! pass per block it needs. In practice it is one pass per *wave* of //! misses: a chunked read asks for all the chunks of a batch at once. //! +//! Parsers also say what they are about to read ([`Storage::hint`]: a +//! node's children, a structure's body, a listed group's child headers). +//! A pass that misses fetches the hinted blocks it lacks too, as far as the +//! operation's fetch budget allows, so what the parser would only have +//! reached on the next pass arrives in the same round trip; a pass that +//! misses nothing ignores its hints, so they never add a round trip, and +//! results never depend on them. +//! //! Blocks are kept in an LRU cache with a byte budget, trimmed only when no //! operation is in flight. Blocks fetched for bulk reads (raw data: a //! `read_ranges` call, or a read longer than a block) go first, so reading @@ -50,6 +58,14 @@ pub const DEFAULT_MAX_FETCH: u64 = 512 << 20; /// the caller of [`LazyStorage::attempt`]: a pass that missed is re-run. pub const NEED_BYTES: &str = "bytes not fetched yet (restartable read)"; +/// Most blocks one pass records as hinted (see [`Storage::hint`]). +const MAX_HINTED: usize = 1 << 16; + +/// Total length of `ranges`. +fn ranges_len(ranges: &[Range]) -> u64 { + ranges.iter().map(|r| r.end - r.start).sum() +} + /// Settings of a [`LazyStorage`]. #[derive(Debug, Clone, PartialEq, Eq)] pub struct LazyConfig { @@ -94,6 +110,9 @@ pub struct LazyStats { pub bytes_fetched: u64, /// Blocks evicted to stay within the budget. pub evictions: u64, + /// Blocks asked for because a parser hinted it would read them + /// ([`Storage::hint`]), not because a pass missed them. + pub hinted_blocks: u64, /// Bytes cached now. pub cached_bytes: u64, } @@ -125,6 +144,9 @@ struct State { /// Blocks the current pass missed, and whether a small read wanted /// them (metadata). missing: HashMap, + /// Blocks the current pass was told it is about to read and that are + /// not cached ([`Storage::hint`]), in file order. + hinted: BTreeSet, /// Blocks a bulk read missed that have not been supplied yet: kept /// as bulk when they arrive. bulk_pending: BTreeSet, @@ -155,6 +177,19 @@ pub struct Operation<'a> { } impl Operation<'_> { + /// One pass of `f`, as [`LazyStorage::attempt`], but a pass that misses + /// also asks for the blocks it was hinted it would read, as many as fit + /// in what is left of the operation's budget ([`LazyConfig::max_fetch`]) + /// after the blocks it missed. + pub fn attempt(&self, f: impl FnOnce() -> T) -> Step { + let left = self + .storage + .config + .max_fetch + .saturating_sub(self.fetched.get()); + self.storage.attempt_within(f, left) + } + /// Count `ranges` against the operation's budget /// ([`LazyConfig::max_fetch`]) before they are fetched: an error, and /// nothing counted, if they would take it past the budget. @@ -225,20 +260,70 @@ impl LazyStorage { /// Run one pass of `f` over this storage. `Done` when `f` read nothing /// that is missing; otherwise `Need` with the ranges to fetch, and `f`'s /// result is dropped (it may be an error caused by the miss, or a - /// result built around one). + /// result built around one). Hints are not followed; see + /// [`Operation::attempt`]. pub fn attempt(&self, f: impl FnOnce() -> T) -> Step { + self.attempt_within(f, 0) + } + + /// [`attempt`](Self::attempt), adding to a pass that misses the hinted + /// blocks it lacks while everything asked for stays within `budget` + /// bytes. + fn attempt_within(&self, f: impl FnOnce() -> T, budget: u64) -> Step { { let mut st = lock(&self.state); st.missing.clear(); + st.hinted.clear(); st.stats.passes += 1; } let out = f(); - let missing = std::mem::take(&mut lock(&self.state).missing); + let (missing, hinted) = { + let mut st = lock(&self.state); + ( + std::mem::take(&mut st.missing), + std::mem::take(&mut st.hinted), + ) + }; if missing.is_empty() { return Step::Done(out); } drop(out); - Step::Need(self.runs(missing)) + let need = self.runs(&missing); + if hinted.is_empty() || budget == 0 { + return Step::Need(need); + } + // The hinted blocks still missing, in file order, while they fit. + let bs = self.config.block_size; + let mut total = ranges_len(&need); + let mut wanted = missing.clone(); + { + let st = lock(&self.state); + for i in hinted { + if wanted.contains_key(&i) || st.blocks.contains_key(&i) { + continue; + } + let n = (self.len - i * bs).min(bs); + if total + n > budget { + break; + } + total += n; + wanted.insert(i, true); + } + } + if wanted.len() == missing.len() { + return Step::Need(need); + } + let with_hints = self.runs(&wanted); + // Filling holes between runs can add blocks: never let the hints + // take the pass past the budget. + if ranges_len(&with_hints) > budget { + return Step::Need(need); + } + { + let mut st = lock(&self.state); + st.stats.hinted_blocks += (wanted.len() - missing.len()) as u64; + } + Step::Need(with_hints) } /// The bytes of the file at `offset`, fetched for a range a pass asked @@ -309,7 +394,7 @@ impl LazyStorage { ) -> Result { let op = self.operation(); loop { - match self.attempt(&mut f) { + match op.attempt(&mut f) { Step::Done(v) => return Ok(v), Step::Need(ranges) => { op.charge(&ranges)?; @@ -350,14 +435,14 @@ impl LazyStorage { /// blocks, a one-block hole between two runs filled so they merge /// (unless the hole is cached: it would be fetched again), each at /// most `max_request` long. - fn runs(&self, missing: HashMap) -> Vec> { + fn runs(&self, missing: &HashMap) -> Vec> { let bs = self.config.block_size; let mut wanted: Vec = missing.keys().copied().collect(); wanted.sort_unstable(); let mut st = lock(&self.state); // Remember which blocks only bulk reads asked for: they are kept // as bulk once supplied. - for (&i, &metadata) in &missing { + for (&i, &metadata) in missing { if metadata { st.bulk_pending.remove(&i); } else { @@ -499,6 +584,25 @@ impl Storage for LazyStorage { self.len } + fn hint(&self, offset: u64, len: usize) { + // A hint is about a structure, not bulk data: at most a block or + // 1 MiB of it is followed, and at most `MAX_HINTED` blocks a pass, + // whatever a hostile file makes a parser hint. + let len = (len as u64).min(self.config.block_size.max(1 << 20)); + let Some(span) = self.span(offset, len) else { + return; + }; + let mut st = lock(&self.state); + for i in span { + if st.hinted.len() >= MAX_HINTED { + return; + } + if !st.blocks.contains_key(&i) { + st.hinted.insert(i); + } + } + } + fn read_ranges(&self, ranges: &[Range]) -> Result>, FormatError> { let mut spans = Vec::with_capacity(ranges.len()); let mut total = 0u64; @@ -797,6 +901,89 @@ mod tests { } } + #[test] + fn hinted_blocks_come_with_a_miss_and_never_alone() { + let data = file(16 * 1024); + let s = LazyStorage::new(data.len() as u64, config(1024, 1 << 20)); + serve(&s, &data, &[3072..4096]); + let op = s.operation(); + // Hints alone: the pass is done, nothing is fetched. + let step = op.attempt(|| { + s.hint(8 * 1024, 100); + owned(s.read_at(3072, 8)) + }); + assert!(matches!(step, Step::Done(Ok(_))), "{step:?}"); + // With a miss, the hinted blocks not cached come too (block 3 is + // cached; 10..=11 is one run, 14 another). + let pass = || { + s.hint(3072, 10); + s.hint(10 * 1024 + 1000, 100); + s.hint(14 * 1024, 1); + owned(s.read_at(0, 8)) + }; + let Step::Need(need) = op.attempt(pass) else { + panic!("block 0 is missing"); + }; + assert_eq!( + need, + vec![0..1024, 10 * 1024..12 * 1024, 14 * 1024..15 * 1024] + ); + // A plain attempt does not follow hints. + let Step::Need(plain) = s.attempt(pass) else { + panic!("block 0 is missing"); + }; + assert_eq!(plain, vec![0..1024]); + serve(&s, &data, &need); + let Step::Done(got) = op.attempt(pass) else { + panic!("everything was supplied"); + }; + assert_eq!(got.unwrap(), &data[..8]); + assert_eq!(s.stats().hinted_blocks, 3); + } + + #[test] + fn hints_stay_within_the_fetch_budget() { + // A budget of 3 blocks: the missed block and the first two hinted + // ones fit, the rest are left out; a hint never makes a call fail. + // (Hinted blocks two apart would be merged with the hole between + // them, which would not fit: then no hint is followed.) + let data = file(64 * 1024); + let mut c = config(1024, 1 << 20); + c.max_fetch = 3 * 1024; + let s = LazyStorage::new(data.len() as u64, c); + let got = s + .run_blocking( + || { + for i in 0..20 { + s.hint(20 * 1024 + i * 3072, 1); + } + owned(s.read_at(0, 8)) + }, + |r| Ok(data[r.start as usize..r.end as usize].to_vec()), + ) + .unwrap() + .unwrap(); + assert_eq!(got, &data[..8]); + let st = s.stats(); + assert_eq!( + (st.bytes_fetched, st.hinted_blocks), + (3 * 1024, 2), + "{st:?}" + ); + // A hint longer than the file, or past its end, is harmless. + let s = LazyStorage::new(data.len() as u64, config(1024, 1 << 20)); + s.run_blocking( + || { + s.hint(0, usize::MAX); + s.hint(u64::MAX, 10); + owned(s.read_at(0, 8)) + }, + |r| Ok(data[r.start as usize..r.end as usize].to_vec()), + ) + .unwrap() + .unwrap(); + } + #[test] fn a_failed_fetch_is_an_error_not_data() { let data = file(4096); diff --git a/crates/clawhdf5-wasm/src/lib.rs b/crates/clawhdf5-wasm/src/lib.rs index 6a512a2..85271d2 100644 --- a/crates/clawhdf5-wasm/src/lib.rs +++ b/crates/clawhdf5-wasm/src/lib.rs @@ -391,7 +391,7 @@ impl Http { ) -> Result { let op = storage.operation(); loop { - match storage.attempt(&mut f) { + match op.attempt(&mut f) { Step::Done(v) => return Ok(v), Step::Need(ranges) => { op.charge(&ranges).map_err(js_err)?; diff --git a/crates/clawhdf5/src/reader.rs b/crates/clawhdf5/src/reader.rs index 4be5f93..7e2ceb5 100644 --- a/crates/clawhdf5/src/reader.rs +++ b/crates/clawhdf5/src/reader.rs @@ -372,6 +372,21 @@ impl Storage for FileData { fn as_contiguous(&self) -> Option<&[u8]> { self.contiguous() } + + fn hint(&self, offset: u64, len: usize) { + if self.contiguous().is_some() { + return; + } + if let Backing::Storage(s) = &self.backing { + // Within the HDF5 data, as a read would be clamped. + let size = Storage::len(self); + let start = offset.min(size); + let len = usize::try_from(size - start).map_or(len, |avail| avail.min(len)); + if len > 0 { + s.hint(self.base + start, len); + } + } + } } /// `bytes`, a backend's answer to a read of `len` bytes, without anything From 6f5d14fd626b413660a9ce37deb8be4d845233cd Mon Sep 17 00:00:00 2001 From: osobh Date: Sun, 27 Sep 2026 22:47:44 -0500 Subject: [PATCH 06/10] format: group walks go on past a failed node and hint what they read next Listing a large group over openUrl still took 6-11 passes (network round trips) for the reviewer's 3000-dataset h5py file: each pass only found the structures the walk reached before its first miss. - The v1 and v2 B-tree collectors descend into every child of a node after one fails (they only read the siblings before, so a sibling's subtree came a pass later), then return the first error: results and errors unchanged. The v2 walk stops once its record budget is spent, so a shared-subtree tree still cannot multiply the work. - Hints (`Storage::hint`, a no-op for every backend but the lazy one): a group B-tree node's and a symbol table node's body (read once their header gives a length, a round trip later when the body is in the next block), an object header's first chunk and its continuation chunks, the symbol table nodes a B-tree leaf names, a dense group's name index header and the heap's root block (both read right after the heap header). A listing also hints every child's object header as its entry is read, even after a failure, and every direct block of a dense group's heap (reading the indirect blocks, at most 4096 entries and 4 levels deep); a lookup does not. - The fractal heap's indirect-block layout (entry sizes, where the first n entries end) is one helper used by the object reads and the hints. Measured with tests/lazy.rs listing_cost_of_a_given_file on an h5py file like the reviewer's (3000 datasets of 64 KiB, 198 MB), list('/'), passes/requests/bytes, before -> after: earliest, 1 MiB: 6/73/192.5 MB -> 4/68/192.5 MB earliest, 64 KiB: 8/531/35.2 MB -> 5/530/35.3 MB latest, 1 MiB: 9/98/196.5 MB -> 5/86/196.5 MB latest, 64 KiB: 11/452/29.6 MB -> 6/454/30.5 MB listing_a_large_group_takes_a_few_passes (512-byte blocks), budgets tightened to the new counts: FileBuilder 600 children 5 -> 4 passes, h5py 2000 children earliest 8 -> 5, latest 11 -> 6. Co-Authored-By: Claude Opus 5.5 (1M context) --- crates/clawhdf5-format/src/btree_v1.rs | 40 ++-- crates/clawhdf5-format/src/btree_v2.rs | 35 ++-- crates/clawhdf5-format/src/fractal_heap.rs | 198 ++++++++++++++++---- crates/clawhdf5-format/src/group_v1.rs | 32 +++- crates/clawhdf5-format/src/group_v2.rs | 86 +++++++-- crates/clawhdf5-format/src/object_header.rs | 35 ++++ crates/clawhdf5-format/src/symbol_table.rs | 4 + crates/clawhdf5-wasm/tests/lazy.rs | 124 ++++++++---- 8 files changed, 443 insertions(+), 111 deletions(-) diff --git a/crates/clawhdf5-format/src/btree_v1.rs b/crates/clawhdf5-format/src/btree_v1.rs index ea4e5b1..7ae2af7 100644 --- a/crates/clawhdf5-format/src/btree_v1.rs +++ b/crates/clawhdf5-format/src/btree_v1.rs @@ -92,6 +92,8 @@ impl BTreeV1Node { // + left_sibling(offset_size) + right_sibling(offset_size) let os = offset_size as usize; let header_size = 8 + os * 2; + // The body is read once the header says how long it is. + file.hint(offset, NODE_HINT_LEN); let header = read_exact_at(file, offset, header_size)?; let file_data: &[u8] = &header; // The header's read checked that `offset + header_size` fits. @@ -158,6 +160,17 @@ impl BTreeV1Node { /// Maximum recursion depth for B-tree traversal (malformed data protection). const MAX_BTREE_DEPTH: usize = 64; +/// What a symbol table node takes with libhdf5's default group leaf K (4): +/// its 8-byte header and 2K entries of 40 bytes (8-byte offsets). Hinted +/// before one is read ([`Storage::hint`]); a node of another size is read +/// all the same. +const SNOD_HINT_LEN: usize = 8 + 8 * 40; + +/// What a group B-tree node takes with libhdf5's default internal K (16): +/// its header (24 bytes with 8-byte offsets), 2K + 1 keys and 2K children +/// of 8 bytes. Hinted before one is read. +const NODE_HINT_LEN: usize = 24 + (2 * 16 + 1 + 2 * 16) * 8; + /// Collect all leaf-level child addresses (SNOD addresses) by traversing the B-tree. pub fn collect_symbol_table_nodes( file_data: &[u8], @@ -196,20 +209,22 @@ fn collect_symbol_table_nodes_inner( } if node.node_level == 0 { - // Leaf: children are SNOD addresses + // Leaf: children are SNOD addresses, read next (see + // `Storage::hint`). + for &snod in &node.children { + file.hint(snod, SNOD_HINT_LEN); + } Ok(node.children) } else { - // Internal: recurse into children. After the first child that - // fails, the others are only read (as `storage::touch` does), not - // descended into; that error is returned. + // Internal: recurse into children. A child that fails does not + // stop the walk: the others are still descended into (reading, not + // using, what they hold), then the first error is returned. The + // result and the error are those of stopping at the first failure; + // a storage that records what it lacks (see `storage::touch`) + // learns every node the walk can reach in one attempt. let mut result = Vec::new(); let mut failed = None; for &child_addr in &node.children { - if failed.is_some() { - // Parsing reads the node's header, then its body. - let _ = BTreeV1Node::parse_in(file, child_addr, offset_size, length_size); - continue; - } match collect_symbol_table_nodes_inner( file, child_addr, @@ -217,8 +232,11 @@ fn collect_symbol_table_nodes_inner( length_size, depth + 1, ) { - Ok(child_snods) => result.extend(child_snods), - Err(e) => failed = Some(e), + Ok(child_snods) if failed.is_none() => result.extend(child_snods), + Ok(_) => {} + Err(e) => { + failed.get_or_insert(e); + } } } match failed { diff --git a/crates/clawhdf5-format/src/btree_v2.rs b/crates/clawhdf5-format/src/btree_v2.rs index f721f40..21010f7 100644 --- a/crates/clawhdf5-format/src/btree_v2.rs +++ b/crates/clawhdf5-format/src/btree_v2.rs @@ -194,10 +194,17 @@ const MAX_DEPTH: u16 = 64; /// Take `n` records from the traversal's budget, or refuse the tree. fn spend(budget: &mut usize, n: usize) -> Result<(), FormatError> { - *budget = budget - .checked_sub(n) - .ok_or(FormatError::NestingDepthExceeded)?; - Ok(()) + match budget.checked_sub(n) { + Some(left) => { + *budget = left; + Ok(()) + } + None => { + // Spent: a walk that goes on after a failure stops here. + *budget = 0; + Err(FormatError::NestingDepthExceeded) + } + } } /// Collect all records from a B-tree v2 by traversing from the root. @@ -504,16 +511,18 @@ fn collect_internal_records( // Interleave: child[0], record[0], child[1], record[1], ..., child[nr] // We collect child[0] records, then record[0], then child[1], etc. - // After the first child that fails, the others are only touched (see - // `storage::touch`); that error is returned. + // A child that fails does not stop the walk: the others are still + // descended into (their records are dropped with the result), then the + // first error is returned, as when stopping there. A storage that + // records what it lacks (see `storage::touch`) so learns every node the + // walk can reach in one attempt. The record budget is spent as before, + // so the walk is no longer than a successful one. let mut failed = None; for (i, &(child_addr, child_nrec)) in node.children.iter().enumerate() { - if failed.is_some() { - let len = usize::try_from(node_size) - .unwrap_or(usize::MAX) - .min(1 << 16); - crate::storage::touch(file, child_addr, len); - continue; + if failed.is_some() && *budget == 0 { + // The record budget is spent: the tree is refused, and a walk + // over what is left could be as long as the one it bounds. + break; } if let Err(e) = (|| -> Result<(), FormatError> { if child_depth == 0 { @@ -553,7 +562,7 @@ fn collect_internal_records( } Ok(()) })() { - failed = Some(e); + failed.get_or_insert(e); } } diff --git a/crates/clawhdf5-format/src/fractal_heap.rs b/crates/clawhdf5-format/src/fractal_heap.rs index 83cfd33..9b39c6a 100644 --- a/crates/clawhdf5-format/src/fractal_heap.rs +++ b/crates/clawhdf5-format/src/fractal_heap.rs @@ -10,7 +10,7 @@ use crate::addr::to_usize; use crate::btree_v2::{BTreeV2Header, find_btree_v2_records_in}; use crate::error::FormatError; use crate::filter_pipeline::FilterPipeline; -use crate::storage::{Storage, Window, len_usize, read_exact_at}; +use crate::storage::{Storage, Window, len_usize, read_exact_at, read_upto}; /// Parsed fractal heap header (signature "FRHP"). #[derive(Debug, Clone)] @@ -132,6 +132,9 @@ fn heap_id_type(first: u8) -> Result { const BTREE_HUGE_INDIRECT: u8 = 1; const BTREE_HUGE_INDIRECT_FILTERED: u8 = 2; +/// Most child entries [`FractalHeapHeader::hint_managed_blocks`] walks. +const MAX_HINTED_BLOCKS: usize = 4096; + impl FractalHeapHeader { /// Parse a fractal heap header at the given offset. pub fn parse( @@ -667,15 +670,8 @@ impl FractalHeapHeader { "fractal heap: maximum recursion depth exceeded".into(), )); } - let block_offset_bytes = (self.max_heap_size as usize).div_ceil(8); - let iblock_header = 5 + offset_size as usize + block_offset_bytes; let nrows_usize = nrows as usize; - // Rows below max_direct_rows hold direct blocks; rows at/above hold - // child indirect blocks. (NOT the FRHP "starting rows" field.) - let start_indirect = self.max_direct_rows(); - let max_direct_rows = nrows_usize.min(start_indirect); - // The block up to its last child entry. The walk below reads // entries in order and stops at the one covering the target, which // the geometry alone locates, so the first window ends there: a @@ -685,32 +681,11 @@ impl FractalHeapHeader { // over the whole block. Either window holds what it was asked for or // ends at the end of the file, so its bounds checks are the // whole-file ones. - let direct_entry = usize::from(offset_size) - + if self.filter_pipeline.is_some() { - usize::from(self.length_size) + 4 - } else { - 0 - }; - let direct_entries = max_direct_rows.saturating_mul(usize::from(self.table_width)); - let entries_len = |n: usize| { - n.min(direct_entries) - .saturating_mul(direct_entry) - .saturating_add( - n.saturating_sub(direct_entries) - .saturating_mul(usize::from(offset_size)), - ) - }; - let all_entries = direct_entries.saturating_add( - nrows_usize - .saturating_sub(start_indirect) - .saturating_mul(usize::from(self.table_width)), - ); - let block_len = iblock_header.saturating_add(entries_len(all_entries)); + let layout = self.indirect_layout(nrows_usize, offset_size); + let block_len = layout.len_upto(layout.all_entries); let target_entry = self.indirect_entry_for(nrows_usize, iblock_heap_offset, target_offset); let first_len = target_entry.map_or(block_len, |i| { - iblock_header - .saturating_add(entries_len(i.saturating_add(1))) - .min(block_len) + layout.len_upto(i.saturating_add(1)).min(block_len) }); let mut next = self.walk_indirect_block( &Window::read(file, iblock_addr as u64, first_len)?, @@ -755,6 +730,140 @@ impl FractalHeapHeader { } } + /// Where the child entries of an indirect block of `nrows` rows are. + fn indirect_layout(&self, nrows: usize, offset_size: u8) -> IndirectLayout { + let block_offset_bytes = (self.max_heap_size as usize).div_ceil(8); + // Rows below max_direct_rows hold direct blocks; rows at/above hold + // child indirect blocks. (NOT the FRHP "starting rows" field.) + let start_indirect = self.max_direct_rows(); + let direct_entries = nrows + .min(start_indirect) + .saturating_mul(usize::from(self.table_width)); + IndirectLayout { + header: 5 + usize::from(offset_size) + block_offset_bytes, + direct_entry: usize::from(offset_size) + + if self.filter_pipeline.is_some() { + usize::from(self.length_size) + 4 + } else { + 0 + }, + direct_entries, + indirect_entry: usize::from(offset_size), + all_entries: direct_entries.saturating_add( + nrows + .saturating_sub(start_indirect) + .saturating_mul(usize::from(self.table_width)), + ), + } + } + + /// Hint the heap's root block (see [`Storage::hint`]): every managed + /// object is read through it, and a storage that fetches between + /// attempts can fetch it along with whatever else the attempt missed + /// (the name index read before any object, say). + pub fn hint_root_block(&self, file: &S) { + if is_undefined(self.root_block_address, self.offset_size) { + return; + } + let len = if self.current_rows_in_root_indirect_block == 0 { + if self.filter_pipeline.is_some() { + self.root_direct_block_filtered_size + } else { + self.starting_block_size + } + } else { + let layout = self.indirect_layout( + usize::from(self.current_rows_in_root_indirect_block), + self.offset_size, + ); + layout.len_upto(layout.all_entries) as u64 + }; + file.hint( + self.root_block_address, + usize::try_from(len).unwrap_or(usize::MAX), + ); + } + + /// Hint every managed direct block of the heap (see + /// [`Storage::hint`]), for a caller about to read all of its objects (a + /// dense group's listing). The indirect blocks leading to them are read + /// here, as every object read goes through them, a few levels deep and + /// up to [`MAX_HINTED_BLOCKS`] entries; direct blocks are only hinted. + /// Nothing is returned and no error: a storage that has the file in + /// memory skips it, and the objects are read (and checked) as before. + pub fn hint_managed_blocks(&self, file: &S) { + if file.as_contiguous().is_some() + || is_undefined(self.root_block_address, self.offset_size) + || self.current_rows_in_root_indirect_block == 0 + { + // A direct root block is what `hint_root_block` hints. + return; + } + let mut budget = MAX_HINTED_BLOCKS; + self.hint_indirect_block( + file, + self.root_block_address, + usize::from(self.current_rows_in_root_indirect_block), + 0, + &mut budget, + ); + } + + fn hint_indirect_block( + &self, + file: &S, + addr: u64, + nrows: usize, + depth: usize, + budget: &mut usize, + ) { + let os = self.offset_size; + let layout = self.indirect_layout(nrows, os); + let len = layout.len_upto(layout.all_entries); + if depth > 4 || len > 1 << 20 { + return; + } + let Ok(bytes) = read_upto(file, addr, len) else { + return; + }; + if bytes.len() < len || bytes.get(..4) != Some(b"FHIB".as_slice()) { + return; + } + let start_indirect = self.max_direct_rows(); + let mut pos = layout.header; + for row in 0..nrows { + for _ in 0..self.table_width { + let Some(left) = budget.checked_sub(1) else { + return; + }; + *budget = left; + let Ok(child) = read_offset(&bytes, pos, os) else { + return; + }; + if row < start_indirect { + let size = if self.filter_pipeline.is_some() { + match read_offset(&bytes, pos + usize::from(os), self.length_size) { + Ok(n) => n, + Err(_) => return, + } + } else { + self.block_size_for_row(row) + }; + pos += layout.direct_entry; + if !is_undefined(child, os) { + file.hint(child, usize::try_from(size).unwrap_or(usize::MAX)); + } + } else { + pos += layout.indirect_entry; + if !is_undefined(child, os) { + let rows = self.rows_for_size(self.block_size_for_row(row)); + self.hint_indirect_block(file, child, usize::from(rows), depth + 1, budget); + } + } + } + } + } + /// Which child entry of an indirect block (numbered in walk order: /// direct rows, then indirect rows) covers `target_offset`, from the /// doubling-table geometry alone — the entry @@ -939,6 +1048,31 @@ impl FractalHeapHeader { } } +/// Where an indirect block's child entries are: after its header, the +/// direct blocks' entries (address, and for a filtered heap the stored +/// size and filter mask), then the child indirect blocks' (address). +struct IndirectLayout { + header: usize, + direct_entry: usize, + direct_entries: usize, + indirect_entry: usize, + all_entries: usize, +} + +impl IndirectLayout { + /// Bytes from the block's start to the end of its first `n` entries. + fn len_upto(&self, n: usize) -> usize { + let direct = n.min(self.direct_entries); + self.header + .saturating_add(direct.saturating_mul(self.direct_entry)) + .saturating_add( + n.min(self.all_entries) + .saturating_sub(direct) + .saturating_mul(self.indirect_entry), + ) + } +} + /// A managed direct block's location, extent and (for a filtered heap) its /// stored size and filter mask. /// The child of an indirect block that covers a heap offset. diff --git a/crates/clawhdf5-format/src/group_v1.rs b/crates/clawhdf5-format/src/group_v1.rs index bf7dcc4..f131acc 100644 --- a/crates/clawhdf5-format/src/group_v1.rs +++ b/crates/clawhdf5-format/src/group_v1.rs @@ -48,7 +48,7 @@ pub fn resolve_v1_group_entries_in( offset_size: u8, length_size: u8, ) -> Result, FormatError> { - let entries = v1_group_entries(file_data, sym_table_msg, offset_size, length_size)?; + let entries = v1_group_entries(file_data, sym_table_msg, offset_size, length_size, true)?; if entries.iter().any(|e| e.name.is_empty()) { return Err(FormatError::InvalidLinkName); } @@ -57,11 +57,16 @@ pub fn resolve_v1_group_entries_in( /// Every entry of a v1 group, empty names included — for looking a name up, /// which never matches an empty name. +/// +/// With `hint_headers` (a listing, whose children are usually opened +/// next), each entry's object header is hinted (see +/// [`Storage::hint`]) as soon as its symbol table node is read. pub(crate) fn v1_group_entries( file_data: &S, sym_table_msg: &SymbolTableMessage, offset_size: u8, length_size: u8, + hint_headers: bool, ) -> Result, FormatError> { // Parse local heap let heap = LocalHeap::parse_in( @@ -93,12 +98,23 @@ pub(crate) fn v1_group_entries( // `storage::touch` does); that error is returned. let mut failed = None; for snod_addr in snod_addrs { + let snod = checked_addr(snod_addr) + .and_then(|a| SymbolTableNode::parse_in(file_data, a, offset_size)); + if hint_headers && let Ok(snod) = &snod { + for entry in &snod.entries { + if entry.object_header_address != u64::MAX { + file_data.hint( + entry.object_header_address, + crate::object_header::OBJECT_HEADER_HINT_LEN, + ); + } + } + } if failed.is_some() { - let _ = SymbolTableNode::parse_in(file_data, snod_addr, offset_size); continue; } - let mut node = || -> Result<(), FormatError> { - let snod = SymbolTableNode::parse_in(file_data, checked_addr(snod_addr)?, offset_size)?; + let node = || -> Result<(), FormatError> { + let snod = snod?; for entry in &snod.entries { // Like libhdf5, look at the heap's free list only once a name // is needed: an empty group with a damaged heap still lists. @@ -298,7 +314,13 @@ pub fn resolve_path_in( let mut current_sym_table = root_sym_table.clone(); for (i, component) in components.iter().enumerate() { - let entries = v1_group_entries(file_data, ¤t_sym_table, offset_size, length_size)?; + let entries = v1_group_entries( + file_data, + ¤t_sym_table, + offset_size, + length_size, + false, + )?; let found = entries.iter().find(|e| e.name == *component); match found { diff --git a/crates/clawhdf5-format/src/group_v2.rs b/crates/clawhdf5-format/src/group_v2.rs index a3e026d..bc1a705 100644 --- a/crates/clawhdf5-format/src/group_v2.rs +++ b/crates/clawhdf5-format/src/group_v2.rs @@ -101,18 +101,50 @@ fn resolve_compact_entries( Ok(entries) } +/// What a version-2 B-tree header takes with 8-byte offsets and lengths +/// (22 bytes of fields, the root node's address and record count, and the +/// checksum), rounded up: hinted before one is read. +const BTREE_V2_HEADER_HINT_LEN: usize = 64; + +/// The fractal heap of a dense group. The name index's header and the +/// heap's root block are read next, whatever the lookup: they are hinted +/// (see [`Storage::hint`]) so that a storage fetching between attempts +/// gets them in the same round trip as the heap's header. +fn dense_heap( + file_data: &S, + link_info: &LinkInfoMessage, + fh_addr: u64, + offset_size: u8, + length_size: u8, +) -> Result { + if let Some(btree_addr) = link_info.btree_name_index_address { + file_data.hint(btree_addr, BTREE_V2_HEADER_HINT_LEN); + } + let fh = + FractalHeapHeader::parse_in(file_data, checked_addr(fh_addr)?, offset_size, length_size)?; + fh.hint_root_block(file_data); + Ok(fh) +} + /// Visit every link in dense storage (fractal heap + B-tree v2 name index). +/// +/// With `hint_headers` (a listing, whose children are usually opened +/// next), the object header of every hard link is hinted (see +/// [`Storage::hint`]) as soon as the link is read, even after a failure. fn for_each_dense_link( file_data: &S, link_info: &LinkInfoMessage, fh_addr: u64, offset_size: u8, length_size: u8, + hint_headers: bool, mut visit: impl FnMut(LinkMessage), ) -> Result<(), FormatError> { - // Parse fractal heap - let fh = - FractalHeapHeader::parse_in(file_data, checked_addr(fh_addr)?, offset_size, length_size)?; + let fh = dense_heap(file_data, link_info, fh_addr, offset_size, length_size)?; + if hint_headers { + // Every link is read: so is every block of the heap. + fh.hint_managed_blocks(file_data); + } // Parse B-tree v2 for name index let btree_addr = link_info @@ -144,11 +176,27 @@ fn for_each_dense_link( let id_bytes = &record.data[id_offset..id_offset + fh.heap_id_length as usize]; // Read managed object from fractal heap - let link_data = fh.read_managed_object_in(file_data, id_bytes, offset_size); + let link = fh + .read_managed_object_in(file_data, id_bytes, offset_size) + .and_then(|d| parse_link(&d, offset_size)); + if hint_headers + && let Ok(Some(LinkMessage { + link_target: + LinkTarget::Hard { + object_header_address, + }, + .. + })) = &link + { + file_data.hint( + *object_header_address, + crate::object_header::OBJECT_HEADER_HINT_LEN, + ); + } if failed.is_some() { continue; } - match link_data.and_then(|d| parse_link(&d, offset_size)) { + match link { Ok(Some(link)) => visit(link), Ok(None) => {} Err(e) => failed = Some(e), @@ -175,6 +223,7 @@ fn resolve_dense_entries( fh_addr, offset_size, length_size, + true, |link| { if let LinkTarget::Hard { object_header_address, @@ -247,8 +296,7 @@ fn links_named( return Ok(found); }; - let fh = - FractalHeapHeader::parse_in(file_data, checked_addr(fh_addr)?, offset_size, length_size)?; + let fh = dense_heap(file_data, &link_info, fh_addr, offset_size, length_size)?; let btree_addr = link_info .btree_name_index_address .ok_or_else(|| FormatError::PathNotFound(String::from("no B-tree v2 name index")))?; @@ -265,6 +313,7 @@ fn links_named( fh_addr, offset_size, length_size, + false, |link| { if link.name == name { found.push(link); @@ -409,7 +458,7 @@ fn resolve_child_core( let not_found = || FormatError::PathNotFound(String::from(name)); let header = ObjectHeader::parse_in(file_data, checked_addr(group_address)?, os, ls)?; if !is_v2_group(&header) || is_v1_group(&header) { - return resolve_group_children_in(file_data, superblock, group_address)? + return group_children(file_data, superblock, group_address, false)? .into_iter() .find(|e| e.name == name) .map(|e| e.object_header_address) @@ -575,6 +624,18 @@ fn resolve_group_children_core( file_data: &S, superblock: &Superblock, group_address: u64, +) -> Result, FormatError> { + group_children(file_data, superblock, group_address, true) +} + +/// [`resolve_group_children`]; with `hint_headers`, every child's object +/// header is hinted (see [`Storage::hint`]) as soon as its address is +/// known, for a listing whose children are opened next. +fn group_children( + file_data: &S, + superblock: &Superblock, + group_address: u64, + hint_headers: bool, ) -> Result, FormatError> { let os = superblock.offset_size; let ls = superblock.length_size; @@ -589,7 +650,10 @@ fn resolve_group_children_core( .find(|m| m.msg_type == MessageType::SymbolTable) .ok_or_else(|| FormatError::PathNotFound(String::from("no symbol table message")))?; let stm = SymbolTableMessage::parse(&sym_msg.data, os)?; - let all = group_v1::resolve_v1_group_entries_in(file_data, &stm, os, ls)?; + let all = group_v1::v1_group_entries(file_data, &stm, os, ls, hint_headers)?; + if all.iter().any(|e| e.name.is_empty()) { + return Err(FormatError::InvalidLinkName); + } if all.iter().any(group_v1::is_v1_soft_link) { soft = group_v1::v1_soft_links_in(file_data, &stm, os, ls)?; } @@ -615,7 +679,7 @@ fn resolve_group_children_core( }; let link_info = find_link_info(&header, os)?; if let Some(fh_addr) = link_info.fractal_heap_address { - for_each_dense_link(file_data, &link_info, fh_addr, os, ls, visit)?; + for_each_dense_link(file_data, &link_info, fh_addr, os, ls, hint_headers, visit)?; } else { for msg in &header.messages { if msg.msg_type == MessageType::Link @@ -737,7 +801,7 @@ fn resolve_group_entries( let stm = SymbolTableMessage::parse(&sym_msg.data, offset_size)?; // A lookup: an entry with an empty name (which fails a listing) is // skipped by the name comparison, as in libhdf5. - group_v1::v1_group_entries(file_data, &stm, offset_size, length_size) + group_v1::v1_group_entries(file_data, &stm, offset_size, length_size, false) } else if is_v2_group(object_header) { resolve_v2_group_entries_in(file_data, object_header, offset_size, length_size) } else { diff --git a/crates/clawhdf5-format/src/object_header.rs b/crates/clawhdf5-format/src/object_header.rs index afcb9c5..5bf2c13 100644 --- a/crates/clawhdf5-format/src/object_header.rs +++ b/crates/clawhdf5-format/src/object_header.rs @@ -163,6 +163,12 @@ impl ObjectHeader { offset_size: u8, length_size: u8, ) -> Result { + // The first chunk is read once the prefix says how long it is: say + // so (see `Storage::hint`), for a storage that fetches between + // attempts. + if file.as_contiguous().is_none() { + file.hint(offset, OBJECT_HEADER_HINT_LEN); + } // The longest prefix of either version, in one read. It holds the // whole prefix or ends at the end of the file, so its bounds checks // are the whole-file ones. @@ -279,10 +285,20 @@ impl ObjectHeader { let mut spans = ChunkSpans::new(file.len(), offset, length)?; let mut chunk0_count = 0usize; let mut next = 0usize; + let hints = file.as_contiguous().is_none(); while let Some((chunk_offset, chunk_length)) = spans.get(next) { let chunk = read_exact_at(file, chunk_offset, chunk_length)?; + let known = spans.len; let count = Self::parse_v1_messages(&chunk, offset_size, length_size, messages, &mut spans)?; + // The continuation chunks this one names are read next. + if hints { + for i in known..spans.len { + if let Some((o, l)) = spans.get(i) { + file.hint(o, l); + } + } + } // Only the first chunk's messages are held to the prefix count. if next == 0 { chunk0_count = count; @@ -477,8 +493,17 @@ impl ObjectHeader { // whenever a message no longer fits), up to the same bound as a // version-1 header. let mut spans = ChunkSpans::new(file.len(), base as u64, chunk0_msg_end.saturating_add(4))?; + // The continuation chunks a chunk names are read next (see + // `Storage::hint`). + let hints = file.as_contiguous().is_none(); + if hints { + for &(o, l) in &continuations { + file.hint(o as u64, l); + } + } while let Some((cont_offset, cont_length)) = continuations.pop() { spans.add(cont_offset as u64, cont_length)?; + let known = continuations.len(); Self::parse_v2_continuation( file, cont_offset as u64, @@ -489,6 +514,11 @@ impl ObjectHeader { &mut messages, &mut continuations, )?; + if hints { + for &(o, l) in &continuations[known..] { + file.hint(o as u64, l); + } + } } Ok(ObjectHeader { @@ -634,6 +664,11 @@ impl ObjectHeader { } } +/// What an object header is hinted to take before its prefix is read (see +/// [`Storage::hint`]): the first chunk of a typical dataset's header. A +/// longer header is read all the same. +pub(crate) const OBJECT_HEADER_HINT_LEN: usize = 512; + /// Longest version-2 object header prefix: signature(4) + version(1) + /// flags(1) + times(16) + attribute phase change(4) + chunk-0 size(8). const V2_PREFIX_MAX: usize = 34; diff --git a/crates/clawhdf5-format/src/symbol_table.rs b/crates/clawhdf5-format/src/symbol_table.rs index 467f43b..4243106 100644 --- a/crates/clawhdf5-format/src/symbol_table.rs +++ b/crates/clawhdf5-format/src/symbol_table.rs @@ -90,6 +90,10 @@ impl SymbolTableNode { offset: u64, offset_size: u8, ) -> Result { + // The entries are read once the header says how many there are: + // say so (see `Storage::hint`), for libhdf5's default node size + // (group leaf K = 4: 8 entries of 40 bytes with 8-byte offsets). + file.hint(offset, 8 + 8 * (2 * usize::from(offset_size) + 24)); // signature(4) + version(1) + reserved(1) + number_of_symbols(2) = 8 let header = read_exact_at(file, offset, 8)?; diff --git a/crates/clawhdf5-wasm/tests/lazy.rs b/crates/clawhdf5-wasm/tests/lazy.rs index 0d0b536..9c972e0 100644 --- a/crates/clawhdf5-wasm/tests/lazy.rs +++ b/crates/clawhdf5-wasm/tests/lazy.rs @@ -146,8 +146,8 @@ fn transcript(api: &impl Api) -> Vec { /// takes), and agrees with the in-memory one: the same values, and an error /// wherever it has one (a malformed file can fail at a different check, /// with a different message, when read by ranges). Returns what the lazy -/// reader fetched and its transcript. -fn check_equal(name: &str, data: &[u8], block: u64) -> (u64, u64, Vec) { +/// reader fetched (requests, bytes, passes) and its transcript. +fn check_equal(name: &str, data: &[u8], block: u64) -> (u64, u64, u64, Vec) { let ctx = format!("{name} (blocks of {block} B)"); let ranged = Reader::open_storage(Arc::new(CountingStorage::new(data.to_vec()))); let local = Reader::open(data.to_vec()); @@ -156,7 +156,7 @@ fn check_equal(name: &str, data: &[u8], block: u64) -> (u64, u64, Vec) { (Ok(r), Ok(l), Ok(z)) => (r, l, z), (Err(r), Err(_), Err(z)) => { assert_eq!(z, r, "{ctx}: open error"); - return (0, 0, Vec::new()); + return (0, 0, 0, Vec::new()); } (r, l, z) => panic!( "{ctx}: opens differently: ranged {:?}, in memory {:?}, lazily {:?}", @@ -184,7 +184,7 @@ fn check_equal(name: &str, data: &[u8], block: u64) -> (u64, u64, Vec) { } assert_eq!(got.len(), local.len(), "{ctx}: transcript length"); let st = lazy.storage.stats(); - (st.requests, st.bytes_fetched, got) + (st.requests, st.bytes_fetched, st.passes, got) } fn config(block: u64) -> LazyConfig { @@ -223,7 +223,7 @@ fn builder_file() -> Vec { fn builder_files_read_the_same_at_every_block_size() { let data = builder_file(); for block in [512, 4096, 1 << 20] { - let (requests, _, lines) = check_equal("builder", &data, block); + let (requests, _, _, lines) = check_equal("builder", &data, block); assert!(requests > 0); // The transcript covers every object, values included. assert!(lines.iter().any(|l| l.starts_with("/grid read: Ok"))); @@ -315,7 +315,7 @@ fn h5py_and_netcdf4_files_read_the_same_lazily() { for name in ["fixture.h5", "fixture.nc"] { let data = std::fs::read(dir.path().join(name)).unwrap(); for block in [512, 64 * 1024] { - let (_, _, lines) = check_equal(name, &data, block); + let (_, _, _, lines) = check_equal(name, &data, block); assert!(lines.iter().filter(|l| l.contains(" read: Ok")).count() >= 2); } } @@ -460,16 +460,17 @@ fn corpus_files_read_the_same_lazily() { } files.sort(); assert!(!files.is_empty(), "no HDF5 files under {dirs}"); - let (mut requests, mut bytes, mut total) = (0u64, 0u64, 0u64); + let (mut requests, mut bytes, mut passes, mut total) = (0u64, 0u64, 0u64, 0u64); for f in &files { let data = std::fs::read(f).unwrap(); total += data.len() as u64; - let (r, b, _) = check_equal(&f.display().to_string(), &data, 64 * 1024); + let (r, b, p, _) = check_equal(&f.display().to_string(), &data, 64 * 1024); requests += r; bytes += b; + passes += p; } eprintln!( - "{} files ({total} bytes): {requests} requests, {bytes} bytes fetched", + "{} files ({total} bytes): {passes} passes, {requests} requests, {bytes} bytes fetched", files.len() ); } @@ -496,13 +497,38 @@ fn listing_cost(data: &[u8], path: &str, block: u64) -> (u64, u64) { ) } +/// An h5py file of `n` datasets of 256 `f32` each (`d0` ... ) in the root +/// group, written with `libver`. +fn h5py_many(dir: &Path, libver: &str, n: usize) -> Vec { + let path = dir.join(format!("{libver}_{n}.h5")); + let script = format!( + "import h5py, numpy as np\n\ + with h5py.File({:?}, 'w', libver='{libver}') as f:\n\ + \x20 for i in range({n}):\n\ + \x20 f.create_dataset('d%d' % i, data=np.full(256, i, np.float32))\n", + path.display().to_string() + ); + let out = Command::new(python()) + .args(["-c", &script]) + .output() + .unwrap(); + assert!( + out.status.success(), + "{}", + String::from_utf8_lossy(&out.stderr) + ); + std::fs::read(&path).unwrap() +} + /// Listing a group reads every child's object header, and its index (B-tree /// and symbol table nodes, or B-tree v2 and heap blocks) before that. Each -/// pass asks for every node of a level it is missing, not the first one -/// only, so the passes (network round trips) grow with the depth of the -/// index, not with the number of children: 2000 children with headers -/// scattered over 512-byte blocks list in a handful of passes, where each -/// header block used to cost its own. +/// pass asks for every node of the index it can reach (a failed node does +/// not stop the walk), the blocks it has been told it reads next (a node's +/// body, a symbol table node's entries, the heap's blocks, each child's +/// header: `Storage::hint`), so the passes (network round trips) follow the +/// depth of the index, not the number of children: 2000 children with +/// headers scattered over 512-byte blocks list in a handful of passes, +/// where each header block used to cost its own. #[test] fn listing_a_large_group_takes_a_few_passes() { let mut b = FileBuilder::new(); @@ -514,39 +540,31 @@ fn listing_a_large_group_takes_a_few_passes() { let data = b.finish().unwrap(); let (passes, requests) = listing_cost(&data, "/many", 512); eprintln!("FileBuilder, 600 children: {passes} passes, {requests} requests"); - assert!(passes <= 6, "{passes} passes"); + // 5 before hints (2026-09-27), 102 before the walks went on past a miss. + assert!(passes <= 4, "{passes} passes"); if !python_available() { + assert!( + !std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1"), + "CLAWHDF5_REQUIRE_INTEROP=1 but {} lacks h5py/netCDF4/numpy", + python() + ); return; } let dir = tempfile::tempdir().unwrap(); - for libver in ["earliest", "latest"] { - let path = dir.path().join(format!("{libver}.h5")); - let script = format!( - "import h5py, numpy as np\n\ - with h5py.File({:?}, 'w', libver='{libver}') as f:\n\ - \x20 for i in range(2000):\n\ - \x20 f.create_dataset('d%d' % i, data=np.full(256, i, np.float32))\n", - path.display().to_string() - ); - let out = Command::new(python()) - .args(["-c", &script]) - .output() - .unwrap(); - assert!( - out.status.success(), - "{}", - String::from_utf8_lossy(&out.stderr) - ); - let data = std::fs::read(&path).unwrap(); + // Most passes each may take: 8 and 11 before hints (2026-09-27). + for (libver, most) in [("earliest", 5), ("latest", 6)] { + let data = h5py_many(dir.path(), libver, 2000); let (passes, requests) = listing_cost(&data, "/", 512); eprintln!("h5py libver={libver}, 2000 children: {passes} passes, {requests} requests"); - assert!(passes <= 12, "{libver}: {passes} passes"); + assert!(passes <= most, "{libver}: {passes} passes"); } } /// `CLAWHDF5_WASM_LIST_FILE=file.h5`: what listing the root group of that -/// file costs lazily, at 1 MiB and 64 KiB blocks (a measurement, printed). +/// file costs lazily, and opening it and reading one dataset whole +/// (`CLAWHDF5_WASM_READ`, by default the middle dataset of the listing), +/// at 1 MiB and 64 KiB blocks (a measurement, printed). #[test] fn listing_cost_of_a_given_file() { let Ok(path) = std::env::var("CLAWHDF5_WASM_LIST_FILE") else { @@ -563,16 +581,44 @@ fn listing_cost_of_a_given_file() { ) .unwrap(); let open = lazy.storage.stats(); - let n = lazy.call(|r| r.list("/")).unwrap().len(); + let list = lazy.call(|r| r.list("/")).unwrap(); let st = lazy.storage.stats(); eprintln!( - "{path} ({} bytes), {block}-byte blocks: open {} requests / {} passes; list('/') of {n}: {} passes, {} requests, {} bytes", + "{path} ({} bytes), {block}-byte blocks: open {} requests / {} passes; list('/') of {}: {} passes, {} requests, {} bytes ({} blocks hinted)", data.len(), open.requests, open.passes, + list.len(), st.passes - open.passes, st.requests - open.requests, - st.bytes_fetched - open.bytes_fetched + st.bytes_fetched - open.bytes_fetched, + st.hinted_blocks - open.hinted_blocks, + ); + let name = std::env::var("CLAWHDF5_WASM_READ").unwrap_or_else(|_| { + let datasets: Vec<_> = list.iter().filter(|c| c.kind == Kind::Dataset).collect(); + format!("/{}", datasets[datasets.len() / 2].name) + }); + // Open and read on a fresh cache: the open's own cost (the probe + // block and its passes) and then the read's. + let fresh = Lazy::open( + data.clone(), + LazyConfig { + block_size: block, + ..LazyConfig::default() + }, + ) + .unwrap(); + let open = fresh.storage.stats(); + let values = fresh.call(|r| r.read(&name, None)).unwrap().data.len(); + let st = fresh.storage.stats(); + eprintln!( + " open + read('{name}') ({values} values): {} passes, {} requests, {} bytes (the open: {} passes, {} requests, {} bytes)", + st.passes, + st.requests, + st.bytes_fetched, + open.passes, + open.requests, + open.bytes_fetched, ); } } From 761bdbf24fbf99e10a5da7c4bb4315da9b1952a9 Mon Sep 17 00:00:00 2001 From: osobh Date: Sun, 27 Sep 2026 22:47:47 -0500 Subject: [PATCH 07/10] format: look a name up in a v1 group down its B-tree, as libhdf5 does Opening one dataset of a v1 (symbol table) group read every symbol table node and every name of the group to find it: over openUrl, 74 requests and 193 MB to read one 64 KiB dataset of the reviewer's 3000-dataset h5py file (libver earliest) at 1 MiB blocks, 515 requests and 34 MB at 64 KiB. Locally it made a lookup O(entries). `group_v1::find_v1_entry` looks the name up as libhdf5's `H5G__stab_lookup` does: `H5B_find`'s binary search at each B-tree node with `H5G__node_cmp3` (left key < name <= right key, keys being names in the local heap, compared bytewise like strcmp), then the one symbol table node, after the heap's free list is checked as the listing does. The heap's data segment (up to 1 MiB) is hinted, since the keys are read one after another. Path resolution uses it for a v1 group; only when it does not find a hard link of that name (a soft link, or a B-tree out of name order, damaged or hand-made, where libhdf5 would report the name missing) does it read every entry as before. A storage error is returned as is (over the lazy reader, a miss: reading every entry would not get further). The result differs from before only in a group holding two entries of one name, where the B-tree's is now the one found, as in libhdf5. Measured with tests/lazy.rs listing_cost_of_a_given_file, open + read one 64 KiB dataset of the reviewer-like file, passes/requests/bytes, before -> after (open included): earliest, 1 MiB: 7/74/193.6 MB -> 6/5/5.2 MB earliest, 64 KiB: 9/515/34.1 MB -> 8/7/524 KB latest (dense groups, already a name-index lookup): 1 MiB 8/7/6.7 MB -> 7/7/6.7 MB, 64 KiB 9/8/581 KB -> 8/8/581 KB (the previous commit's hints: the name index header with the heap header) New tests, failing before: every child of v1_groups_400.h5 resolves to its listed address reading under 1/8 of the listing's bytes, and missing names are not found; a name moved out of B-tree order is still found (by the fallback); reading one of 2000 datasets lazily at 512-byte blocks takes at most 7 passes and 6 requests (earliest; 529 requests, 333 kB before) and 8 passes, 9 requests (latest). Conformance 600 of 697 (baseline 600), no file's class or detail changed against a run of main. Co-Authored-By: Claude Opus 5.5 (1M context) --- crates/clawhdf5-format/src/btree_v1.rs | 2 +- crates/clawhdf5-format/src/group_v1.rs | 91 +++++++++++++++++++++- crates/clawhdf5-format/src/group_v2.rs | 101 +++++++++++++++++++++++++ crates/clawhdf5-wasm/tests/lazy.rs | 81 ++++++++++++++++++-- 4 files changed, 265 insertions(+), 10 deletions(-) diff --git a/crates/clawhdf5-format/src/btree_v1.rs b/crates/clawhdf5-format/src/btree_v1.rs index 7ae2af7..78b2329 100644 --- a/crates/clawhdf5-format/src/btree_v1.rs +++ b/crates/clawhdf5-format/src/btree_v1.rs @@ -158,7 +158,7 @@ impl BTreeV1Node { } /// Maximum recursion depth for B-tree traversal (malformed data protection). -const MAX_BTREE_DEPTH: usize = 64; +pub(crate) const MAX_BTREE_DEPTH: usize = 64; /// What a symbol table node takes with libhdf5's default group leaf K (4): /// its 8-byte header and 2K entries of 40 bytes (8-byte offsets). Hinted diff --git a/crates/clawhdf5-format/src/group_v1.rs b/crates/clawhdf5-format/src/group_v1.rs index f131acc..4a930dc 100644 --- a/crates/clawhdf5-format/src/group_v1.rs +++ b/crates/clawhdf5-format/src/group_v1.rs @@ -4,7 +4,7 @@ use alloc::{string::String, vec::Vec}; use crate::addr::checked_addr; -use crate::btree_v1::collect_symbol_table_nodes_in; +use crate::btree_v1::{BTreeV1Node, collect_symbol_table_nodes_in}; use crate::error::FormatError; use crate::local_heap::LocalHeap; use crate::message_type::MessageType; @@ -142,6 +142,95 @@ pub(crate) fn v1_group_entries( } } +/// The entry called `name` in a v1 group, looked up as libhdf5 looks it up +/// (`H5G__stab_lookup`: `H5B_find` down the group's B-tree, then +/// `H5G__node_found` in one symbol table node) instead of by reading every +/// entry: at each node, the child whose key interval holds the name +/// (left key < name <= right key, keys being names in the local heap, +/// compared bytewise as `strcmp` does) is found by binary search. Reads +/// O(depth) nodes and names, where listing reads the whole group. +/// +/// `Ok(None)` when the search does not lead to the name. In a group whose +/// B-tree is out of name order (damaged, or made by hand) that does not +/// prove it absent, so callers then fall back to reading every entry; +/// libhdf5 would report it missing. +pub(crate) fn find_v1_entry( + file_data: &S, + sym_table_msg: &SymbolTableMessage, + name: &str, + offset_size: u8, + length_size: u8, +) -> Result, FormatError> { + let heap = LocalHeap::parse_in( + file_data, + checked_addr(sym_table_msg.local_heap_address)?, + offset_size, + length_size, + )?; + // The keys and names are read from the heap's data segment one at a + // time: hint it (see `Storage::hint`). + file_data.hint( + heap.data_segment_address, + usize::try_from(heap.data_segment_size).map_or(1 << 20, |n| n.min(1 << 20)), + ); + // As in listing: the heap's free list is checked before a name is used. + let mut heap_checked = false; + let mut name_at = |offset: u64| -> Result { + if !heap_checked { + heap.validate_free_list_in(file_data, length_size)?; + heap_checked = true; + } + heap.read_string_in(file_data, offset) + }; + let want = name.as_bytes(); + let mut address = sym_table_msg.btree_address; + for _ in 0..=crate::btree_v1::MAX_BTREE_DEPTH { + let node = + BTreeV1Node::parse_in(file_data, checked_addr(address)?, offset_size, length_size)?; + if node.node_type != 0 { + return Err(FormatError::InvalidBTreeNodeType(node.node_type)); + } + // H5B_find's binary search with H5G__node_cmp3: go left when the + // name sorts at or before the left key, right when after the right + // key; otherwise this child holds it. + let (mut lo, mut hi) = (0usize, node.children.len()); + let mut child = None; + while lo < hi { + let i = lo + (hi - lo) / 2; + let (Some(&left), Some(&right)) = (node.keys.get(i), node.keys.get(i + 1)) else { + return Ok(None); + }; + if want <= name_at(left)?.as_bytes() { + hi = i; + } else if want > name_at(right)?.as_bytes() { + lo = i + 1; + } else { + child = Some(node.children[i]); + break; + } + } + let Some(child) = child else { + return Ok(None); + }; + if node.node_level > 0 { + address = child; + continue; + } + let snod = SymbolTableNode::parse_in(file_data, checked_addr(child)?, offset_size)?; + for entry in &snod.entries { + if name_at(entry.link_name_offset)?.as_bytes() == want { + return Ok(Some(GroupEntry { + name: String::from(name), + object_header_address: entry.object_header_address, + cache_type: entry.cache_type, + })); + } + } + return Ok(None); + } + Err(FormatError::NestingDepthExceeded) +} + /// Symbol table cache type for a soft link: the scratch pad's first four bytes /// are the local-heap offset of the link's target path, and the entry's object /// header address is undefined. diff --git a/crates/clawhdf5-format/src/group_v2.rs b/crates/clawhdf5-format/src/group_v2.rs index bc1a705..e902239 100644 --- a/crates/clawhdf5-format/src/group_v2.rs +++ b/crates/clawhdf5-format/src/group_v2.rs @@ -385,6 +385,27 @@ fn lookup_link( length_size: u8, ) -> Result, FormatError> { if is_v1_group(object_header) { + // Down the group's B-tree, as libhdf5 looks a name up; only when + // that does not find a hard link of that name is every entry read + // (a soft link, a group whose B-tree is out of order). A storage + // error (a read a restartable storage has not fetched yet) is + // returned as is: reading every entry would not get further. + if let Some(sym_msg) = object_header + .messages + .iter() + .find(|m| m.msg_type == MessageType::SymbolTable) + { + let stm = SymbolTableMessage::parse(&sym_msg.data, offset_size)?; + match group_v1::find_v1_entry(file_data, &stm, name, offset_size, length_size) { + Ok(Some(e)) if e.object_header_address != u64::MAX => { + return Ok(Some(LinkTarget::Hard { + object_header_address: e.object_header_address, + })); + } + Err(e @ FormatError::Storage(_)) => return Err(e), + _ => {} + } + } let entries = resolve_group_entries(file_data, object_header, offset_size, length_size)?; if let Some(e) = entries .iter() @@ -984,6 +1005,86 @@ mod tests { assert_eq!(values, vec![22.5, 23.1, 21.8]); } + /// `v1_groups_400.h5` from its superblock on (it has a user block). + fn v1_groups_400() -> (Vec, Superblock) { + let all: &[u8] = include_bytes!("../tests/fixtures/v1_groups_400.h5"); + let data = all[signature::find_signature(all).unwrap()..].to_vec(); + let sb = Superblock::parse(&data, 0).unwrap(); + (data, sb) + } + + /// Every child of a v1 group resolves by name, down the group's B-tree, + /// to the address the listing gives, reading a small part of what the + /// listing reads; a name the group does not hold is not found. + #[test] + fn v1_lookup_down_the_btree_agrees_with_the_listing() { + let (data, sb) = v1_groups_400(); + let children = resolve_group_children(&data, &sb, sb.root_group_address).unwrap(); + assert_eq!(children.len(), 401); + for c in &children { + let path = format!("/{}", c.name); + assert_eq!( + resolve_path_any(&data, &sb, &path).unwrap(), + c.object_header_address, + "{path}" + ); + } + for missing in ["/g0400", "/a", "/g", "/g00000", "/zz", "/x0"] { + assert!( + matches!( + resolve_path_any(&data, &sb, missing), + Err(FormatError::PathNotFound(_)) + ), + "{missing}" + ); + } + let st = crate::storage::CountingStorage::new(data.clone()); + resolve_group_children_in(&st, &sb, sb.root_group_address).unwrap(); + let listing = st.bytes_read(); + st.reset(); + let last = children.last().unwrap(); + assert_eq!( + resolve_path_any_in(&st, &sb, &format!("/{}", last.name)).unwrap(), + last.object_header_address + ); + assert!( + st.bytes_read() * 8 < listing, + "lookup read {} bytes, listing {listing}", + st.bytes_read() + ); + } + + /// A v1 group whose B-tree is out of name order (a name changed in the + /// heap so that it sorts past every key) is still looked up by reading + /// every entry, as before the lookup went down the B-tree. + #[test] + fn v1_lookup_falls_back_when_the_btree_is_out_of_order() { + let (mut data, sb) = v1_groups_400(); + let at: Vec = data + .windows(6) + .enumerate() + .filter(|(_, w)| *w == b"g0200\0") + .map(|(i, _)| i) + .collect(); + assert_eq!(at.len(), 1, "one heap string"); + data[at[0]] = b'~'; + let children = resolve_group_children(&data, &sb, sb.root_group_address).unwrap(); + let moved = children.iter().find(|c| c.name == "~0200").unwrap(); + assert_eq!( + resolve_path_any(&data, &sb, "/~0200").unwrap(), + moved.object_header_address + ); + assert!(resolve_path_any(&data, &sb, "/g0200").is_err()); + for c in &children { + let path = format!("/{}", c.name); + assert_eq!( + resolve_path_any(&data, &sb, &path).unwrap(), + c.object_header_address, + "{path}" + ); + } + } + #[test] fn path_not_found_v2() { let file_data: &[u8] = include_bytes!("../tests/fixtures/v2_groups.h5"); diff --git a/crates/clawhdf5-wasm/tests/lazy.rs b/crates/clawhdf5-wasm/tests/lazy.rs index 9c972e0..26e10af 100644 --- a/crates/clawhdf5-wasm/tests/lazy.rs +++ b/crates/clawhdf5-wasm/tests/lazy.rs @@ -475,11 +475,17 @@ fn corpus_files_read_the_same_lazily() { ); } -/// Passes and requests `list(path)` takes on a file opened lazily at -/// `block`-byte blocks (the open not counted), checking the listing against -/// the in-memory one. -fn listing_cost(data: &[u8], path: &str, block: u64) -> (u64, u64) { - let want = Reader::open(data.to_vec()).unwrap().list(path).unwrap(); +/// What one call cost on a file opened lazily (the open not counted). +#[derive(Debug, Clone, Copy)] +struct Cost { + passes: u64, + requests: u64, + bytes: u64, +} + +/// Open `data` lazily at `block`-byte blocks, run `op`, and return its +/// result and what it cost (the open not counted). +fn cost_of(data: &[u8], block: u64, op: impl Fn(&Reader) -> T) -> (T, Cost) { let lazy = Lazy::open( data.to_vec(), LazyConfig { @@ -489,14 +495,41 @@ fn listing_cost(data: &[u8], path: &str, block: u64) -> (u64, u64) { ) .unwrap(); let before = lazy.storage.stats(); - assert_eq!(lazy.call(|r| r.list(path)).unwrap(), want); + let out = lazy.call(op); let after = lazy.storage.stats(); ( - after.passes - before.passes, - after.requests - before.requests, + out, + Cost { + passes: after.passes - before.passes, + requests: after.requests - before.requests, + bytes: after.bytes_fetched - before.bytes_fetched, + }, ) } +/// Passes and requests `list(path)` takes on a file opened lazily at +/// `block`-byte blocks (the open not counted), checking the listing against +/// the in-memory one. +fn listing_cost(data: &[u8], path: &str, block: u64) -> (u64, u64) { + let want = Reader::open(data.to_vec()).unwrap().list(path).unwrap(); + let (got, cost) = cost_of(data, block, |r| r.list(path)); + assert_eq!(got.unwrap(), want); + (cost.passes, cost.requests) +} + +/// What reading the dataset at `path` whole takes on a file opened lazily +/// at `block`-byte blocks (the open not counted), checking the values +/// against the in-memory read. +fn read_cost(data: &[u8], path: &str, block: u64) -> Cost { + let want = Reader::open(data.to_vec()) + .unwrap() + .read(path, None) + .unwrap(); + let (got, cost) = cost_of(data, block, |r| r.read(path, None)); + assert_eq!(got.unwrap(), want, "{path}"); + cost +} + /// An h5py file of `n` datasets of 256 `f32` each (`d0` ... ) in the root /// group, written with `libver`. fn h5py_many(dir: &Path, libver: &str, n: usize) -> Vec { @@ -561,6 +594,38 @@ fn listing_a_large_group_takes_a_few_passes() { } } +/// Opening one dataset of a large group looks its name up, not the whole +/// group: in a v1 (symbol table) group down its B-tree as libhdf5 does, in +/// a dense group down its name index. Reading one small dataset of 2000 +/// fetches a few blocks, where it used to read every symbol table node and +/// name of a v1 group (529 requests, 333 kB at 512-byte blocks, before +/// 2026-09-27). +#[test] +fn reading_one_dataset_of_a_large_group_fetches_a_few_blocks() { + if !python_available() { + assert!( + !std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1"), + "CLAWHDF5_REQUIRE_INTEROP=1 but {} lacks h5py/netCDF4/numpy", + python() + ); + eprintln!("skipping: {} lacks h5py/netCDF4/numpy", python()); + return; + } + let dir = tempfile::tempdir().unwrap(); + for (libver, most_passes, most_requests) in [("earliest", 7, 6), ("latest", 8, 9)] { + let data = h5py_many(dir.path(), libver, 2000); + for path in ["/d0", "/d1234", "/d1999"] { + let cost = read_cost(&data, path, 512); + eprintln!("h5py libver={libver}, 2000 children, read {path}: {cost:?}"); + assert!( + cost.passes <= most_passes && cost.requests <= most_requests, + "{libver} {path}: {cost:?}" + ); + assert!(cost.bytes <= 64 * 1024, "{libver} {path}: {cost:?}"); + } + } +} + /// `CLAWHDF5_WASM_LIST_FILE=file.h5`: what listing the root group of that /// file costs lazily, and opening it and reading one dataset whole /// (`CLAWHDF5_WASM_READ`, by default the middle dataset of the listing), From 2e5b0595304c3ecbeb33c096925eceffd1951acd Mon Sep 17 00:00:00 2001 From: osobh Date: Sun, 27 Sep 2026 22:48:47 -0500 Subject: [PATCH 08/10] docs: fewer round trips for remote files in the browser, counted CHANGELOG (Unreleased): the v1 B-tree lookup, `Storage::hint`, the walks that go on past a missing node, and the counts before and after on an h5py file like the reviewer's (3000 datasets, 198 MB, earliest and latest libver, 1 MiB and 64 KiB blocks), the corpus read lazily at 512 B and 64 KiB blocks, and the Node/Chromium suite. known-issues (browser limits, round trips): the new counts, why the passes cannot go lower (the chain of addresses), why merging nearby requests does not help such a file, and that a second listing refetches at 1 MiB blocks when the file's metadata blocks exceed `cacheSize`. range-reads.md M4 status and the viewer README follow. Co-Authored-By: Claude Opus 5.5 (1M context) --- CHANGELOG.md | 55 ++++++++++++++++++++++++++++++++++ docs/design/range-reads.md | 10 +++++++ docs/known-issues.md | 27 ++++++++++++++--- examples/wasm-viewer/README.md | 6 ++-- 4 files changed, 92 insertions(+), 6 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 306dd13..6bc8766 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,61 @@ ## Unreleased +### Remote files in the browser: fewer round trips to list a group or open a dataset (2026-09-27) +- **Opening one dataset of a v1 (symbol table) group no longer reads the + whole group.** A name is looked up down the group's B-tree, as + libhdf5's `H5G__stab_lookup` does (binary search on the node keys, names + in the local heap compared bytewise, then one symbol table node); only + when that finds no hard link of that name (a soft link, or a B-tree out + of name order, where libhdf5 would report it missing) is every entry + read, as before. Local files benefit too (a lookup read O(entries)). + In a group holding two entries of one name, the B-tree's is now the one + found, as in libhdf5. +- **`Storage::hint(offset, len)`** (clawhdf5-format): a parser says what + it reads next — a B-tree node's or symbol table node's body, an object + header's first chunk and continuation chunks, the symbol table nodes a + B-tree leaf names, a dense group's name index header and heap blocks, + and in a listing every child's object header. Every backend ignores it + but the browser's restartable reader (`clawhdf5_wasm::lazy`), which + fetches the hinted blocks it lacks together with the blocks a pass + missed, within the call's `maxFetch` budget; a pass that misses + nothing ignores them, so a hint never adds a round trip, and results + never depend on hints. +- The v1 and v2 B-tree walks of a listing descend into every child after + one fails (before, the siblings were only read, so their subtrees came + a pass later), then return the first error: same results and errors. +- Counted on tank, 2026-09-27, with `CLAWHDF5_WASM_LIST_FILE= + CLAWHDF5_WASM_READ=/d1500 cargo test --release -p clawhdf5-wasm --test + lazy listing_cost_of_a_given_file -- --nocapture` on an h5py file like + the reviewer's (3000 datasets of 16384 `f32`, 198 MB, h5py 3.16 / + HDF5 2.0), passes / requests / bytes, before -> after: + + | file, block size | `list('/')` | open + read one dataset | + |---|---|---| + | earliest, 1 MiB | 6 / 73 / 192.5 MB -> 4 / 68 / 192.5 MB | 7 / 74 / 193.6 MB -> 6 / 5 / 5.2 MB | + | earliest, 64 KiB | 8 / 531 / 35.2 MB -> 5 / 530 / 35.3 MB | 9 / 515 / 34.1 MB -> 8 / 7 / 0.52 MB | + | latest, 1 MiB | 9 / 98 / 196.5 MB -> 5 / 86 / 196.5 MB | 8 / 7 / 6.7 MB -> 7 / 7 / 6.7 MB | + | latest, 64 KiB | 11 / 452 / 29.6 MB -> 6 / 454 / 30.5 MB | 9 / 8 / 0.58 MB -> 8 / 8 / 0.58 MB | + + The listing's passes now follow the depth of the group's index (the + chain index levels -> symbol table nodes or heap objects -> child + headers); its bytes are the child headers, which h5py spreads through + the file (at 1 MiB blocks most of it). The whole corpus read lazily + (`CLAWHDF5_WASM_CORPUS`, 656 files, every object listed, described and + read): 27 513 -> 27 496 passes, 3 001 -> 2 962 requests and 194.8 -> + 195.3 MB at 64 KiB blocks; 41 342 -> 37 517 passes, 32 578 -> 29 511 + requests, 65.4 -> 65.7 MB at 512 B. The Node and Chromium suite + (`examples/wasm-viewer/test/run.sh`) passes unchanged (the 200 MB file + still takes 5 requests, 6 MiB); its corpus comparison fetched 33.54 -> + 33.61 MB. +- Tests: the listing budgets (`listing_a_large_group_takes_a_few_passes`, + 512-byte blocks) are tightened to the new counts (FileBuilder, 600 + children: 5 -> 4 passes; h5py, 2000 children: 8 -> 5 and 11 -> 6); new + `reading_one_dataset_of_a_large_group_fetches_a_few_blocks` (h5py + earliest: 529 requests, 333 kB -> at most 6 requests, 27 kB), v1 + lookups against the listing (and with a name moved out of B-tree + order), hints riding only on misses and within the fetch budget. + ### Deterministic errors on damaged chunked datasets (2026-09-27) - A read through the file's chunk cache listed a damaged dataset's chunks in hash-map order, seeded per `File`, so two opens of the same file could diff --git a/docs/design/range-reads.md b/docs/design/range-reads.md index 4ab9ef5..9dea022 100644 --- a/docs/design/range-reads.md +++ b/docs/design/range-reads.md @@ -607,6 +607,16 @@ fast path within benchmark noise. missing blocks: listing 3000 datasets went from 185 passes to 6. The 32-bit risk below is covered by a Node test that reads data at 3 GiB from a mock server and is refused a 4 GiB file. + - Fewer round trips (2026-09-27, later): the walks descend into every + child after a failure (not only read the siblings), and parsers + call `Storage::hint` for what they read next (node bodies, object + header chunks, a dense group's heap blocks, a listing's child + headers); `LazyStorage` fetches hinted blocks only along with a + pass's real misses and within `maxFetch`, so hints never add a + round trip nor change a result. A v1 group's name is looked up down + its B-tree (`H5G__stab_lookup`), not by listing it. 3000 datasets + list in 4 passes (earliest) and 5 (latest) at 1 MiB blocks, and + opening one of them costs 5 requests, not 74 (CHANGELOG). **M5 — SWMR and growth (later, separate design).** `Storage::len()` may grow; add `File::refresh()` that re-reads the superblock/EOF and invalidates cached diff --git a/docs/known-issues.md b/docs/known-issues.md index 98e1fac..b563aba 100644 --- a/docs/known-issues.md +++ b/docs/known-issues.md @@ -1051,10 +1051,29 @@ cache, but: header, and every node of a level of the group's index, in one pass (since 2026-09-27; it was one round trip per header block): 3000 datasets of an h5py file took 6 passes at 1 MiB blocks, 9 for a - `libver="latest"` file (dense links). Each pass re-parses what the - call reads (CPU, not network). With headers spread through the file - (h5py writes each next to its data) a listing still fetches most of - the file at 1 MiB blocks; a smaller `blockSize` fetches less. + `libver="latest"` file (dense links). **Since 2026-09-27 (later):** + 4 and 5 passes (5 and 6 at 64 KiB, from 8 and 11): the index walks go + on past a missing node, and parsers hint what they read next + (`Storage::hint`: node bodies, the heap's blocks, each child's + header), which the lazy reader fetches with a pass's misses. That is + the depth of the chain (index levels, then symbol table nodes or + heap objects, then headers) plus the pass that finishes; it cannot + go lower without reading structures before their addresses are + known. Opening one dataset of a v1 group looks its name up down the + group's B-tree (it read every entry: 74 requests, 193 MB at 1 MiB + blocks for one 64 KiB dataset of the 3000; now 5 requests, 5 MB). + Each pass re-parses what the call reads (CPU, not network). With + headers spread through the file (h5py writes each next to its data) + a listing still fetches most of the file at 1 MiB blocks (192 of + 198 MB; 35 MB in 530 requests at 64 KiB); a smaller `blockSize` + fetches less. Merging nearby requests does not help such a file: the + blocks a listing needs are five or six apart at 64 KiB, so fewer requests would + mean fetching most of the file. Listing it a second time is free at + 64 KiB blocks, but at 1 MiB its metadata blocks (192 MB) exceed the + 64 MiB `cacheSize`, so they are fetched again (the earliest file: 4 + passes, 50 requests); a larger `cacheSize` keeps them. A file's paged + aggregation (metadata in pages) is not used to fetch its metadata in + one request. - **Memory:** a call keeps every block it reads until it finishes (the cache budget applies between calls). It may fetch at most `maxFetch` bytes (512 MiB by default, at most 1 GiB), and a single read longer diff --git a/examples/wasm-viewer/README.md b/examples/wasm-viewer/README.md index dc9dc05..7eb0849 100644 --- a/examples/wasm-viewer/README.md +++ b/examples/wasm-viewer/README.md @@ -70,8 +70,10 @@ are fetched (in parallel, adjacent blocks in one request), and the pass is run again, until one completes (`docs/design/range-reads.md`, M4). Opening costs one request (the first block, which also gives the file's size); listing a group whose metadata is in blocks already fetched costs none, -and otherwise a round trip per level of the group's index plus one for -its children's headers, all fetched together; +and otherwise about a round trip per level of the group's index plus one +for its children's headers, all fetched together (the reader fetches +what it knows it reads next along with what a pass missed); opening one +object looks its name up in the group's index, not the whole group; reading a chunked dataset costs a round trip for its chunk index (a few for a deep one) and one batch of requests for its chunks. Every answer is checked — a `206` with exactly the bytes asked for, from the same file From 5b45c60c9c440c2b92b2cd873f646c7623a4e35b Mon Sep 17 00:00:00 2001 From: osobh Date: Sun, 27 Sep 2026 23:15:09 -0500 Subject: [PATCH 09/10] docs: ObjectHeader::parse A/B after inlining the v1 message loop (at or below 8f59b2e) Co-Authored-By: Claude Opus 5.5 (1M context) --- BENCHMARKS.md | 40 ++++++++++++++++++++++++++++++++++++++++ CHANGELOG.md | 10 ++++++++++ docs/known-issues.md | 9 ++++++++- 3 files changed, 58 insertions(+), 1 deletion(-) diff --git a/BENCHMARKS.md b/BENCHMARKS.md index f6e8a3c..621f912 100644 --- a/BENCHMARKS.md +++ b/BENCHMARKS.md @@ -500,6 +500,46 @@ explain the slower windows. ## Local file speed after range reads +### `ObjectHeader::parse` back at 8f59b2e's speed (2026-09-27, tank) + +The remaining 4% (below) was the call to the version-1 message loop, which +`4313917` kept out of line with `#[inline(never)]`. Found with A/B builds +changing one piece at a time (perf is not available: `perf_event_paranoid` +4): `#[inline]` on `parse_v1_messages` alone brought +`object_header_parse_x401` from about 24.5–24.9 µs to 23.6–24.0 µs against +8f59b2e's 23.6–24.1 µs (short 4-second rounds); no attribute measured like +`#[inline(never)]`; +creating the chunk list only when a continuation is found measured no +faster on top and was not kept. + +Same method as below: `8f59b2e` built in its own worktree and target +directory, separate binaries alternating, `taskset -c 5 +local_metadata_bench --bench --warm-up-time 3 --measurement-time 10`, every +binary started with the 1-minute load average below 2 (0.19–1.86) and no +`rustc` running. Candidate: `96086ad` (this change). Median (range) of 3 +rounds; run 2 also alternated `main` `425585e`. + +| function | 8f59b2e | 425585e (main) | 96086ad | vs 8f59b2e | +|---|---:|---:|---:|---:| +| run 1: `object_header_parse_x401` | 23.81 µs (23.76–24.00) | | 23.57 µs (23.23–23.89) | **−1.0%** | +| run 1: `snod_parse_all` | 1.840 µs (1.837–1.854) | | 1.839 µs (1.837–1.873) | 0.0% | +| run 1: `btree_v1_walk` | 343 ns (338–349) | | 352 ns (344–363) | +2.7% | +| run 1: `facade_list_400_groups` | 8.03 ms (8.01–8.10) | | 8.05 ms (8.01–8.15) | +0.2% | +| run 2: `object_header_parse_x401` | 24.15 µs (23.74–24.41) | 24.93 µs (24.72–25.05) | 23.52 µs (23.34–23.68) | **−2.6%** | +| run 2: `snod_parse_all` | 1.843 µs (1.830–1.847) | 1.856 µs (1.850–1.865) | 1.861 µs (1.858–1.869) | +1.0% | +| run 2: `btree_v1_walk` | 346 ns (339–366) | 356 ns (347–356) | 357 ns (350–359) | +3.2% | +| run 2: `facade_list_400_groups` | 8.09 ms (8.02–8.10) | 8.12 ms (7.97–8.19) | 8.08 ms (7.98–8.12) | −0.1% | + +- `ObjectHeader::parse` is at or below 8f59b2e (−1.0%, −2.6%) and 5.6% + faster than `main` in the same run. +- `btree_v1_walk` (one walk of a 350 ns B-tree) is 3% above 8f59b2e in + both runs, with overlapping ranges, and is the same on `main` (+0.2% + between `main` and this change): not from this change. The walk's code + changed in `e553153` (after a failed child the siblings are only read, + so the error returns after them; the fixture never takes that path, but + the loop carries the extra state); left as is. +- `snod_parse_all` and the facade listing are within noise. + ### Local metadata and data reads after range-read M2/M3 (2026-09-27, tank) `main` just before range-read M2/M3 (`8f59b2e`, PR #17) against `main` diff --git a/CHANGELOG.md b/CHANGELOG.md index 1518e19..acd24d6 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,16 @@ ## Unreleased +### `ObjectHeader::parse` back at its pre-M2/M3 speed (2026-09-27) +- Parsing a version-1 object header was 4% slower than before range-read + M2/M3 (`docs/known-issues.md`). The cause was the call to the per-chunk + message loop, kept out of line since the allocation fix; it is inlined + again. `object_header_parse_x401` is now 1.0% and 2.6% below `8f59b2e` + and 5.6% below `main` (tank, idle, separate binaries alternating; + `BENCHMARKS.md`). The chunk queue's checks are unchanged (65,536 chunks, + cycle refusal, file-size budget, one chunk buffer at a time, libhdf5's + order, overlapping chunks allowed). + ### Conformance: no our-errors or mismatches left (2026-09-27) - The last 3 our-errors and 2 mismatches were documented as not ours but still counted against us. Re-checked with evidence diff --git a/docs/known-issues.md b/docs/known-issues.md index d87d371..b8e47d6 100644 --- a/docs/known-issues.md +++ b/docs/known-issues.md @@ -9,7 +9,14 @@ deleting it. ## `ObjectHeader::parse` 4% slower after range-read M2/M3 (measured 2026-09-27) -**Status:** open (speed only; values are correct). Measured on an idle tank +**Status:** fixed 2026-09-27 (`96086ad`, branch +`perf/header-parse-and-last-files`): the remaining cost was the call to the +version-1 message loop, kept out of line by `#[inline(never)]`; inlined, +`object_header_parse_x401` is 1.0% and 2.6% below `8f59b2e` in two idle +A/B runs (`BENCHMARKS.md`, "`ObjectHeader::parse` back at 8f59b2e's +speed"), with every chunk-queue check unchanged. + +Was: open (speed only; values are correct). Measured on an idle tank (load below 2 at every round), `main` before range-read M2/M3 (`8f59b2e`) against `main` `7a8fae0`, separate binaries alternating, 3 rounds (`BENCHMARKS.md`, "Local metadata and data reads after range-read M2/M3"): From 694ee0a0909e7a6f06464128583b05aa1dabc2f1 Mon Sep 17 00:00:00 2001 From: osobh Date: Sun, 27 Sep 2026 23:39:41 -0500 Subject: [PATCH 10/10] docs: conformance report after the last non-ok files were classified (602 of 697 ok) Regenerated on tank: ok 600 -> 602 (attr_datatypes.hdf5 and tcomplex_be.h5, compared against h5py's big-endian VL values corrected), mismatch 2 -> 0; cve-2025-2308.h5 and cve-2025-44904.h5 are ref-bug (h5py's values varied across six heaps in this run); bad_nbit_parms_walk.h5 read the same in all six this run, so it stays our-error, as the classification rule requires. Baseline raised. Co-Authored-By: Claude Opus 5.5 (1M context) --- CONFORMANCE.md | 24 ++++++++++++++---------- conformance/baseline.json | 22 +++++++++++----------- 2 files changed, 25 insertions(+), 21 deletions(-) diff --git a/CONFORMANCE.md b/CONFORMANCE.md index a900e17..d5d4219 100644 --- a/CONFORMANCE.md +++ b/CONFORMANCE.md @@ -13,15 +13,15 @@ fatal. This file is generated by `conformance/run.sh`; do not edit it by hand. | | | |---|---| -| date | 2026-09-28 03:33 UTC | -| clawhdf5 commit | `ff0b2f8a4e4ef105d7ff4694349ca475abf3f611` | +| date | 2026-09-28 04:29 UTC | +| clawhdf5 commit | `bf5a163dcf7fe28d651ada6545d8136ffeffc825` | | machine | `tank`: AMD Ryzen 7 7800X3D 8-Core Processor, 16 CPUs, 61 GiB, Linux 7.0.0-34-generic x86_64 | -| command | `conformance/run.sh --no-fetch` | -| rustc | | +| command | `conformance/run.sh --no-fetch --update-baseline` | +| rustc | rustc 1.98.1 (48a229cea 2026-09-01) | | reference | h5py 3.16.0, HDF5 2.0.0, numpy 2.5.3, hdf5plugin 7.1.0, Python 3.14.4 | | h5dump | Version 1.14.6 (CVE corpus only) | | limits | 20 s timeout (SIGKILL), 4096 MiB address space, per process; 16 files in parallel | -| runtime | 21 s probing + comparing (18 s fetch/build before it) | +| runtime | 20 s probing + comparing (5 s fetch/build before it) | ## Results @@ -39,14 +39,14 @@ A file's class is the first that applies: | NCAS-CMS_pyfive | 33 | 33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | | cve_hdf5 | 147 | 113 | 0 | 0 | 32 | 2 | 0 | 0 | 0 | 0 | | h5py_data | 4 | 4 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | -| hdf5 | 466 | 405 | 0 | 0 | 60 | 1 | 0 | 0 | 0 | 0 | +| hdf5 | 466 | 405 | 1 | 0 | 60 | 0 | 0 | 0 | 0 | 0 | | netcdf-c | 20 | 20 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | | netcdf4-python | 18 | 18 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | | usnistgov_h5wasm | 5 | 5 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | | xarray-data | 4 | 4 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | -| **all** | **697** | **602** | **0** | **0** | **92** | **3** | **0** | **0** | **0** | **0** | +| **all** | **697** | **602** | **1** | **0** | **92** | **2** | **0** | **0** | **0** | **0** | -**Our errors and mismatches: 0.** Files not ok: 92 h5py-cannot-read, 3 ref-bug. 2 object(s) were compared against h5py's values corrected for a known h5py bug (2 identical to clawhdf5's; see *Reference bugs*). +**Our errors and mismatches: 1.** Files not ok: 1 our-error, 92 h5py-cannot-read, 2 ref-bug. 2 object(s) were compared against h5py's values corrected for a known h5py bug (2 identical to clawhdf5's; see *Reference bugs*). Corpora (fetched by `conformance/fetch-corpus.sh` into the gitignored `conformance/.cache/`): @@ -67,7 +67,11 @@ None. ## Our-error root causes -None. +Grouped by normalised error message. *files* counts files whose class this cause affects. + +| files | objects | error | examples | +|---:|---:|---|---| +| 1 | 1 | `ChunkedReadError("…")` | `hdf5/test/testfiles/bad_nbit_parms_walk.h5` | ## Mismatch root causes @@ -257,7 +261,7 @@ the same run; an object that reads the same every time goes back to *our-error*. |---|---|---:|---|---| | `cve_hdf5/cvefiles/cve-2025-2308.h5` | `/Scale_offset_long_long_data_le` | 6 | yes | the first chunk records minbits 11: its 12 values need 17 bytes of codes, and the 26-byte chunk holds 5 after its 21-byte header; libhdf5's scale-offset decoder reads past its buffer, and develop refuses the chunk ("Buffer too short") | | `cve_hdf5/cvefiles/cve-2025-44904.h5` | `/Scale_offset_float_data_le` | 6 | yes | unfiltered chunks stored as 38 and 37 bytes for 48-byte chunks: 1.14/2.0 read the stored bytes into a buffer of that size and use it as the whole chunk (H5D__chunk_lock), so the rest is heap memory; develop refuses them ("incorrect chunk size returned from index for unfiltered chunk") | -| `hdf5/test/testfiles/bad_nbit_parms_walk.h5` | `/Nbit_int_data_le` | 3 | yes | the N-Bit parameter list holds 7 values (cd_values[0] = 7) where an integer needs 8: the decoder takes the bit offset from cd_values[7], past the list; libhdf5's own test (`test_filter_bad_params`, test/dsets.c on develop) requires the read to fail | +| `hdf5/test/testfiles/bad_nbit_parms_walk.h5` | `/Nbit_int_data_le` | 1 | **no** | the N-Bit parameter list holds 7 values (cd_values[0] = 7) where an integer needs 8: the decoder takes the bit offset from cd_values[7], past the list; libhdf5's own test (`test_filter_bad_params`, test/dsets.c on develop) requires the read to fail | ### Values corrected for a known h5py bug diff --git a/conformance/baseline.json b/conformance/baseline.json index 0db7eb3..4ed5e2b 100644 --- a/conformance/baseline.json +++ b/conformance/baseline.json @@ -1,33 +1,31 @@ { "comment": "conformance/run.sh fails if the ok count drops below `ok` or a file in `ok_files` stops being ok. Regenerate with `conformance/run.sh --update-baseline` after an intended change.", - "commit": "f37e7ae3263277319dba4bc39be5397194eb00c3", - "date": "2026-09-27 00:34 UTC", + "commit": "bf5a163dcf7fe28d651ada6545d8136ffeffc825", + "date": "2026-09-28 04:29 UTC", "reference": "h5py 3.16.0 / HDF5 2.0.0", "files": 697, - "ok": 600, + "ok": 602, "counts": { "h5py-cannot-read": 92, - "mismatch": 2, - "ok": 600, - "our-error": 3 + "ok": 602, + "our-error": 1, + "ref-bug": 2 }, "per_corpus": { "NCAS-CMS_pyfive": { - "mismatch": 1, - "ok": 32 + "ok": 33 }, "cve_hdf5": { "h5py-cannot-read": 32, "ok": 113, - "our-error": 2 + "ref-bug": 2 }, "h5py_data": { "ok": 4 }, "hdf5": { "h5py-cannot-read": 60, - "mismatch": 1, - "ok": 404, + "ok": 405, "our-error": 1 }, "netcdf-c": { @@ -45,6 +43,7 @@ }, "ok_files": [ "NCAS-CMS_pyfive/tests/compact.hdf5", + "NCAS-CMS_pyfive/tests/data/attr_datatypes.hdf5", "NCAS-CMS_pyfive/tests/data/btreev2.hdf5", "NCAS-CMS_pyfive/tests/data/chunked.hdf5", "NCAS-CMS_pyfive/tests/data/cmip_bad_eg.nc", @@ -455,6 +454,7 @@ "hdf5/tools/test/testfiles/tcmpdints.h5", "hdf5/tools/test/testfiles/tcmpdintsize.h5", "hdf5/tools/test/testfiles/tcomplex.h5", + "hdf5/tools/test/testfiles/tcomplex_be.h5", "hdf5/tools/test/testfiles/tcompound.h5", "hdf5/tools/test/testfiles/tcompound_complex.h5", "hdf5/tools/test/testfiles/tcompound_complex2.h5",