conformance: compare h5py's big-endian VL values corrected, confirm libhdf5 over-reads per run
The last 3 our-errors and 2 mismatches were documented as not ours but
still counted against us, on a heuristic (any big-endian VL mismatch) and
a fixed list.
- ref.py checks that the installed h5py returns big-endian VL elements
with the file's bytes under a little-endian dtype (writing and reading
a vlen('>f4') in memory) and, if so, relabels them with the file's byte
order before hashing, marking the object `ref_fix`. The values are now
compared: attr_datatypes.hdf5 /@vlen_uint64 and tcomplex_be.h5
/VariableLengthDatasetFloatComplex are identical to ours (h5dump 1.14.6
prints the same (1, 2), (3, 4, 5), (42)).
- ref_bugs.py re-reads each object h5py reads only through a libhdf5 bug
in six processes with different heaps (import order, MALLOC_PERTURB_).
Values the file determines are the same every time; these three change
(6, 6 and 3 distinct results), so they are over-read memory, not data
clawhdf5 could match. compare.py classifies a file `ref-bug` only when
every difference is such an object confirmed in the same run.
- report.py: the ref-bug class, the evidence table, the corrected
objects; test_ref.py covers both (run in the nightly job).
Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
+53
-1
@@ -53,6 +53,49 @@ def packed(dt):
|
||||
return dt
|
||||
|
||||
|
||||
# --- reference corrections ---------------------------------------------------
|
||||
# Where h5py is known to return values the file does not hold, and the right
|
||||
# values follow from what it returned, ref.py corrects them and records the
|
||||
# correction on the object ("ref_fix"), so the comparison is still a real
|
||||
# comparison and CONFORMANCE.md lists every corrected object. Each correction
|
||||
# first checks that the installed h5py still has the bug.
|
||||
|
||||
# Corrections applied while encoding the current object.
|
||||
FIXES = set()
|
||||
_BE_VLEN_BUG = None
|
||||
|
||||
|
||||
def be_vlen_bug():
|
||||
"""h5py (3.16 / HDF5 2.0 at least) returns the elements of a
|
||||
variable-length sequence whose base type is big-endian with the file's
|
||||
big-endian bytes under a native (little-endian) dtype: a
|
||||
`vlen_dtype('>f4')` dataset holding [1.0, 2.0] reads back as
|
||||
[4.6e-41, 9.0e-44]. `h5dump` prints the file's values. Checked once per
|
||||
process by writing and reading exactly that dataset in memory."""
|
||||
global _BE_VLEN_BUG
|
||||
if _BE_VLEN_BUG is None:
|
||||
import io
|
||||
try:
|
||||
bio = io.BytesIO()
|
||||
with h5py.File(bio, "w") as f:
|
||||
d = f.create_dataset("v", (1,), dtype=h5py.vlen_dtype(np.dtype(">f4")))
|
||||
d[0] = np.array([1.0, 2.0], dtype=">f4")
|
||||
with h5py.File(bio, "r") as f:
|
||||
got = np.asarray(f["v"][0])
|
||||
_BE_VLEN_BUG = (got.dtype == np.dtype("<f4")
|
||||
and got.view(">f4").tolist() == [1.0, 2.0]
|
||||
and got.tolist() != [1.0, 2.0])
|
||||
except Exception: # noqa: BLE001
|
||||
_BE_VLEN_BUG = False
|
||||
return _BE_VLEN_BUG
|
||||
|
||||
|
||||
def unswapped(got, base):
|
||||
"""`got` is `base` (big-endian somewhere) with every field in native
|
||||
little-endian order instead: the shape of h5py's big-endian VL bug."""
|
||||
return base.newbyteorder("<") == got and base != got
|
||||
|
||||
|
||||
def canon_el(dt, val, out):
|
||||
if dt.fields:
|
||||
for n in dt.names:
|
||||
@@ -79,7 +122,13 @@ def canon_el(dt, val, out):
|
||||
base = h5py.check_vlen_dtype(dt)
|
||||
if base is None:
|
||||
raise TypeError(f"unhandled object dtype {dt!r}")
|
||||
arr = np.asarray(val if val is not None else [], dtype=base).reshape(-1)
|
||||
arr = np.asarray(val if val is not None else [])
|
||||
if arr.dtype != base and be_vlen_bug() and unswapped(arr.dtype, base):
|
||||
# h5py's big-endian VL bug (see be_vlen_bug): the bytes are
|
||||
# the file's, the dtype label is wrong. Relabel, don't convert.
|
||||
arr = arr.view(base)
|
||||
FIXES.add("h5py-be-vlen")
|
||||
arr = np.asarray(arr, dtype=base).reshape(-1)
|
||||
out += b"V" + struct.pack("<I", arr.shape[0])
|
||||
if simple(base):
|
||||
out += arr.astype(packed(base)).tobytes()
|
||||
@@ -118,6 +167,7 @@ def hash_values(arr, dt, rec):
|
||||
while dt.subdtype is not None:
|
||||
dt = dt.subdtype[0]
|
||||
arr = np.asarray(arr, dtype=dt)
|
||||
FIXES.clear()
|
||||
if simple(dt):
|
||||
c = np.ascontiguousarray(arr).astype(packed(dt)).tobytes()
|
||||
else:
|
||||
@@ -125,6 +175,8 @@ def hash_values(arr, dt, rec):
|
||||
for x in arr.reshape(-1):
|
||||
canon_el(dt, x, out)
|
||||
c = bytes(out)
|
||||
if FIXES:
|
||||
rec["ref_fix"] = sorted(FIXES)
|
||||
rec["hash"] = hashlib.sha256(c).hexdigest()
|
||||
rec["head"] = c[:48].hex()
|
||||
|
||||
|
||||
Reference in New Issue
Block a user