"""In-place editing: clawhdf5.File(path, 'r+') against h5py. Every edit is applied twice, to two copies of the same file: once through h5py (libhdf5) and once through clawhdf5 (FileEditor). After every edit both files are read back with h5py and must hold the same shapes, values and attributes; clawhdf5's own view must agree; when h5py refuses an edit, clawhdf5 must refuse it too and leave its file as it was. Files are written by h5py (libver earliest and latest, so every chunk index kind) and by clawhdf5; `h5dump` must read every result.""" import io import os import shutil import subprocess import threading import numpy as np import pytest import clawhdf5 # --------------------------------------------------------------------------- # Files # --------------------------------------------------------------------------- ENUM = {"RED": 0, "GREEN": 1, "BLUE": 7} def _h5py_file(h5py, path, libver): rng = np.random.default_rng(1) with h5py.File(path, "w", libver=libver) as f: f.create_dataset("i4", data=np.arange(60, dtype="u2", "i8", "f8"] def _quiet(f): with np.errstate(all="ignore"): return f() def _libhdf5_undefined(vals, target): """The values whose conversion to `target` libhdf5 2.0 (h5py 3.16) gets wrong even in native byte order, where its C casts are undefined behaviour; clawhdf5 saturates them as libhdf5's range handling intends (docs/known-issues.md): - half floats into unsigned integers: negatives wrap (-1 -> 65535) and +inf becomes 0; into signed integers, +-inf becomes the minimum; - a float equal to the integer maximum rounded up in the float's precision (float32(2**31 - 1) == 2**31 -> int32, float64(2**64 - 1) -> uint64) becomes the minimum (or 0); - a double between 65504 and 65520 into a half float becomes infinity (IEEE rounds it down to 65504, as numpy does). """ bad = np.zeros(vals.shape, dtype=bool) if vals.dtype.kind != "f": return bad if target.kind in "iu": if vals.dtype.itemsize == 2: bad |= np.isinf(vals) if target.kind == "u": bad |= vals <= -1 top = _quiet(lambda: np.array(np.iinfo(target).max).astype(vals.dtype)) if float(top) > np.iinfo(target).max: bad |= vals == top if target.kind == "f" and target.itemsize == 2 and vals.dtype.itemsize > 2: bad |= (np.abs(vals) > 65504) & (np.abs(vals) < 65520) return bad def test_numeric_conversions_match_h5py(h5py, tmp_path): """Every numeric source dtype into every numeric dataset dtype, with values at and beyond the targets' limits, as libhdf5 converts them.""" edge = np.array([0, 1, -1, -0.3, 2.5, -2.5, 3.7, -3.7, 127.9, -128.9, 200.5, 255.5, 256, -129, 32767.5, 40000, 65504, 70000, -70000, 2**31 - 1, 2**31, -2**31 - 1, 4e9, 1e15, -1e15, 1e19, 1e300, -1e300, np.inf, -np.inf]) sources = { "f8": edge, "f4": _quiet(lambda: edge.astype(" {target}" try: with h5py.File(path_t, "r+") as f: f["d"][...] = _native_conversion(h5py, vals, f["d"].dtype) except Exception: # noqa: BLE001 with clawhdf5.File(path_o, "r+") as f, pytest.raises(Exception): f["d"][...] = vals continue with clawhdf5.File(path_o, "r+") as f: f["d"][...] = vals with h5py.File(path_t, "r") as a, h5py.File(path_o, "r") as b: assert a["d"][...].tobytes() == b["d"][...].tobytes(), ( f"{what}: h5py {a['d'][...].tolist()} clawhdf5 {b['d'][...].tolist()}" ) def test_nan_into_an_integer_dataset_is_refused(h5py, tmp_path): """libhdf5 stores NaN as an arbitrary integer (0, the minimum or 2**63, depending on the type); clawhdf5 refuses and writes nothing.""" path = str(tmp_path / "nan.h5") with h5py.File(path, "w") as f: f.create_dataset("d", data=np.arange(4, dtype="f4": np.array([1.5, 2.5], dtype=">f4"), "f2": np.float16(0.5), "b": True, "barr": np.array([True, False]), "c16": np.complex128(1 + 2j), "bytes": b"raw", "sarr": np.array([b"a", b"bcd"]), "str": "héllo", "strs": ["x", "yz"], "2d": np.arange(12, dtype="f4"].dtype == np.dtype(">f4") assert a["f2"].dtype == np.float16 assert a["b"] is np.True_ or a["b"] == True # noqa: E712 assert a["barr"].dtype == np.bool_ assert a["c16"] == 1 + 2j assert a["bytes"] == b"raw" assert list(a["sarr"]) == [b"a", b"bcd"] # str is stored as fixed-length UTF-8: h5py reads bytes. assert a["str"].decode("utf-8") == "héllo" assert [x.decode() for x in a["strs"]] == ["x", "yz"] assert a["2d"].shape == (3, 4) and a["2d"].dtype == np.dtype("