Files
clawhdf5/crates/clawhdf5-py/tests/test_edit.py
T
osobhandClaude Opus 5.5 b0930708b6 py: boolean-mask keys raise NotImplementedError, not TypeError
h5py supports boolean masks for reads and writes; clawhdf5 supports
neither, so a mask is an unsupported operation (NotImplementedError, as
for every other edit the bindings cannot do), not an invalid key.

Tests: test_unsupported_edits_are_clear_errors (1-D, N-D and per-axis
mask writes, file unchanged) and test_boolean_masks_are_refused (reads).

Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
2026-09-27 07:48:04 -05:00

941 lines
41 KiB
Python

"""In-place editing: clawhdf5.File(path, 'r+') against h5py.
Every edit is applied twice, to two copies of the same file: once through
h5py (libhdf5) and once through clawhdf5 (FileEditor). After every edit both
files are read back with h5py and must hold the same shapes, values and
attributes; clawhdf5's own view must agree; when h5py refuses an edit,
clawhdf5 must refuse it too and leave its file as it was. Files are written
by h5py (libver earliest and latest, so every chunk index kind) and by
clawhdf5; `h5dump` must read every result."""
import io
import os
import shutil
import subprocess
import threading
import numpy as np
import pytest
import clawhdf5
# ---------------------------------------------------------------------------
# Files
# ---------------------------------------------------------------------------
ENUM = {"RED": 0, "GREEN": 1, "BLUE": 7}
def _h5py_file(h5py, path, libver):
rng = np.random.default_rng(1)
with h5py.File(path, "w", libver=libver) as f:
f.create_dataset("i4", data=np.arange(60, dtype="<i4").reshape(6, 10))
f.create_dataset("be_i2", data=np.arange(24, dtype=">i2").reshape(4, 6))
f.create_dataset("u1", data=np.arange(16, dtype="u1"))
f.create_dataset("f2", data=rng.standard_normal(12).astype("<f2"))
f.create_dataset("f4_2d", data=rng.standard_normal((8, 9)).astype("<f4"))
f.create_dataset("f8_3d", data=rng.standard_normal((4, 5, 6)))
f.create_dataset("u8", data=np.arange(10, dtype="<u8"))
f.create_dataset("c8", data=(np.arange(6) + 1j * np.arange(6)).astype("<c8"))
f.create_dataset("bool", data=np.array([True, False, True, True]))
f.create_dataset("enum", data=np.array([0, 1, 7, 0], dtype="i1"),
dtype=h5py.enum_dtype(ENUM, basetype="i1"))
f.create_dataset("s5", data=np.array([b"ab", b"cdefg", b""], dtype="S5"))
cmp_dt = np.dtype([("id", "<i4"), ("x", "<f8"), ("tag", "S3")])
f.create_dataset("cmp", data=np.array([(i, i / 2, b"t%d" % i) for i in range(5)], dtype=cmp_dt))
f.create_dataset("scalar", data=np.float64(3.5))
# Chunked: fixed maxshape (v4 fixed array under latest), one
# unlimited dimension (extensible array), two (v2 B-tree), one chunk.
f.create_dataset("chunk_fixed", data=np.arange(100, dtype="<i8").reshape(10, 10), chunks=(3, 4))
f.create_dataset("chunk_ext", data=rng.standard_normal((12, 7)), chunks=(5, 7), maxshape=(None, 7))
f.create_dataset("chunk_bt2", data=np.arange(30, dtype="<i4").reshape(5, 6), chunks=(2, 2),
maxshape=(None, None))
f.create_dataset("chunk_gzip", data=np.arange(400, dtype="<f4").reshape(20, 20), chunks=(6, 6),
compression="gzip", maxshape=(40, 40))
f.create_dataset("chunk_single", data=np.arange(12, dtype="<u2").reshape(3, 4), chunks=(3, 4),
maxshape=(3, 4))
f.create_dataset("chunk_fill", shape=(8,), dtype="<i4", chunks=(3,), maxshape=(20,), fillvalue=-1)
f.create_dataset("vlen", data=["a", "bb"], dtype=h5py.string_dtype())
# Compact layout (low-level API).
dcpl = h5py.h5p.create(h5py.h5p.DATASET_CREATE)
dcpl.set_layout(h5py.h5d.COMPACT)
space = h5py.h5s.create_simple((7,))
dsid = h5py.h5d.create(f.id, b"compact", h5py.h5t.STD_I32LE, space, dcpl=dcpl)
dsid.write(h5py.h5s.ALL, h5py.h5s.ALL, np.arange(7, dtype="<i4"))
g = f.create_group("grp")
g.create_dataset("leaf", data=np.arange(5.0))
g.attrs["units"] = "m"
f.attrs["version"] = np.int32(1)
def _clawhdf5_file(path):
with clawhdf5.File(str(path), "w") as f:
f.create_dataset("i4", data=np.arange(60, dtype="<i4").reshape(6, 10))
f.create_dataset("f8", data=np.linspace(0, 1, 30).reshape(5, 6))
f.create_dataset("chunk_gzip", data=np.arange(400, dtype="<f4").reshape(20, 20),
chunks=[6, 6], compression="gzip")
f.create_dataset("u1", data=np.arange(16, dtype="u1"))
g = f.create_group("grp")
g.create_dataset("leaf", data=np.arange(5.0))
g.attrs["units"] = "m"
f.attrs["version"] = 1
# h5py's libver: "earliest" (v1 B-tree chunk indexes), "v114" (the 1.10+
# indexes: fixed and extensible arrays, v2 B-trees, single chunk) and
# "latest" (HDF5 2.0's newest format, which h5dump 1.14 cannot read).
SOURCES = ["h5py-earliest", "h5py-v114", "h5py-latest", "clawhdf5"]
def _make(h5py, tmp_path, source):
base = tmp_path / f"base-{source}.h5"
if source == "clawhdf5":
_clawhdf5_file(base)
else:
_h5py_file(h5py, str(base), source.split("-")[1])
theirs = tmp_path / f"theirs-{source}.h5"
ours = tmp_path / f"ours-{source}.h5"
shutil.copy(base, theirs)
shutil.copy(base, ours)
return str(theirs), str(ours), str(base)
# ---------------------------------------------------------------------------
# Comparing files through h5py
# ---------------------------------------------------------------------------
def _norm_attr(v):
"""Attribute values comparable across the two writers: clawhdf5 stores
`str` as fixed-length UTF-8 (h5py reads bytes), h5py as variable-length
(h5py reads str)."""
if isinstance(v, bytes):
return ("str", v.decode("utf-8"))
if isinstance(v, str):
return ("str", v)
arr = np.asarray(v)
if arr.dtype.kind in "SO":
return ("strs", [x.decode() if isinstance(x, bytes) else x for x in arr.ravel().tolist()], arr.shape)
return (arr.dtype.str, arr.shape, arr.tobytes())
def snapshot(h5py, path):
"""What h5py sees in the file: every dataset's shape, dtype, bytes and
attributes (read without locking: clawhdf5 may hold the file open)."""
out = {}
with h5py.File(path, "r", locking=False) as f:
def visit(name, obj):
attrs = {k: _norm_attr(obj.attrs[k]) for k in obj.attrs}
if isinstance(obj, h5py.Dataset):
if obj.dtype.kind == "O":
data = [x for x in obj[...].ravel().tolist()]
else:
data = obj[()].tobytes() if obj.shape is not None else None
out[name] = (obj.shape, obj.dtype.str, obj.maxshape, data, attrs)
else:
out[name] = ("group", attrs)
visit("/", f)
f.visititems(visit)
return out
def assert_same_files(h5py, theirs, ours, what):
a, b = snapshot(h5py, theirs), snapshot(h5py, ours)
assert a.keys() == b.keys(), what
for k in a:
assert a[k] == b[k], f"{what}: {k} differs\n h5py: {a[k]}\n clawhdf5: {b[k]}"
def assert_ours_reads_like_h5py(h5py, f, path, what):
"""clawhdf5's own view of the file it is editing matches h5py's."""
with h5py.File(path, "r", locking=False) as t:
for name in ["i4", "chunk_ext", "chunk_bt2", "chunk_gzip", "f8_3d", "cmp", "bool", "enum", "scalar"]:
if name not in t:
continue
o = f[name]
assert o.shape == t[name].shape, f"{what}: {name} shape"
assert o.maxshape == t[name].maxshape, f"{what}: {name} maxshape"
np.testing.assert_array_equal(o[()], t[name][()], err_msg=f"{what}: {name}")
for obj in ["/", "grp"]:
assert sorted(f[obj].attrs.keys()) == sorted(t[obj].attrs.keys()), what
for k in t[obj].attrs:
assert _norm_attr(f[obj].attrs[k]) == _norm_attr(t[obj].attrs[k]), f"{what}: {obj}.attrs[{k}]"
def h5dump_reads(path, base=None):
"""h5dump (libhdf5 1.14) reads every object and value of `path` — when it
reads the unedited `base` (it cannot read HDF5 2.0's newest format)."""
exe = shutil.which("h5dump")
if exe is None:
if os.environ.get("CLAWHDF5_REQUIRE_INTEROP") == "1":
pytest.fail("h5dump is required (CLAWHDF5_REQUIRE_INTEROP=1)")
return
h5rs = os.environ.get("CLAWHDF5_H5RS")
if h5rs:
# clawhdf5's structural and checksum validator (scripts/ci-test.sh
# points this at the h5rs it built).
r = subprocess.run([h5rs, "check", path], capture_output=True, text=True)
assert r.returncode == 0, (r.stdout + r.stderr)[-2000:]
if base is not None and subprocess.run([exe, "-H", base], capture_output=True).returncode != 0:
return
r = subprocess.run([exe, path], capture_output=True, text=True)
assert r.returncode == 0, r.stderr[-2000:]
# ---------------------------------------------------------------------------
# Applying one edit both ways
# ---------------------------------------------------------------------------
def _native_conversion(h5py, value, ds_dtype):
"""`value` as libhdf5 converts it to `ds_dtype` in native byte order.
libhdf5 2.0 (h5py 3.16) converts numbers differently when either side is
not in native byte order (its "soft" conversions): a float in (-1, 0)
becomes the integer type's minimum instead of 0, and an unsigned integer
too large for the signed type of the same size wraps instead of
saturating. clawhdf5 applies the native-order results to every byte
order, so the reference is h5py converting into a native dataset of the
same kind; the result then reaches the real dataset by a plain byte
swap."""
if not isinstance(value, np.ndarray):
return value
if value.dtype.kind not in "biuf" or ds_dtype.kind not in "biuf":
return value
if value.dtype.isnative and ds_dtype.isnative:
return value
with h5py.File(io.BytesIO(), "w") as tmp:
d = tmp.create_dataset("t", shape=value.shape, dtype=ds_dtype.newbyteorder("="))
d[...] = value.astype(value.dtype.newbyteorder("="))
return np.asarray(d[()])
def _apply(f, op, h5py=None):
"""Apply `op` to `f`; with `h5py`, `f` is an h5py file and a numpy array
value is first converted as libhdf5 converts in native byte order (see
`_native_conversion`)."""
kind = op[0]
if kind == "set":
_, name, key, value = op
if h5py is not None:
value = _native_conversion(h5py, value, f[name].dtype)
f[name][key] = value
elif kind == "resize":
_, name, size, axis = op
if axis is None:
f[name].resize(size)
else:
f[name].resize(size, axis=axis)
elif kind == "attr":
_, obj, name, value = op
f[obj].attrs[name] = value
else:
raise AssertionError(op)
def edit_both(h5py, theirs, ours_path, ours, op):
"""Apply `op` with h5py and with clawhdf5 (`ours`, open 'r+'); the two
files must then read the same through h5py. If h5py refuses, clawhdf5
must refuse and its file must be unchanged. Returns h5py's error."""
before = snapshot(h5py, ours_path)
try:
with h5py.File(theirs, "r+") as t:
_apply(t, op, h5py)
except Exception as e: # noqa: BLE001 - h5py refuses: so must we
try:
_apply(ours, op)
except Exception: # noqa: BLE001
pass
else:
pytest.fail(f"{op!r:.300}: h5py refused ({type(e).__name__}: {e}), clawhdf5 did not")
assert snapshot(h5py, ours_path) == before, f"{op!r}: clawhdf5 changed the file while failing"
return e
_apply(ours, op)
assert_same_files(h5py, theirs, ours_path, repr(op)[:200])
return None
# ---------------------------------------------------------------------------
# Tests
# ---------------------------------------------------------------------------
@pytest.mark.parametrize("source", SOURCES)
def test_edit_sequence_matches_h5py(h5py, tmp_path, source):
theirs, ours_path, base = _make(h5py, tmp_path, source)
ops = [
("set", "i4", 0, 99),
("set", "i4", (slice(1, 5, 2), slice(None, None, 3)), np.array([[1.5, -2.5, 1e12, -1e12]])),
("set", "i4", (slice(None), 2), np.arange(6, dtype="<i8") * 1000),
("set", "i4", ([0, 2, 5], slice(4, 6)), np.array([[1, 2], [3, 4], [5, 6]])),
("set", "i4", (Ellipsis, -1), np.int8(-7)),
("set", "i4", (3, 3), 12345.9),
("set", "u1", slice(2, 8), np.array([-5, 0, 300, 255, 256, 1], dtype="<i4")),
("set", "u1", slice(0, 3), [1, 2, 3]),
("set", "chunk_gzip", (slice(0, 20, 7), slice(3, 17)), 42.25),
("set", "chunk_gzip", (slice(5, 11), slice(5, 11)), np.ones((6, 6), dtype="<f8") * np.pi),
("set", "grp/leaf", slice(None), np.array([1, 2, 3, 4, 5], dtype="<u2")),
("attr", "/", "version", np.int32(2)),
("attr", "/", "count", 7),
("attr", "grp", "scale", np.array([0.5, 0.25], dtype="<f4")),
("attr", "grp", "matrix", np.arange(6, dtype=">i8").reshape(2, 3)),
("attr", "i4", "flag", np.bool_(True)),
("attr", "i4", "z", np.complex64(1 - 2j)),
("attr", "i4", "raw", np.bytes_(b"abc")),
("set", "i4", slice(0, 2), np.zeros((3, 10))), # shape mismatch: refused by both
]
if source != "clawhdf5":
ops += [
("set", "be_i2", (slice(None), slice(1, 3)), np.array([70000, -70000], dtype="<i8")),
("set", "f2", slice(None, None, 4), np.array([1e6, -3.25, 0.1])),
("set", "f4_2d", (2, slice(None)), np.linspace(-1, 1, 9)),
("set", "f8_3d", (slice(1, 3), 2, slice(None, None, 2)), np.arange(3, dtype="<i2")),
("set", "u8", slice(None), np.array([-1, 0, 2**63, 1e30, -1e30, 5.5, 2, 3, 4, 5])),
("set", "c8", slice(1, 3), np.array([1 + 1j, 2 - 2j], dtype="<c16")),
("set", "c8", 0, np.float64(1.0)), # h5py: no conversion path
("set", "bool", slice(None), np.array([0, 3, 0, -1], dtype="<i4")),
("set", "bool", 1, np.array(True)),
("set", "enum", slice(0, 2), np.array([7, 1], dtype="<i4")),
("set", "s5", 0, np.bytes_(b"xyzuvw")),
("set", "s5", slice(1, 3), [b"q", b"rs"]),
("set", "s5", 2, np.array("uni")), # h5py: no conversion from 'U'
("set", "cmp", 2, np.array((9, 9.5, b"zz"), dtype=[("id", "<i4"), ("x", "<f8"), ("tag", "S3")])),
("set", "cmp", slice(3, 5), [(1, 0.5, b"a"), (2, 1.5, b"b")]),
("set", "scalar", (), 7.25),
("set", "scalar", Ellipsis, np.float32(-1.5)),
("set", "compact", slice(1, 6, 2), np.array([10, 20, 30])),
("set", "chunk_fixed", (slice(2, 9), slice(1, 10, 4)), np.arange(21).reshape(7, 3)),
("set", "chunk_single", (1, slice(None)), np.array([9, 8, 7, 6])),
("resize", "chunk_ext", (20, 7), None),
("set", "chunk_ext", slice(12, 20), np.full((8, 7), 2.5)),
("resize", "chunk_ext", 9, 0),
("resize", "chunk_ext", 16, 0),
("resize", "chunk_bt2", (9, 11), None),
("set", "chunk_bt2", (slice(4, 9), slice(5, 11)), np.arange(30).reshape(5, 6)),
("resize", "chunk_bt2", (3, 3), None),
("resize", "chunk_bt2", (7, 8), None),
("resize", "chunk_gzip", (40, 25), None),
("resize", "chunk_gzip", (41, 25), None), # beyond maxshape: refused
("resize", "chunk_fill", 15, None), # h5py: a size without axis must be a tuple
("resize", "chunk_fill", (15,), None),
("set", "chunk_fill", slice(10, 12), [1, 2]),
("resize", "chunk_fill", (4,), None),
("resize", "chunk_fill", (20,), None),
("resize", "i4", (7, 10), None), # not chunked: refused
]
with clawhdf5.File(ours_path, "r+") as ours:
assert ours.mode == "r+"
for i, op in enumerate(ops):
edit_both(h5py, theirs, ours_path, ours, op)
if i % 5 == 0:
assert_ours_reads_like_h5py(h5py, ours, ours_path, repr(op))
assert_ours_reads_like_h5py(h5py, ours, ours_path, "end")
h5dump_reads(ours_path, base)
# Reopened, clawhdf5 reads what h5py reads.
with clawhdf5.File(ours_path, "r") as f:
assert f.mode == "r"
assert_ours_reads_like_h5py(h5py, f, ours_path, "reopened")
def _random_key(rng, shape):
key = []
for n in shape:
r = rng.random()
if n == 0 or r < 0.3:
start = int(rng.integers(0, n + 1)) if n else 0
stop = int(rng.integers(start, n + 1)) if n else 0
step = int(rng.integers(1, 4))
key.append(slice(start, stop, step))
elif r < 0.55:
key.append(int(rng.integers(-n, n)))
elif r < 0.7:
k = int(rng.integers(1, min(n, 4) + 1))
key.append(sorted(rng.choice(n, size=k, replace=False).tolist()))
else:
key.append(slice(None))
# Only one index list per key.
lists = [i for i, k in enumerate(key) if isinstance(k, list)]
for i in lists[1:]:
key[i] = slice(None)
return tuple(key)
def _selection_shape(key, shape):
out = []
fancy = False
for k, n in zip(key, shape):
if isinstance(k, slice):
out.append(len(range(*k.indices(n))))
elif isinstance(k, list):
out.append(len(k))
fancy = True
return tuple(out), fancy
def _random_value(rng, sel_shape, fancy, dtype):
r = rng.random()
if r < 0.2 or not sel_shape:
v = rng.standard_normal() * 1000
return np.float64(v) if rng.random() < 0.5 else int(v)
shape = list(sel_shape)
if not fancy and r < 0.35 and shape:
shape[0] = 1 # broadcast along the first axis
kind = rng.choice(["same", "f8", "i8", "u1", "f4"])
if kind == "same" and np.dtype(dtype).kind in "iuf":
dt = np.dtype(dtype)
else:
dt = np.dtype(str(kind) if kind != "same" else "f8")
base = rng.standard_normal(size=shape) * (10 ** rng.integers(0, 6))
with np.errstate(all="ignore"):
return base.astype(dt)
@pytest.mark.parametrize("source", SOURCES)
@pytest.mark.parametrize("seed", [0, 1, 2, 3])
def test_random_edits_match_h5py(h5py, tmp_path, source, seed):
"""Random writes (slices, steps, integers, index lists, broadcasts,
other dtypes and out-of-range values), resizes and attributes, each
compared with h5py applying the same edit."""
rng = np.random.default_rng(1000 * seed + SOURCES.index(source))
theirs, ours_path, base = _make(h5py, tmp_path, source)
with h5py.File(theirs, "r") as t:
names = [n for n in ["i4", "u1", "f8", "f4_2d", "f8_3d", "be_i2", "chunk_fixed", "chunk_ext",
"chunk_bt2", "chunk_gzip", "chunk_fill", "compact", "grp/leaf"] if n in t]
resizable = [n for n in ["chunk_ext", "chunk_bt2", "chunk_gzip", "chunk_fill"] if n in names]
refused = 0
with clawhdf5.File(ours_path, "r+") as ours:
for step in range(40):
r = rng.random()
if r < 0.15 and resizable:
name = str(rng.choice(resizable))
with h5py.File(theirs, "r") as t:
maxshape = t[name].maxshape
shape = t[name].shape
new = tuple(int(rng.integers(0, (m if m is not None else s + 10) + 1)) for m, s in zip(maxshape, shape))
op = ("resize", name, new, None)
elif r < 0.25:
obj = str(rng.choice(["/", "grp", names[0]]))
choices = [np.int16(rng.integers(-100, 100)), rng.standard_normal(3),
np.arange(int(rng.integers(1, 5)), dtype=">u4"), np.float32(0.5)]
value = choices[int(rng.integers(0, len(choices)))]
op = ("attr", obj, f"a{int(rng.integers(0, 4))}", value)
else:
name = str(rng.choice(names))
with h5py.File(theirs, "r") as t:
shape, dtype = t[name].shape, t[name].dtype
key = _random_key(rng, shape)
sel_shape, fancy = _selection_shape(key, shape)
op = ("set", name, key, _random_value(rng, sel_shape, fancy, dtype))
before = None
if op[0] == "resize":
with h5py.File(ours_path, "r", locking=False) as o:
before, fill = o[op[1]][()], o[op[1]].fillvalue
if edit_both(h5py, theirs, ours_path, ours, op) is not None:
refused += 1
elif before is not None:
# Independently of h5py (which a wrong index layout fools
# the same way): the kept elements keep their values.
want = resized_model(before, op[2], fill)
np.testing.assert_array_equal(ours[op[1]][()], want, err_msg=repr(op))
with h5py.File(ours_path, "r", locking=False) as o:
np.testing.assert_array_equal(o[op[1]][()], want, err_msg=repr(op))
assert_ours_reads_like_h5py(h5py, ours, ours_path, "end")
assert refused < 30
h5dump_reads(ours_path, base)
@pytest.mark.parametrize("seed", range(10, 40))
def test_random_edits_on_clawhdf5_files(h5py, tmp_path, seed):
"""More random sequences on a clawhdf5-written file, whose resizes to
zero extents once left files `h5rs check` could not read."""
test_random_edits_match_h5py(h5py, tmp_path, "clawhdf5", seed)
def resized_model(before, shape, fill):
"""`before` resized to `shape` as HDF5 resizes: elements inside both
extents keep their values, the others read as the fill value."""
out = np.full(shape, fill, dtype=before.dtype)
common = tuple(slice(0, min(a, b)) for a, b in zip(before.shape, shape))
out[common] = before[common]
return out
RESIZES = [(15, 15), (3, 2), (20, 20), (1, 1), (1, 0), (0, 0), (7, 20), (20, 13), (20, 20)]
def _check_resizes(h5py, path, name, orig, maxshape=(20, 20), base=None):
"""Resize `name` through RESIZES in 'r+', each checked against a numpy
model with clawhdf5 and h5py and the file with `h5rs check`."""
model = orig
with clawhdf5.File(path, "r+") as f:
ds = f[name]
for shape in RESIZES:
ds.resize(shape)
model = resized_model(model, shape, 0)
np.testing.assert_array_equal(ds[()], model, err_msg=f"{name} {shape}")
with h5py.File(path, "r", locking=False) as t:
np.testing.assert_array_equal(t[name][()], model, err_msg=f"h5py: {name} {shape}")
assert t[name].maxshape == maxshape
h5rs = os.environ.get("CLAWHDF5_H5RS")
if h5rs: # h5dump would wait for the editor's lock
r = subprocess.run([h5rs, "check", "--data", path], capture_output=True, text=True)
assert r.returncode == 0, f"{name} {shape}: " + (r.stdout + r.stderr)[-2000:]
with pytest.raises(ValueError):
ds.resize((maxshape[0] + 1, 20))
# Written values survive a shrink.
ds[...] = orig
ds.resize((15, 15))
with h5py.File(path, "r") as t:
np.testing.assert_array_equal(t[name][()], orig[:15, :15])
h5dump_reads(path, base)
@pytest.mark.parametrize("source", SOURCES)
def test_resizes_keep_values(h5py, tmp_path, source):
"""Shrinking, zero extents and growing back keep the values a numpy
model keeps, on every source (on clawhdf5's own files a shrink once
moved every chunk: h5py read the same wrong values)."""
_, ours, base = _make(h5py, tmp_path, source)
with clawhdf5.File(ours, "r+") as f:
f["chunk_gzip"].resize((20, 20))
maxshape = (20, 20) if source == "clawhdf5" else (40, 40)
_check_resizes(h5py, ours, "chunk_gzip", np.arange(400, dtype="<f4").reshape(20, 20), maxshape, base)
FIXTURES = os.path.join(os.path.dirname(__file__), "..", "..", "clawhdf5", "tests", "fixtures")
@pytest.mark.parametrize("name", ["d", "z"])
def test_resizes_of_a_file_with_no_recorded_maxshape(h5py, tmp_path, name):
"""A file clawhdf5 2.7.0 wrote records no maximum dimensions for chunked
datasets; resizing it must keep the Fixed Array index's layout."""
path = str(tmp_path / "old.h5")
shutil.copy(os.path.join(FIXTURES, "chunked_no_maxshape_v2_7_0.h5"), path)
with h5py.File(path, "r") as t:
orig = t[name][()]
_check_resizes(h5py, path, name, orig)
NUMERIC = ["<i1", "<u1", "<i2", ">u2", "<i4", "<u4", ">i8", "<u8", "<f2", "<f4", ">f8"]
def _quiet(f):
with np.errstate(all="ignore"):
return f()
def _libhdf5_undefined(vals, target):
"""The values whose conversion to `target` libhdf5 2.0 (h5py 3.16) gets
wrong even in native byte order, where its C casts are undefined
behaviour; clawhdf5 saturates them as libhdf5's range handling
intends (docs/known-issues.md):
- half floats into unsigned integers: negatives wrap (-1 -> 65535) and
+inf becomes 0; into signed integers, +-inf becomes the minimum;
- a float equal to the integer maximum rounded up in the float's
precision (float32(2**31 - 1) == 2**31 -> int32, float64(2**64 - 1)
-> uint64) becomes the minimum (or 0);
- a double between 65504 and 65520 into a half float becomes infinity
(IEEE rounds it down to 65504, as numpy does).
"""
bad = np.zeros(vals.shape, dtype=bool)
if vals.dtype.kind != "f":
return bad
if target.kind in "iu":
if vals.dtype.itemsize == 2:
bad |= np.isinf(vals)
if target.kind == "u":
bad |= vals <= -1
top = _quiet(lambda: np.array(np.iinfo(target).max).astype(vals.dtype))
if float(top) > np.iinfo(target).max:
bad |= vals == top
if target.kind == "f" and target.itemsize == 2 and vals.dtype.itemsize > 2:
bad |= (np.abs(vals) > 65504) & (np.abs(vals) < 65520)
return bad
def test_numeric_conversions_match_h5py(h5py, tmp_path):
"""Every numeric source dtype into every numeric dataset dtype, with
values at and beyond the targets' limits, as libhdf5 converts them."""
edge = np.array([0, 1, -1, -0.3, 2.5, -2.5, 3.7, -3.7, 127.9, -128.9, 200.5, 255.5, 256, -129,
32767.5, 40000, 65504, 70000, -70000, 2**31 - 1, 2**31, -2**31 - 1,
4e9, 1e15, -1e15, 1e19, 1e300, -1e300, np.inf, -np.inf])
sources = {
"f8": edge,
"f4": _quiet(lambda: edge.astype("<f4")),
"f2": np.array([0, 1, -1, 2.5, -3.5, 65504, -65504, np.inf, -np.inf, 100.5], dtype="<f2"),
"i8": np.array([0, 1, -1, 127, 128, -129, 255, 256, 32768, -32769, 65536, 2**31, -2**31 - 1,
2**32, 2**62, -2**63, 2**63 - 1], dtype="<i8"),
"u8": np.array([0, 1, 127, 128, 255, 256, 65535, 65536, 2**31, 2**32, 2**63, 2**64 - 1], dtype="<u8"),
"i1": np.array([-128, -1, 0, 1, 127], dtype="i1"),
"u2": np.array([0, 255, 256, 65535], dtype=">u2"),
"b": np.array([True, False, True]),
}
for target in NUMERIC:
for sname, src in sources.items():
vals = src[~_libhdf5_undefined(src, np.dtype(target))]
path_t = str(tmp_path / f"t_{target[1:]}_{sname}.h5")
path_o = str(tmp_path / f"o_{target[1:]}_{sname}.h5")
with h5py.File(path_t, "w") as f:
f.create_dataset("d", shape=vals.shape, dtype=target)
shutil.copy(path_t, path_o)
what = f"{vals.dtype} -> {target}"
try:
with h5py.File(path_t, "r+") as f:
f["d"][...] = _native_conversion(h5py, vals, f["d"].dtype)
except Exception: # noqa: BLE001
with clawhdf5.File(path_o, "r+") as f, pytest.raises(Exception):
f["d"][...] = vals
continue
with clawhdf5.File(path_o, "r+") as f:
f["d"][...] = vals
with h5py.File(path_t, "r") as a, h5py.File(path_o, "r") as b:
assert a["d"][...].tobytes() == b["d"][...].tobytes(), (
f"{what}: h5py {a['d'][...].tolist()} clawhdf5 {b['d'][...].tolist()}"
)
def test_nan_into_an_integer_dataset_is_refused(h5py, tmp_path):
"""libhdf5 stores NaN as an arbitrary integer (0, the minimum or 2**63,
depending on the type); clawhdf5 refuses and writes nothing."""
path = str(tmp_path / "nan.h5")
with h5py.File(path, "w") as f:
f.create_dataset("d", data=np.arange(4, dtype="<i4"))
with clawhdf5.File(path, "r+") as f:
with pytest.raises(ValueError, match="NaN"):
f["d"][...] = np.array([1.0, np.nan, 2.0, 3.0])
# A Python list goes through numpy, which refuses NaN too.
with pytest.raises(ValueError):
f["d"][0:2] = [np.nan, 1.0]
with h5py.File(path, "r") as f:
np.testing.assert_array_equal(f["d"][...], np.arange(4))
def test_unsupported_edits_are_clear_errors(h5py, tmp_path):
path = str(tmp_path / "u.h5")
_h5py_file(h5py, path, "earliest")
before = snapshot(h5py, path)
with clawhdf5.File(path, "r+") as f:
with pytest.raises(NotImplementedError, match="delet"):
del f.attrs["version"]
with pytest.raises(NotImplementedError, match="delet"):
del f["i4"]
with pytest.raises(NotImplementedError, match="delet"):
del f["grp"]["leaf"]
with pytest.raises(NotImplementedError):
f.create_dataset("new", data=np.arange(3.0))
with pytest.raises(NotImplementedError):
f.create_group("newgrp")
with pytest.raises(NotImplementedError):
f["grp"].create_dataset("new", data=np.arange(3.0))
with pytest.raises(NotImplementedError, match="variable-length"):
f["vlen"][0] = "x"
with pytest.raises(NotImplementedError, match="field"):
f["cmp"]["id"] = np.arange(5)
with pytest.raises(NotImplementedError):
f.attrs["empty"] = clawhdf5.Empty("f8")
with pytest.raises(TypeError, match="chunked"):
f["i4"].resize((7, 10))
with pytest.raises(ValueError):
f["chunk_gzip"].resize((41, 20))
with pytest.raises(ValueError, match="axis"):
f["chunk_ext"].resize(3, axis=2)
with pytest.raises(TypeError):
f["i4"][0] = np.array(["a"] * 10)
# h5py writes through boolean masks; clawhdf5 does not.
with pytest.raises(NotImplementedError, match="mask"):
f["u1"][np.arange(16) % 2 == 0] = 5
with pytest.raises(NotImplementedError, match="mask"):
f["i4"][f["i4"][()] > 30] = 0
with pytest.raises(NotImplementedError, match="mask"):
f["i4"][np.ones(6, dtype=bool), 2] = 0
assert snapshot(h5py, path) == before
h5dump_reads(path)
def test_read_only_files_and_modes(h5py, tmp_path):
path = str(tmp_path / "m.h5")
with h5py.File(path, "w") as f:
f.create_dataset("d", data=np.arange(4, dtype="<i4"), chunks=(2,), maxshape=(None,))
with clawhdf5.File(path, "r") as f:
with pytest.raises(OSError, match="r\\+"):
f["d"][0] = 1
with pytest.raises(OSError):
f["d"].resize((8,))
with pytest.raises(OSError):
f.attrs["x"] = 1
with pytest.raises(NotImplementedError, match="does not exist"):
clawhdf5.File(str(tmp_path / "missing.h5"), "a")
with pytest.raises(ValueError, match="mode"):
clawhdf5.File(path, "rw")
with clawhdf5.File(path, "a") as f:
assert f.mode == "r+"
f["d"][1] = 10
with h5py.File(path, "r") as f:
assert f["d"][1] == 10
def test_the_file_is_locked_while_open_for_editing(h5py, tmp_path):
path = str(tmp_path / "lock.h5")
with h5py.File(path, "w") as f:
f.create_dataset("d", data=np.arange(4, dtype="<i4"))
f = clawhdf5.File(path, "r+")
with pytest.raises(OSError):
clawhdf5.File(path, "r+")
with pytest.raises(OSError):
h5py.File(path, "r+")
f.close()
with h5py.File(path, "r+") as g:
g["d"][0] = 5
with clawhdf5.File(path, "r+") as g:
g["d"][1] = 6
with h5py.File(path, "r") as g:
np.testing.assert_array_equal(g["d"][...], [5, 6, 2, 3])
def test_objects_see_edits_made_through_others(h5py, tmp_path):
"""A dataset or attrs object taken before an edit reports the file as
it is after it: the new shape, the new attribute."""
path = str(tmp_path / "live.h5")
with h5py.File(path, "w") as f:
f.create_dataset("d", data=np.arange(6.0), chunks=(4,), maxshape=(None,))
with clawhdf5.File(path, "r+") as f:
d1 = f["d"]
d2 = f["d"]
attrs = d1.attrs
assert len(attrs) == 0 and "u" not in attrs
d2.resize((10,))
assert d1.shape == (10,) and d1.size == 10 and len(d1) == 10
np.testing.assert_array_equal(d1[6:], np.zeros(4))
d2.attrs["u"] = "m/s"
assert "u" in attrs and attrs["u"] == b"m/s" and len(attrs) == 1
f.attrs.create("shaped", np.arange(6), shape=(2, 3), dtype="<i2")
f.attrs.modify("shaped2", [1.5, 2.5])
with h5py.File(path, "r") as f:
assert f["d"].shape == (10,)
assert f["d"].attrs["u"] == b"m/s"
assert f.attrs["shaped"].dtype == np.dtype("<i2") and f.attrs["shaped"].shape == (2, 3)
np.testing.assert_array_equal(f.attrs["shaped2"], [1.5, 2.5])
def test_attribute_types_as_h5py_reads_them(h5py, tmp_path):
path = str(tmp_path / "attrs.h5")
with h5py.File(path, "w") as f:
f.create_group("g")
values = {
"i8": 5,
"f8": 2.5,
"i1": np.int8(-3),
"u8": np.uint64(2**64 - 1),
">f4": np.array([1.5, 2.5], dtype=">f4"),
"f2": np.float16(0.5),
"b": True,
"barr": np.array([True, False]),
"c16": np.complex128(1 + 2j),
"bytes": b"raw",
"sarr": np.array([b"a", b"bcd"]),
"str": "héllo",
"strs": ["x", "yz"],
"2d": np.arange(12, dtype="<u2").reshape(3, 4),
"empty": np.zeros((0,), dtype="<i4"),
}
with clawhdf5.File(path, "r+") as f:
for k, v in values.items():
f["g"].attrs[k] = v
# Replace one, with another type and size.
f["g"].attrs["i8"] = np.arange(100.0)
with h5py.File(path, "r") as f:
a = f["g"].attrs
np.testing.assert_array_equal(a["i8"], np.arange(100.0))
assert a["f8"] == 2.5 and a["f8"].dtype == np.float64
assert a["i1"] == -3 and a["i1"].dtype == np.int8
assert a["u8"] == 2**64 - 1 and a["u8"].dtype == np.uint64
assert a[">f4"].dtype == np.dtype(">f4")
assert a["f2"].dtype == np.float16
assert a["b"] is np.True_ or a["b"] == True # noqa: E712
assert a["barr"].dtype == np.bool_
assert a["c16"] == 1 + 2j
assert a["bytes"] == b"raw"
assert list(a["sarr"]) == [b"a", b"bcd"]
# str is stored as fixed-length UTF-8: h5py reads bytes.
assert a["str"].decode("utf-8") == "héllo"
assert [x.decode() for x in a["strs"]] == ["x", "yz"]
assert a["2d"].shape == (3, 4) and a["2d"].dtype == np.dtype("<u2")
assert a["empty"].shape == (0,)
with clawhdf5.File(path, "r") as f:
assert f["g"].attrs["c16"] == 1 + 2j
assert f["g"].attrs["str"].decode("utf-8") == "héllo"
h5dump_reads(path)
def test_many_attributes_move_to_dense_storage(h5py, tmp_path):
"""Past the compact limit (8 attributes under libver v114) the object's
attributes move to dense storage; h5py reads all of them."""
path = str(tmp_path / "dense.h5")
with h5py.File(path, "w", libver="v114") as f:
f.create_dataset("d", data=np.arange(3))
with clawhdf5.File(path, "r+") as f:
for i in range(20):
f["d"].attrs[f"a{i:02d}"] = np.full(i + 1, i, dtype="<i2")
with h5py.File(path, "r") as f:
assert sorted(f["d"].attrs.keys()) == [f"a{i:02d}" for i in range(20)]
for i in range(20):
np.testing.assert_array_equal(f["d"].attrs[f"a{i:02d}"], np.full(i + 1, i))
h5dump_reads(path)
def test_reads_never_see_a_half_written_edit(h5py, tmp_path):
"""Readers on other threads while one thread rewrites a dataset: every
read returns one whole version (all elements equal), never a mix."""
path = str(tmp_path / "race.h5")
with h5py.File(path, "w") as f:
f.create_dataset("d", data=np.zeros((64, 64)), chunks=(16, 16), compression="gzip")
errors = []
with clawhdf5.File(path, "r+") as f:
stop = threading.Event()
def read():
ds = f["d"]
while not stop.is_set():
a = ds[...]
if not (a == a.flat[0]).all():
errors.append(a)
return
readers = [threading.Thread(target=read) for _ in range(3)]
for t in readers:
t.start()
try:
for k in range(1, 25):
f["d"][...] = float(k)
finally:
stop.set()
for t in readers:
t.join()
np.testing.assert_array_equal(f["d"][...], np.full((64, 64), 24.0))
assert not errors, "a read saw a partly written dataset"
def test_close_releases_the_file(h5py, tmp_path):
path = str(tmp_path / "close.h5")
with h5py.File(path, "w") as f:
f.create_dataset("d", data=np.arange(4, dtype="<i4"))
f = clawhdf5.File(path, "r+")
ds = f["d"]
f.close()
# The handle still reads the file as last written, but cannot edit it.
np.testing.assert_array_equal(ds[...], np.arange(4))
with pytest.raises(OSError, match="closed"):
ds[0] = 1
with h5py.File(path, "r+") as g:
g["d"][0] = 9
def _two_files(h5py, tmp_path):
"""d1/f.h5 and d2/f.h5: same name, different layouts (the review's
repro)."""
(tmp_path / "d1").mkdir()
(tmp_path / "d2").mkdir()
with h5py.File(tmp_path / "d1" / "f.h5", "w") as f:
f.create_dataset("x", data=np.arange(10, dtype="<i4"))
f.create_dataset("big", data=np.full(5000, 1.5))
with h5py.File(tmp_path / "d2" / "f.h5", "w") as f:
f.create_dataset("pad", data=np.full(3000, 2.5))
f.create_dataset("x", data=np.arange(10, dtype="<i4") + 500)
return tmp_path / "d1" / "f.h5", tmp_path / "d2" / "f.h5"
def _held_file_edited(h5py, held, other, other_bytes):
"""The edits went to `held`, planned from its own metadata; `other`
was not touched."""
assert other.read_bytes() == other_bytes, "the other file changed"
with h5py.File(held, "r") as f:
np.testing.assert_array_equal(f["x"][()], np.full(10, 7))
np.testing.assert_array_equal(f["big"][()], np.full(5000, 1.5))
np.testing.assert_array_equal(f.attrs["note"], np.arange(50.0))
h5dump_reads(str(held))
def test_relative_path_and_chdir(h5py, tmp_path, monkeypatch):
"""A file opened by a relative path keeps being the file edited and read
after os.chdir (an edit was planned from the file the path named in the
new directory and written into the one open, corrupting it)."""
held, other = _two_files(h5py, tmp_path)
other_bytes = other.read_bytes()
monkeypatch.chdir(held.parent)
f = clawhdf5.File("f.h5", "r+")
ds = f["x"]
monkeypatch.chdir(other.parent)
ds[:] = np.full(10, 7, "<i4")
np.testing.assert_array_equal(ds[:5], np.full(5, 7)) # read from the held file
f.attrs["note"] = np.arange(50.0)
np.testing.assert_array_equal(f["big"][()], np.full(5000, 1.5))
assert "pad" not in f
f.close()
_held_file_edited(h5py, held, other, other_bytes)
# 'w' writes where the path named when the file was opened.
monkeypatch.chdir(held.parent)
w = clawhdf5.File("new.h5", "w")
monkeypatch.chdir(other.parent)
w.create_dataset("d", data=np.arange(3))
w.close()
assert (held.parent / "new.h5").exists() and not (other.parent / "new.h5").exists()
def test_path_replaced_between_edits(h5py, tmp_path):
"""The path renamed away and another file put in its place between
edits: the edits go to the file held open, never mixed with the other."""
held, other = _two_files(h5py, tmp_path)
path = str(tmp_path / "f.h5")
os.replace(held, path)
moved = tmp_path / "moved.h5"
with clawhdf5.File(path, "r+") as f:
ds = f["x"]
ds[0] = 7
os.replace(path, moved)
shutil.copy(other, path)
other_bytes = open(path, "rb").read()
ds[:] = np.full(10, 7, "<i4")
f.attrs["note"] = np.arange(50.0)
np.testing.assert_array_equal(f["x"][()], np.full(10, 7))
assert "pad" not in f
from pathlib import Path
_held_file_edited(h5py, moved, Path(path), other_bytes)
def test_an_edit_releases_the_gil(h5py, tmp_path):
"""Another Python thread keeps running while a large edit is written:
had the edit held the GIL, the other thread would stall for the whole
edit (Rust code never yields it)."""
import time
path = str(tmp_path / "big.h5")
with h5py.File(path, "w") as f:
f.create_dataset("d", shape=(1024, 1024), dtype="<f8", chunks=(64, 64), compression="gzip")
value = np.random.default_rng(0).standard_normal((1024, 1024))
stamps = []
stop = threading.Event()
def spin():
while not stop.is_set():
stamps.append(time.perf_counter())
with clawhdf5.File(path, "r+") as f:
ds = f["d"]
t = threading.Thread(target=spin)
t.start()
try:
time.sleep(0.05)
t0 = time.perf_counter()
ds[...] = value
t1 = time.perf_counter()
finally:
stop.set()
t.join()
np.testing.assert_array_equal(ds[...], value)
during = [x for x in stamps if t0 <= x <= t1]
gaps = np.diff([t0] + during + [t1])
assert t1 - t0 > 0.1, "the edit is too quick to tell"
assert gaps.max() < 0.5 * (t1 - t0), (
f"the other thread stalled for {gaps.max():.3f} s of a {t1 - t0:.3f} s edit")