Files
clawhdf5/crates/clawhdf5-py/tests/test_edit.py
T
osobhandClaude Opus 5.5 92fb0830e0 py: in-place editing (clawhdf5.File(path, 'r+')) through FileEditor
clawhdf5.File(path, 'r+') (and 'a' on an existing file) holds a
FileEditor, and with it the file's exclusive lock, until close():

- ds[key] = value: h5py's keys and broadcasting (numpy's rules for
  slices and integers with extra leading 1-axes allowed; the exact shape
  for an index list, a scalar only where h5py expands it). Arrays are
  converted as libhdf5 converts them in native byte order (integers
  saturate, floats truncate toward zero and clip, integers go into h5py's
  bool enum by value); other values through
  numpy.asarray(value, dtype=ds.dtype), as h5py does. NaN into an integer
  dataset is a ValueError instead of libhdf5's arbitrary value. The value
  preparation is a small Python module compiled into the extension
  (src/edit_helpers.py).
- ds.resize(shape) / ds.resize(n, axis=k) with h5py's argument rules.
- attrs[name] = value, attrs.create(name, data, shape, dtype),
  attrs.modify: numeric, bool, complex, bytes and str data of any shape,
  with h5py's HDF5 types; str is stored as fixed-length UTF-8 (the editor
  cannot write variable-length strings).
- File.mode, File.flush(), Dataset.chunks.

Each edit runs with the GIL released under the file handle's write lock
(no read sees a half-written edit), then the file is reopened;
datasets and attrs objects re-read their shape and attributes when the
handle's edit generation moved. What the editor cannot do is
NotImplementedError before anything is written: deleting attributes or
objects, creating datasets or groups, compound fields by name,
variable-length data, and FileEditor's own limits.

Where libhdf5 2.0 (h5py 3.16) converts inconsistently -- its soft
conversions in non-native byte order (a float in (-1, 0) becomes the
integer minimum, same-size unsigned->signed wraps) and native casts that
are undefined in C (half floats into unsigned, float(max) rounded up) --
clawhdf5 saturates as libhdf5's native path does; listed in
docs/known-issues.md.

Tests (tests/test_edit.py): every edit applied by h5py and by clawhdf5 to
copies of the same file and both read back through h5py after each edit,
on h5py files (libver earliest, v114, latest) and a clawhdf5 file: a fixed
sequence over every chunk index kind, compact/contiguous/gzip layouts and
numeric, bool, enum, complex, string and compound types, 16 random
sequences of 40 edits, and a numeric conversion matrix; a refused edit
must be refused by both and leave the file unchanged. Also dense
attributes, locking, objects seeing edits, readers racing a writer, and
h5dump (plus h5rs check in ci-test.sh) on every edited file. The
read-vs-h5py suite also runs on a file opened 'r+'.

Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
2026-09-27 06:57:36 -05:00

743 lines
32 KiB
Python

"""In-place editing: clawhdf5.File(path, 'r+') against h5py.
Every edit is applied twice, to two copies of the same file: once through
h5py (libhdf5) and once through clawhdf5 (FileEditor). After every edit both
files are read back with h5py and must hold the same shapes, values and
attributes; clawhdf5's own view must agree; when h5py refuses an edit,
clawhdf5 must refuse it too and leave its file as it was. Files are written
by h5py (libver earliest and latest, so every chunk index kind) and by
clawhdf5; `h5dump` must read every result."""
import io
import os
import shutil
import subprocess
import threading
import numpy as np
import pytest
import clawhdf5
# ---------------------------------------------------------------------------
# Files
# ---------------------------------------------------------------------------
ENUM = {"RED": 0, "GREEN": 1, "BLUE": 7}
def _h5py_file(h5py, path, libver):
rng = np.random.default_rng(1)
with h5py.File(path, "w", libver=libver) as f:
f.create_dataset("i4", data=np.arange(60, dtype="<i4").reshape(6, 10))
f.create_dataset("be_i2", data=np.arange(24, dtype=">i2").reshape(4, 6))
f.create_dataset("u1", data=np.arange(16, dtype="u1"))
f.create_dataset("f2", data=rng.standard_normal(12).astype("<f2"))
f.create_dataset("f4_2d", data=rng.standard_normal((8, 9)).astype("<f4"))
f.create_dataset("f8_3d", data=rng.standard_normal((4, 5, 6)))
f.create_dataset("u8", data=np.arange(10, dtype="<u8"))
f.create_dataset("c8", data=(np.arange(6) + 1j * np.arange(6)).astype("<c8"))
f.create_dataset("bool", data=np.array([True, False, True, True]))
f.create_dataset("enum", data=np.array([0, 1, 7, 0], dtype="i1"),
dtype=h5py.enum_dtype(ENUM, basetype="i1"))
f.create_dataset("s5", data=np.array([b"ab", b"cdefg", b""], dtype="S5"))
cmp_dt = np.dtype([("id", "<i4"), ("x", "<f8"), ("tag", "S3")])
f.create_dataset("cmp", data=np.array([(i, i / 2, b"t%d" % i) for i in range(5)], dtype=cmp_dt))
f.create_dataset("scalar", data=np.float64(3.5))
# Chunked: fixed maxshape (v4 fixed array under latest), one
# unlimited dimension (extensible array), two (v2 B-tree), one chunk.
f.create_dataset("chunk_fixed", data=np.arange(100, dtype="<i8").reshape(10, 10), chunks=(3, 4))
f.create_dataset("chunk_ext", data=rng.standard_normal((12, 7)), chunks=(5, 7), maxshape=(None, 7))
f.create_dataset("chunk_bt2", data=np.arange(30, dtype="<i4").reshape(5, 6), chunks=(2, 2),
maxshape=(None, None))
f.create_dataset("chunk_gzip", data=np.arange(400, dtype="<f4").reshape(20, 20), chunks=(6, 6),
compression="gzip", maxshape=(40, 40))
f.create_dataset("chunk_single", data=np.arange(12, dtype="<u2").reshape(3, 4), chunks=(3, 4),
maxshape=(3, 4))
f.create_dataset("chunk_fill", shape=(8,), dtype="<i4", chunks=(3,), maxshape=(20,), fillvalue=-1)
f.create_dataset("vlen", data=["a", "bb"], dtype=h5py.string_dtype())
# Compact layout (low-level API).
dcpl = h5py.h5p.create(h5py.h5p.DATASET_CREATE)
dcpl.set_layout(h5py.h5d.COMPACT)
space = h5py.h5s.create_simple((7,))
dsid = h5py.h5d.create(f.id, b"compact", h5py.h5t.STD_I32LE, space, dcpl=dcpl)
dsid.write(h5py.h5s.ALL, h5py.h5s.ALL, np.arange(7, dtype="<i4"))
g = f.create_group("grp")
g.create_dataset("leaf", data=np.arange(5.0))
g.attrs["units"] = "m"
f.attrs["version"] = np.int32(1)
def _clawhdf5_file(path):
with clawhdf5.File(str(path), "w") as f:
f.create_dataset("i4", data=np.arange(60, dtype="<i4").reshape(6, 10))
f.create_dataset("f8", data=np.linspace(0, 1, 30).reshape(5, 6))
f.create_dataset("chunk_gzip", data=np.arange(400, dtype="<f4").reshape(20, 20),
chunks=[6, 6], compression="gzip")
f.create_dataset("u1", data=np.arange(16, dtype="u1"))
g = f.create_group("grp")
g.create_dataset("leaf", data=np.arange(5.0))
g.attrs["units"] = "m"
f.attrs["version"] = 1
# h5py's libver: "earliest" (v1 B-tree chunk indexes), "v114" (the 1.10+
# indexes: fixed and extensible arrays, v2 B-trees, single chunk) and
# "latest" (HDF5 2.0's newest format, which h5dump 1.14 cannot read).
SOURCES = ["h5py-earliest", "h5py-v114", "h5py-latest", "clawhdf5"]
def _make(h5py, tmp_path, source):
base = tmp_path / f"base-{source}.h5"
if source == "clawhdf5":
_clawhdf5_file(base)
else:
_h5py_file(h5py, str(base), source.split("-")[1])
theirs = tmp_path / f"theirs-{source}.h5"
ours = tmp_path / f"ours-{source}.h5"
shutil.copy(base, theirs)
shutil.copy(base, ours)
return str(theirs), str(ours), str(base)
# ---------------------------------------------------------------------------
# Comparing files through h5py
# ---------------------------------------------------------------------------
def _norm_attr(v):
"""Attribute values comparable across the two writers: clawhdf5 stores
`str` as fixed-length UTF-8 (h5py reads bytes), h5py as variable-length
(h5py reads str)."""
if isinstance(v, bytes):
return ("str", v.decode("utf-8"))
if isinstance(v, str):
return ("str", v)
arr = np.asarray(v)
if arr.dtype.kind in "SO":
return ("strs", [x.decode() if isinstance(x, bytes) else x for x in arr.ravel().tolist()], arr.shape)
return (arr.dtype.str, arr.shape, arr.tobytes())
def snapshot(h5py, path):
"""What h5py sees in the file: every dataset's shape, dtype, bytes and
attributes (read without locking: clawhdf5 may hold the file open)."""
out = {}
with h5py.File(path, "r", locking=False) as f:
def visit(name, obj):
attrs = {k: _norm_attr(obj.attrs[k]) for k in obj.attrs}
if isinstance(obj, h5py.Dataset):
if obj.dtype.kind == "O":
data = [x for x in obj[...].ravel().tolist()]
else:
data = obj[()].tobytes() if obj.shape is not None else None
out[name] = (obj.shape, obj.dtype.str, obj.maxshape, data, attrs)
else:
out[name] = ("group", attrs)
visit("/", f)
f.visititems(visit)
return out
def assert_same_files(h5py, theirs, ours, what):
a, b = snapshot(h5py, theirs), snapshot(h5py, ours)
assert a.keys() == b.keys(), what
for k in a:
assert a[k] == b[k], f"{what}: {k} differs\n h5py: {a[k]}\n clawhdf5: {b[k]}"
def assert_ours_reads_like_h5py(h5py, f, path, what):
"""clawhdf5's own view of the file it is editing matches h5py's."""
with h5py.File(path, "r", locking=False) as t:
for name in ["i4", "chunk_ext", "chunk_bt2", "chunk_gzip", "f8_3d", "cmp", "bool", "enum", "scalar"]:
if name not in t:
continue
o = f[name]
assert o.shape == t[name].shape, f"{what}: {name} shape"
assert o.maxshape == t[name].maxshape, f"{what}: {name} maxshape"
np.testing.assert_array_equal(o[()], t[name][()], err_msg=f"{what}: {name}")
for obj in ["/", "grp"]:
assert sorted(f[obj].attrs.keys()) == sorted(t[obj].attrs.keys()), what
for k in t[obj].attrs:
assert _norm_attr(f[obj].attrs[k]) == _norm_attr(t[obj].attrs[k]), f"{what}: {obj}.attrs[{k}]"
def h5dump_reads(path, base=None):
"""h5dump (libhdf5 1.14) reads every object and value of `path` — when it
reads the unedited `base` (it cannot read HDF5 2.0's newest format)."""
exe = shutil.which("h5dump")
if exe is None:
if os.environ.get("CLAWHDF5_REQUIRE_INTEROP") == "1":
pytest.fail("h5dump is required (CLAWHDF5_REQUIRE_INTEROP=1)")
return
h5rs = os.environ.get("CLAWHDF5_H5RS")
if h5rs:
# clawhdf5's structural and checksum validator (scripts/ci-test.sh
# points this at the h5rs it built).
r = subprocess.run([h5rs, "check", path], capture_output=True, text=True)
assert r.returncode == 0, (r.stdout + r.stderr)[-2000:]
if base is not None and subprocess.run([exe, "-H", base], capture_output=True).returncode != 0:
return
r = subprocess.run([exe, path], capture_output=True, text=True)
assert r.returncode == 0, r.stderr[-2000:]
# ---------------------------------------------------------------------------
# Applying one edit both ways
# ---------------------------------------------------------------------------
def _native_conversion(h5py, value, ds_dtype):
"""`value` as libhdf5 converts it to `ds_dtype` in native byte order.
libhdf5 2.0 (h5py 3.16) converts numbers differently when either side is
not in native byte order (its "soft" conversions): a float in (-1, 0)
becomes the integer type's minimum instead of 0, and an unsigned integer
too large for the signed type of the same size wraps instead of
saturating. clawhdf5 applies the native-order results to every byte
order, so the reference is h5py converting into a native dataset of the
same kind; the result then reaches the real dataset by a plain byte
swap."""
if not isinstance(value, np.ndarray):
return value
if value.dtype.kind not in "biuf" or ds_dtype.kind not in "biuf":
return value
if value.dtype.isnative and ds_dtype.isnative:
return value
with h5py.File(io.BytesIO(), "w") as tmp:
d = tmp.create_dataset("t", shape=value.shape, dtype=ds_dtype.newbyteorder("="))
d[...] = value.astype(value.dtype.newbyteorder("="))
return np.asarray(d[()])
def _apply(f, op, h5py=None):
"""Apply `op` to `f`; with `h5py`, `f` is an h5py file and a numpy array
value is first converted as libhdf5 converts in native byte order (see
`_native_conversion`)."""
kind = op[0]
if kind == "set":
_, name, key, value = op
if h5py is not None:
value = _native_conversion(h5py, value, f[name].dtype)
f[name][key] = value
elif kind == "resize":
_, name, size, axis = op
if axis is None:
f[name].resize(size)
else:
f[name].resize(size, axis=axis)
elif kind == "attr":
_, obj, name, value = op
f[obj].attrs[name] = value
else:
raise AssertionError(op)
def edit_both(h5py, theirs, ours_path, ours, op):
"""Apply `op` with h5py and with clawhdf5 (`ours`, open 'r+'); the two
files must then read the same through h5py. If h5py refuses, clawhdf5
must refuse and its file must be unchanged. Returns h5py's error."""
before = snapshot(h5py, ours_path)
try:
with h5py.File(theirs, "r+") as t:
_apply(t, op, h5py)
except Exception as e: # noqa: BLE001 - h5py refuses: so must we
try:
_apply(ours, op)
except Exception: # noqa: BLE001
pass
else:
pytest.fail(f"{op!r:.300}: h5py refused ({type(e).__name__}: {e}), clawhdf5 did not")
assert snapshot(h5py, ours_path) == before, f"{op!r}: clawhdf5 changed the file while failing"
return e
_apply(ours, op)
assert_same_files(h5py, theirs, ours_path, repr(op)[:200])
return None
# ---------------------------------------------------------------------------
# Tests
# ---------------------------------------------------------------------------
@pytest.mark.parametrize("source", SOURCES)
def test_edit_sequence_matches_h5py(h5py, tmp_path, source):
theirs, ours_path, base = _make(h5py, tmp_path, source)
ops = [
("set", "i4", 0, 99),
("set", "i4", (slice(1, 5, 2), slice(None, None, 3)), np.array([[1.5, -2.5, 1e12, -1e12]])),
("set", "i4", (slice(None), 2), np.arange(6, dtype="<i8") * 1000),
("set", "i4", ([0, 2, 5], slice(4, 6)), np.array([[1, 2], [3, 4], [5, 6]])),
("set", "i4", (Ellipsis, -1), np.int8(-7)),
("set", "i4", (3, 3), 12345.9),
("set", "u1", slice(2, 8), np.array([-5, 0, 300, 255, 256, 1], dtype="<i4")),
("set", "u1", slice(0, 3), [1, 2, 3]),
("set", "chunk_gzip", (slice(0, 20, 7), slice(3, 17)), 42.25),
("set", "chunk_gzip", (slice(5, 11), slice(5, 11)), np.ones((6, 6), dtype="<f8") * np.pi),
("set", "grp/leaf", slice(None), np.array([1, 2, 3, 4, 5], dtype="<u2")),
("attr", "/", "version", np.int32(2)),
("attr", "/", "count", 7),
("attr", "grp", "scale", np.array([0.5, 0.25], dtype="<f4")),
("attr", "grp", "matrix", np.arange(6, dtype=">i8").reshape(2, 3)),
("attr", "i4", "flag", np.bool_(True)),
("attr", "i4", "z", np.complex64(1 - 2j)),
("attr", "i4", "raw", np.bytes_(b"abc")),
("set", "i4", slice(0, 2), np.zeros((3, 10))), # shape mismatch: refused by both
]
if source != "clawhdf5":
ops += [
("set", "be_i2", (slice(None), slice(1, 3)), np.array([70000, -70000], dtype="<i8")),
("set", "f2", slice(None, None, 4), np.array([1e6, -3.25, 0.1])),
("set", "f4_2d", (2, slice(None)), np.linspace(-1, 1, 9)),
("set", "f8_3d", (slice(1, 3), 2, slice(None, None, 2)), np.arange(3, dtype="<i2")),
("set", "u8", slice(None), np.array([-1, 0, 2**63, 1e30, -1e30, 5.5, 2, 3, 4, 5])),
("set", "c8", slice(1, 3), np.array([1 + 1j, 2 - 2j], dtype="<c16")),
("set", "c8", 0, np.float64(1.0)), # h5py: no conversion path
("set", "bool", slice(None), np.array([0, 3, 0, -1], dtype="<i4")),
("set", "bool", 1, np.array(True)),
("set", "enum", slice(0, 2), np.array([7, 1], dtype="<i4")),
("set", "s5", 0, np.bytes_(b"xyzuvw")),
("set", "s5", slice(1, 3), [b"q", b"rs"]),
("set", "s5", 2, np.array("uni")), # h5py: no conversion from 'U'
("set", "cmp", 2, np.array((9, 9.5, b"zz"), dtype=[("id", "<i4"), ("x", "<f8"), ("tag", "S3")])),
("set", "cmp", slice(3, 5), [(1, 0.5, b"a"), (2, 1.5, b"b")]),
("set", "scalar", (), 7.25),
("set", "scalar", Ellipsis, np.float32(-1.5)),
("set", "compact", slice(1, 6, 2), np.array([10, 20, 30])),
("set", "chunk_fixed", (slice(2, 9), slice(1, 10, 4)), np.arange(21).reshape(7, 3)),
("set", "chunk_single", (1, slice(None)), np.array([9, 8, 7, 6])),
("resize", "chunk_ext", (20, 7), None),
("set", "chunk_ext", slice(12, 20), np.full((8, 7), 2.5)),
("resize", "chunk_ext", 9, 0),
("resize", "chunk_ext", 16, 0),
("resize", "chunk_bt2", (9, 11), None),
("set", "chunk_bt2", (slice(4, 9), slice(5, 11)), np.arange(30).reshape(5, 6)),
("resize", "chunk_bt2", (3, 3), None),
("resize", "chunk_bt2", (7, 8), None),
("resize", "chunk_gzip", (40, 25), None),
("resize", "chunk_gzip", (41, 25), None), # beyond maxshape: refused
("resize", "chunk_fill", 15, None), # h5py: a size without axis must be a tuple
("resize", "chunk_fill", (15,), None),
("set", "chunk_fill", slice(10, 12), [1, 2]),
("resize", "chunk_fill", (4,), None),
("resize", "chunk_fill", (20,), None),
("resize", "i4", (7, 10), None), # not chunked: refused
]
with clawhdf5.File(ours_path, "r+") as ours:
assert ours.mode == "r+"
for i, op in enumerate(ops):
edit_both(h5py, theirs, ours_path, ours, op)
if i % 5 == 0:
assert_ours_reads_like_h5py(h5py, ours, ours_path, repr(op))
assert_ours_reads_like_h5py(h5py, ours, ours_path, "end")
h5dump_reads(ours_path, base)
# Reopened, clawhdf5 reads what h5py reads.
with clawhdf5.File(ours_path, "r") as f:
assert f.mode == "r"
assert_ours_reads_like_h5py(h5py, f, ours_path, "reopened")
def _random_key(rng, shape):
key = []
for n in shape:
r = rng.random()
if n == 0 or r < 0.3:
start = int(rng.integers(0, n + 1)) if n else 0
stop = int(rng.integers(start, n + 1)) if n else 0
step = int(rng.integers(1, 4))
key.append(slice(start, stop, step))
elif r < 0.55:
key.append(int(rng.integers(-n, n)))
elif r < 0.7:
k = int(rng.integers(1, min(n, 4) + 1))
key.append(sorted(rng.choice(n, size=k, replace=False).tolist()))
else:
key.append(slice(None))
# Only one index list per key.
lists = [i for i, k in enumerate(key) if isinstance(k, list)]
for i in lists[1:]:
key[i] = slice(None)
return tuple(key)
def _selection_shape(key, shape):
out = []
fancy = False
for k, n in zip(key, shape):
if isinstance(k, slice):
out.append(len(range(*k.indices(n))))
elif isinstance(k, list):
out.append(len(k))
fancy = True
return tuple(out), fancy
def _random_value(rng, sel_shape, fancy, dtype):
r = rng.random()
if r < 0.2 or not sel_shape:
v = rng.standard_normal() * 1000
return np.float64(v) if rng.random() < 0.5 else int(v)
shape = list(sel_shape)
if not fancy and r < 0.35 and shape:
shape[0] = 1 # broadcast along the first axis
kind = rng.choice(["same", "f8", "i8", "u1", "f4"])
if kind == "same" and np.dtype(dtype).kind in "iuf":
dt = np.dtype(dtype)
else:
dt = np.dtype(str(kind) if kind != "same" else "f8")
base = rng.standard_normal(size=shape) * (10 ** rng.integers(0, 6))
with np.errstate(all="ignore"):
return base.astype(dt)
@pytest.mark.parametrize("source", SOURCES)
@pytest.mark.parametrize("seed", [0, 1, 2, 3])
def test_random_edits_match_h5py(h5py, tmp_path, source, seed):
"""Random writes (slices, steps, integers, index lists, broadcasts,
other dtypes and out-of-range values), resizes and attributes, each
compared with h5py applying the same edit."""
rng = np.random.default_rng(1000 * seed + SOURCES.index(source))
theirs, ours_path, base = _make(h5py, tmp_path, source)
with h5py.File(theirs, "r") as t:
names = [n for n in ["i4", "u1", "f8", "f4_2d", "f8_3d", "be_i2", "chunk_fixed", "chunk_ext",
"chunk_bt2", "chunk_gzip", "chunk_fill", "compact", "grp/leaf"] if n in t]
resizable = [n for n in ["chunk_ext", "chunk_bt2", "chunk_gzip", "chunk_fill"] if n in names]
refused = 0
with clawhdf5.File(ours_path, "r+") as ours:
for step in range(40):
r = rng.random()
if r < 0.15 and resizable:
name = str(rng.choice(resizable))
with h5py.File(theirs, "r") as t:
maxshape = t[name].maxshape
shape = t[name].shape
new = tuple(int(rng.integers(0, (m if m is not None else s + 10) + 1)) for m, s in zip(maxshape, shape))
op = ("resize", name, new, None)
elif r < 0.25:
obj = str(rng.choice(["/", "grp", names[0]]))
choices = [np.int16(rng.integers(-100, 100)), rng.standard_normal(3),
np.arange(int(rng.integers(1, 5)), dtype=">u4"), np.float32(0.5)]
value = choices[int(rng.integers(0, len(choices)))]
op = ("attr", obj, f"a{int(rng.integers(0, 4))}", value)
else:
name = str(rng.choice(names))
with h5py.File(theirs, "r") as t:
shape, dtype = t[name].shape, t[name].dtype
key = _random_key(rng, shape)
sel_shape, fancy = _selection_shape(key, shape)
op = ("set", name, key, _random_value(rng, sel_shape, fancy, dtype))
if edit_both(h5py, theirs, ours_path, ours, op) is not None:
refused += 1
assert_ours_reads_like_h5py(h5py, ours, ours_path, "end")
assert refused < 30
h5dump_reads(ours_path, base)
NUMERIC = ["<i1", "<u1", "<i2", ">u2", "<i4", "<u4", ">i8", "<u8", "<f2", "<f4", ">f8"]
def _quiet(f):
with np.errstate(all="ignore"):
return f()
def _libhdf5_undefined(vals, target):
"""The values whose conversion to `target` libhdf5 2.0 (h5py 3.16) gets
wrong even in native byte order, where its C casts are undefined
behaviour; clawhdf5 saturates them as libhdf5's range handling
intends (docs/known-issues.md):
- half floats into unsigned integers: negatives wrap (-1 -> 65535) and
+inf becomes 0; into signed integers, +-inf becomes the minimum;
- a float equal to the integer maximum rounded up in the float's
precision (float32(2**31 - 1) == 2**31 -> int32, float64(2**64 - 1)
-> uint64) becomes the minimum (or 0);
- a double between 65504 and 65520 into a half float becomes infinity
(IEEE rounds it down to 65504, as numpy does).
"""
bad = np.zeros(vals.shape, dtype=bool)
if vals.dtype.kind != "f":
return bad
if target.kind in "iu":
if vals.dtype.itemsize == 2:
bad |= np.isinf(vals)
if target.kind == "u":
bad |= vals <= -1
top = _quiet(lambda: np.array(np.iinfo(target).max).astype(vals.dtype))
if float(top) > np.iinfo(target).max:
bad |= vals == top
if target.kind == "f" and target.itemsize == 2 and vals.dtype.itemsize > 2:
bad |= (np.abs(vals) > 65504) & (np.abs(vals) < 65520)
return bad
def test_numeric_conversions_match_h5py(h5py, tmp_path):
"""Every numeric source dtype into every numeric dataset dtype, with
values at and beyond the targets' limits, as libhdf5 converts them."""
edge = np.array([0, 1, -1, -0.3, 2.5, -2.5, 3.7, -3.7, 127.9, -128.9, 200.5, 255.5, 256, -129,
32767.5, 40000, 65504, 70000, -70000, 2**31 - 1, 2**31, -2**31 - 1,
4e9, 1e15, -1e15, 1e19, 1e300, -1e300, np.inf, -np.inf])
sources = {
"f8": edge,
"f4": _quiet(lambda: edge.astype("<f4")),
"f2": np.array([0, 1, -1, 2.5, -3.5, 65504, -65504, np.inf, -np.inf, 100.5], dtype="<f2"),
"i8": np.array([0, 1, -1, 127, 128, -129, 255, 256, 32768, -32769, 65536, 2**31, -2**31 - 1,
2**32, 2**62, -2**63, 2**63 - 1], dtype="<i8"),
"u8": np.array([0, 1, 127, 128, 255, 256, 65535, 65536, 2**31, 2**32, 2**63, 2**64 - 1], dtype="<u8"),
"i1": np.array([-128, -1, 0, 1, 127], dtype="i1"),
"u2": np.array([0, 255, 256, 65535], dtype=">u2"),
"b": np.array([True, False, True]),
}
for target in NUMERIC:
for sname, src in sources.items():
vals = src[~_libhdf5_undefined(src, np.dtype(target))]
path_t = str(tmp_path / f"t_{target[1:]}_{sname}.h5")
path_o = str(tmp_path / f"o_{target[1:]}_{sname}.h5")
with h5py.File(path_t, "w") as f:
f.create_dataset("d", shape=vals.shape, dtype=target)
shutil.copy(path_t, path_o)
what = f"{vals.dtype} -> {target}"
try:
with h5py.File(path_t, "r+") as f:
f["d"][...] = _native_conversion(h5py, vals, f["d"].dtype)
except Exception: # noqa: BLE001
with clawhdf5.File(path_o, "r+") as f, pytest.raises(Exception):
f["d"][...] = vals
continue
with clawhdf5.File(path_o, "r+") as f:
f["d"][...] = vals
with h5py.File(path_t, "r") as a, h5py.File(path_o, "r") as b:
assert a["d"][...].tobytes() == b["d"][...].tobytes(), (
f"{what}: h5py {a['d'][...].tolist()} clawhdf5 {b['d'][...].tolist()}"
)
def test_nan_into_an_integer_dataset_is_refused(h5py, tmp_path):
"""libhdf5 stores NaN as an arbitrary integer (0, the minimum or 2**63,
depending on the type); clawhdf5 refuses and writes nothing."""
path = str(tmp_path / "nan.h5")
with h5py.File(path, "w") as f:
f.create_dataset("d", data=np.arange(4, dtype="<i4"))
with clawhdf5.File(path, "r+") as f:
with pytest.raises(ValueError, match="NaN"):
f["d"][...] = np.array([1.0, np.nan, 2.0, 3.0])
# A Python list goes through numpy, which refuses NaN too.
with pytest.raises(ValueError):
f["d"][0:2] = [np.nan, 1.0]
with h5py.File(path, "r") as f:
np.testing.assert_array_equal(f["d"][...], np.arange(4))
def test_unsupported_edits_are_clear_errors(h5py, tmp_path):
path = str(tmp_path / "u.h5")
_h5py_file(h5py, path, "earliest")
before = snapshot(h5py, path)
with clawhdf5.File(path, "r+") as f:
with pytest.raises(NotImplementedError, match="delet"):
del f.attrs["version"]
with pytest.raises(NotImplementedError, match="delet"):
del f["i4"]
with pytest.raises(NotImplementedError, match="delet"):
del f["grp"]["leaf"]
with pytest.raises(NotImplementedError):
f.create_dataset("new", data=np.arange(3.0))
with pytest.raises(NotImplementedError):
f.create_group("newgrp")
with pytest.raises(NotImplementedError):
f["grp"].create_dataset("new", data=np.arange(3.0))
with pytest.raises(NotImplementedError, match="variable-length"):
f["vlen"][0] = "x"
with pytest.raises(NotImplementedError, match="field"):
f["cmp"]["id"] = np.arange(5)
with pytest.raises(NotImplementedError):
f.attrs["empty"] = clawhdf5.Empty("f8")
with pytest.raises(TypeError, match="chunked"):
f["i4"].resize((7, 10))
with pytest.raises(ValueError):
f["chunk_gzip"].resize((41, 20))
with pytest.raises(ValueError, match="axis"):
f["chunk_ext"].resize(3, axis=2)
with pytest.raises(TypeError):
f["i4"][0] = np.array(["a"] * 10)
assert snapshot(h5py, path) == before
h5dump_reads(path)
def test_read_only_files_and_modes(h5py, tmp_path):
path = str(tmp_path / "m.h5")
with h5py.File(path, "w") as f:
f.create_dataset("d", data=np.arange(4, dtype="<i4"), chunks=(2,), maxshape=(None,))
with clawhdf5.File(path, "r") as f:
with pytest.raises(OSError, match="r\\+"):
f["d"][0] = 1
with pytest.raises(OSError):
f["d"].resize((8,))
with pytest.raises(OSError):
f.attrs["x"] = 1
with pytest.raises(NotImplementedError, match="does not exist"):
clawhdf5.File(str(tmp_path / "missing.h5"), "a")
with pytest.raises(ValueError, match="mode"):
clawhdf5.File(path, "rw")
with clawhdf5.File(path, "a") as f:
assert f.mode == "r+"
f["d"][1] = 10
with h5py.File(path, "r") as f:
assert f["d"][1] == 10
def test_the_file_is_locked_while_open_for_editing(h5py, tmp_path):
path = str(tmp_path / "lock.h5")
with h5py.File(path, "w") as f:
f.create_dataset("d", data=np.arange(4, dtype="<i4"))
f = clawhdf5.File(path, "r+")
with pytest.raises(OSError):
clawhdf5.File(path, "r+")
with pytest.raises(OSError):
h5py.File(path, "r+")
f.close()
with h5py.File(path, "r+") as g:
g["d"][0] = 5
with clawhdf5.File(path, "r+") as g:
g["d"][1] = 6
with h5py.File(path, "r") as g:
np.testing.assert_array_equal(g["d"][...], [5, 6, 2, 3])
def test_objects_see_edits_made_through_others(h5py, tmp_path):
"""A dataset or attrs object taken before an edit reports the file as
it is after it: the new shape, the new attribute."""
path = str(tmp_path / "live.h5")
with h5py.File(path, "w") as f:
f.create_dataset("d", data=np.arange(6.0), chunks=(4,), maxshape=(None,))
with clawhdf5.File(path, "r+") as f:
d1 = f["d"]
d2 = f["d"]
attrs = d1.attrs
assert len(attrs) == 0 and "u" not in attrs
d2.resize((10,))
assert d1.shape == (10,) and d1.size == 10 and len(d1) == 10
np.testing.assert_array_equal(d1[6:], np.zeros(4))
d2.attrs["u"] = "m/s"
assert "u" in attrs and attrs["u"] == b"m/s" and len(attrs) == 1
f.attrs.create("shaped", np.arange(6), shape=(2, 3), dtype="<i2")
f.attrs.modify("shaped2", [1.5, 2.5])
with h5py.File(path, "r") as f:
assert f["d"].shape == (10,)
assert f["d"].attrs["u"] == b"m/s"
assert f.attrs["shaped"].dtype == np.dtype("<i2") and f.attrs["shaped"].shape == (2, 3)
np.testing.assert_array_equal(f.attrs["shaped2"], [1.5, 2.5])
def test_attribute_types_as_h5py_reads_them(h5py, tmp_path):
path = str(tmp_path / "attrs.h5")
with h5py.File(path, "w") as f:
f.create_group("g")
values = {
"i8": 5,
"f8": 2.5,
"i1": np.int8(-3),
"u8": np.uint64(2**64 - 1),
">f4": np.array([1.5, 2.5], dtype=">f4"),
"f2": np.float16(0.5),
"b": True,
"barr": np.array([True, False]),
"c16": np.complex128(1 + 2j),
"bytes": b"raw",
"sarr": np.array([b"a", b"bcd"]),
"str": "héllo",
"strs": ["x", "yz"],
"2d": np.arange(12, dtype="<u2").reshape(3, 4),
"empty": np.zeros((0,), dtype="<i4"),
}
with clawhdf5.File(path, "r+") as f:
for k, v in values.items():
f["g"].attrs[k] = v
# Replace one, with another type and size.
f["g"].attrs["i8"] = np.arange(100.0)
with h5py.File(path, "r") as f:
a = f["g"].attrs
np.testing.assert_array_equal(a["i8"], np.arange(100.0))
assert a["f8"] == 2.5 and a["f8"].dtype == np.float64
assert a["i1"] == -3 and a["i1"].dtype == np.int8
assert a["u8"] == 2**64 - 1 and a["u8"].dtype == np.uint64
assert a[">f4"].dtype == np.dtype(">f4")
assert a["f2"].dtype == np.float16
assert a["b"] is np.True_ or a["b"] == True # noqa: E712
assert a["barr"].dtype == np.bool_
assert a["c16"] == 1 + 2j
assert a["bytes"] == b"raw"
assert list(a["sarr"]) == [b"a", b"bcd"]
# str is stored as fixed-length UTF-8: h5py reads bytes.
assert a["str"].decode("utf-8") == "héllo"
assert [x.decode() for x in a["strs"]] == ["x", "yz"]
assert a["2d"].shape == (3, 4) and a["2d"].dtype == np.dtype("<u2")
assert a["empty"].shape == (0,)
with clawhdf5.File(path, "r") as f:
assert f["g"].attrs["c16"] == 1 + 2j
assert f["g"].attrs["str"].decode("utf-8") == "héllo"
h5dump_reads(path)
def test_many_attributes_move_to_dense_storage(h5py, tmp_path):
"""Past the compact limit (8 attributes under libver v114) the object's
attributes move to dense storage; h5py reads all of them."""
path = str(tmp_path / "dense.h5")
with h5py.File(path, "w", libver="v114") as f:
f.create_dataset("d", data=np.arange(3))
with clawhdf5.File(path, "r+") as f:
for i in range(20):
f["d"].attrs[f"a{i:02d}"] = np.full(i + 1, i, dtype="<i2")
with h5py.File(path, "r") as f:
assert sorted(f["d"].attrs.keys()) == [f"a{i:02d}" for i in range(20)]
for i in range(20):
np.testing.assert_array_equal(f["d"].attrs[f"a{i:02d}"], np.full(i + 1, i))
h5dump_reads(path)
def test_reads_never_see_a_half_written_edit(h5py, tmp_path):
"""Readers on other threads while one thread rewrites a dataset: every
read returns one whole version (all elements equal), never a mix."""
path = str(tmp_path / "race.h5")
with h5py.File(path, "w") as f:
f.create_dataset("d", data=np.zeros((64, 64)), chunks=(16, 16), compression="gzip")
errors = []
with clawhdf5.File(path, "r+") as f:
stop = threading.Event()
def read():
ds = f["d"]
while not stop.is_set():
a = ds[...]
if not (a == a.flat[0]).all():
errors.append(a)
return
readers = [threading.Thread(target=read) for _ in range(3)]
for t in readers:
t.start()
try:
for k in range(1, 25):
f["d"][...] = float(k)
finally:
stop.set()
for t in readers:
t.join()
np.testing.assert_array_equal(f["d"][...], np.full((64, 64), 24.0))
assert not errors, "a read saw a partly written dataset"
def test_close_releases_the_file(h5py, tmp_path):
path = str(tmp_path / "close.h5")
with h5py.File(path, "w") as f:
f.create_dataset("d", data=np.arange(4, dtype="<i4"))
f = clawhdf5.File(path, "r+")
ds = f["d"]
f.close()
# The handle still reads the file as last written, but cannot edit it.
np.testing.assert_array_equal(ds[...], np.arange(4))
with pytest.raises(OSError, match="closed"):
ds[0] = 1
with h5py.File(path, "r+") as g:
g["d"][0] = 9