FileEditor::resize (on main since PR #18, a4c2ace) scrambled the values of
a chunked dataset whose dataspace records no maximum dimensions when it
shrank it: clawhdf5's writer stores such a dataspace for every chunked
dataset created without a maxshape, the maximum is then the current
dimensions, and the Fixed Array index linearises chunks by the maximum, so
patching only the current dimensions moved every chunk after the first
row. h5py, h5dump and our reader all read the wrong values; the dataset
could not grow back either.
libhdf5 never writes such a dataspace (H5S_set_extent_simple records the
maximum, equal to the dimensions when none is given); reading one,
H5S_extent_get_dims reports the current dimensions as the maximum and
H5S_set_extent checks against none, so its own H5Dset_extent scrambles
such a file the same way. The editor now records the maximum libhdf5 would
have written (the dimensions the index was built with) before changing the
current ones, moving the grown dataspace message in the header when it
must. The writer records the maximum of every chunked dataset too, so
h5py can resize what clawhdf5 writes (the pinned file hashes of three
no-maxshape cases in plugin_filters_interop change by 8 bytes a dimension).
Tests: edit_resize_interop.rs (a 2.7.0-written fixture, new FileBuilder
files and h5py files through shrinks, zero extents and growth, against a
model with our reader and h5py; h5py resizing a FileBuilder file), and in
test_edit.py resizes checked against a numpy model, independently of h5py.
Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
819 lines
36 KiB
Python
819 lines
36 KiB
Python
"""In-place editing: clawhdf5.File(path, 'r+') against h5py.
|
|
|
|
Every edit is applied twice, to two copies of the same file: once through
|
|
h5py (libhdf5) and once through clawhdf5 (FileEditor). After every edit both
|
|
files are read back with h5py and must hold the same shapes, values and
|
|
attributes; clawhdf5's own view must agree; when h5py refuses an edit,
|
|
clawhdf5 must refuse it too and leave its file as it was. Files are written
|
|
by h5py (libver earliest and latest, so every chunk index kind) and by
|
|
clawhdf5; `h5dump` must read every result."""
|
|
|
|
import io
|
|
import os
|
|
import shutil
|
|
import subprocess
|
|
import threading
|
|
|
|
import numpy as np
|
|
import pytest
|
|
|
|
import clawhdf5
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Files
|
|
# ---------------------------------------------------------------------------
|
|
|
|
ENUM = {"RED": 0, "GREEN": 1, "BLUE": 7}
|
|
|
|
|
|
def _h5py_file(h5py, path, libver):
|
|
rng = np.random.default_rng(1)
|
|
with h5py.File(path, "w", libver=libver) as f:
|
|
f.create_dataset("i4", data=np.arange(60, dtype="<i4").reshape(6, 10))
|
|
f.create_dataset("be_i2", data=np.arange(24, dtype=">i2").reshape(4, 6))
|
|
f.create_dataset("u1", data=np.arange(16, dtype="u1"))
|
|
f.create_dataset("f2", data=rng.standard_normal(12).astype("<f2"))
|
|
f.create_dataset("f4_2d", data=rng.standard_normal((8, 9)).astype("<f4"))
|
|
f.create_dataset("f8_3d", data=rng.standard_normal((4, 5, 6)))
|
|
f.create_dataset("u8", data=np.arange(10, dtype="<u8"))
|
|
f.create_dataset("c8", data=(np.arange(6) + 1j * np.arange(6)).astype("<c8"))
|
|
f.create_dataset("bool", data=np.array([True, False, True, True]))
|
|
f.create_dataset("enum", data=np.array([0, 1, 7, 0], dtype="i1"),
|
|
dtype=h5py.enum_dtype(ENUM, basetype="i1"))
|
|
f.create_dataset("s5", data=np.array([b"ab", b"cdefg", b""], dtype="S5"))
|
|
cmp_dt = np.dtype([("id", "<i4"), ("x", "<f8"), ("tag", "S3")])
|
|
f.create_dataset("cmp", data=np.array([(i, i / 2, b"t%d" % i) for i in range(5)], dtype=cmp_dt))
|
|
f.create_dataset("scalar", data=np.float64(3.5))
|
|
# Chunked: fixed maxshape (v4 fixed array under latest), one
|
|
# unlimited dimension (extensible array), two (v2 B-tree), one chunk.
|
|
f.create_dataset("chunk_fixed", data=np.arange(100, dtype="<i8").reshape(10, 10), chunks=(3, 4))
|
|
f.create_dataset("chunk_ext", data=rng.standard_normal((12, 7)), chunks=(5, 7), maxshape=(None, 7))
|
|
f.create_dataset("chunk_bt2", data=np.arange(30, dtype="<i4").reshape(5, 6), chunks=(2, 2),
|
|
maxshape=(None, None))
|
|
f.create_dataset("chunk_gzip", data=np.arange(400, dtype="<f4").reshape(20, 20), chunks=(6, 6),
|
|
compression="gzip", maxshape=(40, 40))
|
|
f.create_dataset("chunk_single", data=np.arange(12, dtype="<u2").reshape(3, 4), chunks=(3, 4),
|
|
maxshape=(3, 4))
|
|
f.create_dataset("chunk_fill", shape=(8,), dtype="<i4", chunks=(3,), maxshape=(20,), fillvalue=-1)
|
|
f.create_dataset("vlen", data=["a", "bb"], dtype=h5py.string_dtype())
|
|
# Compact layout (low-level API).
|
|
dcpl = h5py.h5p.create(h5py.h5p.DATASET_CREATE)
|
|
dcpl.set_layout(h5py.h5d.COMPACT)
|
|
space = h5py.h5s.create_simple((7,))
|
|
dsid = h5py.h5d.create(f.id, b"compact", h5py.h5t.STD_I32LE, space, dcpl=dcpl)
|
|
dsid.write(h5py.h5s.ALL, h5py.h5s.ALL, np.arange(7, dtype="<i4"))
|
|
g = f.create_group("grp")
|
|
g.create_dataset("leaf", data=np.arange(5.0))
|
|
g.attrs["units"] = "m"
|
|
f.attrs["version"] = np.int32(1)
|
|
|
|
|
|
def _clawhdf5_file(path):
|
|
with clawhdf5.File(str(path), "w") as f:
|
|
f.create_dataset("i4", data=np.arange(60, dtype="<i4").reshape(6, 10))
|
|
f.create_dataset("f8", data=np.linspace(0, 1, 30).reshape(5, 6))
|
|
f.create_dataset("chunk_gzip", data=np.arange(400, dtype="<f4").reshape(20, 20),
|
|
chunks=[6, 6], compression="gzip")
|
|
f.create_dataset("u1", data=np.arange(16, dtype="u1"))
|
|
g = f.create_group("grp")
|
|
g.create_dataset("leaf", data=np.arange(5.0))
|
|
g.attrs["units"] = "m"
|
|
f.attrs["version"] = 1
|
|
|
|
|
|
# h5py's libver: "earliest" (v1 B-tree chunk indexes), "v114" (the 1.10+
|
|
# indexes: fixed and extensible arrays, v2 B-trees, single chunk) and
|
|
# "latest" (HDF5 2.0's newest format, which h5dump 1.14 cannot read).
|
|
SOURCES = ["h5py-earliest", "h5py-v114", "h5py-latest", "clawhdf5"]
|
|
|
|
|
|
def _make(h5py, tmp_path, source):
|
|
base = tmp_path / f"base-{source}.h5"
|
|
if source == "clawhdf5":
|
|
_clawhdf5_file(base)
|
|
else:
|
|
_h5py_file(h5py, str(base), source.split("-")[1])
|
|
theirs = tmp_path / f"theirs-{source}.h5"
|
|
ours = tmp_path / f"ours-{source}.h5"
|
|
shutil.copy(base, theirs)
|
|
shutil.copy(base, ours)
|
|
return str(theirs), str(ours), str(base)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Comparing files through h5py
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def _norm_attr(v):
|
|
"""Attribute values comparable across the two writers: clawhdf5 stores
|
|
`str` as fixed-length UTF-8 (h5py reads bytes), h5py as variable-length
|
|
(h5py reads str)."""
|
|
if isinstance(v, bytes):
|
|
return ("str", v.decode("utf-8"))
|
|
if isinstance(v, str):
|
|
return ("str", v)
|
|
arr = np.asarray(v)
|
|
if arr.dtype.kind in "SO":
|
|
return ("strs", [x.decode() if isinstance(x, bytes) else x for x in arr.ravel().tolist()], arr.shape)
|
|
return (arr.dtype.str, arr.shape, arr.tobytes())
|
|
|
|
|
|
def snapshot(h5py, path):
|
|
"""What h5py sees in the file: every dataset's shape, dtype, bytes and
|
|
attributes (read without locking: clawhdf5 may hold the file open)."""
|
|
out = {}
|
|
with h5py.File(path, "r", locking=False) as f:
|
|
def visit(name, obj):
|
|
attrs = {k: _norm_attr(obj.attrs[k]) for k in obj.attrs}
|
|
if isinstance(obj, h5py.Dataset):
|
|
if obj.dtype.kind == "O":
|
|
data = [x for x in obj[...].ravel().tolist()]
|
|
else:
|
|
data = obj[()].tobytes() if obj.shape is not None else None
|
|
out[name] = (obj.shape, obj.dtype.str, obj.maxshape, data, attrs)
|
|
else:
|
|
out[name] = ("group", attrs)
|
|
visit("/", f)
|
|
f.visititems(visit)
|
|
return out
|
|
|
|
|
|
def assert_same_files(h5py, theirs, ours, what):
|
|
a, b = snapshot(h5py, theirs), snapshot(h5py, ours)
|
|
assert a.keys() == b.keys(), what
|
|
for k in a:
|
|
assert a[k] == b[k], f"{what}: {k} differs\n h5py: {a[k]}\n clawhdf5: {b[k]}"
|
|
|
|
|
|
def assert_ours_reads_like_h5py(h5py, f, path, what):
|
|
"""clawhdf5's own view of the file it is editing matches h5py's."""
|
|
with h5py.File(path, "r", locking=False) as t:
|
|
for name in ["i4", "chunk_ext", "chunk_bt2", "chunk_gzip", "f8_3d", "cmp", "bool", "enum", "scalar"]:
|
|
if name not in t:
|
|
continue
|
|
o = f[name]
|
|
assert o.shape == t[name].shape, f"{what}: {name} shape"
|
|
assert o.maxshape == t[name].maxshape, f"{what}: {name} maxshape"
|
|
np.testing.assert_array_equal(o[()], t[name][()], err_msg=f"{what}: {name}")
|
|
for obj in ["/", "grp"]:
|
|
assert sorted(f[obj].attrs.keys()) == sorted(t[obj].attrs.keys()), what
|
|
for k in t[obj].attrs:
|
|
assert _norm_attr(f[obj].attrs[k]) == _norm_attr(t[obj].attrs[k]), f"{what}: {obj}.attrs[{k}]"
|
|
|
|
|
|
def h5dump_reads(path, base=None):
|
|
"""h5dump (libhdf5 1.14) reads every object and value of `path` — when it
|
|
reads the unedited `base` (it cannot read HDF5 2.0's newest format)."""
|
|
exe = shutil.which("h5dump")
|
|
if exe is None:
|
|
if os.environ.get("CLAWHDF5_REQUIRE_INTEROP") == "1":
|
|
pytest.fail("h5dump is required (CLAWHDF5_REQUIRE_INTEROP=1)")
|
|
return
|
|
h5rs = os.environ.get("CLAWHDF5_H5RS")
|
|
if h5rs:
|
|
# clawhdf5's structural and checksum validator (scripts/ci-test.sh
|
|
# points this at the h5rs it built).
|
|
r = subprocess.run([h5rs, "check", path], capture_output=True, text=True)
|
|
assert r.returncode == 0, (r.stdout + r.stderr)[-2000:]
|
|
if base is not None and subprocess.run([exe, "-H", base], capture_output=True).returncode != 0:
|
|
return
|
|
r = subprocess.run([exe, path], capture_output=True, text=True)
|
|
assert r.returncode == 0, r.stderr[-2000:]
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Applying one edit both ways
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def _native_conversion(h5py, value, ds_dtype):
|
|
"""`value` as libhdf5 converts it to `ds_dtype` in native byte order.
|
|
|
|
libhdf5 2.0 (h5py 3.16) converts numbers differently when either side is
|
|
not in native byte order (its "soft" conversions): a float in (-1, 0)
|
|
becomes the integer type's minimum instead of 0, and an unsigned integer
|
|
too large for the signed type of the same size wraps instead of
|
|
saturating. clawhdf5 applies the native-order results to every byte
|
|
order, so the reference is h5py converting into a native dataset of the
|
|
same kind; the result then reaches the real dataset by a plain byte
|
|
swap."""
|
|
if not isinstance(value, np.ndarray):
|
|
return value
|
|
if value.dtype.kind not in "biuf" or ds_dtype.kind not in "biuf":
|
|
return value
|
|
if value.dtype.isnative and ds_dtype.isnative:
|
|
return value
|
|
with h5py.File(io.BytesIO(), "w") as tmp:
|
|
d = tmp.create_dataset("t", shape=value.shape, dtype=ds_dtype.newbyteorder("="))
|
|
d[...] = value.astype(value.dtype.newbyteorder("="))
|
|
return np.asarray(d[()])
|
|
|
|
|
|
def _apply(f, op, h5py=None):
|
|
"""Apply `op` to `f`; with `h5py`, `f` is an h5py file and a numpy array
|
|
value is first converted as libhdf5 converts in native byte order (see
|
|
`_native_conversion`)."""
|
|
kind = op[0]
|
|
if kind == "set":
|
|
_, name, key, value = op
|
|
if h5py is not None:
|
|
value = _native_conversion(h5py, value, f[name].dtype)
|
|
f[name][key] = value
|
|
elif kind == "resize":
|
|
_, name, size, axis = op
|
|
if axis is None:
|
|
f[name].resize(size)
|
|
else:
|
|
f[name].resize(size, axis=axis)
|
|
elif kind == "attr":
|
|
_, obj, name, value = op
|
|
f[obj].attrs[name] = value
|
|
else:
|
|
raise AssertionError(op)
|
|
|
|
|
|
def edit_both(h5py, theirs, ours_path, ours, op):
|
|
"""Apply `op` with h5py and with clawhdf5 (`ours`, open 'r+'); the two
|
|
files must then read the same through h5py. If h5py refuses, clawhdf5
|
|
must refuse and its file must be unchanged. Returns h5py's error."""
|
|
before = snapshot(h5py, ours_path)
|
|
try:
|
|
with h5py.File(theirs, "r+") as t:
|
|
_apply(t, op, h5py)
|
|
except Exception as e: # noqa: BLE001 - h5py refuses: so must we
|
|
try:
|
|
_apply(ours, op)
|
|
except Exception: # noqa: BLE001
|
|
pass
|
|
else:
|
|
pytest.fail(f"{op!r:.300}: h5py refused ({type(e).__name__}: {e}), clawhdf5 did not")
|
|
assert snapshot(h5py, ours_path) == before, f"{op!r}: clawhdf5 changed the file while failing"
|
|
return e
|
|
_apply(ours, op)
|
|
assert_same_files(h5py, theirs, ours_path, repr(op)[:200])
|
|
return None
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Tests
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
@pytest.mark.parametrize("source", SOURCES)
|
|
def test_edit_sequence_matches_h5py(h5py, tmp_path, source):
|
|
theirs, ours_path, base = _make(h5py, tmp_path, source)
|
|
ops = [
|
|
("set", "i4", 0, 99),
|
|
("set", "i4", (slice(1, 5, 2), slice(None, None, 3)), np.array([[1.5, -2.5, 1e12, -1e12]])),
|
|
("set", "i4", (slice(None), 2), np.arange(6, dtype="<i8") * 1000),
|
|
("set", "i4", ([0, 2, 5], slice(4, 6)), np.array([[1, 2], [3, 4], [5, 6]])),
|
|
("set", "i4", (Ellipsis, -1), np.int8(-7)),
|
|
("set", "i4", (3, 3), 12345.9),
|
|
("set", "u1", slice(2, 8), np.array([-5, 0, 300, 255, 256, 1], dtype="<i4")),
|
|
("set", "u1", slice(0, 3), [1, 2, 3]),
|
|
("set", "chunk_gzip", (slice(0, 20, 7), slice(3, 17)), 42.25),
|
|
("set", "chunk_gzip", (slice(5, 11), slice(5, 11)), np.ones((6, 6), dtype="<f8") * np.pi),
|
|
("set", "grp/leaf", slice(None), np.array([1, 2, 3, 4, 5], dtype="<u2")),
|
|
("attr", "/", "version", np.int32(2)),
|
|
("attr", "/", "count", 7),
|
|
("attr", "grp", "scale", np.array([0.5, 0.25], dtype="<f4")),
|
|
("attr", "grp", "matrix", np.arange(6, dtype=">i8").reshape(2, 3)),
|
|
("attr", "i4", "flag", np.bool_(True)),
|
|
("attr", "i4", "z", np.complex64(1 - 2j)),
|
|
("attr", "i4", "raw", np.bytes_(b"abc")),
|
|
("set", "i4", slice(0, 2), np.zeros((3, 10))), # shape mismatch: refused by both
|
|
]
|
|
if source != "clawhdf5":
|
|
ops += [
|
|
("set", "be_i2", (slice(None), slice(1, 3)), np.array([70000, -70000], dtype="<i8")),
|
|
("set", "f2", slice(None, None, 4), np.array([1e6, -3.25, 0.1])),
|
|
("set", "f4_2d", (2, slice(None)), np.linspace(-1, 1, 9)),
|
|
("set", "f8_3d", (slice(1, 3), 2, slice(None, None, 2)), np.arange(3, dtype="<i2")),
|
|
("set", "u8", slice(None), np.array([-1, 0, 2**63, 1e30, -1e30, 5.5, 2, 3, 4, 5])),
|
|
("set", "c8", slice(1, 3), np.array([1 + 1j, 2 - 2j], dtype="<c16")),
|
|
("set", "c8", 0, np.float64(1.0)), # h5py: no conversion path
|
|
("set", "bool", slice(None), np.array([0, 3, 0, -1], dtype="<i4")),
|
|
("set", "bool", 1, np.array(True)),
|
|
("set", "enum", slice(0, 2), np.array([7, 1], dtype="<i4")),
|
|
("set", "s5", 0, np.bytes_(b"xyzuvw")),
|
|
("set", "s5", slice(1, 3), [b"q", b"rs"]),
|
|
("set", "s5", 2, np.array("uni")), # h5py: no conversion from 'U'
|
|
("set", "cmp", 2, np.array((9, 9.5, b"zz"), dtype=[("id", "<i4"), ("x", "<f8"), ("tag", "S3")])),
|
|
("set", "cmp", slice(3, 5), [(1, 0.5, b"a"), (2, 1.5, b"b")]),
|
|
("set", "scalar", (), 7.25),
|
|
("set", "scalar", Ellipsis, np.float32(-1.5)),
|
|
("set", "compact", slice(1, 6, 2), np.array([10, 20, 30])),
|
|
("set", "chunk_fixed", (slice(2, 9), slice(1, 10, 4)), np.arange(21).reshape(7, 3)),
|
|
("set", "chunk_single", (1, slice(None)), np.array([9, 8, 7, 6])),
|
|
("resize", "chunk_ext", (20, 7), None),
|
|
("set", "chunk_ext", slice(12, 20), np.full((8, 7), 2.5)),
|
|
("resize", "chunk_ext", 9, 0),
|
|
("resize", "chunk_ext", 16, 0),
|
|
("resize", "chunk_bt2", (9, 11), None),
|
|
("set", "chunk_bt2", (slice(4, 9), slice(5, 11)), np.arange(30).reshape(5, 6)),
|
|
("resize", "chunk_bt2", (3, 3), None),
|
|
("resize", "chunk_bt2", (7, 8), None),
|
|
("resize", "chunk_gzip", (40, 25), None),
|
|
("resize", "chunk_gzip", (41, 25), None), # beyond maxshape: refused
|
|
("resize", "chunk_fill", 15, None), # h5py: a size without axis must be a tuple
|
|
("resize", "chunk_fill", (15,), None),
|
|
("set", "chunk_fill", slice(10, 12), [1, 2]),
|
|
("resize", "chunk_fill", (4,), None),
|
|
("resize", "chunk_fill", (20,), None),
|
|
("resize", "i4", (7, 10), None), # not chunked: refused
|
|
]
|
|
with clawhdf5.File(ours_path, "r+") as ours:
|
|
assert ours.mode == "r+"
|
|
for i, op in enumerate(ops):
|
|
edit_both(h5py, theirs, ours_path, ours, op)
|
|
if i % 5 == 0:
|
|
assert_ours_reads_like_h5py(h5py, ours, ours_path, repr(op))
|
|
assert_ours_reads_like_h5py(h5py, ours, ours_path, "end")
|
|
h5dump_reads(ours_path, base)
|
|
# Reopened, clawhdf5 reads what h5py reads.
|
|
with clawhdf5.File(ours_path, "r") as f:
|
|
assert f.mode == "r"
|
|
assert_ours_reads_like_h5py(h5py, f, ours_path, "reopened")
|
|
|
|
|
|
def _random_key(rng, shape):
|
|
key = []
|
|
for n in shape:
|
|
r = rng.random()
|
|
if n == 0 or r < 0.3:
|
|
start = int(rng.integers(0, n + 1)) if n else 0
|
|
stop = int(rng.integers(start, n + 1)) if n else 0
|
|
step = int(rng.integers(1, 4))
|
|
key.append(slice(start, stop, step))
|
|
elif r < 0.55:
|
|
key.append(int(rng.integers(-n, n)))
|
|
elif r < 0.7:
|
|
k = int(rng.integers(1, min(n, 4) + 1))
|
|
key.append(sorted(rng.choice(n, size=k, replace=False).tolist()))
|
|
else:
|
|
key.append(slice(None))
|
|
# Only one index list per key.
|
|
lists = [i for i, k in enumerate(key) if isinstance(k, list)]
|
|
for i in lists[1:]:
|
|
key[i] = slice(None)
|
|
return tuple(key)
|
|
|
|
|
|
def _selection_shape(key, shape):
|
|
out = []
|
|
fancy = False
|
|
for k, n in zip(key, shape):
|
|
if isinstance(k, slice):
|
|
out.append(len(range(*k.indices(n))))
|
|
elif isinstance(k, list):
|
|
out.append(len(k))
|
|
fancy = True
|
|
return tuple(out), fancy
|
|
|
|
|
|
def _random_value(rng, sel_shape, fancy, dtype):
|
|
r = rng.random()
|
|
if r < 0.2 or not sel_shape:
|
|
v = rng.standard_normal() * 1000
|
|
return np.float64(v) if rng.random() < 0.5 else int(v)
|
|
shape = list(sel_shape)
|
|
if not fancy and r < 0.35 and shape:
|
|
shape[0] = 1 # broadcast along the first axis
|
|
kind = rng.choice(["same", "f8", "i8", "u1", "f4"])
|
|
if kind == "same" and np.dtype(dtype).kind in "iuf":
|
|
dt = np.dtype(dtype)
|
|
else:
|
|
dt = np.dtype(str(kind) if kind != "same" else "f8")
|
|
base = rng.standard_normal(size=shape) * (10 ** rng.integers(0, 6))
|
|
with np.errstate(all="ignore"):
|
|
return base.astype(dt)
|
|
|
|
|
|
@pytest.mark.parametrize("source", SOURCES)
|
|
@pytest.mark.parametrize("seed", [0, 1, 2, 3])
|
|
def test_random_edits_match_h5py(h5py, tmp_path, source, seed):
|
|
"""Random writes (slices, steps, integers, index lists, broadcasts,
|
|
other dtypes and out-of-range values), resizes and attributes, each
|
|
compared with h5py applying the same edit."""
|
|
rng = np.random.default_rng(1000 * seed + SOURCES.index(source))
|
|
theirs, ours_path, base = _make(h5py, tmp_path, source)
|
|
with h5py.File(theirs, "r") as t:
|
|
names = [n for n in ["i4", "u1", "f8", "f4_2d", "f8_3d", "be_i2", "chunk_fixed", "chunk_ext",
|
|
"chunk_bt2", "chunk_gzip", "chunk_fill", "compact", "grp/leaf"] if n in t]
|
|
resizable = [n for n in ["chunk_ext", "chunk_bt2", "chunk_gzip", "chunk_fill"] if n in names]
|
|
refused = 0
|
|
with clawhdf5.File(ours_path, "r+") as ours:
|
|
for step in range(40):
|
|
r = rng.random()
|
|
if r < 0.15 and resizable:
|
|
name = str(rng.choice(resizable))
|
|
with h5py.File(theirs, "r") as t:
|
|
maxshape = t[name].maxshape
|
|
shape = t[name].shape
|
|
new = tuple(int(rng.integers(0, (m if m is not None else s + 10) + 1)) for m, s in zip(maxshape, shape))
|
|
op = ("resize", name, new, None)
|
|
elif r < 0.25:
|
|
obj = str(rng.choice(["/", "grp", names[0]]))
|
|
choices = [np.int16(rng.integers(-100, 100)), rng.standard_normal(3),
|
|
np.arange(int(rng.integers(1, 5)), dtype=">u4"), np.float32(0.5)]
|
|
value = choices[int(rng.integers(0, len(choices)))]
|
|
op = ("attr", obj, f"a{int(rng.integers(0, 4))}", value)
|
|
else:
|
|
name = str(rng.choice(names))
|
|
with h5py.File(theirs, "r") as t:
|
|
shape, dtype = t[name].shape, t[name].dtype
|
|
key = _random_key(rng, shape)
|
|
sel_shape, fancy = _selection_shape(key, shape)
|
|
op = ("set", name, key, _random_value(rng, sel_shape, fancy, dtype))
|
|
before = None
|
|
if op[0] == "resize":
|
|
with h5py.File(ours_path, "r", locking=False) as o:
|
|
before, fill = o[op[1]][()], o[op[1]].fillvalue
|
|
if edit_both(h5py, theirs, ours_path, ours, op) is not None:
|
|
refused += 1
|
|
elif before is not None:
|
|
# Independently of h5py (which a wrong index layout fools
|
|
# the same way): the kept elements keep their values.
|
|
want = resized_model(before, op[2], fill)
|
|
np.testing.assert_array_equal(ours[op[1]][()], want, err_msg=repr(op))
|
|
with h5py.File(ours_path, "r", locking=False) as o:
|
|
np.testing.assert_array_equal(o[op[1]][()], want, err_msg=repr(op))
|
|
assert_ours_reads_like_h5py(h5py, ours, ours_path, "end")
|
|
assert refused < 30
|
|
h5dump_reads(ours_path, base)
|
|
|
|
|
|
def resized_model(before, shape, fill):
|
|
"""`before` resized to `shape` as HDF5 resizes: elements inside both
|
|
extents keep their values, the others read as the fill value."""
|
|
out = np.full(shape, fill, dtype=before.dtype)
|
|
common = tuple(slice(0, min(a, b)) for a, b in zip(before.shape, shape))
|
|
out[common] = before[common]
|
|
return out
|
|
|
|
|
|
RESIZES = [(15, 15), (3, 2), (20, 20), (1, 1), (1, 0), (0, 0), (7, 20), (20, 13), (20, 20)]
|
|
|
|
|
|
def _check_resizes(h5py, path, name, orig, maxshape=(20, 20), base=None):
|
|
"""Resize `name` through RESIZES in 'r+', each checked against a numpy
|
|
model with clawhdf5 and h5py and the file with `h5rs check`."""
|
|
model = orig
|
|
with clawhdf5.File(path, "r+") as f:
|
|
ds = f[name]
|
|
for shape in RESIZES:
|
|
ds.resize(shape)
|
|
model = resized_model(model, shape, 0)
|
|
np.testing.assert_array_equal(ds[()], model, err_msg=f"{name} {shape}")
|
|
with h5py.File(path, "r", locking=False) as t:
|
|
np.testing.assert_array_equal(t[name][()], model, err_msg=f"h5py: {name} {shape}")
|
|
assert t[name].maxshape == maxshape
|
|
h5rs = os.environ.get("CLAWHDF5_H5RS")
|
|
if h5rs: # h5dump would wait for the editor's lock
|
|
r = subprocess.run([h5rs, "check", "--data", path], capture_output=True, text=True)
|
|
assert r.returncode == 0, f"{name} {shape}: " + (r.stdout + r.stderr)[-2000:]
|
|
with pytest.raises(ValueError):
|
|
ds.resize((maxshape[0] + 1, 20))
|
|
# Written values survive a shrink.
|
|
ds[...] = orig
|
|
ds.resize((15, 15))
|
|
with h5py.File(path, "r") as t:
|
|
np.testing.assert_array_equal(t[name][()], orig[:15, :15])
|
|
h5dump_reads(path, base)
|
|
|
|
|
|
@pytest.mark.parametrize("source", SOURCES)
|
|
def test_resizes_keep_values(h5py, tmp_path, source):
|
|
"""Shrinking, zero extents and growing back keep the values a numpy
|
|
model keeps, on every source (on clawhdf5's own files a shrink once
|
|
moved every chunk: h5py read the same wrong values)."""
|
|
_, ours, base = _make(h5py, tmp_path, source)
|
|
with clawhdf5.File(ours, "r+") as f:
|
|
f["chunk_gzip"].resize((20, 20))
|
|
maxshape = (20, 20) if source == "clawhdf5" else (40, 40)
|
|
_check_resizes(h5py, ours, "chunk_gzip", np.arange(400, dtype="<f4").reshape(20, 20), maxshape, base)
|
|
|
|
|
|
FIXTURES = os.path.join(os.path.dirname(__file__), "..", "..", "clawhdf5", "tests", "fixtures")
|
|
|
|
|
|
@pytest.mark.parametrize("name", ["d", "z"])
|
|
def test_resizes_of_a_file_with_no_recorded_maxshape(h5py, tmp_path, name):
|
|
"""A file clawhdf5 2.7.0 wrote records no maximum dimensions for chunked
|
|
datasets; resizing it must keep the Fixed Array index's layout."""
|
|
path = str(tmp_path / "old.h5")
|
|
shutil.copy(os.path.join(FIXTURES, "chunked_no_maxshape_v2_7_0.h5"), path)
|
|
with h5py.File(path, "r") as t:
|
|
orig = t[name][()]
|
|
_check_resizes(h5py, path, name, orig)
|
|
|
|
|
|
NUMERIC = ["<i1", "<u1", "<i2", ">u2", "<i4", "<u4", ">i8", "<u8", "<f2", "<f4", ">f8"]
|
|
|
|
|
|
def _quiet(f):
|
|
with np.errstate(all="ignore"):
|
|
return f()
|
|
|
|
|
|
def _libhdf5_undefined(vals, target):
|
|
"""The values whose conversion to `target` libhdf5 2.0 (h5py 3.16) gets
|
|
wrong even in native byte order, where its C casts are undefined
|
|
behaviour; clawhdf5 saturates them as libhdf5's range handling
|
|
intends (docs/known-issues.md):
|
|
|
|
- half floats into unsigned integers: negatives wrap (-1 -> 65535) and
|
|
+inf becomes 0; into signed integers, +-inf becomes the minimum;
|
|
- a float equal to the integer maximum rounded up in the float's
|
|
precision (float32(2**31 - 1) == 2**31 -> int32, float64(2**64 - 1)
|
|
-> uint64) becomes the minimum (or 0);
|
|
- a double between 65504 and 65520 into a half float becomes infinity
|
|
(IEEE rounds it down to 65504, as numpy does).
|
|
"""
|
|
bad = np.zeros(vals.shape, dtype=bool)
|
|
if vals.dtype.kind != "f":
|
|
return bad
|
|
if target.kind in "iu":
|
|
if vals.dtype.itemsize == 2:
|
|
bad |= np.isinf(vals)
|
|
if target.kind == "u":
|
|
bad |= vals <= -1
|
|
top = _quiet(lambda: np.array(np.iinfo(target).max).astype(vals.dtype))
|
|
if float(top) > np.iinfo(target).max:
|
|
bad |= vals == top
|
|
if target.kind == "f" and target.itemsize == 2 and vals.dtype.itemsize > 2:
|
|
bad |= (np.abs(vals) > 65504) & (np.abs(vals) < 65520)
|
|
return bad
|
|
|
|
|
|
def test_numeric_conversions_match_h5py(h5py, tmp_path):
|
|
"""Every numeric source dtype into every numeric dataset dtype, with
|
|
values at and beyond the targets' limits, as libhdf5 converts them."""
|
|
edge = np.array([0, 1, -1, -0.3, 2.5, -2.5, 3.7, -3.7, 127.9, -128.9, 200.5, 255.5, 256, -129,
|
|
32767.5, 40000, 65504, 70000, -70000, 2**31 - 1, 2**31, -2**31 - 1,
|
|
4e9, 1e15, -1e15, 1e19, 1e300, -1e300, np.inf, -np.inf])
|
|
sources = {
|
|
"f8": edge,
|
|
"f4": _quiet(lambda: edge.astype("<f4")),
|
|
"f2": np.array([0, 1, -1, 2.5, -3.5, 65504, -65504, np.inf, -np.inf, 100.5], dtype="<f2"),
|
|
"i8": np.array([0, 1, -1, 127, 128, -129, 255, 256, 32768, -32769, 65536, 2**31, -2**31 - 1,
|
|
2**32, 2**62, -2**63, 2**63 - 1], dtype="<i8"),
|
|
"u8": np.array([0, 1, 127, 128, 255, 256, 65535, 65536, 2**31, 2**32, 2**63, 2**64 - 1], dtype="<u8"),
|
|
"i1": np.array([-128, -1, 0, 1, 127], dtype="i1"),
|
|
"u2": np.array([0, 255, 256, 65535], dtype=">u2"),
|
|
"b": np.array([True, False, True]),
|
|
}
|
|
for target in NUMERIC:
|
|
for sname, src in sources.items():
|
|
vals = src[~_libhdf5_undefined(src, np.dtype(target))]
|
|
path_t = str(tmp_path / f"t_{target[1:]}_{sname}.h5")
|
|
path_o = str(tmp_path / f"o_{target[1:]}_{sname}.h5")
|
|
with h5py.File(path_t, "w") as f:
|
|
f.create_dataset("d", shape=vals.shape, dtype=target)
|
|
shutil.copy(path_t, path_o)
|
|
what = f"{vals.dtype} -> {target}"
|
|
try:
|
|
with h5py.File(path_t, "r+") as f:
|
|
f["d"][...] = _native_conversion(h5py, vals, f["d"].dtype)
|
|
except Exception: # noqa: BLE001
|
|
with clawhdf5.File(path_o, "r+") as f, pytest.raises(Exception):
|
|
f["d"][...] = vals
|
|
continue
|
|
with clawhdf5.File(path_o, "r+") as f:
|
|
f["d"][...] = vals
|
|
with h5py.File(path_t, "r") as a, h5py.File(path_o, "r") as b:
|
|
assert a["d"][...].tobytes() == b["d"][...].tobytes(), (
|
|
f"{what}: h5py {a['d'][...].tolist()} clawhdf5 {b['d'][...].tolist()}"
|
|
)
|
|
|
|
|
|
def test_nan_into_an_integer_dataset_is_refused(h5py, tmp_path):
|
|
"""libhdf5 stores NaN as an arbitrary integer (0, the minimum or 2**63,
|
|
depending on the type); clawhdf5 refuses and writes nothing."""
|
|
path = str(tmp_path / "nan.h5")
|
|
with h5py.File(path, "w") as f:
|
|
f.create_dataset("d", data=np.arange(4, dtype="<i4"))
|
|
with clawhdf5.File(path, "r+") as f:
|
|
with pytest.raises(ValueError, match="NaN"):
|
|
f["d"][...] = np.array([1.0, np.nan, 2.0, 3.0])
|
|
# A Python list goes through numpy, which refuses NaN too.
|
|
with pytest.raises(ValueError):
|
|
f["d"][0:2] = [np.nan, 1.0]
|
|
with h5py.File(path, "r") as f:
|
|
np.testing.assert_array_equal(f["d"][...], np.arange(4))
|
|
|
|
|
|
def test_unsupported_edits_are_clear_errors(h5py, tmp_path):
|
|
path = str(tmp_path / "u.h5")
|
|
_h5py_file(h5py, path, "earliest")
|
|
before = snapshot(h5py, path)
|
|
with clawhdf5.File(path, "r+") as f:
|
|
with pytest.raises(NotImplementedError, match="delet"):
|
|
del f.attrs["version"]
|
|
with pytest.raises(NotImplementedError, match="delet"):
|
|
del f["i4"]
|
|
with pytest.raises(NotImplementedError, match="delet"):
|
|
del f["grp"]["leaf"]
|
|
with pytest.raises(NotImplementedError):
|
|
f.create_dataset("new", data=np.arange(3.0))
|
|
with pytest.raises(NotImplementedError):
|
|
f.create_group("newgrp")
|
|
with pytest.raises(NotImplementedError):
|
|
f["grp"].create_dataset("new", data=np.arange(3.0))
|
|
with pytest.raises(NotImplementedError, match="variable-length"):
|
|
f["vlen"][0] = "x"
|
|
with pytest.raises(NotImplementedError, match="field"):
|
|
f["cmp"]["id"] = np.arange(5)
|
|
with pytest.raises(NotImplementedError):
|
|
f.attrs["empty"] = clawhdf5.Empty("f8")
|
|
with pytest.raises(TypeError, match="chunked"):
|
|
f["i4"].resize((7, 10))
|
|
with pytest.raises(ValueError):
|
|
f["chunk_gzip"].resize((41, 20))
|
|
with pytest.raises(ValueError, match="axis"):
|
|
f["chunk_ext"].resize(3, axis=2)
|
|
with pytest.raises(TypeError):
|
|
f["i4"][0] = np.array(["a"] * 10)
|
|
assert snapshot(h5py, path) == before
|
|
h5dump_reads(path)
|
|
|
|
|
|
def test_read_only_files_and_modes(h5py, tmp_path):
|
|
path = str(tmp_path / "m.h5")
|
|
with h5py.File(path, "w") as f:
|
|
f.create_dataset("d", data=np.arange(4, dtype="<i4"), chunks=(2,), maxshape=(None,))
|
|
with clawhdf5.File(path, "r") as f:
|
|
with pytest.raises(OSError, match="r\\+"):
|
|
f["d"][0] = 1
|
|
with pytest.raises(OSError):
|
|
f["d"].resize((8,))
|
|
with pytest.raises(OSError):
|
|
f.attrs["x"] = 1
|
|
with pytest.raises(NotImplementedError, match="does not exist"):
|
|
clawhdf5.File(str(tmp_path / "missing.h5"), "a")
|
|
with pytest.raises(ValueError, match="mode"):
|
|
clawhdf5.File(path, "rw")
|
|
with clawhdf5.File(path, "a") as f:
|
|
assert f.mode == "r+"
|
|
f["d"][1] = 10
|
|
with h5py.File(path, "r") as f:
|
|
assert f["d"][1] == 10
|
|
|
|
|
|
def test_the_file_is_locked_while_open_for_editing(h5py, tmp_path):
|
|
path = str(tmp_path / "lock.h5")
|
|
with h5py.File(path, "w") as f:
|
|
f.create_dataset("d", data=np.arange(4, dtype="<i4"))
|
|
f = clawhdf5.File(path, "r+")
|
|
with pytest.raises(OSError):
|
|
clawhdf5.File(path, "r+")
|
|
with pytest.raises(OSError):
|
|
h5py.File(path, "r+")
|
|
f.close()
|
|
with h5py.File(path, "r+") as g:
|
|
g["d"][0] = 5
|
|
with clawhdf5.File(path, "r+") as g:
|
|
g["d"][1] = 6
|
|
with h5py.File(path, "r") as g:
|
|
np.testing.assert_array_equal(g["d"][...], [5, 6, 2, 3])
|
|
|
|
|
|
def test_objects_see_edits_made_through_others(h5py, tmp_path):
|
|
"""A dataset or attrs object taken before an edit reports the file as
|
|
it is after it: the new shape, the new attribute."""
|
|
path = str(tmp_path / "live.h5")
|
|
with h5py.File(path, "w") as f:
|
|
f.create_dataset("d", data=np.arange(6.0), chunks=(4,), maxshape=(None,))
|
|
with clawhdf5.File(path, "r+") as f:
|
|
d1 = f["d"]
|
|
d2 = f["d"]
|
|
attrs = d1.attrs
|
|
assert len(attrs) == 0 and "u" not in attrs
|
|
d2.resize((10,))
|
|
assert d1.shape == (10,) and d1.size == 10 and len(d1) == 10
|
|
np.testing.assert_array_equal(d1[6:], np.zeros(4))
|
|
d2.attrs["u"] = "m/s"
|
|
assert "u" in attrs and attrs["u"] == b"m/s" and len(attrs) == 1
|
|
f.attrs.create("shaped", np.arange(6), shape=(2, 3), dtype="<i2")
|
|
f.attrs.modify("shaped2", [1.5, 2.5])
|
|
with h5py.File(path, "r") as f:
|
|
assert f["d"].shape == (10,)
|
|
assert f["d"].attrs["u"] == b"m/s"
|
|
assert f.attrs["shaped"].dtype == np.dtype("<i2") and f.attrs["shaped"].shape == (2, 3)
|
|
np.testing.assert_array_equal(f.attrs["shaped2"], [1.5, 2.5])
|
|
|
|
|
|
def test_attribute_types_as_h5py_reads_them(h5py, tmp_path):
|
|
path = str(tmp_path / "attrs.h5")
|
|
with h5py.File(path, "w") as f:
|
|
f.create_group("g")
|
|
values = {
|
|
"i8": 5,
|
|
"f8": 2.5,
|
|
"i1": np.int8(-3),
|
|
"u8": np.uint64(2**64 - 1),
|
|
">f4": np.array([1.5, 2.5], dtype=">f4"),
|
|
"f2": np.float16(0.5),
|
|
"b": True,
|
|
"barr": np.array([True, False]),
|
|
"c16": np.complex128(1 + 2j),
|
|
"bytes": b"raw",
|
|
"sarr": np.array([b"a", b"bcd"]),
|
|
"str": "héllo",
|
|
"strs": ["x", "yz"],
|
|
"2d": np.arange(12, dtype="<u2").reshape(3, 4),
|
|
"empty": np.zeros((0,), dtype="<i4"),
|
|
}
|
|
with clawhdf5.File(path, "r+") as f:
|
|
for k, v in values.items():
|
|
f["g"].attrs[k] = v
|
|
# Replace one, with another type and size.
|
|
f["g"].attrs["i8"] = np.arange(100.0)
|
|
with h5py.File(path, "r") as f:
|
|
a = f["g"].attrs
|
|
np.testing.assert_array_equal(a["i8"], np.arange(100.0))
|
|
assert a["f8"] == 2.5 and a["f8"].dtype == np.float64
|
|
assert a["i1"] == -3 and a["i1"].dtype == np.int8
|
|
assert a["u8"] == 2**64 - 1 and a["u8"].dtype == np.uint64
|
|
assert a[">f4"].dtype == np.dtype(">f4")
|
|
assert a["f2"].dtype == np.float16
|
|
assert a["b"] is np.True_ or a["b"] == True # noqa: E712
|
|
assert a["barr"].dtype == np.bool_
|
|
assert a["c16"] == 1 + 2j
|
|
assert a["bytes"] == b"raw"
|
|
assert list(a["sarr"]) == [b"a", b"bcd"]
|
|
# str is stored as fixed-length UTF-8: h5py reads bytes.
|
|
assert a["str"].decode("utf-8") == "héllo"
|
|
assert [x.decode() for x in a["strs"]] == ["x", "yz"]
|
|
assert a["2d"].shape == (3, 4) and a["2d"].dtype == np.dtype("<u2")
|
|
assert a["empty"].shape == (0,)
|
|
with clawhdf5.File(path, "r") as f:
|
|
assert f["g"].attrs["c16"] == 1 + 2j
|
|
assert f["g"].attrs["str"].decode("utf-8") == "héllo"
|
|
h5dump_reads(path)
|
|
|
|
|
|
def test_many_attributes_move_to_dense_storage(h5py, tmp_path):
|
|
"""Past the compact limit (8 attributes under libver v114) the object's
|
|
attributes move to dense storage; h5py reads all of them."""
|
|
path = str(tmp_path / "dense.h5")
|
|
with h5py.File(path, "w", libver="v114") as f:
|
|
f.create_dataset("d", data=np.arange(3))
|
|
with clawhdf5.File(path, "r+") as f:
|
|
for i in range(20):
|
|
f["d"].attrs[f"a{i:02d}"] = np.full(i + 1, i, dtype="<i2")
|
|
with h5py.File(path, "r") as f:
|
|
assert sorted(f["d"].attrs.keys()) == [f"a{i:02d}" for i in range(20)]
|
|
for i in range(20):
|
|
np.testing.assert_array_equal(f["d"].attrs[f"a{i:02d}"], np.full(i + 1, i))
|
|
h5dump_reads(path)
|
|
|
|
|
|
def test_reads_never_see_a_half_written_edit(h5py, tmp_path):
|
|
"""Readers on other threads while one thread rewrites a dataset: every
|
|
read returns one whole version (all elements equal), never a mix."""
|
|
path = str(tmp_path / "race.h5")
|
|
with h5py.File(path, "w") as f:
|
|
f.create_dataset("d", data=np.zeros((64, 64)), chunks=(16, 16), compression="gzip")
|
|
errors = []
|
|
with clawhdf5.File(path, "r+") as f:
|
|
stop = threading.Event()
|
|
|
|
def read():
|
|
ds = f["d"]
|
|
while not stop.is_set():
|
|
a = ds[...]
|
|
if not (a == a.flat[0]).all():
|
|
errors.append(a)
|
|
return
|
|
|
|
readers = [threading.Thread(target=read) for _ in range(3)]
|
|
for t in readers:
|
|
t.start()
|
|
try:
|
|
for k in range(1, 25):
|
|
f["d"][...] = float(k)
|
|
finally:
|
|
stop.set()
|
|
for t in readers:
|
|
t.join()
|
|
np.testing.assert_array_equal(f["d"][...], np.full((64, 64), 24.0))
|
|
assert not errors, "a read saw a partly written dataset"
|
|
|
|
|
|
def test_close_releases_the_file(h5py, tmp_path):
|
|
path = str(tmp_path / "close.h5")
|
|
with h5py.File(path, "w") as f:
|
|
f.create_dataset("d", data=np.arange(4, dtype="<i4"))
|
|
f = clawhdf5.File(path, "r+")
|
|
ds = f["d"]
|
|
f.close()
|
|
# The handle still reads the file as last written, but cannot edit it.
|
|
np.testing.assert_array_equal(ds[...], np.arange(4))
|
|
with pytest.raises(OSError, match="closed"):
|
|
ds[0] = 1
|
|
with h5py.File(path, "r+") as g:
|
|
g["d"][0] = 9
|