A thread appending timestamps in a loop runs while a 1024x1024 gzip dataset is rewritten through 'r+': the largest gap between its stamps during the edit must be under half the edit's duration (an edit holding the GIL stalls it for the whole edit; checked with a GIL-holding regex standing in for the edit: one 0.20 s gap in a 0.21 s call). Edits already ran detached; nothing tested it. Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
934 lines
41 KiB
Python
934 lines
41 KiB
Python
"""In-place editing: clawhdf5.File(path, 'r+') against h5py.
|
|
|
|
Every edit is applied twice, to two copies of the same file: once through
|
|
h5py (libhdf5) and once through clawhdf5 (FileEditor). After every edit both
|
|
files are read back with h5py and must hold the same shapes, values and
|
|
attributes; clawhdf5's own view must agree; when h5py refuses an edit,
|
|
clawhdf5 must refuse it too and leave its file as it was. Files are written
|
|
by h5py (libver earliest and latest, so every chunk index kind) and by
|
|
clawhdf5; `h5dump` must read every result."""
|
|
|
|
import io
|
|
import os
|
|
import shutil
|
|
import subprocess
|
|
import threading
|
|
|
|
import numpy as np
|
|
import pytest
|
|
|
|
import clawhdf5
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Files
|
|
# ---------------------------------------------------------------------------
|
|
|
|
ENUM = {"RED": 0, "GREEN": 1, "BLUE": 7}
|
|
|
|
|
|
def _h5py_file(h5py, path, libver):
|
|
rng = np.random.default_rng(1)
|
|
with h5py.File(path, "w", libver=libver) as f:
|
|
f.create_dataset("i4", data=np.arange(60, dtype="<i4").reshape(6, 10))
|
|
f.create_dataset("be_i2", data=np.arange(24, dtype=">i2").reshape(4, 6))
|
|
f.create_dataset("u1", data=np.arange(16, dtype="u1"))
|
|
f.create_dataset("f2", data=rng.standard_normal(12).astype("<f2"))
|
|
f.create_dataset("f4_2d", data=rng.standard_normal((8, 9)).astype("<f4"))
|
|
f.create_dataset("f8_3d", data=rng.standard_normal((4, 5, 6)))
|
|
f.create_dataset("u8", data=np.arange(10, dtype="<u8"))
|
|
f.create_dataset("c8", data=(np.arange(6) + 1j * np.arange(6)).astype("<c8"))
|
|
f.create_dataset("bool", data=np.array([True, False, True, True]))
|
|
f.create_dataset("enum", data=np.array([0, 1, 7, 0], dtype="i1"),
|
|
dtype=h5py.enum_dtype(ENUM, basetype="i1"))
|
|
f.create_dataset("s5", data=np.array([b"ab", b"cdefg", b""], dtype="S5"))
|
|
cmp_dt = np.dtype([("id", "<i4"), ("x", "<f8"), ("tag", "S3")])
|
|
f.create_dataset("cmp", data=np.array([(i, i / 2, b"t%d" % i) for i in range(5)], dtype=cmp_dt))
|
|
f.create_dataset("scalar", data=np.float64(3.5))
|
|
# Chunked: fixed maxshape (v4 fixed array under latest), one
|
|
# unlimited dimension (extensible array), two (v2 B-tree), one chunk.
|
|
f.create_dataset("chunk_fixed", data=np.arange(100, dtype="<i8").reshape(10, 10), chunks=(3, 4))
|
|
f.create_dataset("chunk_ext", data=rng.standard_normal((12, 7)), chunks=(5, 7), maxshape=(None, 7))
|
|
f.create_dataset("chunk_bt2", data=np.arange(30, dtype="<i4").reshape(5, 6), chunks=(2, 2),
|
|
maxshape=(None, None))
|
|
f.create_dataset("chunk_gzip", data=np.arange(400, dtype="<f4").reshape(20, 20), chunks=(6, 6),
|
|
compression="gzip", maxshape=(40, 40))
|
|
f.create_dataset("chunk_single", data=np.arange(12, dtype="<u2").reshape(3, 4), chunks=(3, 4),
|
|
maxshape=(3, 4))
|
|
f.create_dataset("chunk_fill", shape=(8,), dtype="<i4", chunks=(3,), maxshape=(20,), fillvalue=-1)
|
|
f.create_dataset("vlen", data=["a", "bb"], dtype=h5py.string_dtype())
|
|
# Compact layout (low-level API).
|
|
dcpl = h5py.h5p.create(h5py.h5p.DATASET_CREATE)
|
|
dcpl.set_layout(h5py.h5d.COMPACT)
|
|
space = h5py.h5s.create_simple((7,))
|
|
dsid = h5py.h5d.create(f.id, b"compact", h5py.h5t.STD_I32LE, space, dcpl=dcpl)
|
|
dsid.write(h5py.h5s.ALL, h5py.h5s.ALL, np.arange(7, dtype="<i4"))
|
|
g = f.create_group("grp")
|
|
g.create_dataset("leaf", data=np.arange(5.0))
|
|
g.attrs["units"] = "m"
|
|
f.attrs["version"] = np.int32(1)
|
|
|
|
|
|
def _clawhdf5_file(path):
|
|
with clawhdf5.File(str(path), "w") as f:
|
|
f.create_dataset("i4", data=np.arange(60, dtype="<i4").reshape(6, 10))
|
|
f.create_dataset("f8", data=np.linspace(0, 1, 30).reshape(5, 6))
|
|
f.create_dataset("chunk_gzip", data=np.arange(400, dtype="<f4").reshape(20, 20),
|
|
chunks=[6, 6], compression="gzip")
|
|
f.create_dataset("u1", data=np.arange(16, dtype="u1"))
|
|
g = f.create_group("grp")
|
|
g.create_dataset("leaf", data=np.arange(5.0))
|
|
g.attrs["units"] = "m"
|
|
f.attrs["version"] = 1
|
|
|
|
|
|
# h5py's libver: "earliest" (v1 B-tree chunk indexes), "v114" (the 1.10+
|
|
# indexes: fixed and extensible arrays, v2 B-trees, single chunk) and
|
|
# "latest" (HDF5 2.0's newest format, which h5dump 1.14 cannot read).
|
|
SOURCES = ["h5py-earliest", "h5py-v114", "h5py-latest", "clawhdf5"]
|
|
|
|
|
|
def _make(h5py, tmp_path, source):
|
|
base = tmp_path / f"base-{source}.h5"
|
|
if source == "clawhdf5":
|
|
_clawhdf5_file(base)
|
|
else:
|
|
_h5py_file(h5py, str(base), source.split("-")[1])
|
|
theirs = tmp_path / f"theirs-{source}.h5"
|
|
ours = tmp_path / f"ours-{source}.h5"
|
|
shutil.copy(base, theirs)
|
|
shutil.copy(base, ours)
|
|
return str(theirs), str(ours), str(base)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Comparing files through h5py
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def _norm_attr(v):
|
|
"""Attribute values comparable across the two writers: clawhdf5 stores
|
|
`str` as fixed-length UTF-8 (h5py reads bytes), h5py as variable-length
|
|
(h5py reads str)."""
|
|
if isinstance(v, bytes):
|
|
return ("str", v.decode("utf-8"))
|
|
if isinstance(v, str):
|
|
return ("str", v)
|
|
arr = np.asarray(v)
|
|
if arr.dtype.kind in "SO":
|
|
return ("strs", [x.decode() if isinstance(x, bytes) else x for x in arr.ravel().tolist()], arr.shape)
|
|
return (arr.dtype.str, arr.shape, arr.tobytes())
|
|
|
|
|
|
def snapshot(h5py, path):
|
|
"""What h5py sees in the file: every dataset's shape, dtype, bytes and
|
|
attributes (read without locking: clawhdf5 may hold the file open)."""
|
|
out = {}
|
|
with h5py.File(path, "r", locking=False) as f:
|
|
def visit(name, obj):
|
|
attrs = {k: _norm_attr(obj.attrs[k]) for k in obj.attrs}
|
|
if isinstance(obj, h5py.Dataset):
|
|
if obj.dtype.kind == "O":
|
|
data = [x for x in obj[...].ravel().tolist()]
|
|
else:
|
|
data = obj[()].tobytes() if obj.shape is not None else None
|
|
out[name] = (obj.shape, obj.dtype.str, obj.maxshape, data, attrs)
|
|
else:
|
|
out[name] = ("group", attrs)
|
|
visit("/", f)
|
|
f.visititems(visit)
|
|
return out
|
|
|
|
|
|
def assert_same_files(h5py, theirs, ours, what):
|
|
a, b = snapshot(h5py, theirs), snapshot(h5py, ours)
|
|
assert a.keys() == b.keys(), what
|
|
for k in a:
|
|
assert a[k] == b[k], f"{what}: {k} differs\n h5py: {a[k]}\n clawhdf5: {b[k]}"
|
|
|
|
|
|
def assert_ours_reads_like_h5py(h5py, f, path, what):
|
|
"""clawhdf5's own view of the file it is editing matches h5py's."""
|
|
with h5py.File(path, "r", locking=False) as t:
|
|
for name in ["i4", "chunk_ext", "chunk_bt2", "chunk_gzip", "f8_3d", "cmp", "bool", "enum", "scalar"]:
|
|
if name not in t:
|
|
continue
|
|
o = f[name]
|
|
assert o.shape == t[name].shape, f"{what}: {name} shape"
|
|
assert o.maxshape == t[name].maxshape, f"{what}: {name} maxshape"
|
|
np.testing.assert_array_equal(o[()], t[name][()], err_msg=f"{what}: {name}")
|
|
for obj in ["/", "grp"]:
|
|
assert sorted(f[obj].attrs.keys()) == sorted(t[obj].attrs.keys()), what
|
|
for k in t[obj].attrs:
|
|
assert _norm_attr(f[obj].attrs[k]) == _norm_attr(t[obj].attrs[k]), f"{what}: {obj}.attrs[{k}]"
|
|
|
|
|
|
def h5dump_reads(path, base=None):
|
|
"""h5dump (libhdf5 1.14) reads every object and value of `path` — when it
|
|
reads the unedited `base` (it cannot read HDF5 2.0's newest format)."""
|
|
exe = shutil.which("h5dump")
|
|
if exe is None:
|
|
if os.environ.get("CLAWHDF5_REQUIRE_INTEROP") == "1":
|
|
pytest.fail("h5dump is required (CLAWHDF5_REQUIRE_INTEROP=1)")
|
|
return
|
|
h5rs = os.environ.get("CLAWHDF5_H5RS")
|
|
if h5rs:
|
|
# clawhdf5's structural and checksum validator (scripts/ci-test.sh
|
|
# points this at the h5rs it built).
|
|
r = subprocess.run([h5rs, "check", path], capture_output=True, text=True)
|
|
assert r.returncode == 0, (r.stdout + r.stderr)[-2000:]
|
|
if base is not None and subprocess.run([exe, "-H", base], capture_output=True).returncode != 0:
|
|
return
|
|
r = subprocess.run([exe, path], capture_output=True, text=True)
|
|
assert r.returncode == 0, r.stderr[-2000:]
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Applying one edit both ways
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def _native_conversion(h5py, value, ds_dtype):
|
|
"""`value` as libhdf5 converts it to `ds_dtype` in native byte order.
|
|
|
|
libhdf5 2.0 (h5py 3.16) converts numbers differently when either side is
|
|
not in native byte order (its "soft" conversions): a float in (-1, 0)
|
|
becomes the integer type's minimum instead of 0, and an unsigned integer
|
|
too large for the signed type of the same size wraps instead of
|
|
saturating. clawhdf5 applies the native-order results to every byte
|
|
order, so the reference is h5py converting into a native dataset of the
|
|
same kind; the result then reaches the real dataset by a plain byte
|
|
swap."""
|
|
if not isinstance(value, np.ndarray):
|
|
return value
|
|
if value.dtype.kind not in "biuf" or ds_dtype.kind not in "biuf":
|
|
return value
|
|
if value.dtype.isnative and ds_dtype.isnative:
|
|
return value
|
|
with h5py.File(io.BytesIO(), "w") as tmp:
|
|
d = tmp.create_dataset("t", shape=value.shape, dtype=ds_dtype.newbyteorder("="))
|
|
d[...] = value.astype(value.dtype.newbyteorder("="))
|
|
return np.asarray(d[()])
|
|
|
|
|
|
def _apply(f, op, h5py=None):
|
|
"""Apply `op` to `f`; with `h5py`, `f` is an h5py file and a numpy array
|
|
value is first converted as libhdf5 converts in native byte order (see
|
|
`_native_conversion`)."""
|
|
kind = op[0]
|
|
if kind == "set":
|
|
_, name, key, value = op
|
|
if h5py is not None:
|
|
value = _native_conversion(h5py, value, f[name].dtype)
|
|
f[name][key] = value
|
|
elif kind == "resize":
|
|
_, name, size, axis = op
|
|
if axis is None:
|
|
f[name].resize(size)
|
|
else:
|
|
f[name].resize(size, axis=axis)
|
|
elif kind == "attr":
|
|
_, obj, name, value = op
|
|
f[obj].attrs[name] = value
|
|
else:
|
|
raise AssertionError(op)
|
|
|
|
|
|
def edit_both(h5py, theirs, ours_path, ours, op):
|
|
"""Apply `op` with h5py and with clawhdf5 (`ours`, open 'r+'); the two
|
|
files must then read the same through h5py. If h5py refuses, clawhdf5
|
|
must refuse and its file must be unchanged. Returns h5py's error."""
|
|
before = snapshot(h5py, ours_path)
|
|
try:
|
|
with h5py.File(theirs, "r+") as t:
|
|
_apply(t, op, h5py)
|
|
except Exception as e: # noqa: BLE001 - h5py refuses: so must we
|
|
try:
|
|
_apply(ours, op)
|
|
except Exception: # noqa: BLE001
|
|
pass
|
|
else:
|
|
pytest.fail(f"{op!r:.300}: h5py refused ({type(e).__name__}: {e}), clawhdf5 did not")
|
|
assert snapshot(h5py, ours_path) == before, f"{op!r}: clawhdf5 changed the file while failing"
|
|
return e
|
|
_apply(ours, op)
|
|
assert_same_files(h5py, theirs, ours_path, repr(op)[:200])
|
|
return None
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Tests
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
@pytest.mark.parametrize("source", SOURCES)
|
|
def test_edit_sequence_matches_h5py(h5py, tmp_path, source):
|
|
theirs, ours_path, base = _make(h5py, tmp_path, source)
|
|
ops = [
|
|
("set", "i4", 0, 99),
|
|
("set", "i4", (slice(1, 5, 2), slice(None, None, 3)), np.array([[1.5, -2.5, 1e12, -1e12]])),
|
|
("set", "i4", (slice(None), 2), np.arange(6, dtype="<i8") * 1000),
|
|
("set", "i4", ([0, 2, 5], slice(4, 6)), np.array([[1, 2], [3, 4], [5, 6]])),
|
|
("set", "i4", (Ellipsis, -1), np.int8(-7)),
|
|
("set", "i4", (3, 3), 12345.9),
|
|
("set", "u1", slice(2, 8), np.array([-5, 0, 300, 255, 256, 1], dtype="<i4")),
|
|
("set", "u1", slice(0, 3), [1, 2, 3]),
|
|
("set", "chunk_gzip", (slice(0, 20, 7), slice(3, 17)), 42.25),
|
|
("set", "chunk_gzip", (slice(5, 11), slice(5, 11)), np.ones((6, 6), dtype="<f8") * np.pi),
|
|
("set", "grp/leaf", slice(None), np.array([1, 2, 3, 4, 5], dtype="<u2")),
|
|
("attr", "/", "version", np.int32(2)),
|
|
("attr", "/", "count", 7),
|
|
("attr", "grp", "scale", np.array([0.5, 0.25], dtype="<f4")),
|
|
("attr", "grp", "matrix", np.arange(6, dtype=">i8").reshape(2, 3)),
|
|
("attr", "i4", "flag", np.bool_(True)),
|
|
("attr", "i4", "z", np.complex64(1 - 2j)),
|
|
("attr", "i4", "raw", np.bytes_(b"abc")),
|
|
("set", "i4", slice(0, 2), np.zeros((3, 10))), # shape mismatch: refused by both
|
|
]
|
|
if source != "clawhdf5":
|
|
ops += [
|
|
("set", "be_i2", (slice(None), slice(1, 3)), np.array([70000, -70000], dtype="<i8")),
|
|
("set", "f2", slice(None, None, 4), np.array([1e6, -3.25, 0.1])),
|
|
("set", "f4_2d", (2, slice(None)), np.linspace(-1, 1, 9)),
|
|
("set", "f8_3d", (slice(1, 3), 2, slice(None, None, 2)), np.arange(3, dtype="<i2")),
|
|
("set", "u8", slice(None), np.array([-1, 0, 2**63, 1e30, -1e30, 5.5, 2, 3, 4, 5])),
|
|
("set", "c8", slice(1, 3), np.array([1 + 1j, 2 - 2j], dtype="<c16")),
|
|
("set", "c8", 0, np.float64(1.0)), # h5py: no conversion path
|
|
("set", "bool", slice(None), np.array([0, 3, 0, -1], dtype="<i4")),
|
|
("set", "bool", 1, np.array(True)),
|
|
("set", "enum", slice(0, 2), np.array([7, 1], dtype="<i4")),
|
|
("set", "s5", 0, np.bytes_(b"xyzuvw")),
|
|
("set", "s5", slice(1, 3), [b"q", b"rs"]),
|
|
("set", "s5", 2, np.array("uni")), # h5py: no conversion from 'U'
|
|
("set", "cmp", 2, np.array((9, 9.5, b"zz"), dtype=[("id", "<i4"), ("x", "<f8"), ("tag", "S3")])),
|
|
("set", "cmp", slice(3, 5), [(1, 0.5, b"a"), (2, 1.5, b"b")]),
|
|
("set", "scalar", (), 7.25),
|
|
("set", "scalar", Ellipsis, np.float32(-1.5)),
|
|
("set", "compact", slice(1, 6, 2), np.array([10, 20, 30])),
|
|
("set", "chunk_fixed", (slice(2, 9), slice(1, 10, 4)), np.arange(21).reshape(7, 3)),
|
|
("set", "chunk_single", (1, slice(None)), np.array([9, 8, 7, 6])),
|
|
("resize", "chunk_ext", (20, 7), None),
|
|
("set", "chunk_ext", slice(12, 20), np.full((8, 7), 2.5)),
|
|
("resize", "chunk_ext", 9, 0),
|
|
("resize", "chunk_ext", 16, 0),
|
|
("resize", "chunk_bt2", (9, 11), None),
|
|
("set", "chunk_bt2", (slice(4, 9), slice(5, 11)), np.arange(30).reshape(5, 6)),
|
|
("resize", "chunk_bt2", (3, 3), None),
|
|
("resize", "chunk_bt2", (7, 8), None),
|
|
("resize", "chunk_gzip", (40, 25), None),
|
|
("resize", "chunk_gzip", (41, 25), None), # beyond maxshape: refused
|
|
("resize", "chunk_fill", 15, None), # h5py: a size without axis must be a tuple
|
|
("resize", "chunk_fill", (15,), None),
|
|
("set", "chunk_fill", slice(10, 12), [1, 2]),
|
|
("resize", "chunk_fill", (4,), None),
|
|
("resize", "chunk_fill", (20,), None),
|
|
("resize", "i4", (7, 10), None), # not chunked: refused
|
|
]
|
|
with clawhdf5.File(ours_path, "r+") as ours:
|
|
assert ours.mode == "r+"
|
|
for i, op in enumerate(ops):
|
|
edit_both(h5py, theirs, ours_path, ours, op)
|
|
if i % 5 == 0:
|
|
assert_ours_reads_like_h5py(h5py, ours, ours_path, repr(op))
|
|
assert_ours_reads_like_h5py(h5py, ours, ours_path, "end")
|
|
h5dump_reads(ours_path, base)
|
|
# Reopened, clawhdf5 reads what h5py reads.
|
|
with clawhdf5.File(ours_path, "r") as f:
|
|
assert f.mode == "r"
|
|
assert_ours_reads_like_h5py(h5py, f, ours_path, "reopened")
|
|
|
|
|
|
def _random_key(rng, shape):
|
|
key = []
|
|
for n in shape:
|
|
r = rng.random()
|
|
if n == 0 or r < 0.3:
|
|
start = int(rng.integers(0, n + 1)) if n else 0
|
|
stop = int(rng.integers(start, n + 1)) if n else 0
|
|
step = int(rng.integers(1, 4))
|
|
key.append(slice(start, stop, step))
|
|
elif r < 0.55:
|
|
key.append(int(rng.integers(-n, n)))
|
|
elif r < 0.7:
|
|
k = int(rng.integers(1, min(n, 4) + 1))
|
|
key.append(sorted(rng.choice(n, size=k, replace=False).tolist()))
|
|
else:
|
|
key.append(slice(None))
|
|
# Only one index list per key.
|
|
lists = [i for i, k in enumerate(key) if isinstance(k, list)]
|
|
for i in lists[1:]:
|
|
key[i] = slice(None)
|
|
return tuple(key)
|
|
|
|
|
|
def _selection_shape(key, shape):
|
|
out = []
|
|
fancy = False
|
|
for k, n in zip(key, shape):
|
|
if isinstance(k, slice):
|
|
out.append(len(range(*k.indices(n))))
|
|
elif isinstance(k, list):
|
|
out.append(len(k))
|
|
fancy = True
|
|
return tuple(out), fancy
|
|
|
|
|
|
def _random_value(rng, sel_shape, fancy, dtype):
|
|
r = rng.random()
|
|
if r < 0.2 or not sel_shape:
|
|
v = rng.standard_normal() * 1000
|
|
return np.float64(v) if rng.random() < 0.5 else int(v)
|
|
shape = list(sel_shape)
|
|
if not fancy and r < 0.35 and shape:
|
|
shape[0] = 1 # broadcast along the first axis
|
|
kind = rng.choice(["same", "f8", "i8", "u1", "f4"])
|
|
if kind == "same" and np.dtype(dtype).kind in "iuf":
|
|
dt = np.dtype(dtype)
|
|
else:
|
|
dt = np.dtype(str(kind) if kind != "same" else "f8")
|
|
base = rng.standard_normal(size=shape) * (10 ** rng.integers(0, 6))
|
|
with np.errstate(all="ignore"):
|
|
return base.astype(dt)
|
|
|
|
|
|
@pytest.mark.parametrize("source", SOURCES)
|
|
@pytest.mark.parametrize("seed", [0, 1, 2, 3])
|
|
def test_random_edits_match_h5py(h5py, tmp_path, source, seed):
|
|
"""Random writes (slices, steps, integers, index lists, broadcasts,
|
|
other dtypes and out-of-range values), resizes and attributes, each
|
|
compared with h5py applying the same edit."""
|
|
rng = np.random.default_rng(1000 * seed + SOURCES.index(source))
|
|
theirs, ours_path, base = _make(h5py, tmp_path, source)
|
|
with h5py.File(theirs, "r") as t:
|
|
names = [n for n in ["i4", "u1", "f8", "f4_2d", "f8_3d", "be_i2", "chunk_fixed", "chunk_ext",
|
|
"chunk_bt2", "chunk_gzip", "chunk_fill", "compact", "grp/leaf"] if n in t]
|
|
resizable = [n for n in ["chunk_ext", "chunk_bt2", "chunk_gzip", "chunk_fill"] if n in names]
|
|
refused = 0
|
|
with clawhdf5.File(ours_path, "r+") as ours:
|
|
for step in range(40):
|
|
r = rng.random()
|
|
if r < 0.15 and resizable:
|
|
name = str(rng.choice(resizable))
|
|
with h5py.File(theirs, "r") as t:
|
|
maxshape = t[name].maxshape
|
|
shape = t[name].shape
|
|
new = tuple(int(rng.integers(0, (m if m is not None else s + 10) + 1)) for m, s in zip(maxshape, shape))
|
|
op = ("resize", name, new, None)
|
|
elif r < 0.25:
|
|
obj = str(rng.choice(["/", "grp", names[0]]))
|
|
choices = [np.int16(rng.integers(-100, 100)), rng.standard_normal(3),
|
|
np.arange(int(rng.integers(1, 5)), dtype=">u4"), np.float32(0.5)]
|
|
value = choices[int(rng.integers(0, len(choices)))]
|
|
op = ("attr", obj, f"a{int(rng.integers(0, 4))}", value)
|
|
else:
|
|
name = str(rng.choice(names))
|
|
with h5py.File(theirs, "r") as t:
|
|
shape, dtype = t[name].shape, t[name].dtype
|
|
key = _random_key(rng, shape)
|
|
sel_shape, fancy = _selection_shape(key, shape)
|
|
op = ("set", name, key, _random_value(rng, sel_shape, fancy, dtype))
|
|
before = None
|
|
if op[0] == "resize":
|
|
with h5py.File(ours_path, "r", locking=False) as o:
|
|
before, fill = o[op[1]][()], o[op[1]].fillvalue
|
|
if edit_both(h5py, theirs, ours_path, ours, op) is not None:
|
|
refused += 1
|
|
elif before is not None:
|
|
# Independently of h5py (which a wrong index layout fools
|
|
# the same way): the kept elements keep their values.
|
|
want = resized_model(before, op[2], fill)
|
|
np.testing.assert_array_equal(ours[op[1]][()], want, err_msg=repr(op))
|
|
with h5py.File(ours_path, "r", locking=False) as o:
|
|
np.testing.assert_array_equal(o[op[1]][()], want, err_msg=repr(op))
|
|
assert_ours_reads_like_h5py(h5py, ours, ours_path, "end")
|
|
assert refused < 30
|
|
h5dump_reads(ours_path, base)
|
|
|
|
|
|
@pytest.mark.parametrize("seed", range(10, 40))
|
|
def test_random_edits_on_clawhdf5_files(h5py, tmp_path, seed):
|
|
"""More random sequences on a clawhdf5-written file, whose resizes to
|
|
zero extents once left files `h5rs check` could not read."""
|
|
test_random_edits_match_h5py(h5py, tmp_path, "clawhdf5", seed)
|
|
|
|
|
|
def resized_model(before, shape, fill):
|
|
"""`before` resized to `shape` as HDF5 resizes: elements inside both
|
|
extents keep their values, the others read as the fill value."""
|
|
out = np.full(shape, fill, dtype=before.dtype)
|
|
common = tuple(slice(0, min(a, b)) for a, b in zip(before.shape, shape))
|
|
out[common] = before[common]
|
|
return out
|
|
|
|
|
|
RESIZES = [(15, 15), (3, 2), (20, 20), (1, 1), (1, 0), (0, 0), (7, 20), (20, 13), (20, 20)]
|
|
|
|
|
|
def _check_resizes(h5py, path, name, orig, maxshape=(20, 20), base=None):
|
|
"""Resize `name` through RESIZES in 'r+', each checked against a numpy
|
|
model with clawhdf5 and h5py and the file with `h5rs check`."""
|
|
model = orig
|
|
with clawhdf5.File(path, "r+") as f:
|
|
ds = f[name]
|
|
for shape in RESIZES:
|
|
ds.resize(shape)
|
|
model = resized_model(model, shape, 0)
|
|
np.testing.assert_array_equal(ds[()], model, err_msg=f"{name} {shape}")
|
|
with h5py.File(path, "r", locking=False) as t:
|
|
np.testing.assert_array_equal(t[name][()], model, err_msg=f"h5py: {name} {shape}")
|
|
assert t[name].maxshape == maxshape
|
|
h5rs = os.environ.get("CLAWHDF5_H5RS")
|
|
if h5rs: # h5dump would wait for the editor's lock
|
|
r = subprocess.run([h5rs, "check", "--data", path], capture_output=True, text=True)
|
|
assert r.returncode == 0, f"{name} {shape}: " + (r.stdout + r.stderr)[-2000:]
|
|
with pytest.raises(ValueError):
|
|
ds.resize((maxshape[0] + 1, 20))
|
|
# Written values survive a shrink.
|
|
ds[...] = orig
|
|
ds.resize((15, 15))
|
|
with h5py.File(path, "r") as t:
|
|
np.testing.assert_array_equal(t[name][()], orig[:15, :15])
|
|
h5dump_reads(path, base)
|
|
|
|
|
|
@pytest.mark.parametrize("source", SOURCES)
|
|
def test_resizes_keep_values(h5py, tmp_path, source):
|
|
"""Shrinking, zero extents and growing back keep the values a numpy
|
|
model keeps, on every source (on clawhdf5's own files a shrink once
|
|
moved every chunk: h5py read the same wrong values)."""
|
|
_, ours, base = _make(h5py, tmp_path, source)
|
|
with clawhdf5.File(ours, "r+") as f:
|
|
f["chunk_gzip"].resize((20, 20))
|
|
maxshape = (20, 20) if source == "clawhdf5" else (40, 40)
|
|
_check_resizes(h5py, ours, "chunk_gzip", np.arange(400, dtype="<f4").reshape(20, 20), maxshape, base)
|
|
|
|
|
|
FIXTURES = os.path.join(os.path.dirname(__file__), "..", "..", "clawhdf5", "tests", "fixtures")
|
|
|
|
|
|
@pytest.mark.parametrize("name", ["d", "z"])
|
|
def test_resizes_of_a_file_with_no_recorded_maxshape(h5py, tmp_path, name):
|
|
"""A file clawhdf5 2.7.0 wrote records no maximum dimensions for chunked
|
|
datasets; resizing it must keep the Fixed Array index's layout."""
|
|
path = str(tmp_path / "old.h5")
|
|
shutil.copy(os.path.join(FIXTURES, "chunked_no_maxshape_v2_7_0.h5"), path)
|
|
with h5py.File(path, "r") as t:
|
|
orig = t[name][()]
|
|
_check_resizes(h5py, path, name, orig)
|
|
|
|
|
|
NUMERIC = ["<i1", "<u1", "<i2", ">u2", "<i4", "<u4", ">i8", "<u8", "<f2", "<f4", ">f8"]
|
|
|
|
|
|
def _quiet(f):
|
|
with np.errstate(all="ignore"):
|
|
return f()
|
|
|
|
|
|
def _libhdf5_undefined(vals, target):
|
|
"""The values whose conversion to `target` libhdf5 2.0 (h5py 3.16) gets
|
|
wrong even in native byte order, where its C casts are undefined
|
|
behaviour; clawhdf5 saturates them as libhdf5's range handling
|
|
intends (docs/known-issues.md):
|
|
|
|
- half floats into unsigned integers: negatives wrap (-1 -> 65535) and
|
|
+inf becomes 0; into signed integers, +-inf becomes the minimum;
|
|
- a float equal to the integer maximum rounded up in the float's
|
|
precision (float32(2**31 - 1) == 2**31 -> int32, float64(2**64 - 1)
|
|
-> uint64) becomes the minimum (or 0);
|
|
- a double between 65504 and 65520 into a half float becomes infinity
|
|
(IEEE rounds it down to 65504, as numpy does).
|
|
"""
|
|
bad = np.zeros(vals.shape, dtype=bool)
|
|
if vals.dtype.kind != "f":
|
|
return bad
|
|
if target.kind in "iu":
|
|
if vals.dtype.itemsize == 2:
|
|
bad |= np.isinf(vals)
|
|
if target.kind == "u":
|
|
bad |= vals <= -1
|
|
top = _quiet(lambda: np.array(np.iinfo(target).max).astype(vals.dtype))
|
|
if float(top) > np.iinfo(target).max:
|
|
bad |= vals == top
|
|
if target.kind == "f" and target.itemsize == 2 and vals.dtype.itemsize > 2:
|
|
bad |= (np.abs(vals) > 65504) & (np.abs(vals) < 65520)
|
|
return bad
|
|
|
|
|
|
def test_numeric_conversions_match_h5py(h5py, tmp_path):
|
|
"""Every numeric source dtype into every numeric dataset dtype, with
|
|
values at and beyond the targets' limits, as libhdf5 converts them."""
|
|
edge = np.array([0, 1, -1, -0.3, 2.5, -2.5, 3.7, -3.7, 127.9, -128.9, 200.5, 255.5, 256, -129,
|
|
32767.5, 40000, 65504, 70000, -70000, 2**31 - 1, 2**31, -2**31 - 1,
|
|
4e9, 1e15, -1e15, 1e19, 1e300, -1e300, np.inf, -np.inf])
|
|
sources = {
|
|
"f8": edge,
|
|
"f4": _quiet(lambda: edge.astype("<f4")),
|
|
"f2": np.array([0, 1, -1, 2.5, -3.5, 65504, -65504, np.inf, -np.inf, 100.5], dtype="<f2"),
|
|
"i8": np.array([0, 1, -1, 127, 128, -129, 255, 256, 32768, -32769, 65536, 2**31, -2**31 - 1,
|
|
2**32, 2**62, -2**63, 2**63 - 1], dtype="<i8"),
|
|
"u8": np.array([0, 1, 127, 128, 255, 256, 65535, 65536, 2**31, 2**32, 2**63, 2**64 - 1], dtype="<u8"),
|
|
"i1": np.array([-128, -1, 0, 1, 127], dtype="i1"),
|
|
"u2": np.array([0, 255, 256, 65535], dtype=">u2"),
|
|
"b": np.array([True, False, True]),
|
|
}
|
|
for target in NUMERIC:
|
|
for sname, src in sources.items():
|
|
vals = src[~_libhdf5_undefined(src, np.dtype(target))]
|
|
path_t = str(tmp_path / f"t_{target[1:]}_{sname}.h5")
|
|
path_o = str(tmp_path / f"o_{target[1:]}_{sname}.h5")
|
|
with h5py.File(path_t, "w") as f:
|
|
f.create_dataset("d", shape=vals.shape, dtype=target)
|
|
shutil.copy(path_t, path_o)
|
|
what = f"{vals.dtype} -> {target}"
|
|
try:
|
|
with h5py.File(path_t, "r+") as f:
|
|
f["d"][...] = _native_conversion(h5py, vals, f["d"].dtype)
|
|
except Exception: # noqa: BLE001
|
|
with clawhdf5.File(path_o, "r+") as f, pytest.raises(Exception):
|
|
f["d"][...] = vals
|
|
continue
|
|
with clawhdf5.File(path_o, "r+") as f:
|
|
f["d"][...] = vals
|
|
with h5py.File(path_t, "r") as a, h5py.File(path_o, "r") as b:
|
|
assert a["d"][...].tobytes() == b["d"][...].tobytes(), (
|
|
f"{what}: h5py {a['d'][...].tolist()} clawhdf5 {b['d'][...].tolist()}"
|
|
)
|
|
|
|
|
|
def test_nan_into_an_integer_dataset_is_refused(h5py, tmp_path):
|
|
"""libhdf5 stores NaN as an arbitrary integer (0, the minimum or 2**63,
|
|
depending on the type); clawhdf5 refuses and writes nothing."""
|
|
path = str(tmp_path / "nan.h5")
|
|
with h5py.File(path, "w") as f:
|
|
f.create_dataset("d", data=np.arange(4, dtype="<i4"))
|
|
with clawhdf5.File(path, "r+") as f:
|
|
with pytest.raises(ValueError, match="NaN"):
|
|
f["d"][...] = np.array([1.0, np.nan, 2.0, 3.0])
|
|
# A Python list goes through numpy, which refuses NaN too.
|
|
with pytest.raises(ValueError):
|
|
f["d"][0:2] = [np.nan, 1.0]
|
|
with h5py.File(path, "r") as f:
|
|
np.testing.assert_array_equal(f["d"][...], np.arange(4))
|
|
|
|
|
|
def test_unsupported_edits_are_clear_errors(h5py, tmp_path):
|
|
path = str(tmp_path / "u.h5")
|
|
_h5py_file(h5py, path, "earliest")
|
|
before = snapshot(h5py, path)
|
|
with clawhdf5.File(path, "r+") as f:
|
|
with pytest.raises(NotImplementedError, match="delet"):
|
|
del f.attrs["version"]
|
|
with pytest.raises(NotImplementedError, match="delet"):
|
|
del f["i4"]
|
|
with pytest.raises(NotImplementedError, match="delet"):
|
|
del f["grp"]["leaf"]
|
|
with pytest.raises(NotImplementedError):
|
|
f.create_dataset("new", data=np.arange(3.0))
|
|
with pytest.raises(NotImplementedError):
|
|
f.create_group("newgrp")
|
|
with pytest.raises(NotImplementedError):
|
|
f["grp"].create_dataset("new", data=np.arange(3.0))
|
|
with pytest.raises(NotImplementedError, match="variable-length"):
|
|
f["vlen"][0] = "x"
|
|
with pytest.raises(NotImplementedError, match="field"):
|
|
f["cmp"]["id"] = np.arange(5)
|
|
with pytest.raises(NotImplementedError):
|
|
f.attrs["empty"] = clawhdf5.Empty("f8")
|
|
with pytest.raises(TypeError, match="chunked"):
|
|
f["i4"].resize((7, 10))
|
|
with pytest.raises(ValueError):
|
|
f["chunk_gzip"].resize((41, 20))
|
|
with pytest.raises(ValueError, match="axis"):
|
|
f["chunk_ext"].resize(3, axis=2)
|
|
with pytest.raises(TypeError):
|
|
f["i4"][0] = np.array(["a"] * 10)
|
|
assert snapshot(h5py, path) == before
|
|
h5dump_reads(path)
|
|
|
|
|
|
def test_read_only_files_and_modes(h5py, tmp_path):
|
|
path = str(tmp_path / "m.h5")
|
|
with h5py.File(path, "w") as f:
|
|
f.create_dataset("d", data=np.arange(4, dtype="<i4"), chunks=(2,), maxshape=(None,))
|
|
with clawhdf5.File(path, "r") as f:
|
|
with pytest.raises(OSError, match="r\\+"):
|
|
f["d"][0] = 1
|
|
with pytest.raises(OSError):
|
|
f["d"].resize((8,))
|
|
with pytest.raises(OSError):
|
|
f.attrs["x"] = 1
|
|
with pytest.raises(NotImplementedError, match="does not exist"):
|
|
clawhdf5.File(str(tmp_path / "missing.h5"), "a")
|
|
with pytest.raises(ValueError, match="mode"):
|
|
clawhdf5.File(path, "rw")
|
|
with clawhdf5.File(path, "a") as f:
|
|
assert f.mode == "r+"
|
|
f["d"][1] = 10
|
|
with h5py.File(path, "r") as f:
|
|
assert f["d"][1] == 10
|
|
|
|
|
|
def test_the_file_is_locked_while_open_for_editing(h5py, tmp_path):
|
|
path = str(tmp_path / "lock.h5")
|
|
with h5py.File(path, "w") as f:
|
|
f.create_dataset("d", data=np.arange(4, dtype="<i4"))
|
|
f = clawhdf5.File(path, "r+")
|
|
with pytest.raises(OSError):
|
|
clawhdf5.File(path, "r+")
|
|
with pytest.raises(OSError):
|
|
h5py.File(path, "r+")
|
|
f.close()
|
|
with h5py.File(path, "r+") as g:
|
|
g["d"][0] = 5
|
|
with clawhdf5.File(path, "r+") as g:
|
|
g["d"][1] = 6
|
|
with h5py.File(path, "r") as g:
|
|
np.testing.assert_array_equal(g["d"][...], [5, 6, 2, 3])
|
|
|
|
|
|
def test_objects_see_edits_made_through_others(h5py, tmp_path):
|
|
"""A dataset or attrs object taken before an edit reports the file as
|
|
it is after it: the new shape, the new attribute."""
|
|
path = str(tmp_path / "live.h5")
|
|
with h5py.File(path, "w") as f:
|
|
f.create_dataset("d", data=np.arange(6.0), chunks=(4,), maxshape=(None,))
|
|
with clawhdf5.File(path, "r+") as f:
|
|
d1 = f["d"]
|
|
d2 = f["d"]
|
|
attrs = d1.attrs
|
|
assert len(attrs) == 0 and "u" not in attrs
|
|
d2.resize((10,))
|
|
assert d1.shape == (10,) and d1.size == 10 and len(d1) == 10
|
|
np.testing.assert_array_equal(d1[6:], np.zeros(4))
|
|
d2.attrs["u"] = "m/s"
|
|
assert "u" in attrs and attrs["u"] == b"m/s" and len(attrs) == 1
|
|
f.attrs.create("shaped", np.arange(6), shape=(2, 3), dtype="<i2")
|
|
f.attrs.modify("shaped2", [1.5, 2.5])
|
|
with h5py.File(path, "r") as f:
|
|
assert f["d"].shape == (10,)
|
|
assert f["d"].attrs["u"] == b"m/s"
|
|
assert f.attrs["shaped"].dtype == np.dtype("<i2") and f.attrs["shaped"].shape == (2, 3)
|
|
np.testing.assert_array_equal(f.attrs["shaped2"], [1.5, 2.5])
|
|
|
|
|
|
def test_attribute_types_as_h5py_reads_them(h5py, tmp_path):
|
|
path = str(tmp_path / "attrs.h5")
|
|
with h5py.File(path, "w") as f:
|
|
f.create_group("g")
|
|
values = {
|
|
"i8": 5,
|
|
"f8": 2.5,
|
|
"i1": np.int8(-3),
|
|
"u8": np.uint64(2**64 - 1),
|
|
">f4": np.array([1.5, 2.5], dtype=">f4"),
|
|
"f2": np.float16(0.5),
|
|
"b": True,
|
|
"barr": np.array([True, False]),
|
|
"c16": np.complex128(1 + 2j),
|
|
"bytes": b"raw",
|
|
"sarr": np.array([b"a", b"bcd"]),
|
|
"str": "héllo",
|
|
"strs": ["x", "yz"],
|
|
"2d": np.arange(12, dtype="<u2").reshape(3, 4),
|
|
"empty": np.zeros((0,), dtype="<i4"),
|
|
}
|
|
with clawhdf5.File(path, "r+") as f:
|
|
for k, v in values.items():
|
|
f["g"].attrs[k] = v
|
|
# Replace one, with another type and size.
|
|
f["g"].attrs["i8"] = np.arange(100.0)
|
|
with h5py.File(path, "r") as f:
|
|
a = f["g"].attrs
|
|
np.testing.assert_array_equal(a["i8"], np.arange(100.0))
|
|
assert a["f8"] == 2.5 and a["f8"].dtype == np.float64
|
|
assert a["i1"] == -3 and a["i1"].dtype == np.int8
|
|
assert a["u8"] == 2**64 - 1 and a["u8"].dtype == np.uint64
|
|
assert a[">f4"].dtype == np.dtype(">f4")
|
|
assert a["f2"].dtype == np.float16
|
|
assert a["b"] is np.True_ or a["b"] == True # noqa: E712
|
|
assert a["barr"].dtype == np.bool_
|
|
assert a["c16"] == 1 + 2j
|
|
assert a["bytes"] == b"raw"
|
|
assert list(a["sarr"]) == [b"a", b"bcd"]
|
|
# str is stored as fixed-length UTF-8: h5py reads bytes.
|
|
assert a["str"].decode("utf-8") == "héllo"
|
|
assert [x.decode() for x in a["strs"]] == ["x", "yz"]
|
|
assert a["2d"].shape == (3, 4) and a["2d"].dtype == np.dtype("<u2")
|
|
assert a["empty"].shape == (0,)
|
|
with clawhdf5.File(path, "r") as f:
|
|
assert f["g"].attrs["c16"] == 1 + 2j
|
|
assert f["g"].attrs["str"].decode("utf-8") == "héllo"
|
|
h5dump_reads(path)
|
|
|
|
|
|
def test_many_attributes_move_to_dense_storage(h5py, tmp_path):
|
|
"""Past the compact limit (8 attributes under libver v114) the object's
|
|
attributes move to dense storage; h5py reads all of them."""
|
|
path = str(tmp_path / "dense.h5")
|
|
with h5py.File(path, "w", libver="v114") as f:
|
|
f.create_dataset("d", data=np.arange(3))
|
|
with clawhdf5.File(path, "r+") as f:
|
|
for i in range(20):
|
|
f["d"].attrs[f"a{i:02d}"] = np.full(i + 1, i, dtype="<i2")
|
|
with h5py.File(path, "r") as f:
|
|
assert sorted(f["d"].attrs.keys()) == [f"a{i:02d}" for i in range(20)]
|
|
for i in range(20):
|
|
np.testing.assert_array_equal(f["d"].attrs[f"a{i:02d}"], np.full(i + 1, i))
|
|
h5dump_reads(path)
|
|
|
|
|
|
def test_reads_never_see_a_half_written_edit(h5py, tmp_path):
|
|
"""Readers on other threads while one thread rewrites a dataset: every
|
|
read returns one whole version (all elements equal), never a mix."""
|
|
path = str(tmp_path / "race.h5")
|
|
with h5py.File(path, "w") as f:
|
|
f.create_dataset("d", data=np.zeros((64, 64)), chunks=(16, 16), compression="gzip")
|
|
errors = []
|
|
with clawhdf5.File(path, "r+") as f:
|
|
stop = threading.Event()
|
|
|
|
def read():
|
|
ds = f["d"]
|
|
while not stop.is_set():
|
|
a = ds[...]
|
|
if not (a == a.flat[0]).all():
|
|
errors.append(a)
|
|
return
|
|
|
|
readers = [threading.Thread(target=read) for _ in range(3)]
|
|
for t in readers:
|
|
t.start()
|
|
try:
|
|
for k in range(1, 25):
|
|
f["d"][...] = float(k)
|
|
finally:
|
|
stop.set()
|
|
for t in readers:
|
|
t.join()
|
|
np.testing.assert_array_equal(f["d"][...], np.full((64, 64), 24.0))
|
|
assert not errors, "a read saw a partly written dataset"
|
|
|
|
|
|
def test_close_releases_the_file(h5py, tmp_path):
|
|
path = str(tmp_path / "close.h5")
|
|
with h5py.File(path, "w") as f:
|
|
f.create_dataset("d", data=np.arange(4, dtype="<i4"))
|
|
f = clawhdf5.File(path, "r+")
|
|
ds = f["d"]
|
|
f.close()
|
|
# The handle still reads the file as last written, but cannot edit it.
|
|
np.testing.assert_array_equal(ds[...], np.arange(4))
|
|
with pytest.raises(OSError, match="closed"):
|
|
ds[0] = 1
|
|
with h5py.File(path, "r+") as g:
|
|
g["d"][0] = 9
|
|
|
|
|
|
def _two_files(h5py, tmp_path):
|
|
"""d1/f.h5 and d2/f.h5: same name, different layouts (the review's
|
|
repro)."""
|
|
(tmp_path / "d1").mkdir()
|
|
(tmp_path / "d2").mkdir()
|
|
with h5py.File(tmp_path / "d1" / "f.h5", "w") as f:
|
|
f.create_dataset("x", data=np.arange(10, dtype="<i4"))
|
|
f.create_dataset("big", data=np.full(5000, 1.5))
|
|
with h5py.File(tmp_path / "d2" / "f.h5", "w") as f:
|
|
f.create_dataset("pad", data=np.full(3000, 2.5))
|
|
f.create_dataset("x", data=np.arange(10, dtype="<i4") + 500)
|
|
return tmp_path / "d1" / "f.h5", tmp_path / "d2" / "f.h5"
|
|
|
|
|
|
def _held_file_edited(h5py, held, other, other_bytes):
|
|
"""The edits went to `held`, planned from its own metadata; `other`
|
|
was not touched."""
|
|
assert other.read_bytes() == other_bytes, "the other file changed"
|
|
with h5py.File(held, "r") as f:
|
|
np.testing.assert_array_equal(f["x"][()], np.full(10, 7))
|
|
np.testing.assert_array_equal(f["big"][()], np.full(5000, 1.5))
|
|
np.testing.assert_array_equal(f.attrs["note"], np.arange(50.0))
|
|
h5dump_reads(str(held))
|
|
|
|
|
|
def test_relative_path_and_chdir(h5py, tmp_path, monkeypatch):
|
|
"""A file opened by a relative path keeps being the file edited and read
|
|
after os.chdir (an edit was planned from the file the path named in the
|
|
new directory and written into the one open, corrupting it)."""
|
|
held, other = _two_files(h5py, tmp_path)
|
|
other_bytes = other.read_bytes()
|
|
monkeypatch.chdir(held.parent)
|
|
f = clawhdf5.File("f.h5", "r+")
|
|
ds = f["x"]
|
|
monkeypatch.chdir(other.parent)
|
|
ds[:] = np.full(10, 7, "<i4")
|
|
np.testing.assert_array_equal(ds[:5], np.full(5, 7)) # read from the held file
|
|
f.attrs["note"] = np.arange(50.0)
|
|
np.testing.assert_array_equal(f["big"][()], np.full(5000, 1.5))
|
|
assert "pad" not in f
|
|
f.close()
|
|
_held_file_edited(h5py, held, other, other_bytes)
|
|
# 'w' writes where the path named when the file was opened.
|
|
monkeypatch.chdir(held.parent)
|
|
w = clawhdf5.File("new.h5", "w")
|
|
monkeypatch.chdir(other.parent)
|
|
w.create_dataset("d", data=np.arange(3))
|
|
w.close()
|
|
assert (held.parent / "new.h5").exists() and not (other.parent / "new.h5").exists()
|
|
|
|
|
|
def test_path_replaced_between_edits(h5py, tmp_path):
|
|
"""The path renamed away and another file put in its place between
|
|
edits: the edits go to the file held open, never mixed with the other."""
|
|
held, other = _two_files(h5py, tmp_path)
|
|
path = str(tmp_path / "f.h5")
|
|
os.replace(held, path)
|
|
moved = tmp_path / "moved.h5"
|
|
with clawhdf5.File(path, "r+") as f:
|
|
ds = f["x"]
|
|
ds[0] = 7
|
|
os.replace(path, moved)
|
|
shutil.copy(other, path)
|
|
other_bytes = open(path, "rb").read()
|
|
ds[:] = np.full(10, 7, "<i4")
|
|
f.attrs["note"] = np.arange(50.0)
|
|
np.testing.assert_array_equal(f["x"][()], np.full(10, 7))
|
|
assert "pad" not in f
|
|
from pathlib import Path
|
|
_held_file_edited(h5py, moved, Path(path), other_bytes)
|
|
|
|
|
|
def test_an_edit_releases_the_gil(h5py, tmp_path):
|
|
"""Another Python thread keeps running while a large edit is written:
|
|
had the edit held the GIL, the other thread would stall for the whole
|
|
edit (Rust code never yields it)."""
|
|
import time
|
|
path = str(tmp_path / "big.h5")
|
|
with h5py.File(path, "w") as f:
|
|
f.create_dataset("d", shape=(1024, 1024), dtype="<f8", chunks=(64, 64), compression="gzip")
|
|
value = np.random.default_rng(0).standard_normal((1024, 1024))
|
|
stamps = []
|
|
stop = threading.Event()
|
|
|
|
def spin():
|
|
while not stop.is_set():
|
|
stamps.append(time.perf_counter())
|
|
|
|
with clawhdf5.File(path, "r+") as f:
|
|
ds = f["d"]
|
|
t = threading.Thread(target=spin)
|
|
t.start()
|
|
try:
|
|
time.sleep(0.05)
|
|
t0 = time.perf_counter()
|
|
ds[...] = value
|
|
t1 = time.perf_counter()
|
|
finally:
|
|
stop.set()
|
|
t.join()
|
|
np.testing.assert_array_equal(ds[...], value)
|
|
during = [x for x in stamps if t0 <= x <= t1]
|
|
gaps = np.diff([t0] + during + [t1])
|
|
assert t1 - t0 > 0.1, "the edit is too quick to tell"
|
|
assert gaps.max() < 0.5 * (t1 - t0), (
|
|
f"the other thread stalled for {gaps.max():.3f} s of a {t1 - t0:.3f} s edit")
|