clawhdf5.File(path, 'r+') (and 'a' on an existing file) holds a FileEditor, and with it the file's exclusive lock, until close(): - ds[key] = value: h5py's keys and broadcasting (numpy's rules for slices and integers with extra leading 1-axes allowed; the exact shape for an index list, a scalar only where h5py expands it). Arrays are converted as libhdf5 converts them in native byte order (integers saturate, floats truncate toward zero and clip, integers go into h5py's bool enum by value); other values through numpy.asarray(value, dtype=ds.dtype), as h5py does. NaN into an integer dataset is a ValueError instead of libhdf5's arbitrary value. The value preparation is a small Python module compiled into the extension (src/edit_helpers.py). - ds.resize(shape) / ds.resize(n, axis=k) with h5py's argument rules. - attrs[name] = value, attrs.create(name, data, shape, dtype), attrs.modify: numeric, bool, complex, bytes and str data of any shape, with h5py's HDF5 types; str is stored as fixed-length UTF-8 (the editor cannot write variable-length strings). - File.mode, File.flush(), Dataset.chunks. Each edit runs with the GIL released under the file handle's write lock (no read sees a half-written edit), then the file is reopened; datasets and attrs objects re-read their shape and attributes when the handle's edit generation moved. What the editor cannot do is NotImplementedError before anything is written: deleting attributes or objects, creating datasets or groups, compound fields by name, variable-length data, and FileEditor's own limits. Where libhdf5 2.0 (h5py 3.16) converts inconsistently -- its soft conversions in non-native byte order (a float in (-1, 0) becomes the integer minimum, same-size unsigned->signed wraps) and native casts that are undefined in C (half floats into unsigned, float(max) rounded up) -- clawhdf5 saturates as libhdf5's native path does; listed in docs/known-issues.md. Tests (tests/test_edit.py): every edit applied by h5py and by clawhdf5 to copies of the same file and both read back through h5py after each edit, on h5py files (libver earliest, v114, latest) and a clawhdf5 file: a fixed sequence over every chunk index kind, compact/contiguous/gzip layouts and numeric, bool, enum, complex, string and compound types, 16 random sequences of 40 edits, and a numeric conversion matrix; a refused edit must be refused by both and leave the file unchanged. Also dense attributes, locking, objects seeing edits, readers racing a writer, and h5dump (plus h5rs check in ci-test.sh) on every edited file. The read-vs-h5py suite also runs on a file opened 'r+'. Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
743 lines
32 KiB
Python
743 lines
32 KiB
Python
"""In-place editing: clawhdf5.File(path, 'r+') against h5py.
|
|
|
|
Every edit is applied twice, to two copies of the same file: once through
|
|
h5py (libhdf5) and once through clawhdf5 (FileEditor). After every edit both
|
|
files are read back with h5py and must hold the same shapes, values and
|
|
attributes; clawhdf5's own view must agree; when h5py refuses an edit,
|
|
clawhdf5 must refuse it too and leave its file as it was. Files are written
|
|
by h5py (libver earliest and latest, so every chunk index kind) and by
|
|
clawhdf5; `h5dump` must read every result."""
|
|
|
|
import io
|
|
import os
|
|
import shutil
|
|
import subprocess
|
|
import threading
|
|
|
|
import numpy as np
|
|
import pytest
|
|
|
|
import clawhdf5
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Files
|
|
# ---------------------------------------------------------------------------
|
|
|
|
ENUM = {"RED": 0, "GREEN": 1, "BLUE": 7}
|
|
|
|
|
|
def _h5py_file(h5py, path, libver):
|
|
rng = np.random.default_rng(1)
|
|
with h5py.File(path, "w", libver=libver) as f:
|
|
f.create_dataset("i4", data=np.arange(60, dtype="<i4").reshape(6, 10))
|
|
f.create_dataset("be_i2", data=np.arange(24, dtype=">i2").reshape(4, 6))
|
|
f.create_dataset("u1", data=np.arange(16, dtype="u1"))
|
|
f.create_dataset("f2", data=rng.standard_normal(12).astype("<f2"))
|
|
f.create_dataset("f4_2d", data=rng.standard_normal((8, 9)).astype("<f4"))
|
|
f.create_dataset("f8_3d", data=rng.standard_normal((4, 5, 6)))
|
|
f.create_dataset("u8", data=np.arange(10, dtype="<u8"))
|
|
f.create_dataset("c8", data=(np.arange(6) + 1j * np.arange(6)).astype("<c8"))
|
|
f.create_dataset("bool", data=np.array([True, False, True, True]))
|
|
f.create_dataset("enum", data=np.array([0, 1, 7, 0], dtype="i1"),
|
|
dtype=h5py.enum_dtype(ENUM, basetype="i1"))
|
|
f.create_dataset("s5", data=np.array([b"ab", b"cdefg", b""], dtype="S5"))
|
|
cmp_dt = np.dtype([("id", "<i4"), ("x", "<f8"), ("tag", "S3")])
|
|
f.create_dataset("cmp", data=np.array([(i, i / 2, b"t%d" % i) for i in range(5)], dtype=cmp_dt))
|
|
f.create_dataset("scalar", data=np.float64(3.5))
|
|
# Chunked: fixed maxshape (v4 fixed array under latest), one
|
|
# unlimited dimension (extensible array), two (v2 B-tree), one chunk.
|
|
f.create_dataset("chunk_fixed", data=np.arange(100, dtype="<i8").reshape(10, 10), chunks=(3, 4))
|
|
f.create_dataset("chunk_ext", data=rng.standard_normal((12, 7)), chunks=(5, 7), maxshape=(None, 7))
|
|
f.create_dataset("chunk_bt2", data=np.arange(30, dtype="<i4").reshape(5, 6), chunks=(2, 2),
|
|
maxshape=(None, None))
|
|
f.create_dataset("chunk_gzip", data=np.arange(400, dtype="<f4").reshape(20, 20), chunks=(6, 6),
|
|
compression="gzip", maxshape=(40, 40))
|
|
f.create_dataset("chunk_single", data=np.arange(12, dtype="<u2").reshape(3, 4), chunks=(3, 4),
|
|
maxshape=(3, 4))
|
|
f.create_dataset("chunk_fill", shape=(8,), dtype="<i4", chunks=(3,), maxshape=(20,), fillvalue=-1)
|
|
f.create_dataset("vlen", data=["a", "bb"], dtype=h5py.string_dtype())
|
|
# Compact layout (low-level API).
|
|
dcpl = h5py.h5p.create(h5py.h5p.DATASET_CREATE)
|
|
dcpl.set_layout(h5py.h5d.COMPACT)
|
|
space = h5py.h5s.create_simple((7,))
|
|
dsid = h5py.h5d.create(f.id, b"compact", h5py.h5t.STD_I32LE, space, dcpl=dcpl)
|
|
dsid.write(h5py.h5s.ALL, h5py.h5s.ALL, np.arange(7, dtype="<i4"))
|
|
g = f.create_group("grp")
|
|
g.create_dataset("leaf", data=np.arange(5.0))
|
|
g.attrs["units"] = "m"
|
|
f.attrs["version"] = np.int32(1)
|
|
|
|
|
|
def _clawhdf5_file(path):
|
|
with clawhdf5.File(str(path), "w") as f:
|
|
f.create_dataset("i4", data=np.arange(60, dtype="<i4").reshape(6, 10))
|
|
f.create_dataset("f8", data=np.linspace(0, 1, 30).reshape(5, 6))
|
|
f.create_dataset("chunk_gzip", data=np.arange(400, dtype="<f4").reshape(20, 20),
|
|
chunks=[6, 6], compression="gzip")
|
|
f.create_dataset("u1", data=np.arange(16, dtype="u1"))
|
|
g = f.create_group("grp")
|
|
g.create_dataset("leaf", data=np.arange(5.0))
|
|
g.attrs["units"] = "m"
|
|
f.attrs["version"] = 1
|
|
|
|
|
|
# h5py's libver: "earliest" (v1 B-tree chunk indexes), "v114" (the 1.10+
|
|
# indexes: fixed and extensible arrays, v2 B-trees, single chunk) and
|
|
# "latest" (HDF5 2.0's newest format, which h5dump 1.14 cannot read).
|
|
SOURCES = ["h5py-earliest", "h5py-v114", "h5py-latest", "clawhdf5"]
|
|
|
|
|
|
def _make(h5py, tmp_path, source):
|
|
base = tmp_path / f"base-{source}.h5"
|
|
if source == "clawhdf5":
|
|
_clawhdf5_file(base)
|
|
else:
|
|
_h5py_file(h5py, str(base), source.split("-")[1])
|
|
theirs = tmp_path / f"theirs-{source}.h5"
|
|
ours = tmp_path / f"ours-{source}.h5"
|
|
shutil.copy(base, theirs)
|
|
shutil.copy(base, ours)
|
|
return str(theirs), str(ours), str(base)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Comparing files through h5py
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def _norm_attr(v):
|
|
"""Attribute values comparable across the two writers: clawhdf5 stores
|
|
`str` as fixed-length UTF-8 (h5py reads bytes), h5py as variable-length
|
|
(h5py reads str)."""
|
|
if isinstance(v, bytes):
|
|
return ("str", v.decode("utf-8"))
|
|
if isinstance(v, str):
|
|
return ("str", v)
|
|
arr = np.asarray(v)
|
|
if arr.dtype.kind in "SO":
|
|
return ("strs", [x.decode() if isinstance(x, bytes) else x for x in arr.ravel().tolist()], arr.shape)
|
|
return (arr.dtype.str, arr.shape, arr.tobytes())
|
|
|
|
|
|
def snapshot(h5py, path):
|
|
"""What h5py sees in the file: every dataset's shape, dtype, bytes and
|
|
attributes (read without locking: clawhdf5 may hold the file open)."""
|
|
out = {}
|
|
with h5py.File(path, "r", locking=False) as f:
|
|
def visit(name, obj):
|
|
attrs = {k: _norm_attr(obj.attrs[k]) for k in obj.attrs}
|
|
if isinstance(obj, h5py.Dataset):
|
|
if obj.dtype.kind == "O":
|
|
data = [x for x in obj[...].ravel().tolist()]
|
|
else:
|
|
data = obj[()].tobytes() if obj.shape is not None else None
|
|
out[name] = (obj.shape, obj.dtype.str, obj.maxshape, data, attrs)
|
|
else:
|
|
out[name] = ("group", attrs)
|
|
visit("/", f)
|
|
f.visititems(visit)
|
|
return out
|
|
|
|
|
|
def assert_same_files(h5py, theirs, ours, what):
|
|
a, b = snapshot(h5py, theirs), snapshot(h5py, ours)
|
|
assert a.keys() == b.keys(), what
|
|
for k in a:
|
|
assert a[k] == b[k], f"{what}: {k} differs\n h5py: {a[k]}\n clawhdf5: {b[k]}"
|
|
|
|
|
|
def assert_ours_reads_like_h5py(h5py, f, path, what):
|
|
"""clawhdf5's own view of the file it is editing matches h5py's."""
|
|
with h5py.File(path, "r", locking=False) as t:
|
|
for name in ["i4", "chunk_ext", "chunk_bt2", "chunk_gzip", "f8_3d", "cmp", "bool", "enum", "scalar"]:
|
|
if name not in t:
|
|
continue
|
|
o = f[name]
|
|
assert o.shape == t[name].shape, f"{what}: {name} shape"
|
|
assert o.maxshape == t[name].maxshape, f"{what}: {name} maxshape"
|
|
np.testing.assert_array_equal(o[()], t[name][()], err_msg=f"{what}: {name}")
|
|
for obj in ["/", "grp"]:
|
|
assert sorted(f[obj].attrs.keys()) == sorted(t[obj].attrs.keys()), what
|
|
for k in t[obj].attrs:
|
|
assert _norm_attr(f[obj].attrs[k]) == _norm_attr(t[obj].attrs[k]), f"{what}: {obj}.attrs[{k}]"
|
|
|
|
|
|
def h5dump_reads(path, base=None):
|
|
"""h5dump (libhdf5 1.14) reads every object and value of `path` — when it
|
|
reads the unedited `base` (it cannot read HDF5 2.0's newest format)."""
|
|
exe = shutil.which("h5dump")
|
|
if exe is None:
|
|
if os.environ.get("CLAWHDF5_REQUIRE_INTEROP") == "1":
|
|
pytest.fail("h5dump is required (CLAWHDF5_REQUIRE_INTEROP=1)")
|
|
return
|
|
h5rs = os.environ.get("CLAWHDF5_H5RS")
|
|
if h5rs:
|
|
# clawhdf5's structural and checksum validator (scripts/ci-test.sh
|
|
# points this at the h5rs it built).
|
|
r = subprocess.run([h5rs, "check", path], capture_output=True, text=True)
|
|
assert r.returncode == 0, (r.stdout + r.stderr)[-2000:]
|
|
if base is not None and subprocess.run([exe, "-H", base], capture_output=True).returncode != 0:
|
|
return
|
|
r = subprocess.run([exe, path], capture_output=True, text=True)
|
|
assert r.returncode == 0, r.stderr[-2000:]
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Applying one edit both ways
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def _native_conversion(h5py, value, ds_dtype):
|
|
"""`value` as libhdf5 converts it to `ds_dtype` in native byte order.
|
|
|
|
libhdf5 2.0 (h5py 3.16) converts numbers differently when either side is
|
|
not in native byte order (its "soft" conversions): a float in (-1, 0)
|
|
becomes the integer type's minimum instead of 0, and an unsigned integer
|
|
too large for the signed type of the same size wraps instead of
|
|
saturating. clawhdf5 applies the native-order results to every byte
|
|
order, so the reference is h5py converting into a native dataset of the
|
|
same kind; the result then reaches the real dataset by a plain byte
|
|
swap."""
|
|
if not isinstance(value, np.ndarray):
|
|
return value
|
|
if value.dtype.kind not in "biuf" or ds_dtype.kind not in "biuf":
|
|
return value
|
|
if value.dtype.isnative and ds_dtype.isnative:
|
|
return value
|
|
with h5py.File(io.BytesIO(), "w") as tmp:
|
|
d = tmp.create_dataset("t", shape=value.shape, dtype=ds_dtype.newbyteorder("="))
|
|
d[...] = value.astype(value.dtype.newbyteorder("="))
|
|
return np.asarray(d[()])
|
|
|
|
|
|
def _apply(f, op, h5py=None):
|
|
"""Apply `op` to `f`; with `h5py`, `f` is an h5py file and a numpy array
|
|
value is first converted as libhdf5 converts in native byte order (see
|
|
`_native_conversion`)."""
|
|
kind = op[0]
|
|
if kind == "set":
|
|
_, name, key, value = op
|
|
if h5py is not None:
|
|
value = _native_conversion(h5py, value, f[name].dtype)
|
|
f[name][key] = value
|
|
elif kind == "resize":
|
|
_, name, size, axis = op
|
|
if axis is None:
|
|
f[name].resize(size)
|
|
else:
|
|
f[name].resize(size, axis=axis)
|
|
elif kind == "attr":
|
|
_, obj, name, value = op
|
|
f[obj].attrs[name] = value
|
|
else:
|
|
raise AssertionError(op)
|
|
|
|
|
|
def edit_both(h5py, theirs, ours_path, ours, op):
|
|
"""Apply `op` with h5py and with clawhdf5 (`ours`, open 'r+'); the two
|
|
files must then read the same through h5py. If h5py refuses, clawhdf5
|
|
must refuse and its file must be unchanged. Returns h5py's error."""
|
|
before = snapshot(h5py, ours_path)
|
|
try:
|
|
with h5py.File(theirs, "r+") as t:
|
|
_apply(t, op, h5py)
|
|
except Exception as e: # noqa: BLE001 - h5py refuses: so must we
|
|
try:
|
|
_apply(ours, op)
|
|
except Exception: # noqa: BLE001
|
|
pass
|
|
else:
|
|
pytest.fail(f"{op!r:.300}: h5py refused ({type(e).__name__}: {e}), clawhdf5 did not")
|
|
assert snapshot(h5py, ours_path) == before, f"{op!r}: clawhdf5 changed the file while failing"
|
|
return e
|
|
_apply(ours, op)
|
|
assert_same_files(h5py, theirs, ours_path, repr(op)[:200])
|
|
return None
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Tests
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
@pytest.mark.parametrize("source", SOURCES)
|
|
def test_edit_sequence_matches_h5py(h5py, tmp_path, source):
|
|
theirs, ours_path, base = _make(h5py, tmp_path, source)
|
|
ops = [
|
|
("set", "i4", 0, 99),
|
|
("set", "i4", (slice(1, 5, 2), slice(None, None, 3)), np.array([[1.5, -2.5, 1e12, -1e12]])),
|
|
("set", "i4", (slice(None), 2), np.arange(6, dtype="<i8") * 1000),
|
|
("set", "i4", ([0, 2, 5], slice(4, 6)), np.array([[1, 2], [3, 4], [5, 6]])),
|
|
("set", "i4", (Ellipsis, -1), np.int8(-7)),
|
|
("set", "i4", (3, 3), 12345.9),
|
|
("set", "u1", slice(2, 8), np.array([-5, 0, 300, 255, 256, 1], dtype="<i4")),
|
|
("set", "u1", slice(0, 3), [1, 2, 3]),
|
|
("set", "chunk_gzip", (slice(0, 20, 7), slice(3, 17)), 42.25),
|
|
("set", "chunk_gzip", (slice(5, 11), slice(5, 11)), np.ones((6, 6), dtype="<f8") * np.pi),
|
|
("set", "grp/leaf", slice(None), np.array([1, 2, 3, 4, 5], dtype="<u2")),
|
|
("attr", "/", "version", np.int32(2)),
|
|
("attr", "/", "count", 7),
|
|
("attr", "grp", "scale", np.array([0.5, 0.25], dtype="<f4")),
|
|
("attr", "grp", "matrix", np.arange(6, dtype=">i8").reshape(2, 3)),
|
|
("attr", "i4", "flag", np.bool_(True)),
|
|
("attr", "i4", "z", np.complex64(1 - 2j)),
|
|
("attr", "i4", "raw", np.bytes_(b"abc")),
|
|
("set", "i4", slice(0, 2), np.zeros((3, 10))), # shape mismatch: refused by both
|
|
]
|
|
if source != "clawhdf5":
|
|
ops += [
|
|
("set", "be_i2", (slice(None), slice(1, 3)), np.array([70000, -70000], dtype="<i8")),
|
|
("set", "f2", slice(None, None, 4), np.array([1e6, -3.25, 0.1])),
|
|
("set", "f4_2d", (2, slice(None)), np.linspace(-1, 1, 9)),
|
|
("set", "f8_3d", (slice(1, 3), 2, slice(None, None, 2)), np.arange(3, dtype="<i2")),
|
|
("set", "u8", slice(None), np.array([-1, 0, 2**63, 1e30, -1e30, 5.5, 2, 3, 4, 5])),
|
|
("set", "c8", slice(1, 3), np.array([1 + 1j, 2 - 2j], dtype="<c16")),
|
|
("set", "c8", 0, np.float64(1.0)), # h5py: no conversion path
|
|
("set", "bool", slice(None), np.array([0, 3, 0, -1], dtype="<i4")),
|
|
("set", "bool", 1, np.array(True)),
|
|
("set", "enum", slice(0, 2), np.array([7, 1], dtype="<i4")),
|
|
("set", "s5", 0, np.bytes_(b"xyzuvw")),
|
|
("set", "s5", slice(1, 3), [b"q", b"rs"]),
|
|
("set", "s5", 2, np.array("uni")), # h5py: no conversion from 'U'
|
|
("set", "cmp", 2, np.array((9, 9.5, b"zz"), dtype=[("id", "<i4"), ("x", "<f8"), ("tag", "S3")])),
|
|
("set", "cmp", slice(3, 5), [(1, 0.5, b"a"), (2, 1.5, b"b")]),
|
|
("set", "scalar", (), 7.25),
|
|
("set", "scalar", Ellipsis, np.float32(-1.5)),
|
|
("set", "compact", slice(1, 6, 2), np.array([10, 20, 30])),
|
|
("set", "chunk_fixed", (slice(2, 9), slice(1, 10, 4)), np.arange(21).reshape(7, 3)),
|
|
("set", "chunk_single", (1, slice(None)), np.array([9, 8, 7, 6])),
|
|
("resize", "chunk_ext", (20, 7), None),
|
|
("set", "chunk_ext", slice(12, 20), np.full((8, 7), 2.5)),
|
|
("resize", "chunk_ext", 9, 0),
|
|
("resize", "chunk_ext", 16, 0),
|
|
("resize", "chunk_bt2", (9, 11), None),
|
|
("set", "chunk_bt2", (slice(4, 9), slice(5, 11)), np.arange(30).reshape(5, 6)),
|
|
("resize", "chunk_bt2", (3, 3), None),
|
|
("resize", "chunk_bt2", (7, 8), None),
|
|
("resize", "chunk_gzip", (40, 25), None),
|
|
("resize", "chunk_gzip", (41, 25), None), # beyond maxshape: refused
|
|
("resize", "chunk_fill", 15, None), # h5py: a size without axis must be a tuple
|
|
("resize", "chunk_fill", (15,), None),
|
|
("set", "chunk_fill", slice(10, 12), [1, 2]),
|
|
("resize", "chunk_fill", (4,), None),
|
|
("resize", "chunk_fill", (20,), None),
|
|
("resize", "i4", (7, 10), None), # not chunked: refused
|
|
]
|
|
with clawhdf5.File(ours_path, "r+") as ours:
|
|
assert ours.mode == "r+"
|
|
for i, op in enumerate(ops):
|
|
edit_both(h5py, theirs, ours_path, ours, op)
|
|
if i % 5 == 0:
|
|
assert_ours_reads_like_h5py(h5py, ours, ours_path, repr(op))
|
|
assert_ours_reads_like_h5py(h5py, ours, ours_path, "end")
|
|
h5dump_reads(ours_path, base)
|
|
# Reopened, clawhdf5 reads what h5py reads.
|
|
with clawhdf5.File(ours_path, "r") as f:
|
|
assert f.mode == "r"
|
|
assert_ours_reads_like_h5py(h5py, f, ours_path, "reopened")
|
|
|
|
|
|
def _random_key(rng, shape):
|
|
key = []
|
|
for n in shape:
|
|
r = rng.random()
|
|
if n == 0 or r < 0.3:
|
|
start = int(rng.integers(0, n + 1)) if n else 0
|
|
stop = int(rng.integers(start, n + 1)) if n else 0
|
|
step = int(rng.integers(1, 4))
|
|
key.append(slice(start, stop, step))
|
|
elif r < 0.55:
|
|
key.append(int(rng.integers(-n, n)))
|
|
elif r < 0.7:
|
|
k = int(rng.integers(1, min(n, 4) + 1))
|
|
key.append(sorted(rng.choice(n, size=k, replace=False).tolist()))
|
|
else:
|
|
key.append(slice(None))
|
|
# Only one index list per key.
|
|
lists = [i for i, k in enumerate(key) if isinstance(k, list)]
|
|
for i in lists[1:]:
|
|
key[i] = slice(None)
|
|
return tuple(key)
|
|
|
|
|
|
def _selection_shape(key, shape):
|
|
out = []
|
|
fancy = False
|
|
for k, n in zip(key, shape):
|
|
if isinstance(k, slice):
|
|
out.append(len(range(*k.indices(n))))
|
|
elif isinstance(k, list):
|
|
out.append(len(k))
|
|
fancy = True
|
|
return tuple(out), fancy
|
|
|
|
|
|
def _random_value(rng, sel_shape, fancy, dtype):
|
|
r = rng.random()
|
|
if r < 0.2 or not sel_shape:
|
|
v = rng.standard_normal() * 1000
|
|
return np.float64(v) if rng.random() < 0.5 else int(v)
|
|
shape = list(sel_shape)
|
|
if not fancy and r < 0.35 and shape:
|
|
shape[0] = 1 # broadcast along the first axis
|
|
kind = rng.choice(["same", "f8", "i8", "u1", "f4"])
|
|
if kind == "same" and np.dtype(dtype).kind in "iuf":
|
|
dt = np.dtype(dtype)
|
|
else:
|
|
dt = np.dtype(str(kind) if kind != "same" else "f8")
|
|
base = rng.standard_normal(size=shape) * (10 ** rng.integers(0, 6))
|
|
with np.errstate(all="ignore"):
|
|
return base.astype(dt)
|
|
|
|
|
|
@pytest.mark.parametrize("source", SOURCES)
|
|
@pytest.mark.parametrize("seed", [0, 1, 2, 3])
|
|
def test_random_edits_match_h5py(h5py, tmp_path, source, seed):
|
|
"""Random writes (slices, steps, integers, index lists, broadcasts,
|
|
other dtypes and out-of-range values), resizes and attributes, each
|
|
compared with h5py applying the same edit."""
|
|
rng = np.random.default_rng(1000 * seed + SOURCES.index(source))
|
|
theirs, ours_path, base = _make(h5py, tmp_path, source)
|
|
with h5py.File(theirs, "r") as t:
|
|
names = [n for n in ["i4", "u1", "f8", "f4_2d", "f8_3d", "be_i2", "chunk_fixed", "chunk_ext",
|
|
"chunk_bt2", "chunk_gzip", "chunk_fill", "compact", "grp/leaf"] if n in t]
|
|
resizable = [n for n in ["chunk_ext", "chunk_bt2", "chunk_gzip", "chunk_fill"] if n in names]
|
|
refused = 0
|
|
with clawhdf5.File(ours_path, "r+") as ours:
|
|
for step in range(40):
|
|
r = rng.random()
|
|
if r < 0.15 and resizable:
|
|
name = str(rng.choice(resizable))
|
|
with h5py.File(theirs, "r") as t:
|
|
maxshape = t[name].maxshape
|
|
shape = t[name].shape
|
|
new = tuple(int(rng.integers(0, (m if m is not None else s + 10) + 1)) for m, s in zip(maxshape, shape))
|
|
op = ("resize", name, new, None)
|
|
elif r < 0.25:
|
|
obj = str(rng.choice(["/", "grp", names[0]]))
|
|
choices = [np.int16(rng.integers(-100, 100)), rng.standard_normal(3),
|
|
np.arange(int(rng.integers(1, 5)), dtype=">u4"), np.float32(0.5)]
|
|
value = choices[int(rng.integers(0, len(choices)))]
|
|
op = ("attr", obj, f"a{int(rng.integers(0, 4))}", value)
|
|
else:
|
|
name = str(rng.choice(names))
|
|
with h5py.File(theirs, "r") as t:
|
|
shape, dtype = t[name].shape, t[name].dtype
|
|
key = _random_key(rng, shape)
|
|
sel_shape, fancy = _selection_shape(key, shape)
|
|
op = ("set", name, key, _random_value(rng, sel_shape, fancy, dtype))
|
|
if edit_both(h5py, theirs, ours_path, ours, op) is not None:
|
|
refused += 1
|
|
assert_ours_reads_like_h5py(h5py, ours, ours_path, "end")
|
|
assert refused < 30
|
|
h5dump_reads(ours_path, base)
|
|
|
|
|
|
NUMERIC = ["<i1", "<u1", "<i2", ">u2", "<i4", "<u4", ">i8", "<u8", "<f2", "<f4", ">f8"]
|
|
|
|
|
|
def _quiet(f):
|
|
with np.errstate(all="ignore"):
|
|
return f()
|
|
|
|
|
|
def _libhdf5_undefined(vals, target):
|
|
"""The values whose conversion to `target` libhdf5 2.0 (h5py 3.16) gets
|
|
wrong even in native byte order, where its C casts are undefined
|
|
behaviour; clawhdf5 saturates them as libhdf5's range handling
|
|
intends (docs/known-issues.md):
|
|
|
|
- half floats into unsigned integers: negatives wrap (-1 -> 65535) and
|
|
+inf becomes 0; into signed integers, +-inf becomes the minimum;
|
|
- a float equal to the integer maximum rounded up in the float's
|
|
precision (float32(2**31 - 1) == 2**31 -> int32, float64(2**64 - 1)
|
|
-> uint64) becomes the minimum (or 0);
|
|
- a double between 65504 and 65520 into a half float becomes infinity
|
|
(IEEE rounds it down to 65504, as numpy does).
|
|
"""
|
|
bad = np.zeros(vals.shape, dtype=bool)
|
|
if vals.dtype.kind != "f":
|
|
return bad
|
|
if target.kind in "iu":
|
|
if vals.dtype.itemsize == 2:
|
|
bad |= np.isinf(vals)
|
|
if target.kind == "u":
|
|
bad |= vals <= -1
|
|
top = _quiet(lambda: np.array(np.iinfo(target).max).astype(vals.dtype))
|
|
if float(top) > np.iinfo(target).max:
|
|
bad |= vals == top
|
|
if target.kind == "f" and target.itemsize == 2 and vals.dtype.itemsize > 2:
|
|
bad |= (np.abs(vals) > 65504) & (np.abs(vals) < 65520)
|
|
return bad
|
|
|
|
|
|
def test_numeric_conversions_match_h5py(h5py, tmp_path):
|
|
"""Every numeric source dtype into every numeric dataset dtype, with
|
|
values at and beyond the targets' limits, as libhdf5 converts them."""
|
|
edge = np.array([0, 1, -1, -0.3, 2.5, -2.5, 3.7, -3.7, 127.9, -128.9, 200.5, 255.5, 256, -129,
|
|
32767.5, 40000, 65504, 70000, -70000, 2**31 - 1, 2**31, -2**31 - 1,
|
|
4e9, 1e15, -1e15, 1e19, 1e300, -1e300, np.inf, -np.inf])
|
|
sources = {
|
|
"f8": edge,
|
|
"f4": _quiet(lambda: edge.astype("<f4")),
|
|
"f2": np.array([0, 1, -1, 2.5, -3.5, 65504, -65504, np.inf, -np.inf, 100.5], dtype="<f2"),
|
|
"i8": np.array([0, 1, -1, 127, 128, -129, 255, 256, 32768, -32769, 65536, 2**31, -2**31 - 1,
|
|
2**32, 2**62, -2**63, 2**63 - 1], dtype="<i8"),
|
|
"u8": np.array([0, 1, 127, 128, 255, 256, 65535, 65536, 2**31, 2**32, 2**63, 2**64 - 1], dtype="<u8"),
|
|
"i1": np.array([-128, -1, 0, 1, 127], dtype="i1"),
|
|
"u2": np.array([0, 255, 256, 65535], dtype=">u2"),
|
|
"b": np.array([True, False, True]),
|
|
}
|
|
for target in NUMERIC:
|
|
for sname, src in sources.items():
|
|
vals = src[~_libhdf5_undefined(src, np.dtype(target))]
|
|
path_t = str(tmp_path / f"t_{target[1:]}_{sname}.h5")
|
|
path_o = str(tmp_path / f"o_{target[1:]}_{sname}.h5")
|
|
with h5py.File(path_t, "w") as f:
|
|
f.create_dataset("d", shape=vals.shape, dtype=target)
|
|
shutil.copy(path_t, path_o)
|
|
what = f"{vals.dtype} -> {target}"
|
|
try:
|
|
with h5py.File(path_t, "r+") as f:
|
|
f["d"][...] = _native_conversion(h5py, vals, f["d"].dtype)
|
|
except Exception: # noqa: BLE001
|
|
with clawhdf5.File(path_o, "r+") as f, pytest.raises(Exception):
|
|
f["d"][...] = vals
|
|
continue
|
|
with clawhdf5.File(path_o, "r+") as f:
|
|
f["d"][...] = vals
|
|
with h5py.File(path_t, "r") as a, h5py.File(path_o, "r") as b:
|
|
assert a["d"][...].tobytes() == b["d"][...].tobytes(), (
|
|
f"{what}: h5py {a['d'][...].tolist()} clawhdf5 {b['d'][...].tolist()}"
|
|
)
|
|
|
|
|
|
def test_nan_into_an_integer_dataset_is_refused(h5py, tmp_path):
|
|
"""libhdf5 stores NaN as an arbitrary integer (0, the minimum or 2**63,
|
|
depending on the type); clawhdf5 refuses and writes nothing."""
|
|
path = str(tmp_path / "nan.h5")
|
|
with h5py.File(path, "w") as f:
|
|
f.create_dataset("d", data=np.arange(4, dtype="<i4"))
|
|
with clawhdf5.File(path, "r+") as f:
|
|
with pytest.raises(ValueError, match="NaN"):
|
|
f["d"][...] = np.array([1.0, np.nan, 2.0, 3.0])
|
|
# A Python list goes through numpy, which refuses NaN too.
|
|
with pytest.raises(ValueError):
|
|
f["d"][0:2] = [np.nan, 1.0]
|
|
with h5py.File(path, "r") as f:
|
|
np.testing.assert_array_equal(f["d"][...], np.arange(4))
|
|
|
|
|
|
def test_unsupported_edits_are_clear_errors(h5py, tmp_path):
|
|
path = str(tmp_path / "u.h5")
|
|
_h5py_file(h5py, path, "earliest")
|
|
before = snapshot(h5py, path)
|
|
with clawhdf5.File(path, "r+") as f:
|
|
with pytest.raises(NotImplementedError, match="delet"):
|
|
del f.attrs["version"]
|
|
with pytest.raises(NotImplementedError, match="delet"):
|
|
del f["i4"]
|
|
with pytest.raises(NotImplementedError, match="delet"):
|
|
del f["grp"]["leaf"]
|
|
with pytest.raises(NotImplementedError):
|
|
f.create_dataset("new", data=np.arange(3.0))
|
|
with pytest.raises(NotImplementedError):
|
|
f.create_group("newgrp")
|
|
with pytest.raises(NotImplementedError):
|
|
f["grp"].create_dataset("new", data=np.arange(3.0))
|
|
with pytest.raises(NotImplementedError, match="variable-length"):
|
|
f["vlen"][0] = "x"
|
|
with pytest.raises(NotImplementedError, match="field"):
|
|
f["cmp"]["id"] = np.arange(5)
|
|
with pytest.raises(NotImplementedError):
|
|
f.attrs["empty"] = clawhdf5.Empty("f8")
|
|
with pytest.raises(TypeError, match="chunked"):
|
|
f["i4"].resize((7, 10))
|
|
with pytest.raises(ValueError):
|
|
f["chunk_gzip"].resize((41, 20))
|
|
with pytest.raises(ValueError, match="axis"):
|
|
f["chunk_ext"].resize(3, axis=2)
|
|
with pytest.raises(TypeError):
|
|
f["i4"][0] = np.array(["a"] * 10)
|
|
assert snapshot(h5py, path) == before
|
|
h5dump_reads(path)
|
|
|
|
|
|
def test_read_only_files_and_modes(h5py, tmp_path):
|
|
path = str(tmp_path / "m.h5")
|
|
with h5py.File(path, "w") as f:
|
|
f.create_dataset("d", data=np.arange(4, dtype="<i4"), chunks=(2,), maxshape=(None,))
|
|
with clawhdf5.File(path, "r") as f:
|
|
with pytest.raises(OSError, match="r\\+"):
|
|
f["d"][0] = 1
|
|
with pytest.raises(OSError):
|
|
f["d"].resize((8,))
|
|
with pytest.raises(OSError):
|
|
f.attrs["x"] = 1
|
|
with pytest.raises(NotImplementedError, match="does not exist"):
|
|
clawhdf5.File(str(tmp_path / "missing.h5"), "a")
|
|
with pytest.raises(ValueError, match="mode"):
|
|
clawhdf5.File(path, "rw")
|
|
with clawhdf5.File(path, "a") as f:
|
|
assert f.mode == "r+"
|
|
f["d"][1] = 10
|
|
with h5py.File(path, "r") as f:
|
|
assert f["d"][1] == 10
|
|
|
|
|
|
def test_the_file_is_locked_while_open_for_editing(h5py, tmp_path):
|
|
path = str(tmp_path / "lock.h5")
|
|
with h5py.File(path, "w") as f:
|
|
f.create_dataset("d", data=np.arange(4, dtype="<i4"))
|
|
f = clawhdf5.File(path, "r+")
|
|
with pytest.raises(OSError):
|
|
clawhdf5.File(path, "r+")
|
|
with pytest.raises(OSError):
|
|
h5py.File(path, "r+")
|
|
f.close()
|
|
with h5py.File(path, "r+") as g:
|
|
g["d"][0] = 5
|
|
with clawhdf5.File(path, "r+") as g:
|
|
g["d"][1] = 6
|
|
with h5py.File(path, "r") as g:
|
|
np.testing.assert_array_equal(g["d"][...], [5, 6, 2, 3])
|
|
|
|
|
|
def test_objects_see_edits_made_through_others(h5py, tmp_path):
|
|
"""A dataset or attrs object taken before an edit reports the file as
|
|
it is after it: the new shape, the new attribute."""
|
|
path = str(tmp_path / "live.h5")
|
|
with h5py.File(path, "w") as f:
|
|
f.create_dataset("d", data=np.arange(6.0), chunks=(4,), maxshape=(None,))
|
|
with clawhdf5.File(path, "r+") as f:
|
|
d1 = f["d"]
|
|
d2 = f["d"]
|
|
attrs = d1.attrs
|
|
assert len(attrs) == 0 and "u" not in attrs
|
|
d2.resize((10,))
|
|
assert d1.shape == (10,) and d1.size == 10 and len(d1) == 10
|
|
np.testing.assert_array_equal(d1[6:], np.zeros(4))
|
|
d2.attrs["u"] = "m/s"
|
|
assert "u" in attrs and attrs["u"] == b"m/s" and len(attrs) == 1
|
|
f.attrs.create("shaped", np.arange(6), shape=(2, 3), dtype="<i2")
|
|
f.attrs.modify("shaped2", [1.5, 2.5])
|
|
with h5py.File(path, "r") as f:
|
|
assert f["d"].shape == (10,)
|
|
assert f["d"].attrs["u"] == b"m/s"
|
|
assert f.attrs["shaped"].dtype == np.dtype("<i2") and f.attrs["shaped"].shape == (2, 3)
|
|
np.testing.assert_array_equal(f.attrs["shaped2"], [1.5, 2.5])
|
|
|
|
|
|
def test_attribute_types_as_h5py_reads_them(h5py, tmp_path):
|
|
path = str(tmp_path / "attrs.h5")
|
|
with h5py.File(path, "w") as f:
|
|
f.create_group("g")
|
|
values = {
|
|
"i8": 5,
|
|
"f8": 2.5,
|
|
"i1": np.int8(-3),
|
|
"u8": np.uint64(2**64 - 1),
|
|
">f4": np.array([1.5, 2.5], dtype=">f4"),
|
|
"f2": np.float16(0.5),
|
|
"b": True,
|
|
"barr": np.array([True, False]),
|
|
"c16": np.complex128(1 + 2j),
|
|
"bytes": b"raw",
|
|
"sarr": np.array([b"a", b"bcd"]),
|
|
"str": "héllo",
|
|
"strs": ["x", "yz"],
|
|
"2d": np.arange(12, dtype="<u2").reshape(3, 4),
|
|
"empty": np.zeros((0,), dtype="<i4"),
|
|
}
|
|
with clawhdf5.File(path, "r+") as f:
|
|
for k, v in values.items():
|
|
f["g"].attrs[k] = v
|
|
# Replace one, with another type and size.
|
|
f["g"].attrs["i8"] = np.arange(100.0)
|
|
with h5py.File(path, "r") as f:
|
|
a = f["g"].attrs
|
|
np.testing.assert_array_equal(a["i8"], np.arange(100.0))
|
|
assert a["f8"] == 2.5 and a["f8"].dtype == np.float64
|
|
assert a["i1"] == -3 and a["i1"].dtype == np.int8
|
|
assert a["u8"] == 2**64 - 1 and a["u8"].dtype == np.uint64
|
|
assert a[">f4"].dtype == np.dtype(">f4")
|
|
assert a["f2"].dtype == np.float16
|
|
assert a["b"] is np.True_ or a["b"] == True # noqa: E712
|
|
assert a["barr"].dtype == np.bool_
|
|
assert a["c16"] == 1 + 2j
|
|
assert a["bytes"] == b"raw"
|
|
assert list(a["sarr"]) == [b"a", b"bcd"]
|
|
# str is stored as fixed-length UTF-8: h5py reads bytes.
|
|
assert a["str"].decode("utf-8") == "héllo"
|
|
assert [x.decode() for x in a["strs"]] == ["x", "yz"]
|
|
assert a["2d"].shape == (3, 4) and a["2d"].dtype == np.dtype("<u2")
|
|
assert a["empty"].shape == (0,)
|
|
with clawhdf5.File(path, "r") as f:
|
|
assert f["g"].attrs["c16"] == 1 + 2j
|
|
assert f["g"].attrs["str"].decode("utf-8") == "héllo"
|
|
h5dump_reads(path)
|
|
|
|
|
|
def test_many_attributes_move_to_dense_storage(h5py, tmp_path):
|
|
"""Past the compact limit (8 attributes under libver v114) the object's
|
|
attributes move to dense storage; h5py reads all of them."""
|
|
path = str(tmp_path / "dense.h5")
|
|
with h5py.File(path, "w", libver="v114") as f:
|
|
f.create_dataset("d", data=np.arange(3))
|
|
with clawhdf5.File(path, "r+") as f:
|
|
for i in range(20):
|
|
f["d"].attrs[f"a{i:02d}"] = np.full(i + 1, i, dtype="<i2")
|
|
with h5py.File(path, "r") as f:
|
|
assert sorted(f["d"].attrs.keys()) == [f"a{i:02d}" for i in range(20)]
|
|
for i in range(20):
|
|
np.testing.assert_array_equal(f["d"].attrs[f"a{i:02d}"], np.full(i + 1, i))
|
|
h5dump_reads(path)
|
|
|
|
|
|
def test_reads_never_see_a_half_written_edit(h5py, tmp_path):
|
|
"""Readers on other threads while one thread rewrites a dataset: every
|
|
read returns one whole version (all elements equal), never a mix."""
|
|
path = str(tmp_path / "race.h5")
|
|
with h5py.File(path, "w") as f:
|
|
f.create_dataset("d", data=np.zeros((64, 64)), chunks=(16, 16), compression="gzip")
|
|
errors = []
|
|
with clawhdf5.File(path, "r+") as f:
|
|
stop = threading.Event()
|
|
|
|
def read():
|
|
ds = f["d"]
|
|
while not stop.is_set():
|
|
a = ds[...]
|
|
if not (a == a.flat[0]).all():
|
|
errors.append(a)
|
|
return
|
|
|
|
readers = [threading.Thread(target=read) for _ in range(3)]
|
|
for t in readers:
|
|
t.start()
|
|
try:
|
|
for k in range(1, 25):
|
|
f["d"][...] = float(k)
|
|
finally:
|
|
stop.set()
|
|
for t in readers:
|
|
t.join()
|
|
np.testing.assert_array_equal(f["d"][...], np.full((64, 64), 24.0))
|
|
assert not errors, "a read saw a partly written dataset"
|
|
|
|
|
|
def test_close_releases_the_file(h5py, tmp_path):
|
|
path = str(tmp_path / "close.h5")
|
|
with h5py.File(path, "w") as f:
|
|
f.create_dataset("d", data=np.arange(4, dtype="<i4"))
|
|
f = clawhdf5.File(path, "r+")
|
|
ds = f["d"]
|
|
f.close()
|
|
# The handle still reads the file as last written, but cannot edit it.
|
|
np.testing.assert_array_equal(ds[...], np.arange(4))
|
|
with pytest.raises(OSError, match="closed"):
|
|
ds[0] = 1
|
|
with h5py.File(path, "r+") as g:
|
|
g["d"][0] = 9
|