clawhdf5.File(path, 'r+') (and 'a' on an existing file) holds a FileEditor, and with it the file's exclusive lock, until close(): - ds[key] = value: h5py's keys and broadcasting (numpy's rules for slices and integers with extra leading 1-axes allowed; the exact shape for an index list, a scalar only where h5py expands it). Arrays are converted as libhdf5 converts them in native byte order (integers saturate, floats truncate toward zero and clip, integers go into h5py's bool enum by value); other values through numpy.asarray(value, dtype=ds.dtype), as h5py does. NaN into an integer dataset is a ValueError instead of libhdf5's arbitrary value. The value preparation is a small Python module compiled into the extension (src/edit_helpers.py). - ds.resize(shape) / ds.resize(n, axis=k) with h5py's argument rules. - attrs[name] = value, attrs.create(name, data, shape, dtype), attrs.modify: numeric, bool, complex, bytes and str data of any shape, with h5py's HDF5 types; str is stored as fixed-length UTF-8 (the editor cannot write variable-length strings). - File.mode, File.flush(), Dataset.chunks. Each edit runs with the GIL released under the file handle's write lock (no read sees a half-written edit), then the file is reopened; datasets and attrs objects re-read their shape and attributes when the handle's edit generation moved. What the editor cannot do is NotImplementedError before anything is written: deleting attributes or objects, creating datasets or groups, compound fields by name, variable-length data, and FileEditor's own limits. Where libhdf5 2.0 (h5py 3.16) converts inconsistently -- its soft conversions in non-native byte order (a float in (-1, 0) becomes the integer minimum, same-size unsigned->signed wraps) and native casts that are undefined in C (half floats into unsigned, float(max) rounded up) -- clawhdf5 saturates as libhdf5's native path does; listed in docs/known-issues.md. Tests (tests/test_edit.py): every edit applied by h5py and by clawhdf5 to copies of the same file and both read back through h5py after each edit, on h5py files (libver earliest, v114, latest) and a clawhdf5 file: a fixed sequence over every chunk index kind, compact/contiguous/gzip layouts and numeric, bool, enum, complex, string and compound types, 16 random sequences of 40 edits, and a numeric conversion matrix; a refused edit must be refused by both and leave the file unchanged. Also dense attributes, locking, objects seeing edits, readers racing a writer, and h5dump (plus h5rs check in ci-test.sh) on every edited file. The read-vs-h5py suite also runs on a file opened 'r+'. Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
678 lines
28 KiB
Python
678 lines
28 KiB
Python
"""Every read through clawhdf5 compared against h5py (libhdf5) on files h5py
|
|
writes: dtypes, shapes, values and the type of what comes back (array,
|
|
numpy scalar, bytes, str, Empty), for every datatype the bindings map and a
|
|
spread of index expressions; plus the errors h5py gives for the same keys."""
|
|
|
|
import threading
|
|
from concurrent.futures import ThreadPoolExecutor
|
|
|
|
import numpy as np
|
|
import pytest
|
|
|
|
import clawhdf5
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# The generated file
|
|
# ---------------------------------------------------------------------------
|
|
|
|
NUMERIC = [
|
|
"<i1", "<i2", "<i4", "<i8", "<u1", "<u2", "<u4", "<u8",
|
|
">i2", ">i4", ">i8", ">u2", ">u4", ">u8",
|
|
"<f2", "<f4", "<f8", ">f2", ">f4", ">f8",
|
|
]
|
|
|
|
|
|
def _values(dtype, shape, seed):
|
|
rng = np.random.default_rng(seed)
|
|
dt = np.dtype(dtype)
|
|
n = int(np.prod(shape))
|
|
if dt.kind == "f":
|
|
return rng.standard_normal(n).astype(dt).reshape(shape)
|
|
info = np.iinfo(dt)
|
|
return rng.integers(info.min, info.max, size=n, dtype=dt.newbyteorder("=")).astype(dt).reshape(shape)
|
|
|
|
|
|
def _compound_dtype():
|
|
return np.dtype(
|
|
{
|
|
"names": ["id", "pos", "label", "flag", "vec"],
|
|
"formats": ["<i4", ">f8", "S6", "u1", ("<f4", (3,))],
|
|
"offsets": [0, 8, 16, 22, 24],
|
|
"itemsize": 40,
|
|
}
|
|
)
|
|
|
|
|
|
def _nested_dtype():
|
|
inner = np.dtype([("a", "<i2"), ("b", "<f4")])
|
|
return np.dtype([("x", "<u8"), ("inner", inner), ("c", "<c8")])
|
|
|
|
|
|
def _write_fixture(h5py, path):
|
|
str_dt = h5py.string_dtype()
|
|
ascii_dt = h5py.string_dtype("ascii")
|
|
with h5py.File(path, "w") as f:
|
|
# Numeric types, both byte orders, 1-D contiguous and 2-D chunked+gzip.
|
|
for i, dt in enumerate(NUMERIC):
|
|
name = dt.replace("<", "le_").replace(">", "be_")
|
|
f.create_dataset(f"num/{name}_1d", data=_values(dt, (37,), i))
|
|
f.create_dataset(
|
|
f"num/{name}_2d_gzip",
|
|
data=_values(dt, (13, 11), 100 + i),
|
|
chunks=(4, 5),
|
|
compression="gzip",
|
|
)
|
|
f.create_dataset("num/f8_3d", data=_values("<f8", (6, 7, 5), 7), chunks=(2, 3, 5))
|
|
f.create_dataset(
|
|
"num/i4_3d_shuffle",
|
|
data=_values("<i4", (5, 9, 4), 8),
|
|
chunks=(3, 4, 2),
|
|
shuffle=True,
|
|
fletcher32=True,
|
|
compression="gzip",
|
|
)
|
|
f.create_dataset("num/scalar_f8", data=np.float64(3.25))
|
|
f.create_dataset("num/scalar_i2_be", data=np.array(-7, dtype=">i2"))
|
|
f.create_dataset("num/zero_size", shape=(0, 3), dtype="<f4")
|
|
f.create_dataset("num/empty", data=h5py.Empty("<f8"))
|
|
sparse = f.create_dataset("num/sparse_fill", shape=(40,), chunks=(8,), dtype="<i4", fillvalue=-3)
|
|
sparse[10:14] = [1, 2, 3, 4]
|
|
f.create_dataset("num/resizable", data=np.arange(12.0), maxshape=(None,), chunks=(5,))
|
|
# Compact layout (low level: h5py's create_dataset cannot ask for it).
|
|
dcpl = h5py.h5p.create(h5py.h5p.DATASET_CREATE)
|
|
dcpl.set_layout(h5py.h5d.COMPACT)
|
|
space = h5py.h5s.create_simple((9,))
|
|
dsid = h5py.h5d.create(f.id, b"num/compact_u2", h5py.h5t.py_create(np.dtype("<u2")), space, dcpl=dcpl)
|
|
dsid.write(h5py.h5s.ALL, h5py.h5s.ALL, np.arange(9, dtype="<u2") * 7)
|
|
|
|
# bool, enum, complex.
|
|
f.create_dataset("misc/bool", data=np.array([True, False, True, True, False]))
|
|
enum_dt = h5py.enum_dtype({"RED": 0, "GREEN": 1, "BLUE": 42}, basetype="i2")
|
|
f.create_dataset("misc/enum", data=np.array([0, 42, 1, 1, 0], dtype="<i2"), dtype=enum_dt)
|
|
uenum_dt = h5py.enum_dtype({"LO": 0, "HI": 2**40}, basetype="u8")
|
|
f.create_dataset("misc/enum_u8", data=np.array([0, 2**40], dtype="<u8"), dtype=uenum_dt)
|
|
f.create_dataset("misc/c8", data=(np.arange(6) + 1j * np.arange(6)).astype("<c8"))
|
|
f.create_dataset("misc/c16_2d", data=(np.arange(12) - 2j).reshape(3, 4).astype("<c16"))
|
|
f.create_dataset("misc/opaque", data=np.array([b"\x00\x01\x02", b"\xff\xfe\xfd"], dtype="V3"))
|
|
|
|
# Strings.
|
|
f.create_dataset("str/fixed", data=np.array([b"alpha", b"be", b"", b"gamma!"], dtype="S6"))
|
|
utf8_4 = h5py.string_dtype("utf-8", 4)
|
|
f.create_dataset(
|
|
"str/fixed_2d_utf8",
|
|
data=np.array([["é".encode(), b"b"], [b"c", b"dd"]], dtype=utf8_4),
|
|
)
|
|
f.create_dataset("str/vlen", data=["", "one", "twø", "a" * 300, "ünïcödé"], dtype=str_dt)
|
|
f.create_dataset("str/vlen_ascii", data=[b"x", b"yy", b"zzz"], dtype=ascii_dt)
|
|
f.create_dataset(
|
|
"str/vlen_2d_gzip",
|
|
data=np.array([[f"r{r}c{c}" * (r + c) for c in range(5)] for r in range(6)], dtype=object),
|
|
dtype=str_dt,
|
|
chunks=(2, 2),
|
|
compression="gzip",
|
|
)
|
|
f.create_dataset("str/vlen_scalar", data="just one", dtype=str_dt)
|
|
|
|
# Variable-length sequences.
|
|
vl_i = h5py.vlen_dtype(np.dtype("<i4"))
|
|
seqs = np.empty(4, dtype=object)
|
|
seqs[:] = [np.arange(3, dtype="<i4"), np.array([], dtype="<i4"), np.arange(10, dtype="<i4") * -1, np.array([7], dtype="<i4")]
|
|
f.create_dataset("vlen/i4", data=seqs, dtype=vl_i)
|
|
vl_f = h5py.vlen_dtype(np.dtype(">f8"))
|
|
fseqs = np.empty(3, dtype=object)
|
|
fseqs[:] = [np.linspace(0, 1, 5).astype(">f8"), np.array([2.5], dtype=">f8"), np.array([], dtype=">f8")]
|
|
f.create_dataset("vlen/f8_be", data=fseqs, dtype=vl_f)
|
|
|
|
# Compounds.
|
|
cdt = _compound_dtype()
|
|
rec = np.zeros(10, dtype=cdt)
|
|
rec["id"] = np.arange(10) * 3
|
|
rec["pos"] = np.linspace(-1, 1, 10)
|
|
rec["label"] = [f"n{i}".encode() for i in range(10)]
|
|
rec["flag"] = np.arange(10) % 2
|
|
rec["vec"] = np.arange(30, dtype="<f4").reshape(10, 3)
|
|
f.create_dataset("cmp/padded", data=rec)
|
|
f.create_dataset("cmp/padded_chunked", data=rec, chunks=(3,), compression="gzip")
|
|
ndt = _nested_dtype()
|
|
nrec = np.zeros((4, 3), dtype=ndt)
|
|
nrec["x"] = np.arange(12).reshape(4, 3)
|
|
nrec["inner"]["a"] = -np.arange(12).reshape(4, 3)
|
|
nrec["inner"]["b"] = np.arange(12).reshape(4, 3) / 4
|
|
nrec["c"] = np.arange(12).reshape(4, 3) * (1 + 1j)
|
|
f.create_dataset("cmp/nested_2d", data=nrec)
|
|
vdt = np.dtype([("n", "<i4"), ("s", h5py.string_dtype())])
|
|
vrec = np.array([(1, "a"), (2, "bb")], dtype=vdt)
|
|
f.create_dataset("cmp/with_vlen", data=vrec)
|
|
|
|
# A true HDF5 array datatype (h5py's high level would widen the shape).
|
|
tid = h5py.h5t.array_create(h5py.h5t.py_create(np.dtype("<i4")), (2, 3))
|
|
space = h5py.h5s.create_simple((4,))
|
|
dsid = h5py.h5d.create(f.id, b"cmp/array_type", tid, space)
|
|
dsid.write(h5py.h5s.ALL, h5py.h5s.ALL, np.arange(24, dtype="<i4").reshape(4, 2, 3), mtype=tid)
|
|
|
|
# Unsupported: references.
|
|
f.create_dataset("unsupported/refs", data=[f["num"].ref, f["misc"].ref], dtype=h5py.ref_dtype)
|
|
|
|
# Groups and attributes.
|
|
g = f.create_group("deep/er/est")
|
|
g.create_dataset("leaf", data=np.arange(5))
|
|
f.create_group("empty_group")
|
|
f.attrs["i4"] = np.int32(-5)
|
|
f.attrs["u8"] = np.uint64(2**63 + 1)
|
|
f.attrs["f2"] = np.float16(1.5)
|
|
f.attrs["f8_arr"] = np.array([1.0, 2.5, -3.0])
|
|
f.attrs["f8_one"] = np.array([4.0])
|
|
f.attrs["i2_2d_be"] = np.arange(6, dtype=">i2").reshape(2, 3)
|
|
f.attrs["bool"] = True
|
|
f.attrs["bool_arr"] = np.array([True, False])
|
|
f.attrs["vstr"] = "héllo"
|
|
f.attrs["vstr_arr"] = ["a", "bcd", ""]
|
|
f.attrs["fstr"] = np.bytes_(b"fixed")
|
|
f.attrs["fstr_arr"] = np.array([b"x", b"yz"], dtype="S2")
|
|
f.attrs["empty"] = h5py.Empty("<i4")
|
|
f.attrs["complex"] = np.complex128(1 - 2j)
|
|
f.attrs["compound"] = np.array((7, 2.5), dtype=[("a", "<i2"), ("b", "<f8")])
|
|
f.attrs["enum"] = np.array(42, dtype=enum_dt)
|
|
f["num/le_f8_1d"].attrs["units"] = "m/s"
|
|
f["num/le_f8_1d"].attrs["scale"] = np.float32(0.5)
|
|
g.attrs["depth"] = np.int8(3)
|
|
|
|
|
|
@pytest.fixture(scope="module", params=["local", "r+", "http", "http-1k-blocks"])
|
|
def pair(request, h5py, tmp_path_factory):
|
|
"""The fixture file through h5py and through clawhdf5: opened locally
|
|
(read-only, and for editing: a copy, since editing locks the file), and
|
|
over HTTP range requests (a local server in this process) with the
|
|
default 1 MiB blocks and with 1 KiB blocks, so every structure is read
|
|
through many small ranges."""
|
|
import shutil
|
|
|
|
from conftest import RangeServer
|
|
|
|
root = tmp_path_factory.mktemp("h5")
|
|
path = str(root / "fixture.h5")
|
|
_write_fixture(h5py, path)
|
|
theirs = h5py.File(path, "r")
|
|
server = None
|
|
if request.param == "local":
|
|
ours = clawhdf5.File(path, "r")
|
|
elif request.param == "r+":
|
|
copy = str(root / "editable.h5")
|
|
shutil.copy(path, copy)
|
|
ours = clawhdf5.File(copy, "r+")
|
|
else:
|
|
server = RangeServer(root)
|
|
if request.param == "http":
|
|
ours = clawhdf5.File(server.url("fixture.h5"))
|
|
else:
|
|
ours = clawhdf5.File.open_url(server.url("fixture.h5"), block_size=1024)
|
|
yield ours, theirs, path
|
|
theirs.close()
|
|
ours.close()
|
|
if server is not None:
|
|
server.close()
|
|
|
|
|
|
def _all_datasets(h5py, f):
|
|
names = []
|
|
f.visititems(lambda n, o: names.append(n) if isinstance(o, h5py.Dataset) else None)
|
|
return sorted(names)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Comparison helpers
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def assert_same(ours, theirs, what=""):
|
|
if type(theirs).__name__ == "Empty":
|
|
assert isinstance(ours, clawhdf5.Empty), what
|
|
assert ours.dtype == theirs.dtype, what
|
|
return
|
|
assert type(ours) is type(theirs), f"{what}: {type(ours)} vs {type(theirs)}"
|
|
if isinstance(theirs, np.ndarray):
|
|
assert ours.shape == theirs.shape, what
|
|
assert ours.dtype == theirs.dtype, f"{what}: {ours.dtype} vs {theirs.dtype}"
|
|
if theirs.dtype == object:
|
|
for a, b in zip(ours.ravel(), theirs.ravel()):
|
|
assert_same(a, b, what)
|
|
elif theirs.dtype.kind == "V":
|
|
# Structured and opaque: every byte, padding included (h5py's
|
|
# padding is zero; uninitialised memory there would leak).
|
|
if theirs.dtype.names is not None:
|
|
np.testing.assert_array_equal(ours, theirs, err_msg=what)
|
|
assert ours.tobytes() == theirs.tobytes(), f"{what}: bytes differ"
|
|
else:
|
|
np.testing.assert_array_equal(ours, theirs, err_msg=what)
|
|
elif isinstance(theirs, np.generic):
|
|
assert ours.dtype == theirs.dtype, what
|
|
if theirs.dtype.names is not None:
|
|
np.testing.assert_array_equal(np.asarray(ours), np.asarray(theirs), err_msg=what)
|
|
assert ours.tobytes() == theirs.tobytes(), f"{what}: bytes differ"
|
|
else:
|
|
assert ours == theirs or (ours != ours and theirs != theirs), what
|
|
else:
|
|
assert ours == theirs, what
|
|
|
|
|
|
def keys_for(shape):
|
|
if shape == ():
|
|
return [(), Ellipsis]
|
|
keys = [(), Ellipsis, 0, -1, slice(None), slice(None, None, 2), slice(1, None, 3), slice(0, 0), np.int64(0),
|
|
np.array(0), np.array(-1, dtype="i1")]
|
|
n0 = shape[0]
|
|
if n0 == 0:
|
|
return [(), Ellipsis, slice(None), slice(None, None, 2), slice(0, 0)]
|
|
keys += [slice(n0 // 2, None), slice(-3, None), [0, n0 - 1] if n0 > 1 else [0], (Ellipsis,)]
|
|
if n0 > 5:
|
|
keys += [[1, 2, 3, 5], [0, 4, 5]]
|
|
if len(shape) >= 2:
|
|
n1 = shape[1]
|
|
keys += [
|
|
(0, 0),
|
|
(-1, -1),
|
|
(slice(1, 3), slice(None, None, 2)),
|
|
(Ellipsis, 1),
|
|
(1, Ellipsis),
|
|
(slice(None), [0, n1 - 1] if n1 > 1 else [0]),
|
|
(slice(None, None, 2), 1),
|
|
(slice(0, 2), slice(3, 1)),
|
|
(np.array(1) if n0 > 1 else np.array(0), slice(None)),
|
|
(slice(None), np.array(n1 - 1, dtype="u2")),
|
|
]
|
|
if len(shape) >= 3:
|
|
keys += [(0, slice(None), -1), (slice(1, None, 2), 2, slice(None, None, 3)), (Ellipsis, 0, 0), (0, Ellipsis, 1)]
|
|
return keys
|
|
|
|
|
|
# h5py 3.16 (HDF5 2.0) returns the elements of a variable-length sequence of
|
|
# big-endian floats unswapped (0.25 comes back as 2.6e-319), so these are
|
|
# checked against the values written instead (test_vlen_big_endian).
|
|
H5PY_MISREADS = {"vlen/f8_be"}
|
|
|
|
ERROR_KEYS_1D = [
|
|
slice(None, None, -1), 10**6, -(10**6), None, (0, 0, 0, 0, 0), [3, 1], (Ellipsis, Ellipsis), 1.5, "nope",
|
|
np.array(1.0), np.array(True), np.array(10**6), [0, 0], [], (), Ellipsis, (0,), [-1], np.array([1, 2]),
|
|
]
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Tests
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def test_every_dataset_matches_h5py(h5py, pair):
|
|
ours, theirs, _ = pair
|
|
checked = 0
|
|
for name in _all_datasets(h5py, theirs):
|
|
if name.startswith("unsupported/") or name in H5PY_MISREADS or name == "cmp/with_vlen":
|
|
continue
|
|
t = theirs[name]
|
|
o = ours[name]
|
|
assert o.shape == t.shape, name
|
|
assert o.dtype == t.dtype, f"{name}: {o.dtype} vs {t.dtype}"
|
|
assert dict(o.dtype.metadata or {}) == dict(t.dtype.metadata or {}), name
|
|
assert o.ndim == t.ndim and o.size == t.size, name
|
|
assert o.maxshape == t.maxshape, name
|
|
assert o.name == t.name, name
|
|
if t.shape is None:
|
|
assert_same(o[()], t[()], name)
|
|
continue
|
|
for key in keys_for(t.shape):
|
|
what = f"{name}[{key!r}]"
|
|
try:
|
|
expected = t[key]
|
|
except Exception as e: # noqa: BLE001 - h5py refuses: so must we
|
|
with pytest.raises(type(e)):
|
|
o[key]
|
|
continue
|
|
assert_same(o[key], expected, what)
|
|
checked += 1
|
|
assert checked > 500
|
|
|
|
|
|
def test_vlen_big_endian(pair):
|
|
ours, _, _ = pair
|
|
ds = ours["vlen/f8_be"]
|
|
assert ds.dtype.metadata["vlen"] == np.dtype(">f8")
|
|
got = ds[()]
|
|
expected = [np.linspace(0, 1, 5), np.array([2.5]), np.array([])]
|
|
assert got.shape == (3,)
|
|
for g, e in zip(got, expected):
|
|
assert g.dtype == np.dtype(">f8")
|
|
np.testing.assert_array_equal(g, e)
|
|
np.testing.assert_array_equal(ds[1], [2.5])
|
|
|
|
|
|
def test_errors_match_h5py(h5py, pair):
|
|
ours, theirs, _ = pair
|
|
for name in ["num/le_i4_1d", "num/be_f8_2d_gzip", "str/vlen", "num/scalar_f8"]:
|
|
for key in ERROR_KEYS_1D:
|
|
try:
|
|
expected = theirs[name][key]
|
|
except Exception as e: # noqa: BLE001
|
|
with pytest.raises(type(e)):
|
|
ours[name][key]
|
|
else:
|
|
# h5py reads it, so must we (and the same values).
|
|
assert_same(ours[name][key], expected, f"{name}[{key!r}]")
|
|
|
|
|
|
def test_compound_fields_match_h5py(pair):
|
|
ours, theirs, _ = pair
|
|
for name in ["cmp/padded", "cmp/padded_chunked", "cmp/nested_2d"]:
|
|
t, o = theirs[name], ours[name]
|
|
for field in t.dtype.names:
|
|
assert_same(o[field], t[field], f"{name}[{field}]")
|
|
assert_same(o[field, 1:3], t[field, 1:3], f"{name}[{field}, 1:3]")
|
|
two = list(t.dtype.names[:2])
|
|
expected = t[tuple(two)]
|
|
got = o[tuple(two)]
|
|
assert got.dtype.names == expected.dtype.names
|
|
for field in two:
|
|
np.testing.assert_array_equal(got[field], expected[field])
|
|
with pytest.raises(ValueError):
|
|
ours["cmp/padded"]["no_such_field"]
|
|
with pytest.raises(ValueError):
|
|
ours["num/le_i4_1d"]["id"]
|
|
|
|
|
|
def test_numpy_asarray_and_len(pair):
|
|
ours, theirs, _ = pair
|
|
assert_same(np.asarray(ours["num/f8_3d"]), np.asarray(theirs["num/f8_3d"]))
|
|
assert len(ours["num/f8_3d"]) == len(theirs["num/f8_3d"])
|
|
with pytest.raises(TypeError):
|
|
len(ours["num/scalar_f8"])
|
|
|
|
|
|
def test_attributes_match_h5py(h5py, pair):
|
|
ours, theirs, _ = pair
|
|
for path in ["/", "num/le_f8_1d", "deep/er/est"]:
|
|
t = theirs[path].attrs
|
|
o = ours[path].attrs
|
|
assert list(o.keys()) == sorted(t.keys()), path
|
|
assert len(o) == len(t)
|
|
for k in t.keys():
|
|
assert k in o
|
|
assert_same(o[k], t[k], f"{path}.attrs[{k}]")
|
|
assert o.get("missing", 5) == 5
|
|
with pytest.raises(KeyError):
|
|
o["missing"]
|
|
assert [k for k, _ in o.items()] == list(o.keys())
|
|
|
|
|
|
def test_groups_match_h5py(h5py, pair):
|
|
ours, theirs, _ = pair
|
|
assert list(ours.keys()) == list(theirs.keys())
|
|
assert len(ours) == len(theirs)
|
|
for path in ["num", "deep", "deep/er", "deep/er/est", "empty_group"]:
|
|
o, t = ours[path], theirs[path]
|
|
assert isinstance(o, clawhdf5.Group)
|
|
assert list(o.keys()) == list(t.keys()), path
|
|
assert list(o) == list(t), path
|
|
assert len(o) == len(t), path
|
|
assert o.name == t.name
|
|
g = ours["deep/er"]
|
|
assert isinstance(g["est"], clawhdf5.Group)
|
|
assert isinstance(g["est/leaf"], clawhdf5.Dataset)
|
|
assert g["/deep/er/est/leaf"].name == "/deep/er/est/leaf"
|
|
assert "est/leaf" in g and "/num" in g and "nope" not in g
|
|
assert ours["/"].name == "/"
|
|
assert "deep/er/est/leaf" in ours
|
|
assert ours.get("nope") is None
|
|
with pytest.raises(KeyError):
|
|
ours["deep/nope"]
|
|
names = [k for k, v in ours["deep/er/est"].items()]
|
|
assert names == ["leaf"]
|
|
assert isinstance(ours["deep/er/est"].values()[0], clawhdf5.Dataset)
|
|
|
|
|
|
def test_unsupported_types_are_errors_not_data(pair):
|
|
ours, _, _ = pair
|
|
ds = ours["unsupported/refs"] # opening works
|
|
with pytest.raises(TypeError):
|
|
ds.dtype
|
|
with pytest.raises(TypeError):
|
|
ds[()]
|
|
with pytest.raises(TypeError):
|
|
ours["cmp/with_vlen"][()]
|
|
|
|
|
|
def test_boolean_masks_are_refused(pair):
|
|
ours, _, _ = pair
|
|
with pytest.raises(TypeError):
|
|
ours["num/le_i4_1d"][np.ones(37, dtype=bool)]
|
|
|
|
|
|
def test_reads_hand_numpy_the_rust_buffer(pair):
|
|
"""A fixed-size read is a view over the buffer the library filled, not a
|
|
copy of it."""
|
|
ours, _, _ = pair
|
|
arr = ours["num/le_f8_2d_gzip"][2:9, 1:4]
|
|
assert not arr.flags.owndata
|
|
assert arr.base is not None
|
|
assert arr.flags.aligned and arr.flags.c_contiguous
|
|
|
|
|
|
def test_only_the_selected_chunks_are_read(h5py, tmp_path):
|
|
"""Damage one chunk: a selection that avoids it still reads, one that
|
|
touches it fails. Reading everything and slicing afterwards (what the
|
|
bindings did before) failed both. (The library reads the whole dataset
|
|
anyway when a selection's bounding box covers more than half of it, so
|
|
the selections here stay below that.)"""
|
|
path = str(tmp_path / "damaged.h5")
|
|
data = np.arange(1000, dtype="<f8")
|
|
with h5py.File(path, "w") as f:
|
|
f.create_dataset("d", data=data, chunks=(100,), compression="gzip")
|
|
with h5py.File(path, "r") as f:
|
|
info = f["d"].id.get_chunk_info(9) # the last chunk
|
|
with open(path, "r+b") as fh:
|
|
fh.seek(info.byte_offset)
|
|
fh.write(b"\xff" * info.size)
|
|
with clawhdf5.File(path, "r") as f:
|
|
ds = f["d"]
|
|
np.testing.assert_array_equal(ds[0:400], data[0:400])
|
|
np.testing.assert_array_equal(ds[805:900], data[805:900])
|
|
np.testing.assert_array_equal(ds[5:450:7], data[5:450:7])
|
|
np.testing.assert_array_equal(ds[[3, 450, 899]], data[[3, 450, 899]])
|
|
with pytest.raises(Exception):
|
|
ds[950]
|
|
with pytest.raises(Exception):
|
|
ds[:]
|
|
|
|
|
|
def test_threads_read_the_same_file(pair):
|
|
"""Reads from many threads at once return exactly what h5py returns."""
|
|
ours, theirs, _ = pair
|
|
names = ["num/le_f8_2d_gzip", "num/i4_3d_shuffle", "str/vlen_2d_gzip", "cmp/padded_chunked", "num/be_i8_1d"]
|
|
expected = {n: theirs[n][()] for n in names}
|
|
errors = []
|
|
barrier = threading.Barrier(8)
|
|
|
|
def work(i):
|
|
barrier.wait()
|
|
for r in range(40):
|
|
n = names[(i + r) % len(names)]
|
|
got = ours[n][()]
|
|
try:
|
|
assert_same(got, expected[n], n)
|
|
except AssertionError as e:
|
|
errors.append(e)
|
|
|
|
with ThreadPoolExecutor(8) as pool:
|
|
list(pool.map(work, range(8)))
|
|
assert not errors, errors[:3]
|
|
|
|
|
|
def test_reads_release_the_gil(h5py, tmp_path):
|
|
"""While one thread is inside a long read, another Python thread keeps
|
|
running. With the GIL held for the read, the other thread would stall
|
|
for the whole read; the test measures its longest stall."""
|
|
import sys
|
|
import time
|
|
|
|
path = str(tmp_path / "gil.h5")
|
|
data = np.arange(2048 * 4096, dtype="<f4").reshape(2048, 4096)
|
|
with h5py.File(path, "w") as f:
|
|
f.create_dataset("d", data=data, chunks=(64, 4096), compression="gzip", compression_opts=1)
|
|
ds = clawhdf5.File(path, "r")["d"]
|
|
key = (slice(0, 1000), slice(None)) # under half: the uncached selection path
|
|
t0 = time.perf_counter()
|
|
ds[key]
|
|
one_read = time.perf_counter() - t0
|
|
assert one_read > 0.03, f"a read took only {one_read:.3f} s; too short to measure"
|
|
|
|
old = sys.getswitchinterval()
|
|
sys.setswitchinterval(0.001)
|
|
stop = threading.Event()
|
|
gaps = []
|
|
|
|
def spin():
|
|
last = time.perf_counter()
|
|
worst = 0.0
|
|
while not stop.is_set():
|
|
now = time.perf_counter()
|
|
worst = max(worst, now - last)
|
|
last = now
|
|
gaps.append(worst)
|
|
|
|
try:
|
|
t = threading.Thread(target=spin)
|
|
t.start()
|
|
time.sleep(0.01)
|
|
t0 = time.perf_counter()
|
|
for _ in range(2):
|
|
ds[key]
|
|
reading = time.perf_counter() - t0
|
|
stop.set()
|
|
t.join()
|
|
finally:
|
|
sys.setswitchinterval(old)
|
|
np.testing.assert_array_equal(ds[key], data[:1000])
|
|
# Held, the spinner would stall for about one read.
|
|
assert gaps[0] < one_read / 3, f"spinner stalled {gaps[0]:.3f} s during reads of {one_read:.3f} s ({reading:.3f} s)"
|
|
|
|
|
|
def _v4_index_fixture(h5py, path):
|
|
"""One 2-D dataset per v4 chunk index (HDF5 1.10+ layout, libver='latest')."""
|
|
data = (np.arange(37 * 23, dtype="<i4") * 7 - 1000).reshape(37, 23)
|
|
early = h5py.h5p.create(h5py.h5p.DATASET_CREATE)
|
|
early.set_alloc_time(h5py.h5d.ALLOC_TIME_EARLY)
|
|
with h5py.File(path, "w", libver="latest") as f:
|
|
f.create_dataset("implicit", data=data, chunks=(5, 4), dcpl=early)
|
|
f.create_dataset("fixed_array", data=data, chunks=(5, 4), compression="gzip")
|
|
f.create_dataset("extensible_array", data=data, chunks=(5, 4), maxshape=(None, 23), compression="gzip")
|
|
f.create_dataset("btree2", data=data, chunks=(5, 4), maxshape=(None, None), compression="gzip")
|
|
f.create_dataset("single_chunk", data=data, chunks=(37, 23), compression="gzip")
|
|
|
|
|
|
V4_KEYS = [
|
|
slice(0, 3), slice(0, 30), (slice(7, 16), slice(3, 9)), (36, 22), (slice(None, None, 3), slice(1, None, 4)),
|
|
(slice(1, None, 2), Ellipsis), (Ellipsis, slice(2, 22)), [0, 5, 6, 36], (slice(None), [0, 3, 22]), -1, (),
|
|
]
|
|
|
|
|
|
def test_every_v4_chunk_index_matches_h5py(h5py, tmp_path):
|
|
"""Partial reads of each v4 chunk index. The implicit index (early
|
|
allocation, no filters) used to panic in the library for any selection
|
|
covering more than half the dataset, e.g. ds[0:30]."""
|
|
path = str(tmp_path / "v4.h5")
|
|
_v4_index_fixture(h5py, path)
|
|
with h5py.File(path, "r") as theirs, clawhdf5.File(path, "r") as ours:
|
|
for name in theirs:
|
|
for key in V4_KEYS:
|
|
assert_same(ours[name][key], theirs[name][key], f"{name}[{key!r}]")
|
|
|
|
|
|
def test_a_library_panic_is_an_ordinary_exception():
|
|
"""PyO3 turns a Rust panic into PanicException, a BaseException that
|
|
`except Exception` does not catch. Every call into the library is
|
|
guarded, so a panic surfaces as clawhdf5.InternalError instead."""
|
|
assert issubclass(clawhdf5.InternalError, RuntimeError)
|
|
with pytest.raises(clawhdf5.InternalError, match="deliberate panic"):
|
|
clawhdf5._panic_for_test()
|
|
try:
|
|
clawhdf5._panic_for_test()
|
|
except Exception: # noqa: BLE001 - the point of the test
|
|
pass
|
|
|
|
|
|
def test_compound_padding_bytes_match_h5py(pair):
|
|
"""Every byte of a padded compound, padding included, is h5py's, for
|
|
index lists with many runs as well as slices. Joining the runs with
|
|
np.concatenate left the padding uninitialised: process memory ended up
|
|
in tobytes()."""
|
|
ours, theirs, _ = pair
|
|
keys = [[0, 3, 6], [1, 2, 5, 9], [0, 2, 4, 6, 8], slice(None), slice(1, 9, 3), 4, [9]]
|
|
for name in ["cmp/padded", "cmp/padded_chunked"]:
|
|
for _ in range(20): # garbage varies between runs; zeros do not
|
|
for key in keys:
|
|
assert ours[name][key].tobytes() == theirs[name][key].tobytes(), f"{name}[{key!r}]"
|
|
for key in [(slice(None), [0, 2]), ([0, 2, 3], slice(None)), ([1, 3], 1)]:
|
|
assert ours["cmp/nested_2d"][key].tobytes() == theirs["cmp/nested_2d"][key].tobytes(), key
|
|
|
|
|
|
def test_a_long_index_list_decodes_each_chunk_once(h5py, tmp_path):
|
|
"""An index list is read one group of chunks at a time, not one
|
|
hyperslab per run of indices: 5000 runs over 20 gzip chunks used to
|
|
decode the chunks 5000 times (8 s, against h5py's 0.014 s)."""
|
|
import time
|
|
|
|
path = str(tmp_path / "long_list.h5")
|
|
data = np.arange(200000, dtype="<f8")
|
|
grid = np.arange(400 * 3000, dtype="<i4").reshape(400, 3000)
|
|
with h5py.File(path, "w") as f:
|
|
f.create_dataset("d", data=data, chunks=(10000,), compression="gzip")
|
|
f.create_dataset("grid", data=grid, chunks=(50, 100), compression="gzip")
|
|
f.create_dataset("flat", data=data) # contiguous
|
|
rng = np.random.default_rng(3)
|
|
cases = [
|
|
("d", list(range(0, 200000, 40))),
|
|
("d", sorted(rng.choice(200000, 3000, replace=False).tolist())),
|
|
("flat", list(range(0, 200000, 40))),
|
|
("flat", [0, 7, 199999]),
|
|
("grid", (slice(None), list(range(0, 3000, 3)))),
|
|
("grid", (sorted(rng.choice(400, 150, replace=False).tolist()), slice(5, 2900, 7))),
|
|
("grid", (7, [0, 1, 2, 2000, 2999])),
|
|
]
|
|
with h5py.File(path, "r") as theirs, clawhdf5.File(path, "r") as ours:
|
|
for name, key in cases:
|
|
t0 = time.perf_counter()
|
|
got = ours[name][key]
|
|
took = time.perf_counter() - t0
|
|
assert_same(got, theirs[name][key], f"{name}[{len(key)}-key]")
|
|
assert took < 2.0, f"{name}: {took:.2f} s"
|
|
|
|
|
|
@pytest.mark.parametrize("libver", ["earliest", "latest"])
|
|
def test_big_groups_are_not_quadratic(h5py, tmp_path, libver):
|
|
"""A dataset or group remembers where its object is, and a group its
|
|
links, so reads and walks over a large group do not resolve every path
|
|
from the root again (it was O(n) per access: O(n^2) to visit a group)."""
|
|
import time
|
|
|
|
path = str(tmp_path / f"big_{libver}.h5")
|
|
n = 4000
|
|
with h5py.File(path, "w", libver=libver) as f:
|
|
g = f.create_group("g")
|
|
for i in range(n):
|
|
g.create_dataset(f"d{i:05d}", data=np.int32(i))
|
|
g.create_group("sub").create_dataset("leaf", data=np.arange(3))
|
|
with h5py.File(path, "r") as theirs, clawhdf5.File(path, "r") as ours:
|
|
t0 = time.perf_counter()
|
|
g = ours["g"]
|
|
assert list(g.keys()) == list(theirs["g"].keys())
|
|
total = sum(int(v[()]) for k, v in g.items() if k.startswith("d"))
|
|
assert total == n * (n - 1) // 2
|
|
seen = 0
|
|
for k in g:
|
|
if k.startswith("d"):
|
|
seen += int(g[k][()]) == int(k[1:])
|
|
assert seen == n
|
|
ds = g["d00007"]
|
|
assert all(ds[()] == 7 for _ in range(2000))
|
|
assert list(g["sub"]["leaf"][:]) == [0, 1, 2]
|
|
assert ours["/g/sub/leaf"][1] == 1 and g["/g/d00003"][()] == 3
|
|
took = time.perf_counter() - t0
|
|
assert took < 5.0, f"{took:.2f} s"
|