- Datatype::Complex serializes class 11 version 5 byte-identically to
libhdf5 2.2.0; containers holding it are written as version 5.
- DatasetBuilder::with_complex_f32/f64_data (h5py's {r, i} compound,
default) and with_native_complex_f32/f64_data (class 11, opt-in);
make_(native_)complex_f32/f64_type for attributes.
- Dataset::read_complex_f64/f32 read either form.
- Python create_dataset accepts complex64/complex128 (compound form).
- Parsing unchanged: class 11 still surfaces as {r, i}.
- Tests vs h5py 3.16 / libhdf5 2.0.0 and h5dump 2.2.0; docs.
Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
381 lines
13 KiB
Python
381 lines
13 KiB
Python
"""Tests for clawhdf5 Python bindings."""
|
|
|
|
import os
|
|
import tempfile
|
|
|
|
import numpy as np
|
|
import pytest
|
|
|
|
import clawhdf5
|
|
|
|
|
|
@pytest.fixture
|
|
def tmp_h5(tmp_path):
|
|
"""Return a temporary HDF5 file path."""
|
|
return str(tmp_path / "test.h5")
|
|
|
|
|
|
@pytest.fixture
|
|
def sample_read_file(tmp_h5):
|
|
"""Create a sample HDF5 file for reading tests."""
|
|
with clawhdf5.File(tmp_h5, "w") as f:
|
|
f.create_dataset("temperatures", data=np.array([22.5, 23.1, 21.8]))
|
|
f.create_dataset("counts", data=np.array([10, 20, 30], dtype=np.int32))
|
|
f.attrs["version"] = 1
|
|
f.attrs["description"] = "test file"
|
|
return tmp_h5
|
|
|
|
|
|
@pytest.fixture
|
|
def grouped_read_file(tmp_h5):
|
|
"""Create an HDF5 file with groups for reading tests."""
|
|
with clawhdf5.File(tmp_h5, "w") as f:
|
|
f.create_dataset("root_data", data=np.array([0.0, 1.0]))
|
|
grp = f.create_group("sensors")
|
|
grp.create_dataset("temperature", data=np.array([22.5, 23.1, 21.8]))
|
|
grp.create_dataset("humidity", data=np.array([45, 50, 55], dtype=np.int32))
|
|
grp.attrs["location"] = "lab"
|
|
grp2 = f.create_group("metadata")
|
|
grp2.create_dataset("timestamps", data=np.array([1000, 2000, 3000], dtype=np.int64))
|
|
return tmp_h5
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Test: open and read datasets
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def test_open_and_read_f64(sample_read_file):
|
|
f = clawhdf5.File(sample_read_file, "r")
|
|
ds = f["temperatures"]
|
|
data = ds[:]
|
|
np.testing.assert_array_almost_equal(data, [22.5, 23.1, 21.8])
|
|
f.close()
|
|
|
|
|
|
def test_open_and_read_i32(sample_read_file):
|
|
f = clawhdf5.File(sample_read_file, "r")
|
|
ds = f["counts"]
|
|
data = ds[:]
|
|
np.testing.assert_array_equal(data, [10, 20, 30])
|
|
assert data.dtype == np.int32
|
|
f.close()
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Test: dataset properties (shape, dtype)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def test_dataset_shape(sample_read_file):
|
|
with clawhdf5.File(sample_read_file, "r") as f:
|
|
ds = f["temperatures"]
|
|
assert ds.shape == (3,)
|
|
|
|
|
|
def test_dataset_dtype(sample_read_file):
|
|
with clawhdf5.File(sample_read_file, "r") as f:
|
|
assert f["temperatures"].dtype == "float64"
|
|
assert f["counts"].dtype == "int32"
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Test: read attributes
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def test_read_root_attrs(sample_read_file):
|
|
with clawhdf5.File(sample_read_file, "r") as f:
|
|
assert f.attrs["version"] == 1
|
|
assert f.attrs["description"] == b"test file" # fixed-length string: numpy.bytes_, as in h5py
|
|
|
|
|
|
def test_attrs_len(sample_read_file):
|
|
with clawhdf5.File(sample_read_file, "r") as f:
|
|
assert len(f.attrs) >= 2
|
|
|
|
|
|
def test_attrs_contains(sample_read_file):
|
|
with clawhdf5.File(sample_read_file, "r") as f:
|
|
assert "version" in f.attrs
|
|
assert "nonexistent" not in f.attrs
|
|
|
|
|
|
def test_attrs_keys(sample_read_file):
|
|
with clawhdf5.File(sample_read_file, "r") as f:
|
|
keys = f.attrs.keys()
|
|
assert "version" in keys
|
|
assert "description" in keys
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Test: read groups
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def test_read_group_keys(grouped_read_file):
|
|
with clawhdf5.File(grouped_read_file, "r") as f:
|
|
keys = f.keys()
|
|
assert "sensors" in keys
|
|
assert "metadata" in keys
|
|
assert "root_data" in keys
|
|
|
|
|
|
def test_read_group_dataset(grouped_read_file):
|
|
with clawhdf5.File(grouped_read_file, "r") as f:
|
|
grp = f["sensors"]
|
|
ds = grp["temperature"]
|
|
data = ds[:]
|
|
np.testing.assert_array_almost_equal(data, [22.5, 23.1, 21.8])
|
|
|
|
|
|
def test_read_group_attrs(grouped_read_file):
|
|
with clawhdf5.File(grouped_read_file, "r") as f:
|
|
grp = f["sensors"]
|
|
assert grp.attrs["location"] == b"lab"
|
|
|
|
|
|
def test_nested_path_access(grouped_read_file):
|
|
"""Test f['group/dataset'] path navigation."""
|
|
with clawhdf5.File(grouped_read_file, "r") as f:
|
|
ds = f["sensors/temperature"]
|
|
data = ds[:]
|
|
np.testing.assert_array_almost_equal(data, [22.5, 23.1, 21.8])
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Test: context manager
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def test_context_manager(sample_read_file):
|
|
with clawhdf5.File(sample_read_file, "r") as f:
|
|
data = f["temperatures"][:]
|
|
np.testing.assert_array_almost_equal(data, [22.5, 23.1, 21.8])
|
|
# File should be closed after with block
|
|
assert repr(f) == "<HDF5 File (closed)>"
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Test: create files / write mode
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def test_write_simple(tmp_h5):
|
|
with clawhdf5.File(tmp_h5, "w") as f:
|
|
f.create_dataset("data", data=np.array([1.0, 2.0, 3.0]))
|
|
# Verify by reading back
|
|
with clawhdf5.File(tmp_h5, "r") as f:
|
|
data = f["data"][:]
|
|
np.testing.assert_array_almost_equal(data, [1.0, 2.0, 3.0])
|
|
|
|
|
|
def test_write_with_attrs(tmp_h5):
|
|
with clawhdf5.File(tmp_h5, "w") as f:
|
|
f.create_dataset("values", data=np.array([10, 20], dtype=np.int32))
|
|
f.attrs["author"] = "test"
|
|
f.attrs["count"] = 42
|
|
with clawhdf5.File(tmp_h5, "r") as f:
|
|
assert f.attrs["author"] == b"test"
|
|
assert f.attrs["count"] == 42
|
|
|
|
|
|
def test_write_with_group(tmp_h5):
|
|
with clawhdf5.File(tmp_h5, "w") as f:
|
|
grp = f.create_group("experiment")
|
|
grp.create_dataset("results", data=np.array([3.14, 2.72]))
|
|
grp.attrs["version"] = 1
|
|
with clawhdf5.File(tmp_h5, "r") as f:
|
|
ds = f["experiment/results"]
|
|
np.testing.assert_array_almost_equal(ds[:], [3.14, 2.72])
|
|
grp = f["experiment"]
|
|
assert grp.attrs["version"] == 1
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Test: numpy array types round-trip
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def test_roundtrip_float64(tmp_h5):
|
|
original = np.array([1.1, 2.2, 3.3], dtype=np.float64)
|
|
with clawhdf5.File(tmp_h5, "w") as f:
|
|
f.create_dataset("data", data=original)
|
|
with clawhdf5.File(tmp_h5, "r") as f:
|
|
result = f["data"][:]
|
|
np.testing.assert_array_almost_equal(result, original)
|
|
assert result.dtype == np.float64
|
|
|
|
|
|
def test_roundtrip_float32(tmp_h5):
|
|
original = np.array([1.5, 2.5, 3.5], dtype=np.float32)
|
|
with clawhdf5.File(tmp_h5, "w") as f:
|
|
f.create_dataset("data", data=original)
|
|
with clawhdf5.File(tmp_h5, "r") as f:
|
|
result = f["data"][:]
|
|
np.testing.assert_array_almost_equal(result, original)
|
|
assert result.dtype == np.float32
|
|
|
|
|
|
def test_roundtrip_int32(tmp_h5):
|
|
original = np.array([-10, 0, 10, 100], dtype=np.int32)
|
|
with clawhdf5.File(tmp_h5, "w") as f:
|
|
f.create_dataset("data", data=original)
|
|
with clawhdf5.File(tmp_h5, "r") as f:
|
|
result = f["data"][:]
|
|
np.testing.assert_array_equal(result, original)
|
|
assert result.dtype == np.int32
|
|
|
|
|
|
def test_roundtrip_int64(tmp_h5):
|
|
original = np.array([-1, 0, 1, 2**40], dtype=np.int64)
|
|
with clawhdf5.File(tmp_h5, "w") as f:
|
|
f.create_dataset("data", data=original)
|
|
with clawhdf5.File(tmp_h5, "r") as f:
|
|
result = f["data"][:]
|
|
np.testing.assert_array_equal(result, original)
|
|
assert result.dtype == np.int64
|
|
|
|
|
|
def test_roundtrip_uint8(tmp_h5):
|
|
original = np.array([0, 127, 255], dtype=np.uint8)
|
|
with clawhdf5.File(tmp_h5, "w") as f:
|
|
f.create_dataset("data", data=original)
|
|
with clawhdf5.File(tmp_h5, "r") as f:
|
|
result = f["data"][:]
|
|
np.testing.assert_array_equal(result, original)
|
|
assert result.dtype == np.uint8
|
|
|
|
|
|
@pytest.mark.parametrize("dtype", [np.complex64, np.complex128])
|
|
def test_roundtrip_complex(tmp_h5, dtype):
|
|
"""Complex arrays are written as h5py writes them (a compound {r, i});
|
|
h5py and clawhdf5 both read them back as the same numpy complex dtype."""
|
|
import h5py
|
|
|
|
original = (np.arange(12).reshape(3, 4) * (1.5 - 0.25j)).astype(dtype)
|
|
with clawhdf5.File(tmp_h5, "w") as f:
|
|
f.create_dataset("z", data=original)
|
|
f.create_dataset("zc", data=original, chunks=(2, 2), compression="gzip")
|
|
with clawhdf5.File(tmp_h5, "r") as f:
|
|
for name in ["z", "zc"]:
|
|
result = f[name][:]
|
|
assert result.dtype == dtype
|
|
np.testing.assert_array_equal(result, original)
|
|
with h5py.File(tmp_h5, "r") as f:
|
|
for name in ["z", "zc"]:
|
|
assert f[name].dtype == dtype
|
|
assert f[name].id.get_type().get_class() == h5py.h5t.COMPOUND
|
|
np.testing.assert_array_equal(f[name][:], original)
|
|
|
|
|
|
def test_read_native_complex_from_h5py(tmp_h5):
|
|
"""HDF5 2.0's native complex type (class 11), written through h5py's
|
|
low-level API, reads as numpy complex."""
|
|
import h5py
|
|
from h5py import h5s, h5t
|
|
|
|
if not getattr(h5py.get_config(), "has_native_complex", False):
|
|
pytest.skip("h5py's libhdf5 predates 2.0")
|
|
original = np.array([1 + 2j, -3.5 + 0j, 0 - 1e-3j])
|
|
with h5py.File(tmp_h5, "w") as f:
|
|
for name, t, dt in [
|
|
(b"n64", h5t.COMPLEX_IEEE_F32LE, np.complex64),
|
|
(b"n128", h5t.COMPLEX_IEEE_F64LE, np.complex128),
|
|
]:
|
|
d = h5py.h5d.create(f.id, name, t, h5s.create_simple((3,)))
|
|
d.write(h5s.ALL, h5s.ALL, original.astype(dt), mtype=t)
|
|
with clawhdf5.File(tmp_h5, "r") as f:
|
|
for name, dt in [("n64", np.complex64), ("n128", np.complex128)]:
|
|
result = f[name][:]
|
|
assert result.dtype == dt
|
|
np.testing.assert_array_equal(result, original.astype(dt))
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Test: chunked + compressed datasets
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def test_chunked_gzip(tmp_h5):
|
|
original = np.arange(100, dtype=np.float64)
|
|
with clawhdf5.File(tmp_h5, "w") as f:
|
|
f.create_dataset(
|
|
"compressed",
|
|
data=original,
|
|
chunks=(50,),
|
|
compression="gzip",
|
|
compression_opts=6,
|
|
)
|
|
with clawhdf5.File(tmp_h5, "r") as f:
|
|
result = f["compressed"][:]
|
|
np.testing.assert_array_equal(result, original)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Test: h5py interoperability
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def test_h5py_can_read_our_file(tmp_h5):
|
|
"""Verify that h5py can read files we create."""
|
|
import h5py
|
|
|
|
with clawhdf5.File(tmp_h5, "w") as f:
|
|
f.create_dataset("values", data=np.array([1.0, 2.0, 3.0]))
|
|
f.attrs["meta"] = "hello"
|
|
with h5py.File(tmp_h5, "r") as f:
|
|
np.testing.assert_array_equal(f["values"][:], [1.0, 2.0, 3.0])
|
|
# h5py reads fixed-length strings as bytes
|
|
assert f.attrs["meta"] == b"hello"
|
|
|
|
|
|
def test_we_can_read_h5py_file(tmp_h5):
|
|
"""Verify that we can read files created by h5py."""
|
|
import h5py
|
|
|
|
with h5py.File(tmp_h5, "w") as f:
|
|
f.create_dataset("data", data=np.array([10.0, 20.0, 30.0]))
|
|
f.attrs["version"] = 2
|
|
with clawhdf5.File(tmp_h5, "r") as f:
|
|
data = f["data"][:]
|
|
np.testing.assert_array_equal(data, [10.0, 20.0, 30.0])
|
|
assert f.attrs["version"] == 2
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Test: 2D array shape
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def test_2d_array_roundtrip(tmp_h5):
|
|
original = np.array([[1.0, 2.0, 3.0], [4.0, 5.0, 6.0]], dtype=np.float64)
|
|
with clawhdf5.File(tmp_h5, "w") as f:
|
|
f.create_dataset("matrix", data=original)
|
|
with clawhdf5.File(tmp_h5, "r") as f:
|
|
ds = f["matrix"]
|
|
assert ds.shape == (2, 3)
|
|
result = ds[:]
|
|
np.testing.assert_array_almost_equal(result, original)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Test: error handling
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def test_open_nonexistent_file():
|
|
with pytest.raises(OSError):
|
|
clawhdf5.File("/nonexistent/path.h5", "r")
|
|
|
|
|
|
def test_invalid_mode(tmp_h5):
|
|
with pytest.raises(ValueError):
|
|
clawhdf5.File(tmp_h5, "x")
|
|
|
|
|
|
def test_key_error_on_missing_dataset(sample_read_file):
|
|
with clawhdf5.File(sample_read_file, "r") as f:
|
|
with pytest.raises(KeyError):
|
|
f["nonexistent"]
|