Files
clawhdf5/crates/clawhdf5-py/tests/test_write_read.py
T
osobhandClaude Opus 5.5 d0d5347cd9 feat: write complex numbers, incl. HDF5 2.0 native complex (class 11)
- Datatype::Complex serializes class 11 version 5 byte-identically to
  libhdf5 2.2.0; containers holding it are written as version 5.
- DatasetBuilder::with_complex_f32/f64_data (h5py's {r, i} compound,
  default) and with_native_complex_f32/f64_data (class 11, opt-in);
  make_(native_)complex_f32/f64_type for attributes.
- Dataset::read_complex_f64/f32 read either form.
- Python create_dataset accepts complex64/complex128 (compound form).
- Parsing unchanged: class 11 still surfaces as {r, i}.
- Tests vs h5py 3.16 / libhdf5 2.0.0 and h5dump 2.2.0; docs.

Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
2026-09-28 21:25:03 -05:00

381 lines
13 KiB
Python

"""Tests for clawhdf5 Python bindings."""
import os
import tempfile
import numpy as np
import pytest
import clawhdf5
@pytest.fixture
def tmp_h5(tmp_path):
"""Return a temporary HDF5 file path."""
return str(tmp_path / "test.h5")
@pytest.fixture
def sample_read_file(tmp_h5):
"""Create a sample HDF5 file for reading tests."""
with clawhdf5.File(tmp_h5, "w") as f:
f.create_dataset("temperatures", data=np.array([22.5, 23.1, 21.8]))
f.create_dataset("counts", data=np.array([10, 20, 30], dtype=np.int32))
f.attrs["version"] = 1
f.attrs["description"] = "test file"
return tmp_h5
@pytest.fixture
def grouped_read_file(tmp_h5):
"""Create an HDF5 file with groups for reading tests."""
with clawhdf5.File(tmp_h5, "w") as f:
f.create_dataset("root_data", data=np.array([0.0, 1.0]))
grp = f.create_group("sensors")
grp.create_dataset("temperature", data=np.array([22.5, 23.1, 21.8]))
grp.create_dataset("humidity", data=np.array([45, 50, 55], dtype=np.int32))
grp.attrs["location"] = "lab"
grp2 = f.create_group("metadata")
grp2.create_dataset("timestamps", data=np.array([1000, 2000, 3000], dtype=np.int64))
return tmp_h5
# ---------------------------------------------------------------------------
# Test: open and read datasets
# ---------------------------------------------------------------------------
def test_open_and_read_f64(sample_read_file):
f = clawhdf5.File(sample_read_file, "r")
ds = f["temperatures"]
data = ds[:]
np.testing.assert_array_almost_equal(data, [22.5, 23.1, 21.8])
f.close()
def test_open_and_read_i32(sample_read_file):
f = clawhdf5.File(sample_read_file, "r")
ds = f["counts"]
data = ds[:]
np.testing.assert_array_equal(data, [10, 20, 30])
assert data.dtype == np.int32
f.close()
# ---------------------------------------------------------------------------
# Test: dataset properties (shape, dtype)
# ---------------------------------------------------------------------------
def test_dataset_shape(sample_read_file):
with clawhdf5.File(sample_read_file, "r") as f:
ds = f["temperatures"]
assert ds.shape == (3,)
def test_dataset_dtype(sample_read_file):
with clawhdf5.File(sample_read_file, "r") as f:
assert f["temperatures"].dtype == "float64"
assert f["counts"].dtype == "int32"
# ---------------------------------------------------------------------------
# Test: read attributes
# ---------------------------------------------------------------------------
def test_read_root_attrs(sample_read_file):
with clawhdf5.File(sample_read_file, "r") as f:
assert f.attrs["version"] == 1
assert f.attrs["description"] == b"test file" # fixed-length string: numpy.bytes_, as in h5py
def test_attrs_len(sample_read_file):
with clawhdf5.File(sample_read_file, "r") as f:
assert len(f.attrs) >= 2
def test_attrs_contains(sample_read_file):
with clawhdf5.File(sample_read_file, "r") as f:
assert "version" in f.attrs
assert "nonexistent" not in f.attrs
def test_attrs_keys(sample_read_file):
with clawhdf5.File(sample_read_file, "r") as f:
keys = f.attrs.keys()
assert "version" in keys
assert "description" in keys
# ---------------------------------------------------------------------------
# Test: read groups
# ---------------------------------------------------------------------------
def test_read_group_keys(grouped_read_file):
with clawhdf5.File(grouped_read_file, "r") as f:
keys = f.keys()
assert "sensors" in keys
assert "metadata" in keys
assert "root_data" in keys
def test_read_group_dataset(grouped_read_file):
with clawhdf5.File(grouped_read_file, "r") as f:
grp = f["sensors"]
ds = grp["temperature"]
data = ds[:]
np.testing.assert_array_almost_equal(data, [22.5, 23.1, 21.8])
def test_read_group_attrs(grouped_read_file):
with clawhdf5.File(grouped_read_file, "r") as f:
grp = f["sensors"]
assert grp.attrs["location"] == b"lab"
def test_nested_path_access(grouped_read_file):
"""Test f['group/dataset'] path navigation."""
with clawhdf5.File(grouped_read_file, "r") as f:
ds = f["sensors/temperature"]
data = ds[:]
np.testing.assert_array_almost_equal(data, [22.5, 23.1, 21.8])
# ---------------------------------------------------------------------------
# Test: context manager
# ---------------------------------------------------------------------------
def test_context_manager(sample_read_file):
with clawhdf5.File(sample_read_file, "r") as f:
data = f["temperatures"][:]
np.testing.assert_array_almost_equal(data, [22.5, 23.1, 21.8])
# File should be closed after with block
assert repr(f) == "<HDF5 File (closed)>"
# ---------------------------------------------------------------------------
# Test: create files / write mode
# ---------------------------------------------------------------------------
def test_write_simple(tmp_h5):
with clawhdf5.File(tmp_h5, "w") as f:
f.create_dataset("data", data=np.array([1.0, 2.0, 3.0]))
# Verify by reading back
with clawhdf5.File(tmp_h5, "r") as f:
data = f["data"][:]
np.testing.assert_array_almost_equal(data, [1.0, 2.0, 3.0])
def test_write_with_attrs(tmp_h5):
with clawhdf5.File(tmp_h5, "w") as f:
f.create_dataset("values", data=np.array([10, 20], dtype=np.int32))
f.attrs["author"] = "test"
f.attrs["count"] = 42
with clawhdf5.File(tmp_h5, "r") as f:
assert f.attrs["author"] == b"test"
assert f.attrs["count"] == 42
def test_write_with_group(tmp_h5):
with clawhdf5.File(tmp_h5, "w") as f:
grp = f.create_group("experiment")
grp.create_dataset("results", data=np.array([3.14, 2.72]))
grp.attrs["version"] = 1
with clawhdf5.File(tmp_h5, "r") as f:
ds = f["experiment/results"]
np.testing.assert_array_almost_equal(ds[:], [3.14, 2.72])
grp = f["experiment"]
assert grp.attrs["version"] == 1
# ---------------------------------------------------------------------------
# Test: numpy array types round-trip
# ---------------------------------------------------------------------------
def test_roundtrip_float64(tmp_h5):
original = np.array([1.1, 2.2, 3.3], dtype=np.float64)
with clawhdf5.File(tmp_h5, "w") as f:
f.create_dataset("data", data=original)
with clawhdf5.File(tmp_h5, "r") as f:
result = f["data"][:]
np.testing.assert_array_almost_equal(result, original)
assert result.dtype == np.float64
def test_roundtrip_float32(tmp_h5):
original = np.array([1.5, 2.5, 3.5], dtype=np.float32)
with clawhdf5.File(tmp_h5, "w") as f:
f.create_dataset("data", data=original)
with clawhdf5.File(tmp_h5, "r") as f:
result = f["data"][:]
np.testing.assert_array_almost_equal(result, original)
assert result.dtype == np.float32
def test_roundtrip_int32(tmp_h5):
original = np.array([-10, 0, 10, 100], dtype=np.int32)
with clawhdf5.File(tmp_h5, "w") as f:
f.create_dataset("data", data=original)
with clawhdf5.File(tmp_h5, "r") as f:
result = f["data"][:]
np.testing.assert_array_equal(result, original)
assert result.dtype == np.int32
def test_roundtrip_int64(tmp_h5):
original = np.array([-1, 0, 1, 2**40], dtype=np.int64)
with clawhdf5.File(tmp_h5, "w") as f:
f.create_dataset("data", data=original)
with clawhdf5.File(tmp_h5, "r") as f:
result = f["data"][:]
np.testing.assert_array_equal(result, original)
assert result.dtype == np.int64
def test_roundtrip_uint8(tmp_h5):
original = np.array([0, 127, 255], dtype=np.uint8)
with clawhdf5.File(tmp_h5, "w") as f:
f.create_dataset("data", data=original)
with clawhdf5.File(tmp_h5, "r") as f:
result = f["data"][:]
np.testing.assert_array_equal(result, original)
assert result.dtype == np.uint8
@pytest.mark.parametrize("dtype", [np.complex64, np.complex128])
def test_roundtrip_complex(tmp_h5, dtype):
"""Complex arrays are written as h5py writes them (a compound {r, i});
h5py and clawhdf5 both read them back as the same numpy complex dtype."""
import h5py
original = (np.arange(12).reshape(3, 4) * (1.5 - 0.25j)).astype(dtype)
with clawhdf5.File(tmp_h5, "w") as f:
f.create_dataset("z", data=original)
f.create_dataset("zc", data=original, chunks=(2, 2), compression="gzip")
with clawhdf5.File(tmp_h5, "r") as f:
for name in ["z", "zc"]:
result = f[name][:]
assert result.dtype == dtype
np.testing.assert_array_equal(result, original)
with h5py.File(tmp_h5, "r") as f:
for name in ["z", "zc"]:
assert f[name].dtype == dtype
assert f[name].id.get_type().get_class() == h5py.h5t.COMPOUND
np.testing.assert_array_equal(f[name][:], original)
def test_read_native_complex_from_h5py(tmp_h5):
"""HDF5 2.0's native complex type (class 11), written through h5py's
low-level API, reads as numpy complex."""
import h5py
from h5py import h5s, h5t
if not getattr(h5py.get_config(), "has_native_complex", False):
pytest.skip("h5py's libhdf5 predates 2.0")
original = np.array([1 + 2j, -3.5 + 0j, 0 - 1e-3j])
with h5py.File(tmp_h5, "w") as f:
for name, t, dt in [
(b"n64", h5t.COMPLEX_IEEE_F32LE, np.complex64),
(b"n128", h5t.COMPLEX_IEEE_F64LE, np.complex128),
]:
d = h5py.h5d.create(f.id, name, t, h5s.create_simple((3,)))
d.write(h5s.ALL, h5s.ALL, original.astype(dt), mtype=t)
with clawhdf5.File(tmp_h5, "r") as f:
for name, dt in [("n64", np.complex64), ("n128", np.complex128)]:
result = f[name][:]
assert result.dtype == dt
np.testing.assert_array_equal(result, original.astype(dt))
# ---------------------------------------------------------------------------
# Test: chunked + compressed datasets
# ---------------------------------------------------------------------------
def test_chunked_gzip(tmp_h5):
original = np.arange(100, dtype=np.float64)
with clawhdf5.File(tmp_h5, "w") as f:
f.create_dataset(
"compressed",
data=original,
chunks=(50,),
compression="gzip",
compression_opts=6,
)
with clawhdf5.File(tmp_h5, "r") as f:
result = f["compressed"][:]
np.testing.assert_array_equal(result, original)
# ---------------------------------------------------------------------------
# Test: h5py interoperability
# ---------------------------------------------------------------------------
def test_h5py_can_read_our_file(tmp_h5):
"""Verify that h5py can read files we create."""
import h5py
with clawhdf5.File(tmp_h5, "w") as f:
f.create_dataset("values", data=np.array([1.0, 2.0, 3.0]))
f.attrs["meta"] = "hello"
with h5py.File(tmp_h5, "r") as f:
np.testing.assert_array_equal(f["values"][:], [1.0, 2.0, 3.0])
# h5py reads fixed-length strings as bytes
assert f.attrs["meta"] == b"hello"
def test_we_can_read_h5py_file(tmp_h5):
"""Verify that we can read files created by h5py."""
import h5py
with h5py.File(tmp_h5, "w") as f:
f.create_dataset("data", data=np.array([10.0, 20.0, 30.0]))
f.attrs["version"] = 2
with clawhdf5.File(tmp_h5, "r") as f:
data = f["data"][:]
np.testing.assert_array_equal(data, [10.0, 20.0, 30.0])
assert f.attrs["version"] == 2
# ---------------------------------------------------------------------------
# Test: 2D array shape
# ---------------------------------------------------------------------------
def test_2d_array_roundtrip(tmp_h5):
original = np.array([[1.0, 2.0, 3.0], [4.0, 5.0, 6.0]], dtype=np.float64)
with clawhdf5.File(tmp_h5, "w") as f:
f.create_dataset("matrix", data=original)
with clawhdf5.File(tmp_h5, "r") as f:
ds = f["matrix"]
assert ds.shape == (2, 3)
result = ds[:]
np.testing.assert_array_almost_equal(result, original)
# ---------------------------------------------------------------------------
# Test: error handling
# ---------------------------------------------------------------------------
def test_open_nonexistent_file():
with pytest.raises(OSError):
clawhdf5.File("/nonexistent/path.h5", "r")
def test_invalid_mode(tmp_h5):
with pytest.raises(ValueError):
clawhdf5.File(tmp_h5, "x")
def test_key_error_on_missing_dataset(sample_read_file):
with clawhdf5.File(sample_read_file, "r") as f:
with pytest.raises(KeyError):
f["nonexistent"]