"""Tests for clawhdf5 Python bindings.""" import os import tempfile import numpy as np import pytest import clawhdf5 @pytest.fixture def tmp_h5(tmp_path): """Return a temporary HDF5 file path.""" return str(tmp_path / "test.h5") @pytest.fixture def sample_read_file(tmp_h5): """Create a sample HDF5 file for reading tests.""" with clawhdf5.File(tmp_h5, "w") as f: f.create_dataset("temperatures", data=np.array([22.5, 23.1, 21.8])) f.create_dataset("counts", data=np.array([10, 20, 30], dtype=np.int32)) f.attrs["version"] = 1 f.attrs["description"] = "test file" return tmp_h5 @pytest.fixture def grouped_read_file(tmp_h5): """Create an HDF5 file with groups for reading tests.""" with clawhdf5.File(tmp_h5, "w") as f: f.create_dataset("root_data", data=np.array([0.0, 1.0])) grp = f.create_group("sensors") grp.create_dataset("temperature", data=np.array([22.5, 23.1, 21.8])) grp.create_dataset("humidity", data=np.array([45, 50, 55], dtype=np.int32)) grp.attrs["location"] = "lab" grp2 = f.create_group("metadata") grp2.create_dataset("timestamps", data=np.array([1000, 2000, 3000], dtype=np.int64)) return tmp_h5 # --------------------------------------------------------------------------- # Test: open and read datasets # --------------------------------------------------------------------------- def test_open_and_read_f64(sample_read_file): f = clawhdf5.File(sample_read_file, "r") ds = f["temperatures"] data = ds[:] np.testing.assert_array_almost_equal(data, [22.5, 23.1, 21.8]) f.close() def test_open_and_read_i32(sample_read_file): f = clawhdf5.File(sample_read_file, "r") ds = f["counts"] data = ds[:] np.testing.assert_array_equal(data, [10, 20, 30]) assert data.dtype == np.int32 f.close() # --------------------------------------------------------------------------- # Test: dataset properties (shape, dtype) # --------------------------------------------------------------------------- def test_dataset_shape(sample_read_file): with clawhdf5.File(sample_read_file, "r") as f: ds = f["temperatures"] assert ds.shape == (3,) def test_dataset_dtype(sample_read_file): with clawhdf5.File(sample_read_file, "r") as f: assert f["temperatures"].dtype == "float64" assert f["counts"].dtype == "int32" # --------------------------------------------------------------------------- # Test: read attributes # --------------------------------------------------------------------------- def test_read_root_attrs(sample_read_file): with clawhdf5.File(sample_read_file, "r") as f: assert f.attrs["version"] == 1 assert f.attrs["description"] == b"test file" # fixed-length string: numpy.bytes_, as in h5py def test_attrs_len(sample_read_file): with clawhdf5.File(sample_read_file, "r") as f: assert len(f.attrs) >= 2 def test_attrs_contains(sample_read_file): with clawhdf5.File(sample_read_file, "r") as f: assert "version" in f.attrs assert "nonexistent" not in f.attrs def test_attrs_keys(sample_read_file): with clawhdf5.File(sample_read_file, "r") as f: keys = f.attrs.keys() assert "version" in keys assert "description" in keys # --------------------------------------------------------------------------- # Test: read groups # --------------------------------------------------------------------------- def test_read_group_keys(grouped_read_file): with clawhdf5.File(grouped_read_file, "r") as f: keys = f.keys() assert "sensors" in keys assert "metadata" in keys assert "root_data" in keys def test_read_group_dataset(grouped_read_file): with clawhdf5.File(grouped_read_file, "r") as f: grp = f["sensors"] ds = grp["temperature"] data = ds[:] np.testing.assert_array_almost_equal(data, [22.5, 23.1, 21.8]) def test_read_group_attrs(grouped_read_file): with clawhdf5.File(grouped_read_file, "r") as f: grp = f["sensors"] assert grp.attrs["location"] == b"lab" def test_nested_path_access(grouped_read_file): """Test f['group/dataset'] path navigation.""" with clawhdf5.File(grouped_read_file, "r") as f: ds = f["sensors/temperature"] data = ds[:] np.testing.assert_array_almost_equal(data, [22.5, 23.1, 21.8]) # --------------------------------------------------------------------------- # Test: context manager # --------------------------------------------------------------------------- def test_context_manager(sample_read_file): with clawhdf5.File(sample_read_file, "r") as f: data = f["temperatures"][:] np.testing.assert_array_almost_equal(data, [22.5, 23.1, 21.8]) # File should be closed after with block assert repr(f) == "" # --------------------------------------------------------------------------- # Test: create files / write mode # --------------------------------------------------------------------------- def test_write_simple(tmp_h5): with clawhdf5.File(tmp_h5, "w") as f: f.create_dataset("data", data=np.array([1.0, 2.0, 3.0])) # Verify by reading back with clawhdf5.File(tmp_h5, "r") as f: data = f["data"][:] np.testing.assert_array_almost_equal(data, [1.0, 2.0, 3.0]) def test_write_with_attrs(tmp_h5): with clawhdf5.File(tmp_h5, "w") as f: f.create_dataset("values", data=np.array([10, 20], dtype=np.int32)) f.attrs["author"] = "test" f.attrs["count"] = 42 with clawhdf5.File(tmp_h5, "r") as f: assert f.attrs["author"] == b"test" assert f.attrs["count"] == 42 def test_write_with_group(tmp_h5): with clawhdf5.File(tmp_h5, "w") as f: grp = f.create_group("experiment") grp.create_dataset("results", data=np.array([3.14, 2.72])) grp.attrs["version"] = 1 with clawhdf5.File(tmp_h5, "r") as f: ds = f["experiment/results"] np.testing.assert_array_almost_equal(ds[:], [3.14, 2.72]) grp = f["experiment"] assert grp.attrs["version"] == 1 # --------------------------------------------------------------------------- # Test: numpy array types round-trip # --------------------------------------------------------------------------- def test_roundtrip_float64(tmp_h5): original = np.array([1.1, 2.2, 3.3], dtype=np.float64) with clawhdf5.File(tmp_h5, "w") as f: f.create_dataset("data", data=original) with clawhdf5.File(tmp_h5, "r") as f: result = f["data"][:] np.testing.assert_array_almost_equal(result, original) assert result.dtype == np.float64 def test_roundtrip_float32(tmp_h5): original = np.array([1.5, 2.5, 3.5], dtype=np.float32) with clawhdf5.File(tmp_h5, "w") as f: f.create_dataset("data", data=original) with clawhdf5.File(tmp_h5, "r") as f: result = f["data"][:] np.testing.assert_array_almost_equal(result, original) assert result.dtype == np.float32 def test_roundtrip_int32(tmp_h5): original = np.array([-10, 0, 10, 100], dtype=np.int32) with clawhdf5.File(tmp_h5, "w") as f: f.create_dataset("data", data=original) with clawhdf5.File(tmp_h5, "r") as f: result = f["data"][:] np.testing.assert_array_equal(result, original) assert result.dtype == np.int32 def test_roundtrip_int64(tmp_h5): original = np.array([-1, 0, 1, 2**40], dtype=np.int64) with clawhdf5.File(tmp_h5, "w") as f: f.create_dataset("data", data=original) with clawhdf5.File(tmp_h5, "r") as f: result = f["data"][:] np.testing.assert_array_equal(result, original) assert result.dtype == np.int64 def test_roundtrip_uint8(tmp_h5): original = np.array([0, 127, 255], dtype=np.uint8) with clawhdf5.File(tmp_h5, "w") as f: f.create_dataset("data", data=original) with clawhdf5.File(tmp_h5, "r") as f: result = f["data"][:] np.testing.assert_array_equal(result, original) assert result.dtype == np.uint8 @pytest.mark.parametrize("dtype", [np.complex64, np.complex128]) def test_roundtrip_complex(tmp_h5, dtype): """Complex arrays are written as h5py writes them (a compound {r, i}); h5py and clawhdf5 both read them back as the same numpy complex dtype.""" import h5py original = (np.arange(12).reshape(3, 4) * (1.5 - 0.25j)).astype(dtype) with clawhdf5.File(tmp_h5, "w") as f: f.create_dataset("z", data=original) f.create_dataset("zc", data=original, chunks=(2, 2), compression="gzip") with clawhdf5.File(tmp_h5, "r") as f: for name in ["z", "zc"]: result = f[name][:] assert result.dtype == dtype np.testing.assert_array_equal(result, original) with h5py.File(tmp_h5, "r") as f: for name in ["z", "zc"]: assert f[name].dtype == dtype assert f[name].id.get_type().get_class() == h5py.h5t.COMPOUND np.testing.assert_array_equal(f[name][:], original) def test_read_native_complex_from_h5py(tmp_h5): """HDF5 2.0's native complex type (class 11), written through h5py's low-level API, reads as numpy complex.""" import h5py from h5py import h5s, h5t if not getattr(h5py.get_config(), "has_native_complex", False): pytest.skip("h5py's libhdf5 predates 2.0") original = np.array([1 + 2j, -3.5 + 0j, 0 - 1e-3j]) with h5py.File(tmp_h5, "w") as f: for name, t, dt in [ (b"n64", h5t.COMPLEX_IEEE_F32LE, np.complex64), (b"n128", h5t.COMPLEX_IEEE_F64LE, np.complex128), ]: d = h5py.h5d.create(f.id, name, t, h5s.create_simple((3,))) d.write(h5s.ALL, h5s.ALL, original.astype(dt), mtype=t) with clawhdf5.File(tmp_h5, "r") as f: for name, dt in [("n64", np.complex64), ("n128", np.complex128)]: result = f[name][:] assert result.dtype == dt np.testing.assert_array_equal(result, original.astype(dt)) # --------------------------------------------------------------------------- # Test: chunked + compressed datasets # --------------------------------------------------------------------------- def test_chunked_gzip(tmp_h5): original = np.arange(100, dtype=np.float64) with clawhdf5.File(tmp_h5, "w") as f: f.create_dataset( "compressed", data=original, chunks=(50,), compression="gzip", compression_opts=6, ) with clawhdf5.File(tmp_h5, "r") as f: result = f["compressed"][:] np.testing.assert_array_equal(result, original) # --------------------------------------------------------------------------- # Test: h5py interoperability # --------------------------------------------------------------------------- def test_h5py_can_read_our_file(tmp_h5): """Verify that h5py can read files we create.""" import h5py with clawhdf5.File(tmp_h5, "w") as f: f.create_dataset("values", data=np.array([1.0, 2.0, 3.0])) f.attrs["meta"] = "hello" with h5py.File(tmp_h5, "r") as f: np.testing.assert_array_equal(f["values"][:], [1.0, 2.0, 3.0]) # h5py reads fixed-length strings as bytes assert f.attrs["meta"] == b"hello" def test_we_can_read_h5py_file(tmp_h5): """Verify that we can read files created by h5py.""" import h5py with h5py.File(tmp_h5, "w") as f: f.create_dataset("data", data=np.array([10.0, 20.0, 30.0])) f.attrs["version"] = 2 with clawhdf5.File(tmp_h5, "r") as f: data = f["data"][:] np.testing.assert_array_equal(data, [10.0, 20.0, 30.0]) assert f.attrs["version"] == 2 # --------------------------------------------------------------------------- # Test: 2D array shape # --------------------------------------------------------------------------- def test_2d_array_roundtrip(tmp_h5): original = np.array([[1.0, 2.0, 3.0], [4.0, 5.0, 6.0]], dtype=np.float64) with clawhdf5.File(tmp_h5, "w") as f: f.create_dataset("matrix", data=original) with clawhdf5.File(tmp_h5, "r") as f: ds = f["matrix"] assert ds.shape == (2, 3) result = ds[:] np.testing.assert_array_almost_equal(result, original) # --------------------------------------------------------------------------- # Test: error handling # --------------------------------------------------------------------------- def test_open_nonexistent_file(): with pytest.raises(OSError): clawhdf5.File("/nonexistent/path.h5", "r") def test_invalid_mode(tmp_h5): with pytest.raises(ValueError): clawhdf5.File(tmp_h5, "x") def test_key_error_on_missing_dataset(sample_read_file): with clawhdf5.File(sample_read_file, "r") as f: with pytest.raises(KeyError): f["nonexistent"]