HDF5 2.0 native complex as a first-class type on read; Python libver=
CI / test-arm64 (pull_request) Successful in 1m38s
CI / test (pull_request) Successful in 21m11s

- Datatype::parse returns Datatype::Complex for class 11 (also inside
  compounds, arrays and VL types) instead of the {r, i} compound view.
- Facade: DType::Complex(Box<DType>); read_complex_f32/f64 accept it.
- h5rs dump/ls/diff print native complex as h5dump/h5ls/h5diff 2.2.0 do
  (checked against a fixture written by h5py 3.16 / libhdf5 2.0.0);
  dump --json keeps the {r, i} compound (hdf5-json has no complex class).
- clawhdf5-wasm reads native complex datasets as [re, im] pairs.
- Python: clawhdf5.File(path, 'w', libver=...) with h5py's values,
  mapped to FileBuilder::libver_bounds; 'v108' output opens in HDF5 1.8.23.
- Docs: known-issues entry moved to Fixed (history), CHANGELOG, READMEs.

Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
osobh
2026-09-29 20:47:33 -05:00
co-authored by Claude Opus 5.5
parent e5d6f59e12
commit e4ba09946f
22 changed files with 988 additions and 81 deletions
+16 -1
View File
@@ -104,7 +104,22 @@ writes `float64`, `float32`, `int64`, `int32`, `uint8`, `complex64` and
`complex128` arrays; the file is written on `close()`. Complex arrays are
stored as h5py stores them, a compound `{r, i}` that every libhdf5 reads
(not HDF5 2.0's native complex type, which only libhdf5 2.0+ reads; the
Rust API writes that on request).
Rust API writes that on request). Both forms read back as numpy
`complex64`/`complex128`.
`libver=` sets the library version bounds as in h5py: `'v108'`,
`'v110'`, `'v112'`, `'v114'`, `'v200'` or `'latest'` (the low bound; the
high bound is then `'latest'`), or a `(low, high)` tuple. With `'v108'`
the file is written in the HDF5 1.8 format (version-2 superblock,
version-1 B-tree chunk indexes), which HDF5 1.8 reads; the default is the
HDF5 1.10 format. `'earliest'` as the low bound writes the 1.8 format too,
with a `UserWarning`: clawhdf5 cannot write the pre-1.8 format. The
argument is ignored when reading and refused for `'r+'`/`'a'`.
```python
with clawhdf5.File("old.h5", "w", libver="v108") as f: # HDF5 1.8 reads it
f.create_dataset("x", data=np.arange(10.0), chunks=(5,), compression="gzip")
```
## Editing a file in place
+90 -2
View File
@@ -13,10 +13,13 @@ use crate::attrs::PyAttrs;
use crate::group::{PyGroup, ReadGroup, WriteGroupState, finalize_write_group};
use crate::handle::Handle;
use crate::{DatasetSpec, OwnedAttrValue, apply_dataset_spec, extract_numpy_data, to_py_err};
use clawhdf5_rs::LibVer;
/// Internal state for write mode.
struct WriteState {
path: PathBuf,
/// `libver=` as (low, high); `None` keeps the writer's default.
libver: Option<(LibVer, LibVer)>,
root_datasets: Vec<DatasetSpec>,
root_attrs: Arc<Mutex<Vec<(String, OwnedAttrValue)>>>,
groups: Vec<Arc<Mutex<WriteGroupState>>>,
@@ -80,10 +83,32 @@ impl PyFile {
/// `az://`; which schemes work depends on how the wheel was built)
/// to read the file remotely with default options (see `open_url`)
/// mode: 'r' for read (default), 'w' for write
/// libver: library version bounds for mode 'w', as h5py's: one of
/// 'earliest', 'v108', 'v110', 'v112', 'v114', 'v200', 'latest' (the
/// low bound; the high bound is then 'latest') or a (low, high)
/// tuple of them. The low bound is the oldest HDF5 release whose
/// format the file uses ('v108': HDF5 1.8 can read it); the high
/// bound the newest whose features it may use. clawhdf5 cannot write
/// the pre-1.8 format, so a low bound of 'earliest' writes the 1.8
/// format (with a warning) and a high bound of 'earliest' is an
/// error. Default (None): the HDF5 1.10 format clawhdf5 has always
/// written. Ignored for reading; not supported with 'r+' / 'a'.
#[new]
#[pyo3(signature = (path, mode="r"))]
fn new(py: Python<'_>, path: &str, mode: &str) -> PyResult<Self> {
#[pyo3(signature = (path, mode="r", libver=None))]
fn new(
py: Python<'_>,
path: &str,
mode: &str,
libver: Option<&Bound<'_, PyAny>>,
) -> PyResult<Self> {
let filename = path.to_string();
let libver = libver.map(|v| parse_libver(py, v)).transpose()?;
if libver.is_some() && matches!(mode, "r+" | "a") {
return Err(PyNotImplementedError::new_err(format!(
"libver with mode '{mode}': clawhdf5's in-place editor keeps the format \
versions the file already uses"
)));
}
if is_url(path) {
if mode != "r" {
return Err(PyValueError::new_err(format!(
@@ -113,6 +138,7 @@ impl PyFile {
// Absolute now: the file is written at close, possibly
// after the working directory changed.
path: std::path::absolute(path).unwrap_or_else(|_| PathBuf::from(path)),
libver,
root_datasets: Vec::new(),
root_attrs: Arc::new(Mutex::new(Vec::new())),
groups: Vec::new(),
@@ -460,10 +486,71 @@ fn parse_compression(
}
}
/// One of h5py's `libver` names as a bound; `high` says which end it is.
/// `Ok(None)` is 'earliest' as the low bound: the pre-1.8 format, which
/// clawhdf5 cannot write.
fn libver_name(name: &str, high: bool) -> PyResult<Option<LibVer>> {
Ok(Some(match name {
"earliest" if high => {
return Err(PyValueError::new_err(
"libver high bound 'earliest' (the pre-1.8 format) cannot be written by \
clawhdf5; the oldest format it writes is 'v108'",
));
}
"earliest" => return Ok(None),
"v108" => LibVer::V18,
"v110" => LibVer::V110,
"v112" => LibVer::V112,
"v114" => LibVer::V114,
"v200" => LibVer::V200,
"latest" => LibVer::Latest,
other => {
return Err(PyValueError::new_err(format!(
"unknown libver '{other}'; expected 'earliest', 'v108', 'v110', 'v112', \
'v114', 'v200' or 'latest'"
)));
}
}))
}
/// h5py's `libver=`: a name (the low bound, high bound 'latest') or a
/// `(low, high)` pair.
fn parse_libver(py: Python<'_>, v: &Bound<'_, PyAny>) -> PyResult<(LibVer, LibVer)> {
let (low, high): (String, String) = match v.extract::<String>() {
Ok(name) => (name, "latest".into()),
Err(_) => v.extract().map_err(|_| {
PyValueError::new_err("libver must be a string or a (low, high) tuple of strings")
})?,
};
let Some(high) = libver_name(&high, true)? else {
unreachable!("a high bound is never None")
};
let low = match libver_name(&low, false)? {
Some(low) => low,
None => {
PyModule::import(py, "warnings")?.getattr("warn")?.call1((
"libver 'earliest': clawhdf5 cannot write the pre-1.8 format; the file \
is written in the HDF5 1.8 format ('v108') instead",
py.get_type::<pyo3::exceptions::PyUserWarning>(),
))?;
LibVer::V18
}
};
if low > high {
return Err(PyValueError::new_err(format!(
"libver low bound {low} is newer than the high bound {high}"
)));
}
Ok((low, high))
}
/// Build and write the HDF5 file from accumulated write state.
fn finalize_write(state: WriteState) -> PyResult<()> {
crate::no_panic(|| {
let mut builder = clawhdf5_rs::FileBuilder::new();
if let Some((low, high)) = state.libver {
builder.libver_bounds(low, high);
}
// Root attributes
let root_attrs = state.root_attrs.lock().unwrap_or_else(|e| e.into_inner());
@@ -520,6 +607,7 @@ mod tests {
let state = WriteState {
path: path.clone(),
libver: None,
root_datasets: vec![DatasetSpec {
name: "data".into(),
data: crate::DatasetData::F64(vec![1.0, 2.0, 3.0]),
+143
View File
@@ -0,0 +1,143 @@
"""`clawhdf5.File(path, 'w', libver=...)`: h5py's library version bounds.
A file written with libver='v108' must open in HDF5 1.8. Its h5dump is
found through CLAWHDF5_H5DUMP18 or at ~/.cache/hdf5-1.8.23/bin/h5dump
(scripts/build-hdf5-1.8.sh builds it); without it that check is skipped.
"""
import os
import subprocess
import numpy as np
import pytest
import clawhdf5
def superblock_version(path):
with open(path, "rb") as f:
head = f.read(9)
assert head[:8] == b"\x89HDF\r\n\x1a\n"
return head[8]
def h5dump18():
path = os.environ.get("CLAWHDF5_H5DUMP18") or os.path.expanduser(
"~/.cache/hdf5-1.8.23/bin/h5dump"
)
try:
out = subprocess.run([path, "--version"], capture_output=True, text=True)
except OSError:
return None
return path if "1.8." in out.stdout else None
def write_sample(path, libver):
with clawhdf5.File(path, "w", libver=libver) as f:
f.create_dataset("x", data=np.arange(10, dtype=np.float64))
f.create_dataset(
"chunked",
data=np.arange(100, dtype=np.int32),
chunks=(30,),
compression="gzip",
)
f.create_dataset("z", data=np.array([1 + 2j, -3j], dtype=np.complex128))
g = f.create_group("g")
g.create_dataset("y", data=np.ones(3, dtype=np.float32))
f.attrs["version"] = 1
@pytest.mark.parametrize(
"libver, sb",
[
(None, 3),
("v108", 2),
(("v108", "v108"), 2),
(("v108", "latest"), 2),
("v110", 3),
("v112", 3),
("v114", 3),
("v200", 3),
("latest", 3),
(("v110", "v200"), 3),
],
)
def test_libver_bounds(h5py, tmp_path, libver, sb):
path = str(tmp_path / "libver.h5")
write_sample(path, libver)
# v108 low bound: HDF5 1.8's version-2 superblock; 1.10 and later: 3.
assert superblock_version(path) == sb
with h5py.File(path, "r") as f:
np.testing.assert_array_equal(f["x"][:], np.arange(10.0))
np.testing.assert_array_equal(f["chunked"][:], np.arange(100))
np.testing.assert_array_equal(f["z"][:], [1 + 2j, -3j])
np.testing.assert_array_equal(f["g/y"][:], np.ones(3))
assert f.attrs["version"] == 1
with clawhdf5.File(path, "r") as f:
np.testing.assert_array_equal(f["chunked"][:], np.arange(100))
def test_libver_v108_opens_in_hdf5_1_8(tmp_path):
h5dump = h5dump18()
if h5dump is None:
pytest.skip("no HDF5 1.8 h5dump (CLAWHDF5_H5DUMP18)")
path = str(tmp_path / "v108.h5")
write_sample(path, "v108")
out = subprocess.run([h5dump, path], capture_output=True, text=True)
assert out.returncode == 0, out.stderr
assert "h5dump error" not in out.stderr
assert 'DATASET "y"' in out.stdout
# The chunked, deflated dataset (a version-1 B-tree) reads in full.
out = subprocess.run(
[h5dump, "-w", "0", "-d", "/chunked", path], capture_output=True, text=True
)
assert out.returncode == 0, out.stderr
assert "(0): " + ", ".join(str(k) for k in range(100)) + "\n" in out.stdout
# The default (1.10 format) does not open in 1.8: the bound matters.
path = str(tmp_path / "default.h5")
write_sample(path, None)
out = subprocess.run([h5dump, path], capture_output=True, text=True)
assert out.returncode != 0
def test_libver_earliest_writes_v108_with_a_warning(h5py, tmp_path):
path = str(tmp_path / "earliest.h5")
with pytest.warns(UserWarning, match="pre-1.8"):
write_sample(path, "earliest")
assert superblock_version(path) == 2
with h5py.File(path, "r") as f:
np.testing.assert_array_equal(f["x"][:], np.arange(10.0))
@pytest.mark.parametrize(
"libver, match",
[
(("v108", "earliest"), "high bound 'earliest'"),
("v109", "unknown libver 'v109'"),
(("latest", "v108"), "newer than the high bound"),
(("v108",), "tuple"),
(3, "tuple"),
],
)
def test_libver_errors(tmp_path, libver, match):
with pytest.raises(ValueError, match=match):
clawhdf5.File(str(tmp_path / "bad.h5"), "w", libver=libver)
def test_libver_v108_high_bound_writes_everything(tmp_path):
# Native complex numbers need HDF5 2.0; the Python writer stores complex
# as h5py's {r, i} compound, which 1.8 reads, so the 1.8 high bound is
# fine for everything clawhdf5.File writes.
path = str(tmp_path / "v18only.h5")
write_sample(path, ("v108", "v108"))
assert superblock_version(path) == 2
def test_libver_not_for_editing(tmp_path):
path = str(tmp_path / "e.h5")
write_sample(path, None)
with pytest.raises(NotImplementedError, match="libver"):
clawhdf5.File(path, "r+", libver="latest")
# Reading ignores it, as the bounds only affect what is written.
with clawhdf5.File(path, "r", libver="v108") as f:
assert f["x"].shape == (10,)