HDF5 2.0 native complex as a first-class type on read; Python libver=
- Datatype::parse returns Datatype::Complex for class 11 (also inside
compounds, arrays and VL types) instead of the {r, i} compound view.
- Facade: DType::Complex(Box<DType>); read_complex_f32/f64 accept it.
- h5rs dump/ls/diff print native complex as h5dump/h5ls/h5diff 2.2.0 do
(checked against a fixture written by h5py 3.16 / libhdf5 2.0.0);
dump --json keeps the {r, i} compound (hdf5-json has no complex class).
- clawhdf5-wasm reads native complex datasets as [re, im] pairs.
- Python: clawhdf5.File(path, 'w', libver=...) with h5py's values,
mapped to FileBuilder::libver_bounds; 'v108' output opens in HDF5 1.8.23.
- Docs: known-issues entry moved to Fixed (history), CHANGELOG, READMEs.
Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
@@ -104,7 +104,22 @@ writes `float64`, `float32`, `int64`, `int32`, `uint8`, `complex64` and
|
||||
`complex128` arrays; the file is written on `close()`. Complex arrays are
|
||||
stored as h5py stores them, a compound `{r, i}` that every libhdf5 reads
|
||||
(not HDF5 2.0's native complex type, which only libhdf5 2.0+ reads; the
|
||||
Rust API writes that on request).
|
||||
Rust API writes that on request). Both forms read back as numpy
|
||||
`complex64`/`complex128`.
|
||||
|
||||
`libver=` sets the library version bounds as in h5py: `'v108'`,
|
||||
`'v110'`, `'v112'`, `'v114'`, `'v200'` or `'latest'` (the low bound; the
|
||||
high bound is then `'latest'`), or a `(low, high)` tuple. With `'v108'`
|
||||
the file is written in the HDF5 1.8 format (version-2 superblock,
|
||||
version-1 B-tree chunk indexes), which HDF5 1.8 reads; the default is the
|
||||
HDF5 1.10 format. `'earliest'` as the low bound writes the 1.8 format too,
|
||||
with a `UserWarning`: clawhdf5 cannot write the pre-1.8 format. The
|
||||
argument is ignored when reading and refused for `'r+'`/`'a'`.
|
||||
|
||||
```python
|
||||
with clawhdf5.File("old.h5", "w", libver="v108") as f: # HDF5 1.8 reads it
|
||||
f.create_dataset("x", data=np.arange(10.0), chunks=(5,), compression="gzip")
|
||||
```
|
||||
|
||||
## Editing a file in place
|
||||
|
||||
|
||||
@@ -13,10 +13,13 @@ use crate::attrs::PyAttrs;
|
||||
use crate::group::{PyGroup, ReadGroup, WriteGroupState, finalize_write_group};
|
||||
use crate::handle::Handle;
|
||||
use crate::{DatasetSpec, OwnedAttrValue, apply_dataset_spec, extract_numpy_data, to_py_err};
|
||||
use clawhdf5_rs::LibVer;
|
||||
|
||||
/// Internal state for write mode.
|
||||
struct WriteState {
|
||||
path: PathBuf,
|
||||
/// `libver=` as (low, high); `None` keeps the writer's default.
|
||||
libver: Option<(LibVer, LibVer)>,
|
||||
root_datasets: Vec<DatasetSpec>,
|
||||
root_attrs: Arc<Mutex<Vec<(String, OwnedAttrValue)>>>,
|
||||
groups: Vec<Arc<Mutex<WriteGroupState>>>,
|
||||
@@ -80,10 +83,32 @@ impl PyFile {
|
||||
/// `az://`; which schemes work depends on how the wheel was built)
|
||||
/// to read the file remotely with default options (see `open_url`)
|
||||
/// mode: 'r' for read (default), 'w' for write
|
||||
/// libver: library version bounds for mode 'w', as h5py's: one of
|
||||
/// 'earliest', 'v108', 'v110', 'v112', 'v114', 'v200', 'latest' (the
|
||||
/// low bound; the high bound is then 'latest') or a (low, high)
|
||||
/// tuple of them. The low bound is the oldest HDF5 release whose
|
||||
/// format the file uses ('v108': HDF5 1.8 can read it); the high
|
||||
/// bound the newest whose features it may use. clawhdf5 cannot write
|
||||
/// the pre-1.8 format, so a low bound of 'earliest' writes the 1.8
|
||||
/// format (with a warning) and a high bound of 'earliest' is an
|
||||
/// error. Default (None): the HDF5 1.10 format clawhdf5 has always
|
||||
/// written. Ignored for reading; not supported with 'r+' / 'a'.
|
||||
#[new]
|
||||
#[pyo3(signature = (path, mode="r"))]
|
||||
fn new(py: Python<'_>, path: &str, mode: &str) -> PyResult<Self> {
|
||||
#[pyo3(signature = (path, mode="r", libver=None))]
|
||||
fn new(
|
||||
py: Python<'_>,
|
||||
path: &str,
|
||||
mode: &str,
|
||||
libver: Option<&Bound<'_, PyAny>>,
|
||||
) -> PyResult<Self> {
|
||||
let filename = path.to_string();
|
||||
let libver = libver.map(|v| parse_libver(py, v)).transpose()?;
|
||||
if libver.is_some() && matches!(mode, "r+" | "a") {
|
||||
return Err(PyNotImplementedError::new_err(format!(
|
||||
"libver with mode '{mode}': clawhdf5's in-place editor keeps the format \
|
||||
versions the file already uses"
|
||||
)));
|
||||
}
|
||||
if is_url(path) {
|
||||
if mode != "r" {
|
||||
return Err(PyValueError::new_err(format!(
|
||||
@@ -113,6 +138,7 @@ impl PyFile {
|
||||
// Absolute now: the file is written at close, possibly
|
||||
// after the working directory changed.
|
||||
path: std::path::absolute(path).unwrap_or_else(|_| PathBuf::from(path)),
|
||||
libver,
|
||||
root_datasets: Vec::new(),
|
||||
root_attrs: Arc::new(Mutex::new(Vec::new())),
|
||||
groups: Vec::new(),
|
||||
@@ -460,10 +486,71 @@ fn parse_compression(
|
||||
}
|
||||
}
|
||||
|
||||
/// One of h5py's `libver` names as a bound; `high` says which end it is.
|
||||
/// `Ok(None)` is 'earliest' as the low bound: the pre-1.8 format, which
|
||||
/// clawhdf5 cannot write.
|
||||
fn libver_name(name: &str, high: bool) -> PyResult<Option<LibVer>> {
|
||||
Ok(Some(match name {
|
||||
"earliest" if high => {
|
||||
return Err(PyValueError::new_err(
|
||||
"libver high bound 'earliest' (the pre-1.8 format) cannot be written by \
|
||||
clawhdf5; the oldest format it writes is 'v108'",
|
||||
));
|
||||
}
|
||||
"earliest" => return Ok(None),
|
||||
"v108" => LibVer::V18,
|
||||
"v110" => LibVer::V110,
|
||||
"v112" => LibVer::V112,
|
||||
"v114" => LibVer::V114,
|
||||
"v200" => LibVer::V200,
|
||||
"latest" => LibVer::Latest,
|
||||
other => {
|
||||
return Err(PyValueError::new_err(format!(
|
||||
"unknown libver '{other}'; expected 'earliest', 'v108', 'v110', 'v112', \
|
||||
'v114', 'v200' or 'latest'"
|
||||
)));
|
||||
}
|
||||
}))
|
||||
}
|
||||
|
||||
/// h5py's `libver=`: a name (the low bound, high bound 'latest') or a
|
||||
/// `(low, high)` pair.
|
||||
fn parse_libver(py: Python<'_>, v: &Bound<'_, PyAny>) -> PyResult<(LibVer, LibVer)> {
|
||||
let (low, high): (String, String) = match v.extract::<String>() {
|
||||
Ok(name) => (name, "latest".into()),
|
||||
Err(_) => v.extract().map_err(|_| {
|
||||
PyValueError::new_err("libver must be a string or a (low, high) tuple of strings")
|
||||
})?,
|
||||
};
|
||||
let Some(high) = libver_name(&high, true)? else {
|
||||
unreachable!("a high bound is never None")
|
||||
};
|
||||
let low = match libver_name(&low, false)? {
|
||||
Some(low) => low,
|
||||
None => {
|
||||
PyModule::import(py, "warnings")?.getattr("warn")?.call1((
|
||||
"libver 'earliest': clawhdf5 cannot write the pre-1.8 format; the file \
|
||||
is written in the HDF5 1.8 format ('v108') instead",
|
||||
py.get_type::<pyo3::exceptions::PyUserWarning>(),
|
||||
))?;
|
||||
LibVer::V18
|
||||
}
|
||||
};
|
||||
if low > high {
|
||||
return Err(PyValueError::new_err(format!(
|
||||
"libver low bound {low} is newer than the high bound {high}"
|
||||
)));
|
||||
}
|
||||
Ok((low, high))
|
||||
}
|
||||
|
||||
/// Build and write the HDF5 file from accumulated write state.
|
||||
fn finalize_write(state: WriteState) -> PyResult<()> {
|
||||
crate::no_panic(|| {
|
||||
let mut builder = clawhdf5_rs::FileBuilder::new();
|
||||
if let Some((low, high)) = state.libver {
|
||||
builder.libver_bounds(low, high);
|
||||
}
|
||||
|
||||
// Root attributes
|
||||
let root_attrs = state.root_attrs.lock().unwrap_or_else(|e| e.into_inner());
|
||||
@@ -520,6 +607,7 @@ mod tests {
|
||||
|
||||
let state = WriteState {
|
||||
path: path.clone(),
|
||||
libver: None,
|
||||
root_datasets: vec![DatasetSpec {
|
||||
name: "data".into(),
|
||||
data: crate::DatasetData::F64(vec![1.0, 2.0, 3.0]),
|
||||
|
||||
@@ -0,0 +1,143 @@
|
||||
"""`clawhdf5.File(path, 'w', libver=...)`: h5py's library version bounds.
|
||||
|
||||
A file written with libver='v108' must open in HDF5 1.8. Its h5dump is
|
||||
found through CLAWHDF5_H5DUMP18 or at ~/.cache/hdf5-1.8.23/bin/h5dump
|
||||
(scripts/build-hdf5-1.8.sh builds it); without it that check is skipped.
|
||||
"""
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
|
||||
import clawhdf5
|
||||
|
||||
|
||||
def superblock_version(path):
|
||||
with open(path, "rb") as f:
|
||||
head = f.read(9)
|
||||
assert head[:8] == b"\x89HDF\r\n\x1a\n"
|
||||
return head[8]
|
||||
|
||||
|
||||
def h5dump18():
|
||||
path = os.environ.get("CLAWHDF5_H5DUMP18") or os.path.expanduser(
|
||||
"~/.cache/hdf5-1.8.23/bin/h5dump"
|
||||
)
|
||||
try:
|
||||
out = subprocess.run([path, "--version"], capture_output=True, text=True)
|
||||
except OSError:
|
||||
return None
|
||||
return path if "1.8." in out.stdout else None
|
||||
|
||||
|
||||
def write_sample(path, libver):
|
||||
with clawhdf5.File(path, "w", libver=libver) as f:
|
||||
f.create_dataset("x", data=np.arange(10, dtype=np.float64))
|
||||
f.create_dataset(
|
||||
"chunked",
|
||||
data=np.arange(100, dtype=np.int32),
|
||||
chunks=(30,),
|
||||
compression="gzip",
|
||||
)
|
||||
f.create_dataset("z", data=np.array([1 + 2j, -3j], dtype=np.complex128))
|
||||
g = f.create_group("g")
|
||||
g.create_dataset("y", data=np.ones(3, dtype=np.float32))
|
||||
f.attrs["version"] = 1
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"libver, sb",
|
||||
[
|
||||
(None, 3),
|
||||
("v108", 2),
|
||||
(("v108", "v108"), 2),
|
||||
(("v108", "latest"), 2),
|
||||
("v110", 3),
|
||||
("v112", 3),
|
||||
("v114", 3),
|
||||
("v200", 3),
|
||||
("latest", 3),
|
||||
(("v110", "v200"), 3),
|
||||
],
|
||||
)
|
||||
def test_libver_bounds(h5py, tmp_path, libver, sb):
|
||||
path = str(tmp_path / "libver.h5")
|
||||
write_sample(path, libver)
|
||||
# v108 low bound: HDF5 1.8's version-2 superblock; 1.10 and later: 3.
|
||||
assert superblock_version(path) == sb
|
||||
with h5py.File(path, "r") as f:
|
||||
np.testing.assert_array_equal(f["x"][:], np.arange(10.0))
|
||||
np.testing.assert_array_equal(f["chunked"][:], np.arange(100))
|
||||
np.testing.assert_array_equal(f["z"][:], [1 + 2j, -3j])
|
||||
np.testing.assert_array_equal(f["g/y"][:], np.ones(3))
|
||||
assert f.attrs["version"] == 1
|
||||
with clawhdf5.File(path, "r") as f:
|
||||
np.testing.assert_array_equal(f["chunked"][:], np.arange(100))
|
||||
|
||||
|
||||
def test_libver_v108_opens_in_hdf5_1_8(tmp_path):
|
||||
h5dump = h5dump18()
|
||||
if h5dump is None:
|
||||
pytest.skip("no HDF5 1.8 h5dump (CLAWHDF5_H5DUMP18)")
|
||||
path = str(tmp_path / "v108.h5")
|
||||
write_sample(path, "v108")
|
||||
out = subprocess.run([h5dump, path], capture_output=True, text=True)
|
||||
assert out.returncode == 0, out.stderr
|
||||
assert "h5dump error" not in out.stderr
|
||||
assert 'DATASET "y"' in out.stdout
|
||||
# The chunked, deflated dataset (a version-1 B-tree) reads in full.
|
||||
out = subprocess.run(
|
||||
[h5dump, "-w", "0", "-d", "/chunked", path], capture_output=True, text=True
|
||||
)
|
||||
assert out.returncode == 0, out.stderr
|
||||
assert "(0): " + ", ".join(str(k) for k in range(100)) + "\n" in out.stdout
|
||||
# The default (1.10 format) does not open in 1.8: the bound matters.
|
||||
path = str(tmp_path / "default.h5")
|
||||
write_sample(path, None)
|
||||
out = subprocess.run([h5dump, path], capture_output=True, text=True)
|
||||
assert out.returncode != 0
|
||||
|
||||
|
||||
def test_libver_earliest_writes_v108_with_a_warning(h5py, tmp_path):
|
||||
path = str(tmp_path / "earliest.h5")
|
||||
with pytest.warns(UserWarning, match="pre-1.8"):
|
||||
write_sample(path, "earliest")
|
||||
assert superblock_version(path) == 2
|
||||
with h5py.File(path, "r") as f:
|
||||
np.testing.assert_array_equal(f["x"][:], np.arange(10.0))
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"libver, match",
|
||||
[
|
||||
(("v108", "earliest"), "high bound 'earliest'"),
|
||||
("v109", "unknown libver 'v109'"),
|
||||
(("latest", "v108"), "newer than the high bound"),
|
||||
(("v108",), "tuple"),
|
||||
(3, "tuple"),
|
||||
],
|
||||
)
|
||||
def test_libver_errors(tmp_path, libver, match):
|
||||
with pytest.raises(ValueError, match=match):
|
||||
clawhdf5.File(str(tmp_path / "bad.h5"), "w", libver=libver)
|
||||
|
||||
|
||||
def test_libver_v108_high_bound_writes_everything(tmp_path):
|
||||
# Native complex numbers need HDF5 2.0; the Python writer stores complex
|
||||
# as h5py's {r, i} compound, which 1.8 reads, so the 1.8 high bound is
|
||||
# fine for everything clawhdf5.File writes.
|
||||
path = str(tmp_path / "v18only.h5")
|
||||
write_sample(path, ("v108", "v108"))
|
||||
assert superblock_version(path) == 2
|
||||
|
||||
|
||||
def test_libver_not_for_editing(tmp_path):
|
||||
path = str(tmp_path / "e.h5")
|
||||
write_sample(path, None)
|
||||
with pytest.raises(NotImplementedError, match="libver"):
|
||||
clawhdf5.File(path, "r+", libver="latest")
|
||||
# Reading ignores it, as the bounds only affect what is written.
|
||||
with clawhdf5.File(path, "r", libver="v108") as f:
|
||||
assert f["x"].shape == (10,)
|
||||
Reference in New Issue
Block a user