py: in-place editing (clawhdf5.File(path, 'r+')) through FileEditor
clawhdf5.File(path, 'r+') (and 'a' on an existing file) holds a FileEditor, and with it the file's exclusive lock, until close(): - ds[key] = value: h5py's keys and broadcasting (numpy's rules for slices and integers with extra leading 1-axes allowed; the exact shape for an index list, a scalar only where h5py expands it). Arrays are converted as libhdf5 converts them in native byte order (integers saturate, floats truncate toward zero and clip, integers go into h5py's bool enum by value); other values through numpy.asarray(value, dtype=ds.dtype), as h5py does. NaN into an integer dataset is a ValueError instead of libhdf5's arbitrary value. The value preparation is a small Python module compiled into the extension (src/edit_helpers.py). - ds.resize(shape) / ds.resize(n, axis=k) with h5py's argument rules. - attrs[name] = value, attrs.create(name, data, shape, dtype), attrs.modify: numeric, bool, complex, bytes and str data of any shape, with h5py's HDF5 types; str is stored as fixed-length UTF-8 (the editor cannot write variable-length strings). - File.mode, File.flush(), Dataset.chunks. Each edit runs with the GIL released under the file handle's write lock (no read sees a half-written edit), then the file is reopened; datasets and attrs objects re-read their shape and attributes when the handle's edit generation moved. What the editor cannot do is NotImplementedError before anything is written: deleting attributes or objects, creating datasets or groups, compound fields by name, variable-length data, and FileEditor's own limits. Where libhdf5 2.0 (h5py 3.16) converts inconsistently -- its soft conversions in non-native byte order (a float in (-1, 0) becomes the integer minimum, same-size unsigned->signed wraps) and native casts that are undefined in C (half floats into unsigned, float(max) rounded up) -- clawhdf5 saturates as libhdf5's native path does; listed in docs/known-issues.md. Tests (tests/test_edit.py): every edit applied by h5py and by clawhdf5 to copies of the same file and both read back through h5py after each edit, on h5py files (libver earliest, v114, latest) and a clawhdf5 file: a fixed sequence over every chunk index kind, compact/contiguous/gzip layouts and numeric, bool, enum, complex, string and compound types, 16 random sequences of 40 edits, and a numeric conversion matrix; a refused edit must be refused by both and leave the file unchanged. Also dense attributes, locking, objects seeing edits, readers racing a writer, and h5dump (plus h5rs check in ci-test.sh) on every edited file. The read-vs-h5py suite also runs on a file opened 'r+'. Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
@@ -9,12 +9,12 @@
|
||||
//! Python threads reading the same or different datasets run in parallel,
|
||||
//! and a remote file's network reads never hold the GIL.
|
||||
|
||||
use std::sync::Arc;
|
||||
use std::sync::{Arc, Mutex, PoisonError};
|
||||
|
||||
use clawhdf5_format::datatype::Datatype;
|
||||
use clawhdf5_format::object_header::ObjectHeader;
|
||||
use clawhdf5_rs::File;
|
||||
use pyo3::exceptions::{PyTypeError, PyValueError};
|
||||
use pyo3::exceptions::{PyNotImplementedError, PyOSError, PyTypeError, PyValueError};
|
||||
use pyo3::prelude::*;
|
||||
use pyo3::types::{PyList, PyTuple};
|
||||
|
||||
@@ -22,7 +22,7 @@ use crate::attrs::PyAttrs;
|
||||
use crate::convert::{Converter, Elements, VlError, resolve_vl};
|
||||
use crate::handle::Handle;
|
||||
use crate::select::{self, Plan};
|
||||
use crate::{PyEmpty, node, to_py_err};
|
||||
use crate::{PyEmpty, edit, node, to_py_err};
|
||||
|
||||
/// What opening a dataset reads from the file (without the GIL).
|
||||
pub(crate) struct DatasetMeta {
|
||||
@@ -67,8 +67,9 @@ pub struct PyDataset {
|
||||
/// Where the dataset's object header is: reads open it from here rather
|
||||
/// than resolve `path` again.
|
||||
addr: u64,
|
||||
/// `None` for a dataset with a null dataspace (h5py's `Empty`).
|
||||
shape: Option<Vec<u64>>,
|
||||
/// The shape (`None` for a null dataspace, h5py's `Empty`), with the
|
||||
/// file generation it was read at: an edit (a resize) may change it.
|
||||
shape: Mutex<(u64, Option<Vec<u64>>)>,
|
||||
/// The chunk shape, for a chunked dataset.
|
||||
chunks: Option<Vec<u64>>,
|
||||
datatype: Datatype,
|
||||
@@ -84,13 +85,14 @@ impl PyDataset {
|
||||
addr: u64,
|
||||
meta: DatasetMeta,
|
||||
) -> Self {
|
||||
let generation = handle.generation();
|
||||
let conv = crate::no_panic(|| Converter::new(py, &meta.datatype, handle.offset_size))
|
||||
.map_err(|e| e.value(py).to_string());
|
||||
Self {
|
||||
handle,
|
||||
path,
|
||||
addr,
|
||||
shape: meta.shape,
|
||||
shape: Mutex::new((generation, meta.shape)),
|
||||
chunks: meta.chunks,
|
||||
datatype: meta.datatype,
|
||||
conv,
|
||||
@@ -103,10 +105,54 @@ impl PyDataset {
|
||||
.map_err(|msg| PyTypeError::new_err(format!("{}: {msg}", node::name(&self.path))))
|
||||
}
|
||||
|
||||
/// The current shape: the one read at open, or re-read after an edit.
|
||||
fn dims(&self, py: Python<'_>) -> PyResult<Option<Vec<u64>>> {
|
||||
let generation = self.handle.generation();
|
||||
{
|
||||
let cached = self.shape.lock().unwrap_or_else(PoisonError::into_inner);
|
||||
if cached.0 == generation {
|
||||
return Ok(cached.1.clone());
|
||||
}
|
||||
}
|
||||
let addr = self.addr;
|
||||
let null = self
|
||||
.shape
|
||||
.lock()
|
||||
.unwrap_or_else(PoisonError::into_inner)
|
||||
.1
|
||||
.is_none();
|
||||
let shape = if null {
|
||||
None
|
||||
} else {
|
||||
Some(self.handle.with(py, |f| {
|
||||
f.dataset_at(addr)
|
||||
.and_then(|ds| ds.shape())
|
||||
.map_err(to_py_err)
|
||||
})?)
|
||||
};
|
||||
*self.shape.lock().unwrap_or_else(PoisonError::into_inner) = (generation, shape.clone());
|
||||
Ok(shape)
|
||||
}
|
||||
|
||||
fn check_writable(&self) -> PyResult<()> {
|
||||
if self.handle.is_writable() {
|
||||
Ok(())
|
||||
} else {
|
||||
Err(PyOSError::new_err(format!(
|
||||
"{}: the file is open read-only; open it with mode 'r+' to change it",
|
||||
node::name(&self.path)
|
||||
)))
|
||||
}
|
||||
}
|
||||
|
||||
/// Read the selection described by `plan` into a numpy array.
|
||||
fn read_plan<'py>(&self, py: Python<'py>, plan: &Plan) -> PyResult<Bound<'py, PyAny>> {
|
||||
fn read_plan<'py>(
|
||||
&self,
|
||||
py: Python<'py>,
|
||||
plan: &Plan,
|
||||
dims: &[u64],
|
||||
) -> PyResult<Bound<'py, PyAny>> {
|
||||
let conv = self.converter()?;
|
||||
let dims = self.shape.as_deref().unwrap_or(&[]);
|
||||
let out_shape = plan.out_shape();
|
||||
|
||||
let arr = if plan.is_empty() {
|
||||
@@ -272,7 +318,7 @@ impl PyDataset {
|
||||
/// The shape of the dataset (`None` for an empty/null dataspace).
|
||||
#[getter]
|
||||
fn shape<'py>(&self, py: Python<'py>) -> PyResult<Bound<'py, PyAny>> {
|
||||
match &self.shape {
|
||||
match self.dims(py)? {
|
||||
Some(s) => Ok(PyTuple::new(py, s)?.into_any()),
|
||||
None => Ok(py.None().into_bound(py)),
|
||||
}
|
||||
@@ -281,7 +327,7 @@ impl PyDataset {
|
||||
/// The maximum shape (`None` per unlimited dimension), like h5py.
|
||||
#[getter]
|
||||
fn maxshape<'py>(&self, py: Python<'py>) -> PyResult<Bound<'py, PyAny>> {
|
||||
let Some(shape) = &self.shape else {
|
||||
let Some(shape) = self.dims(py)? else {
|
||||
return Ok(py.None().into_bound(py));
|
||||
};
|
||||
let addr = self.addr;
|
||||
@@ -292,7 +338,7 @@ impl PyDataset {
|
||||
.and_then(|ds| ds.max_dimensions())
|
||||
.map_err(to_py_err)
|
||||
})?
|
||||
.unwrap_or_else(|| shape.clone());
|
||||
.unwrap_or(shape);
|
||||
let items: Vec<Option<u64>> = max
|
||||
.into_iter()
|
||||
.map(|d| (d != u64::MAX).then_some(d))
|
||||
@@ -300,6 +346,15 @@ impl PyDataset {
|
||||
Ok(PyTuple::new(py, items)?.into_any())
|
||||
}
|
||||
|
||||
/// The chunk shape, or `None` for a dataset that is not chunked.
|
||||
#[getter]
|
||||
fn chunks<'py>(&self, py: Python<'py>) -> PyResult<Bound<'py, PyAny>> {
|
||||
match &self.chunks {
|
||||
Some(c) => Ok(PyTuple::new(py, c)?.into_any()),
|
||||
None => Ok(py.None().into_bound(py)),
|
||||
}
|
||||
}
|
||||
|
||||
/// The dataset's numpy dtype, as h5py reports it.
|
||||
#[getter]
|
||||
fn dtype<'py>(&self, py: Python<'py>) -> PyResult<Bound<'py, PyAny>> {
|
||||
@@ -307,14 +362,14 @@ impl PyDataset {
|
||||
}
|
||||
|
||||
#[getter]
|
||||
fn ndim(&self) -> usize {
|
||||
self.shape.as_ref().map_or(0, Vec::len)
|
||||
fn ndim(&self, py: Python<'_>) -> PyResult<usize> {
|
||||
Ok(self.dims(py)?.map_or(0, |s| s.len()))
|
||||
}
|
||||
|
||||
/// Number of elements (`None` for an empty/null dataspace, as h5py).
|
||||
#[getter]
|
||||
fn size(&self) -> Option<u64> {
|
||||
self.shape.as_ref().map(|s| s.iter().product())
|
||||
fn size(&self, py: Python<'_>) -> PyResult<Option<u64>> {
|
||||
Ok(self.dims(py)?.map(|s| s.iter().product()))
|
||||
}
|
||||
|
||||
/// The dataset's full name, e.g. `/group/data`.
|
||||
@@ -323,7 +378,8 @@ impl PyDataset {
|
||||
node::name(&self.path)
|
||||
}
|
||||
|
||||
/// The dataset's attributes (read-only, dict-like).
|
||||
/// The dataset's attributes (dict-like; writable in a file opened with
|
||||
/// `'r+'`).
|
||||
#[getter]
|
||||
fn attrs(&self, py: Python<'_>) -> PyResult<PyAttrs> {
|
||||
PyAttrs::read(py, Arc::clone(&self.handle), self.addr, &self.path)
|
||||
@@ -338,7 +394,7 @@ impl PyDataset {
|
||||
py: Python<'py>,
|
||||
key: &Bound<'py, PyAny>,
|
||||
) -> PyResult<Bound<'py, PyAny>> {
|
||||
let Some(dims) = &self.shape else {
|
||||
let Some(dims) = self.dims(py)? else {
|
||||
let is_empty_tuple = key.cast::<PyTuple>().is_ok_and(|t| t.is_empty());
|
||||
let is_ellipsis = key.is_instance_of::<pyo3::types::PyEllipsis>();
|
||||
if is_empty_tuple || is_ellipsis {
|
||||
@@ -347,8 +403,109 @@ impl PyDataset {
|
||||
}
|
||||
return Err(PyValueError::new_err("Empty datasets cannot be sliced"));
|
||||
};
|
||||
let plan = select::parse(key, dims)?;
|
||||
self.read_plan(py, &plan)
|
||||
let plan = select::parse(key, &dims)?;
|
||||
self.read_plan(py, &plan, &dims)
|
||||
}
|
||||
|
||||
/// Write with h5py indexing (file opened with `'r+'`): `ds[key] = value`.
|
||||
///
|
||||
/// The key is what `ds[key]` reads (without compound field names). The
|
||||
/// value is converted to the dataset's dtype as h5py converts it (a
|
||||
/// numpy array as libhdf5 does, clipping out-of-range numbers; anything
|
||||
/// else through `numpy.asarray(value, dtype=ds.dtype)`), and broadcast
|
||||
/// to the selection as h5py broadcasts. The edit is written and synced
|
||||
/// before this returns; what the in-place editor cannot write raises
|
||||
/// `NotImplementedError` and leaves the file as it was.
|
||||
fn __setitem__(
|
||||
&self,
|
||||
py: Python<'_>,
|
||||
key: &Bound<'_, PyAny>,
|
||||
value: &Bound<'_, PyAny>,
|
||||
) -> PyResult<()> {
|
||||
self.check_writable()?;
|
||||
let Some(dims) = self.dims(py)? else {
|
||||
return Err(PyNotImplementedError::new_err(
|
||||
"writing to an empty (null dataspace) dataset is not supported",
|
||||
));
|
||||
};
|
||||
let plan = select::parse(key, &dims)?;
|
||||
if !plan.fields.is_empty() {
|
||||
return Err(PyNotImplementedError::new_err(
|
||||
"writing compound fields by name is not supported by clawhdf5's in-place editor; \
|
||||
write whole elements",
|
||||
));
|
||||
}
|
||||
let category = edit::category(&self.datatype)?;
|
||||
let conv = self.converter()?;
|
||||
let bytes = edit::dataset_bytes(
|
||||
py,
|
||||
value,
|
||||
conv.dtype.bind(py),
|
||||
category,
|
||||
&plan,
|
||||
self.chunks.as_deref(),
|
||||
)?;
|
||||
if plan.is_empty() {
|
||||
return Ok(());
|
||||
}
|
||||
let sel = edit::selection(&plan, &dims)?;
|
||||
let path = node::name(&self.path);
|
||||
self.handle
|
||||
.edit(py, |ed| ed.write_selection(&path, &sel, &bytes))
|
||||
}
|
||||
|
||||
/// Change the dataset's shape (file opened with `'r+'`), as h5py's
|
||||
/// `Dataset.resize`: `ds.resize((100, 20))`, or `ds.resize(100, axis=0)`.
|
||||
/// Only chunked datasets, within their maximum shape; new elements read
|
||||
/// as the fill value.
|
||||
#[pyo3(signature = (size, axis=None))]
|
||||
fn resize(&self, py: Python<'_>, size: &Bound<'_, PyAny>, axis: Option<isize>) -> PyResult<()> {
|
||||
self.check_writable()?;
|
||||
let Some(dims) = self.dims(py)? else {
|
||||
return Err(PyTypeError::new_err("Empty datasets cannot be resized"));
|
||||
};
|
||||
if self.chunks.is_none() {
|
||||
return Err(PyTypeError::new_err("Only chunked datasets can be resized"));
|
||||
}
|
||||
let shape: Vec<u64> = match axis {
|
||||
Some(axis) => {
|
||||
let rank = dims.len();
|
||||
let a = usize::try_from(axis)
|
||||
.ok()
|
||||
.filter(|&a| a < rank)
|
||||
.ok_or_else(|| {
|
||||
PyValueError::new_err(format!(
|
||||
"Invalid axis (0 to {} allowed)",
|
||||
rank.saturating_sub(1)
|
||||
))
|
||||
})?;
|
||||
let n: u64 = size.extract().map_err(|_| {
|
||||
PyTypeError::new_err("Argument must be a single int if axis is specified")
|
||||
})?;
|
||||
let mut s = dims.clone();
|
||||
s[a] = n;
|
||||
s
|
||||
}
|
||||
// As h5py: without `axis` the size is a sequence (`tuple(size)`).
|
||||
None => size.extract().map_err(|_| {
|
||||
PyTypeError::new_err(format!(
|
||||
"'{}' object is not iterable",
|
||||
size.get_type()
|
||||
.name()
|
||||
.map(|n| n.to_string())
|
||||
.unwrap_or_default()
|
||||
))
|
||||
})?,
|
||||
};
|
||||
if shape.len() != dims.len() {
|
||||
return Err(PyValueError::new_err(format!(
|
||||
"new shape {shape:?} has {} dimensions, the dataset {}",
|
||||
shape.len(),
|
||||
dims.len()
|
||||
)));
|
||||
}
|
||||
let path = node::name(&self.path);
|
||||
self.handle.edit(py, |ed| ed.resize(&path, &shape))
|
||||
}
|
||||
|
||||
/// `numpy.asarray(ds)` reads the whole dataset.
|
||||
@@ -360,20 +517,20 @@ impl PyDataset {
|
||||
copy: Option<bool>,
|
||||
) -> PyResult<Bound<'py, PyAny>> {
|
||||
let _ = copy; // every read is a fresh array
|
||||
let Some(dims) = &self.shape else {
|
||||
let Some(dims) = self.dims(py)? else {
|
||||
return Err(PyValueError::new_err("an empty dataset has no array value"));
|
||||
};
|
||||
let ellipsis = pyo3::types::PyEllipsis::get(py).to_owned().into_any();
|
||||
let plan = select::parse(&ellipsis, dims)?;
|
||||
let arr = self.read_plan(py, &plan)?;
|
||||
let plan = select::parse(&ellipsis, &dims)?;
|
||||
let arr = self.read_plan(py, &plan, &dims)?;
|
||||
match dtype {
|
||||
Some(dt) => arr.call_method1("astype", (dt,)),
|
||||
None => Ok(arr),
|
||||
}
|
||||
}
|
||||
|
||||
fn __len__(&self) -> PyResult<usize> {
|
||||
match self.shape.as_deref() {
|
||||
fn __len__(&self, py: Python<'_>) -> PyResult<usize> {
|
||||
match self.dims(py)?.as_deref() {
|
||||
Some([first, ..]) => Ok(*first as usize),
|
||||
_ => Err(PyTypeError::new_err(
|
||||
"Attempt to take len() of scalar dataset",
|
||||
@@ -391,9 +548,10 @@ impl PyDataset {
|
||||
.unwrap_or_default(),
|
||||
Err(_) => format!("{:?}", self.datatype),
|
||||
};
|
||||
let shape = match &self.shape {
|
||||
Some(s) => format!("{s:?}"),
|
||||
None => "None".to_string(),
|
||||
let shape = match self.dims(py) {
|
||||
Ok(Some(s)) => format!("{s:?}"),
|
||||
Ok(None) => "None".to_string(),
|
||||
Err(_) => "?".to_string(),
|
||||
};
|
||||
format!(
|
||||
"<HDF5 dataset \"{}\": shape {shape}, type \"{dtype}\">",
|
||||
|
||||
Reference in New Issue
Block a user