Merge branch 'feat/p3-python-remote-edit' into feat/p3-wasm-swmr-python
# Conflicts: # CHANGELOG.md # docs/design/range-reads.md # docs/known-issues.md
This commit is contained in:
@@ -65,7 +65,8 @@ const MSG_FLAG_DONTSHARE: u8 = 0x04;
|
||||
/// that may be mid-update.
|
||||
///
|
||||
/// Every method is one self-contained edit: it re-reads the file's
|
||||
/// metadata, applies the change, and syncs the file before returning.
|
||||
/// metadata (from the file it holds open, never by path), applies the
|
||||
/// change, and syncs the file before returning.
|
||||
///
|
||||
/// # What it can change
|
||||
///
|
||||
@@ -750,9 +751,17 @@ impl FileEditor {
|
||||
/// consistent: a metadata cache image, paged or persistent free-space
|
||||
/// management, a multi-file driver, a file another writer has marked
|
||||
/// open (superblock version 3 consistency flags).
|
||||
///
|
||||
/// The path is only used to open the file: every edit is planned from
|
||||
/// and written to the file opened here, even if the path is renamed,
|
||||
/// replaced or (relative) resolved from another working directory
|
||||
/// later. [`path`](Self::path) is the absolute path it had at open.
|
||||
pub fn open<P: AsRef<Path>>(path: P) -> Result<Self, Error> {
|
||||
let path = path.as_ref().to_path_buf();
|
||||
let file = OpenOptions::new().read(true).write(true).open(&path)?;
|
||||
let file = OpenOptions::new()
|
||||
.read(true)
|
||||
.write(true)
|
||||
.open(path.as_ref())?;
|
||||
let path = std::fs::canonicalize(path.as_ref())?;
|
||||
match file.try_lock() {
|
||||
Ok(()) => {}
|
||||
Err(TryLockError::WouldBlock) => {
|
||||
@@ -768,16 +777,81 @@ impl FileEditor {
|
||||
file,
|
||||
free: FreeList::default(),
|
||||
};
|
||||
let f = File::open(&ed.path)?;
|
||||
let f = ed.plan_reader()?;
|
||||
check_editable(&f)?;
|
||||
Ok(ed)
|
||||
}
|
||||
|
||||
/// The file's path.
|
||||
/// The file's absolute path when it was opened (it may have been
|
||||
/// renamed since; the editor keeps editing the file it opened).
|
||||
pub fn path(&self) -> &Path {
|
||||
&self.path
|
||||
}
|
||||
|
||||
/// A reader over the file this editor holds, as last written: the file
|
||||
/// opened by [`open`](Self::open), not whatever its path names now.
|
||||
///
|
||||
/// It opens the file anew (read-only), so it does not share the
|
||||
/// editor's lock and stays usable after the editor is dropped: on Linux
|
||||
/// through `/proc/self/fd`, which reaches the held file even after its
|
||||
/// path was renamed or replaced; elsewhere by the path the file had at
|
||||
/// open, refused with [`Error::Io`] when that path no longer names the
|
||||
/// held file (on Unix, compared by device and inode; Windows cannot
|
||||
/// check). With the `mmap` feature the reader maps the file: edits
|
||||
/// through the editor change the bytes it sees, so take a new reader
|
||||
/// after each edit rather than reading through an old one while an edit
|
||||
/// runs.
|
||||
pub fn reader(&self) -> Result<File, Error> {
|
||||
let dir = self.path.parent().map(Path::to_path_buf);
|
||||
File::from_std_file(self.reopen()?, dir)
|
||||
}
|
||||
|
||||
/// A new read-only open file description of the held file (see
|
||||
/// [`reader`](Self::reader)).
|
||||
fn reopen(&self) -> Result<std::fs::File, Error> {
|
||||
#[cfg(target_os = "linux")]
|
||||
{
|
||||
use std::os::fd::AsRawFd;
|
||||
let proc = format!("/proc/self/fd/{}", self.file.as_raw_fd());
|
||||
if let Ok(f) = std::fs::File::open(proc) {
|
||||
return Ok(f);
|
||||
}
|
||||
}
|
||||
let f = std::fs::File::open(&self.path)?;
|
||||
#[cfg(unix)]
|
||||
{
|
||||
use std::os::unix::fs::MetadataExt;
|
||||
let (a, b) = (self.file.metadata()?, f.metadata()?);
|
||||
if (a.dev(), a.ino()) != (b.dev(), b.ino()) {
|
||||
return Err(Error::Io(std::io::Error::other(format!(
|
||||
"{} no longer names the file being edited (renamed or replaced)",
|
||||
self.path.display()
|
||||
))));
|
||||
}
|
||||
}
|
||||
Ok(f)
|
||||
}
|
||||
|
||||
/// A reader over the held file for planning an edit (dropped before the
|
||||
/// edit writes). On Linux a new open file description through
|
||||
/// `/proc/self/fd`: a mapping of a clone of the held descriptor would
|
||||
/// share its `flock`, and a process forked meanwhile (any
|
||||
/// `std::process::Command` on another thread) would briefly keep the
|
||||
/// lock alive after the editor is dropped. Elsewhere a clone of the
|
||||
/// held descriptor, which follows the file wherever its path goes.
|
||||
fn plan_reader(&self) -> Result<File, Error> {
|
||||
let dir = self.path.parent().map(Path::to_path_buf);
|
||||
#[cfg(target_os = "linux")]
|
||||
{
|
||||
use std::os::fd::AsRawFd;
|
||||
let proc = format!("/proc/self/fd/{}", self.file.as_raw_fd());
|
||||
if let Ok(f) = std::fs::File::open(proc) {
|
||||
return File::from_std_file(f, dir);
|
||||
}
|
||||
}
|
||||
File::from_std_file(self.file.try_clone()?, dir)
|
||||
}
|
||||
|
||||
/// Bytes earlier edits of this editor freed that later ones can still
|
||||
/// reuse.
|
||||
pub fn reusable_bytes(&self) -> u64 {
|
||||
@@ -794,7 +868,7 @@ impl FileEditor {
|
||||
&mut self,
|
||||
op: impl FnOnce(&File, &mut Image<'_>) -> Result<R, Error>,
|
||||
) -> Result<R, Error> {
|
||||
let f = File::open(&self.path)?;
|
||||
let f = self.plan_reader()?;
|
||||
check_editable(&f)?;
|
||||
let sb = f.superblock().clone();
|
||||
let user_block = f.user_block_size();
|
||||
@@ -877,7 +951,7 @@ impl FileEditor {
|
||||
/// the fill value (`H5D__chunk_prune_by_extent`).
|
||||
pub fn resize(&mut self, path: &str, shape: &[u64]) -> Result<(), Error> {
|
||||
self.edit(|f, img| {
|
||||
let t = Target::load(f, path)?;
|
||||
let mut t = Target::load(f, path)?;
|
||||
let dims = t.dims().to_vec();
|
||||
if shape.len() != dims.len() {
|
||||
return Err(Error::InvalidArgument(format!(
|
||||
@@ -889,7 +963,12 @@ impl FileEditor {
|
||||
if shape == dims.as_slice() {
|
||||
return Ok(());
|
||||
}
|
||||
let max = t.ds.max_dimensions.clone().unwrap_or_else(|| dims.clone());
|
||||
// No maximum recorded means the current dimensions (see below).
|
||||
let record_max = t.ds.max_dimensions.is_none();
|
||||
let max =
|
||||
t.ds.max_dimensions
|
||||
.get_or_insert_with(|| dims.clone())
|
||||
.clone();
|
||||
for d in 0..dims.len() {
|
||||
if shape[d] > max[d] {
|
||||
return Err(Error::InvalidArgument(format!(
|
||||
@@ -928,7 +1007,36 @@ impl FileEditor {
|
||||
}
|
||||
put_uint(&mut dims_bytes[d * ls..], n, img.ls);
|
||||
}
|
||||
hdr.patch(img, i, first, &dims_bytes)?;
|
||||
if !record_max {
|
||||
hdr.patch(img, i, first, &dims_bytes)?;
|
||||
} else {
|
||||
// No maximum recorded (clawhdf5's writer, for a dataset
|
||||
// created without a maxshape). libhdf5 never writes such a
|
||||
// dataspace: `H5S_set_extent_simple` records the maximum,
|
||||
// equal to the dimensions when none is given. Reading one,
|
||||
// libhdf5 takes the maximum to be the *current* dimensions
|
||||
// (`H5S_extent_get_dims`), so changing them would also
|
||||
// change the maximum the chunk index was built with — the
|
||||
// Fixed Array linearises chunks by it — and move every
|
||||
// existing chunk. Record the maximum libhdf5 would have
|
||||
// written, the dimensions before this resize, so the index
|
||||
// keeps its layout and the dataset can grow back to them.
|
||||
let body_len = first + dims.len() * ls;
|
||||
if body.len() < body_len || body[2] & !0x01 != 0 {
|
||||
return Err(Error::Unsupported("dataspace message layout".into()));
|
||||
}
|
||||
let mut new_body = body[..first].to_vec();
|
||||
new_body[2] |= 0x01;
|
||||
new_body.extend_from_slice(&dims_bytes);
|
||||
let at = new_body.len();
|
||||
new_body.resize(at + dims.len() * ls, 0);
|
||||
for (d, &n) in dims.iter().enumerate() {
|
||||
put_uint(&mut new_body[at + d * ls..], n, img.ls);
|
||||
}
|
||||
let (flags, corder) = (hdr.msgs[i].flags, hdr.msgs[i].corder);
|
||||
hdr.delete(img, i)?;
|
||||
hdr.insert(img, MSG_DATASPACE, flags, &new_body, corder)?;
|
||||
}
|
||||
let fill = fill_info(img, &hdr)?;
|
||||
hdr.finish(img)?;
|
||||
let expand = shape.iter().zip(&dims).any(|(n, o)| n > o);
|
||||
|
||||
@@ -446,6 +446,38 @@ impl File {
|
||||
}
|
||||
}
|
||||
|
||||
/// A reader over an already open file (the file itself, not whatever
|
||||
/// its path names now): mapped with the `mmap` feature, else read into
|
||||
/// memory. `base_dir` resolves external Virtual Dataset sources.
|
||||
pub(crate) fn from_std_file(
|
||||
file: std::fs::File,
|
||||
base_dir: Option<std::path::PathBuf>,
|
||||
) -> Result<Self, Error> {
|
||||
#[cfg(feature = "mmap")]
|
||||
let mut f = {
|
||||
let reader = clawhdf5_io::MmapReader::from_file(file).map_err(Error::Io)?;
|
||||
let (data, superblock) = FileData::new(Backing::Mmap(reader))?;
|
||||
Self {
|
||||
data,
|
||||
superblock,
|
||||
chunk_cache: ChunkCache::new(),
|
||||
base_dir: None,
|
||||
vds_resolver: None,
|
||||
}
|
||||
};
|
||||
#[cfg(not(feature = "mmap"))]
|
||||
let mut f = {
|
||||
use std::io::{Read, Seek, SeekFrom};
|
||||
let mut file = file;
|
||||
let mut bytes = Vec::new();
|
||||
file.seek(SeekFrom::Start(0)).map_err(Error::Io)?;
|
||||
file.read_to_end(&mut bytes).map_err(Error::Io)?;
|
||||
Self::from_bytes(bytes)?
|
||||
};
|
||||
f.base_dir = base_dir;
|
||||
Ok(f)
|
||||
}
|
||||
|
||||
/// Open an HDF5 file by reading it entirely into memory.
|
||||
///
|
||||
/// This is the pre-mmap behaviour and is useful when memory-mapping is
|
||||
|
||||
@@ -0,0 +1,287 @@
|
||||
//! `FileEditor::resize` on chunked datasets whose dataspace records no
|
||||
//! maximum dimensions, as clawhdf5's writer stored a dataset created without
|
||||
//! a `maxshape` up to 2.7.0 (`fixtures/chunked_no_maxshape_v2_7_0.h5`). libhdf5 never writes such a dataspace (`H5S_set_extent_simple`
|
||||
//! always records the maximum, equal to the dimensions when none is given),
|
||||
//! and its Fixed Array chunk index linearises chunks by the maximum
|
||||
//! dimensions. The editor therefore records the maximum libhdf5 would have
|
||||
//! written (the dimensions the index was built with) before it changes the
|
||||
//! current ones, so existing chunks stay where the index put them and the
|
||||
//! dataset can grow back to its original extent.
|
||||
//!
|
||||
//! The writer now records the maximum too, so h5py can resize what it writes.
|
||||
//!
|
||||
//! Checked against a model of the expected values, with our reader and with
|
||||
//! h5py (`CLAWHDF5_PYTHON`; skipped without it unless
|
||||
//! `CLAWHDF5_REQUIRE_INTEROP=1`), on files clawhdf5 (old and new) and h5py
|
||||
//! wrote.
|
||||
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::process::Command;
|
||||
|
||||
use clawhdf5::{Error, File, FileBuilder, FileEditor};
|
||||
|
||||
fn python() -> String {
|
||||
std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string())
|
||||
}
|
||||
|
||||
fn h5py_ok() -> bool {
|
||||
let ok = Command::new(python())
|
||||
.args(["-c", "import h5py, numpy"])
|
||||
.output()
|
||||
.is_ok_and(|o| o.status.success());
|
||||
if !ok {
|
||||
assert!(
|
||||
!std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1"),
|
||||
"CLAWHDF5_REQUIRE_INTEROP=1 but h5py/numpy is not available"
|
||||
);
|
||||
eprintln!("SKIP (h5py part): h5py/numpy not available");
|
||||
}
|
||||
ok
|
||||
}
|
||||
|
||||
fn py(script: &str) -> String {
|
||||
let o = Command::new(python())
|
||||
.args(["-c", script])
|
||||
.output()
|
||||
.expect("run python");
|
||||
assert!(
|
||||
o.status.success(),
|
||||
"python failed:\n{script}\nSTDERR: {}",
|
||||
String::from_utf8_lossy(&o.stderr)
|
||||
);
|
||||
String::from_utf8_lossy(&o.stdout).trim().to_string()
|
||||
}
|
||||
|
||||
/// Row-major values of a 2-D model after resizing `data` (shape `old`) to
|
||||
/// `new`: kept elements keep their values, new ones are 0 (the fill value).
|
||||
fn resized(data: &[f32], old: [u64; 2], new: [u64; 2]) -> Vec<f32> {
|
||||
let mut out = vec![0f32; (new[0] * new[1]) as usize];
|
||||
for r in 0..old[0].min(new[0]) {
|
||||
for c in 0..old[1].min(new[1]) {
|
||||
out[(r * new[1] + c) as usize] = data[(r * old[1] + c) as usize];
|
||||
}
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
/// Our reader and (when available) h5py read `expect` at `shape`.
|
||||
fn check(path: &Path, name: &str, shape: [u64; 2], expect: &[f32], with_h5py: bool) {
|
||||
let f = File::open(path).unwrap();
|
||||
let d = f.dataset(name).unwrap();
|
||||
assert_eq!(d.shape().unwrap(), shape);
|
||||
assert_eq!(
|
||||
d.read_f32().unwrap(),
|
||||
expect,
|
||||
"{name}: our reader at {shape:?}"
|
||||
);
|
||||
if with_h5py {
|
||||
let got = py(&format!(
|
||||
"import h5py, numpy as np\n\
|
||||
with h5py.File({p:?}, 'r') as f:\n\
|
||||
\x20 d = f[{name:?}][()]\n\
|
||||
print(d.shape, ','.join(repr(float(x)) for x in d.ravel()))",
|
||||
p = path.to_str().unwrap()
|
||||
));
|
||||
let want = format!(
|
||||
"({}, {}) {}",
|
||||
shape[0],
|
||||
shape[1],
|
||||
expect
|
||||
.iter()
|
||||
.map(|x| format!("{:?}", f64::from(*x)))
|
||||
.collect::<Vec<_>>()
|
||||
.join(",")
|
||||
);
|
||||
assert_eq!(got, want.trim(), "{name}: h5py at {shape:?}");
|
||||
}
|
||||
}
|
||||
|
||||
/// Resize `name` (20 x 20, values 0..400) through a sequence of shrinks,
|
||||
/// zero extents and growth back, checking every step.
|
||||
fn run(path: &Path, name: &str, with_h5py: bool) {
|
||||
let orig: Vec<f32> = (0..400).map(|i| i as f32).collect();
|
||||
let mut shape = [20u64, 20];
|
||||
let mut data = orig.clone();
|
||||
check(path, name, shape, &data, with_h5py);
|
||||
for next in [
|
||||
[15, 15],
|
||||
[3, 2],
|
||||
[20, 20],
|
||||
[1, 1],
|
||||
[1, 0],
|
||||
[0, 0],
|
||||
[7, 20],
|
||||
[20, 13],
|
||||
[20, 20],
|
||||
] {
|
||||
let mut ed = FileEditor::open(path).unwrap();
|
||||
ed.resize(name, &next).unwrap();
|
||||
drop(ed);
|
||||
data = resized(&data, shape, next);
|
||||
shape = next;
|
||||
check(path, name, shape, &data, with_h5py);
|
||||
}
|
||||
// The maximum is the extent the dataset was created with.
|
||||
let mut ed = FileEditor::open(path).unwrap();
|
||||
assert!(matches!(
|
||||
ed.resize(name, &[21, 20]),
|
||||
Err(Error::InvalidArgument(_))
|
||||
));
|
||||
drop(ed);
|
||||
let f = File::open(path).unwrap();
|
||||
assert_eq!(
|
||||
f.dataset(name).unwrap().max_dimensions().unwrap(),
|
||||
Some(vec![20, 20])
|
||||
);
|
||||
drop(f);
|
||||
// A shrink keeps the values it keeps.
|
||||
let mut ed = FileEditor::open(path).unwrap();
|
||||
let vals: Vec<f32> = orig.iter().map(|v| v + 0.5).collect();
|
||||
ed.write_values(name, &clawhdf5::Selection::All, &vals)
|
||||
.unwrap();
|
||||
ed.resize(name, &[15, 15]).unwrap();
|
||||
drop(ed);
|
||||
check(
|
||||
path,
|
||||
name,
|
||||
[15, 15],
|
||||
&resized(&vals, [20, 20], [15, 15]),
|
||||
with_h5py,
|
||||
);
|
||||
}
|
||||
|
||||
fn fixture(dir: &Path, name: &str) -> PathBuf {
|
||||
let path = dir.join(name);
|
||||
std::fs::copy(
|
||||
Path::new(env!("CARGO_MANIFEST_DIR"))
|
||||
.join("tests/fixtures")
|
||||
.join(name),
|
||||
&path,
|
||||
)
|
||||
.unwrap();
|
||||
path
|
||||
}
|
||||
|
||||
/// Files clawhdf5 2.7.0 wrote: a Fixed Array index (Single Chunk for `s`)
|
||||
/// and a dataspace with no maximum. Shrinking scrambled the values
|
||||
/// (released in 2.7.0's `FileEditor`, PR #18).
|
||||
#[test]
|
||||
fn resize_without_stored_maxshape_keeps_values() {
|
||||
let with_h5py = h5py_ok();
|
||||
for name in ["d", "z"] {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let path = fixture(dir.path(), "chunked_no_maxshape_v2_7_0.h5");
|
||||
run(&path, name, with_h5py);
|
||||
}
|
||||
// A single-chunk dataset and an empty one keep their extents as maxima.
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let path = fixture(dir.path(), "chunked_no_maxshape_v2_7_0.h5");
|
||||
let mut ed = FileEditor::open(&path).unwrap();
|
||||
ed.resize("s", &[2, 4]).unwrap();
|
||||
ed.resize("s", &[3, 4]).unwrap();
|
||||
assert!(matches!(
|
||||
ed.resize("s", &[4, 4]),
|
||||
Err(Error::InvalidArgument(_))
|
||||
));
|
||||
assert!(matches!(
|
||||
ed.resize("e", &[1, 5]),
|
||||
Err(Error::InvalidArgument(_))
|
||||
));
|
||||
ed.resize("e", &[0, 3]).unwrap();
|
||||
drop(ed);
|
||||
let f = File::open(&path).unwrap();
|
||||
let s = f.dataset("s").unwrap();
|
||||
assert_eq!(s.max_dimensions().unwrap(), Some(vec![3, 4]));
|
||||
let mut want: Vec<i32> = (0..12).collect();
|
||||
want[8..].fill(0);
|
||||
assert_eq!(s.read_i32().unwrap(), want);
|
||||
assert_eq!(
|
||||
f.dataset("e").unwrap().max_dimensions().unwrap(),
|
||||
Some(vec![0, 5])
|
||||
);
|
||||
}
|
||||
|
||||
fn written(path: &Path, deflate: bool) {
|
||||
let data: Vec<f32> = (0..400).map(|i| i as f32).collect();
|
||||
let mut b = FileBuilder::new();
|
||||
let d = b
|
||||
.create_dataset("d")
|
||||
.with_f32_data(&data)
|
||||
.with_shape(&[20, 20])
|
||||
.with_chunks(&[6, 6]);
|
||||
if deflate {
|
||||
d.with_deflate(4);
|
||||
}
|
||||
b.write(path).unwrap();
|
||||
}
|
||||
|
||||
/// clawhdf5's writer now records the maximum, as libhdf5 does.
|
||||
#[test]
|
||||
fn resize_file_written_without_maxshape_keeps_values() {
|
||||
let with_h5py = h5py_ok();
|
||||
for deflate in [false, true] {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let path = dir.path().join("cw.h5");
|
||||
written(&path, deflate);
|
||||
let f = File::open(&path).unwrap();
|
||||
assert_eq!(
|
||||
f.dataset("d").unwrap().max_dimensions().unwrap(),
|
||||
Some(vec![20, 20])
|
||||
);
|
||||
drop(f);
|
||||
run(&path, "d", with_h5py);
|
||||
}
|
||||
}
|
||||
|
||||
/// h5py resizing a file clawhdf5 wrote without a maxshape keeps its values
|
||||
/// (it scrambled them while the writer recorded no maximum).
|
||||
#[test]
|
||||
fn h5py_resizes_what_clawhdf5_writes() {
|
||||
if !h5py_ok() {
|
||||
return;
|
||||
}
|
||||
for deflate in [false, true] {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let path = dir.path().join("cw.h5");
|
||||
written(&path, deflate);
|
||||
let out = py(&format!(
|
||||
"import h5py, numpy as np\n\
|
||||
exp = np.arange(400, dtype='f4').reshape(20, 20)\n\
|
||||
with h5py.File({p:?}, 'r+') as f:\n\
|
||||
\x20 f['d'].resize((15, 15))\n\
|
||||
\x20 ok = np.array_equal(f['d'][()], exp[:15, :15])\n\
|
||||
\x20 f['d'].resize((20, 20))\n\
|
||||
\x20 back = f['d'][()]\n\
|
||||
want = np.zeros((20, 20), 'f4'); want[:15, :15] = exp[:15, :15]\n\
|
||||
print(ok and np.array_equal(back, want))",
|
||||
p = path.to_str().unwrap()
|
||||
));
|
||||
assert_eq!(out, "True");
|
||||
let mut want = vec![0f32; 400];
|
||||
for r in 0..15 {
|
||||
for c in 0..15 {
|
||||
want[r * 20 + c] = (r * 20 + c) as f32;
|
||||
}
|
||||
}
|
||||
check(&path, "d", [20, 20], &want, false);
|
||||
}
|
||||
}
|
||||
|
||||
/// h5py's files record the maximum; the same sequence must hold.
|
||||
#[test]
|
||||
fn resize_h5py_file_without_maxshape_keeps_values() {
|
||||
if !h5py_ok() {
|
||||
return;
|
||||
}
|
||||
for libver in ["earliest", "v110", "latest"] {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let path = dir.path().join("hp.h5");
|
||||
py(&format!(
|
||||
"import h5py, numpy as np\n\
|
||||
with h5py.File({p:?}, 'w', libver=({libver:?}, 'latest')) as f:\n\
|
||||
\x20 f.create_dataset('d', data=np.arange(400, dtype='f4').reshape(20, 20), chunks=(6, 6))",
|
||||
p = path.to_str().unwrap()
|
||||
));
|
||||
run(&path, "d", true);
|
||||
}
|
||||
}
|
||||
@@ -162,3 +162,59 @@ fn shrink_then_grow_reads_fill() {
|
||||
raw[..4].fill(0.5);
|
||||
assert_eq!(f.dataset("raw").unwrap().read_f64().unwrap(), raw);
|
||||
}
|
||||
|
||||
/// The editor plans every edit from the file it holds open, never by
|
||||
/// re-opening its path: with the path renamed away and another file put in
|
||||
/// its place, edits go to the held file, planned from its own metadata,
|
||||
/// and the file now at the path is untouched (planning from it and writing
|
||||
/// into the held file corrupted the held one).
|
||||
#[test]
|
||||
fn edits_go_to_the_file_held_not_the_path() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let a = dir.path().join("a.h5");
|
||||
let b = dir.path().join("b.h5");
|
||||
let mut fb = FileBuilder::new();
|
||||
fb.create_dataset("x")
|
||||
.with_i32_data(&[0; 10])
|
||||
.with_shape(&[10]);
|
||||
fb.create_dataset("big")
|
||||
.with_f64_data(&[1.5; 5000])
|
||||
.with_shape(&[5000]);
|
||||
fb.set_attr("title", AttrValue::String("a".into()));
|
||||
fb.write(&a).unwrap();
|
||||
let mut fb = FileBuilder::new();
|
||||
fb.create_dataset("pad")
|
||||
.with_f64_data(&[2.5; 3000])
|
||||
.with_shape(&[3000]);
|
||||
fb.create_dataset("x")
|
||||
.with_i32_data(&[500; 10])
|
||||
.with_shape(&[10]);
|
||||
fb.write(&b).unwrap();
|
||||
|
||||
let mut ed = FileEditor::open(&a).unwrap();
|
||||
assert!(ed.path().is_absolute());
|
||||
let moved = dir.path().join("moved.h5");
|
||||
std::fs::rename(&a, &moved).unwrap();
|
||||
std::fs::rename(&b, &a).unwrap();
|
||||
let other = std::fs::read(&a).unwrap();
|
||||
|
||||
ed.write_values("x", &Selection::All, &[7i32; 10]).unwrap();
|
||||
let vals: Vec<f64> = (0..50).map(f64::from).collect();
|
||||
ed.set_attr("/", "note", &AttrValue::F64Array(vals.clone()))
|
||||
.unwrap();
|
||||
// The editor's own reader sees the held file.
|
||||
let r = ed.reader().unwrap();
|
||||
assert_eq!(r.dataset("x").unwrap().read_i32().unwrap(), [7; 10]);
|
||||
assert!(r.dataset("pad").is_err());
|
||||
drop(r);
|
||||
drop(ed);
|
||||
|
||||
assert!(
|
||||
std::fs::read(&a).unwrap() == other,
|
||||
"the file at the path changed"
|
||||
);
|
||||
let f = File::open(&moved).unwrap();
|
||||
assert_eq!(f.dataset("x").unwrap().read_i32().unwrap(), [7; 10]);
|
||||
assert_eq!(f.dataset("big").unwrap().read_f64().unwrap(), [1.5; 5000]);
|
||||
assert!(matches!(f.root().attr("note").unwrap(), Some(AttrValue::F64Array(v)) if v == vals));
|
||||
}
|
||||
|
||||
Binary file not shown.
Binary file not shown.
@@ -966,7 +966,10 @@ fn skipped_optional_filters_are_masked_as_libhdf5_masks_them() {
|
||||
|
||||
/// Files whose chunks all compress are written exactly as before optional
|
||||
/// filters could be skipped: every mask is 0 and nothing else changed. The
|
||||
/// hashes are of the files the writer produced before that change.
|
||||
/// hashes are of the files the writer produced before that change, except
|
||||
/// that a chunked dataset without a maxshape now records its maximum
|
||||
/// dimensions (8 bytes per dimension; `lzf_fixed`, `lzf_single`, and
|
||||
/// `blosc_fixed`).
|
||||
#[cfg(feature = "lzf")]
|
||||
#[test]
|
||||
fn files_whose_chunks_all_compress_are_unchanged() {
|
||||
@@ -985,7 +988,7 @@ fn files_whose_chunks_all_compress_are_unchanged() {
|
||||
.with_chunks(&[500])
|
||||
.with_lzf();
|
||||
},
|
||||
(3965, 449169442),
|
||||
(3973, 452644487),
|
||||
),
|
||||
(
|
||||
"lzf_ea_noshuffle",
|
||||
@@ -1017,7 +1020,7 @@ fn files_whose_chunks_all_compress_are_unchanged() {
|
||||
.with_chunks(&[3000])
|
||||
.with_lzf();
|
||||
},
|
||||
(546, 690805477),
|
||||
(554, 1394027497),
|
||||
),
|
||||
];
|
||||
#[cfg(feature = "blosc")]
|
||||
@@ -1029,7 +1032,7 @@ fn files_whose_chunks_all_compress_are_unchanged() {
|
||||
.with_chunks(&[1024])
|
||||
.with_blosc(BloscCodec::Lz4, 5, BloscShuffle::Byte);
|
||||
},
|
||||
(2776, 4278611376),
|
||||
(2784, 1180133244),
|
||||
));
|
||||
for (name, build, want) in &cases {
|
||||
let mut fb = clawhdf5::FileBuilder::new();
|
||||
|
||||
Reference in New Issue
Block a user