Files
clawhdf5/crates/clawhdf5-tools/tests/edit_interop.rs
T
osobhandClaude Opus 5.5 dc9cfba6bb edit: reuse space freed earlier in the editing session
A FileEditor now keeps the space its edits free — a filtered chunk that
moved, chunks a shrink removed, B-tree nodes merged away, a heap's
replaced root indirect block or free-space section info, huge objects
replaced — and later edits allocate from it (best fit, lowest address
among equals, zeroed) before growing the file. An edit never reuses what
it frees itself: until it is committed the file still refers to that
space. Reused blocks are written in the commit's first phase with the
space past the old end of file (nothing on disk refers to them yet),
before any existing byte changes, so the crash-safety ordering holds.
Space still free when the editor is dropped is leaked, as libhdf5 leaks
it without a persistent free-space manager (files that have one, or use
paged aggregation, are still refused at open).
FileEditor::reusable_bytes reports what is left to reuse.

Tests: FreeList merging and best fit, the edit-local rule and the commit
split (image unit tests); a shrink followed by regrowth writing the same
data reuses every removed chunk and leaves the file size unchanged,
while one editor per edit grows the file, h5py/h5dump/h5rs check read
both and h5py continues (freed_space_is_reused_within_a_session).

measure_append_waste (edit_interop, ignored), same workload, one editor,
before -> after (bytes; libhdf5 in brackets), on tank 2026-09-26:
gzip chunks 1024, 1000 appends of 100: 307210 -> 306780 (306058);
gzip chunks 4096, 2000 appends of 10: 119684 -> 79829 (50292);
unfiltered unchanged (no space is freed).

Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
2026-09-26 17:04:15 -05:00

1437 lines
54 KiB
Rust

//! `clawhdf5::FileEditor` against libhdf5: files written by h5py (default
//! `libver`, `v114`, and HDF5 2.0's `latest`) and by clawhdf5 are modified
//! in place — values overwritten, datasets grown and appended to many
//! times (crossing Extensible Array super-block / data-block boundaries and
//! version-1 B-tree node splits), attributes added and replaced — and after
//! each round h5py must read exactly the expected values, h5dump must read
//! the file, `h5rs check --data` must find nothing, and our reader must
//! agree. At the end h5py opens the file `r+` and modifies it further.
//!
//! Needs python3 with h5py and numpy (`CLAWHDF5_PYTHON`) and h5dump; skips
//! when they are missing, unless `CLAWHDF5_REQUIRE_INTEROP=1`.
use std::path::Path;
use std::process::{Command, Output};
use clawhdf5::{AttrValue, Error, File, FileBuilder, FileEditor, Selection};
fn python() -> String {
std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string())
}
fn interop_required() -> bool {
std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1")
}
fn available(cmd: &str, args: &[&str]) -> bool {
Command::new(cmd)
.args(args)
.output()
.map(|o| o.status.success())
.unwrap_or(false)
}
/// Whether the tools are here; false (skip) or a panic when interop is
/// required.
fn tools_ok() -> bool {
let ok =
available(&python(), &["-c", "import h5py, numpy"]) && available("h5dump", &["--version"]);
if !ok {
assert!(
!interop_required(),
"CLAWHDF5_REQUIRE_INTEROP=1 but h5py/numpy or h5dump is not available"
);
eprintln!("SKIP: h5py/numpy or h5dump not available");
}
ok
}
fn py(script: &str) -> String {
let o = Command::new(python())
.args(["-c", script])
.output()
.expect("run python");
assert!(
o.status.success(),
"python failed:\n{script}\nSTDOUT: {}\nSTDERR: {}",
String::from_utf8_lossy(&o.stdout),
String::from_utf8_lossy(&o.stderr)
);
String::from_utf8_lossy(&o.stdout).trim().to_string()
}
fn text(o: &Output) -> String {
format!(
"{}{}",
String::from_utf8_lossy(&o.stdout),
String::from_utf8_lossy(&o.stderr)
)
}
/// The file is valid for libhdf5's tools and for `h5rs check --data`.
/// HDF5 2.0 `latest` files are beyond h5dump 1.14.
fn check_tools(path: &Path, h5dump: bool) {
let p = path.to_str().unwrap();
let o = Command::new(env!("CARGO_BIN_EXE_h5rs"))
.args(["check", "--data", "-q", p])
.output()
.unwrap();
assert!(o.status.success(), "h5rs check --data {p}:\n{}", text(&o));
if h5dump {
let o = Command::new("h5dump")
.args(["-o", "/dev/null", p])
.output()
.unwrap();
assert!(
o.status.success() && o.stderr.is_empty(),
"h5dump {p}:\n{}",
text(&o)
);
}
}
/// A dataset's expected contents.
#[derive(Clone, Debug)]
struct Model {
shape: Vec<u64>,
/// Row-major values.
data: Vec<i32>,
}
impl Model {
fn new(shape: &[u64], f: impl Fn(u64) -> i32) -> Self {
let n: u64 = shape.iter().product();
Self {
shape: shape.to_vec(),
data: (0..n).map(f).collect(),
}
}
fn index(&self, c: &[u64]) -> usize {
c.iter()
.zip(&self.shape)
.fold(0u64, |a, (&x, &d)| a * d + x) as usize
}
/// Grow to `shape`, new elements `fill`.
fn resize(&mut self, shape: &[u64], fill: i32) {
let old = self.clone();
*self = Self::new(shape, |_| fill);
let n: u64 = old.shape.iter().product();
for flat in 0..n {
let mut c = vec![0u64; old.shape.len()];
let mut r = flat;
for d in (0..c.len()).rev() {
c[d] = r % old.shape[d];
r /= old.shape[d];
}
let i = self.index(&c);
self.data[i] = old.data[flat as usize];
}
}
/// Apply a hyperslab write of `vals` (row-major over the block).
fn write_block(&mut self, start: &[u64], count: &[u64], vals: &[i32]) {
let n: u64 = count.iter().product();
for flat in 0..n {
let mut c = vec![0u64; count.len()];
let mut r = flat;
for d in (0..c.len()).rev() {
c[d] = start[d] + r % count[d];
r /= count[d];
}
let i = self.index(&c);
self.data[i] = vals[flat as usize];
}
}
}
fn block(start: &[u64], count: &[u64]) -> Selection {
Selection::Hyperslab {
start: start.to_vec(),
stride: vec![1; start.len()],
count: count.to_vec(),
block: vec![1; start.len()],
}
}
/// h5py and our reader both read `m` from dataset `name`.
fn verify(path: &Path, name: &str, m: &Model) {
let f = File::open(path).unwrap();
let ds = f.dataset(name).unwrap();
assert_eq!(ds.shape().unwrap(), m.shape, "our shape of {name}");
assert!(ds.read_i32().unwrap() == m.data, "our values of {name}");
let exp = path.with_extension("expect");
let bytes: Vec<u8> = m.data.iter().flat_map(|v| v.to_le_bytes()).collect();
std::fs::write(&exp, bytes).unwrap();
let shape: Vec<String> = m.shape.iter().map(|d| d.to_string()).collect();
py(&format!(
"import h5py, numpy as np\n\
f = h5py.File({p:?}, 'r')\n\
a = f[{name:?}][()]\n\
e = np.fromfile({e:?}, dtype='<i4').reshape(({s},))\n\
assert a.shape == e.shape, (a.shape, e.shape)\n\
bad = np.argwhere(a != e)\n\
assert len(bad) == 0, ('mismatch at', bad[:5], a[tuple(bad[0])], e[tuple(bad[0])])\n",
p = path.to_str().unwrap(),
e = exp.to_str().unwrap(),
s = shape.join(",")
));
}
/// A deterministic pseudo-random sequence (SplitMix64).
struct Rng(u64);
impl Rng {
fn next(&mut self) -> u64 {
self.0 = self.0.wrapping_add(0x9E37_79B9_7F4A_7C15);
let mut z = self.0;
z = (z ^ (z >> 30)).wrapping_mul(0xBF58_476D_1CE4_E5B9);
z = (z ^ (z >> 27)).wrapping_mul(0x94D0_49BB_1331_11EB);
z ^ (z >> 31)
}
fn below(&mut self, n: u64) -> u64 {
self.next() % n
}
}
/// `h5py.File(..., libver=...)` argument for each flavour tested; the last
/// is HDF5 2.0's own format, which h5dump 1.14 cannot read.
const LIBVERS: &[(&str, bool)] = &[("'earliest'", true), ("'v114'", true), ("'latest'", false)];
fn tmpdir() -> tempfile::TempDir {
let base = std::path::PathBuf::from(env!("CARGO_TARGET_TMPDIR"));
tempfile::TempDir::new_in(base).unwrap()
}
/// Append to a 1-D unlimited dataset hundreds of times, in runs of random
/// length: enough chunks to fill an Extensible Array's index block, direct
/// data blocks and several super blocks (paged data blocks included), and
/// to split version-1 B-tree nodes several levels deep.
fn append_many(libver: &str, h5dump: bool, compression: &str, tag: &str) {
let dir = tmpdir();
let path = dir.path().join(format!("append_{tag}.h5"));
py(&format!(
"import h5py, numpy as np\n\
with h5py.File({p:?}, 'w', libver={libver}) as f:\n\
\x20 f.create_dataset('x', shape=(5,), maxshape=(None,), chunks=(3,), dtype='<i4', \
data=np.arange(5, dtype='<i4'){compression})\n",
p = path.to_str().unwrap()
));
let mut m = Model::new(&[5], |i| i as i32);
verify(&path, "x", &m);
let mut rng = Rng(0x5eed ^ tag.len() as u64);
let mut ed = FileEditor::open(&path).unwrap();
for round in 0..600 {
let n = m.shape[0];
let add = 1 + rng.below(12);
ed.resize("x", &[n + add]).unwrap();
m.resize(&[n + add], 0);
let vals: Vec<i32> = (0..add).map(|k| (n + k) as i32 * 7 - 3).collect();
ed.write_values("x", &block(&[n], &[add]), &vals).unwrap();
m.write_block(&[n], &[add], &vals);
// Occasionally rewrite something earlier (in-place, or a filtered
// chunk that has to move).
if round % 7 == 0 {
let s = rng.below(m.shape[0]);
let c = 1 + rng.below((m.shape[0] - s).min(9));
let vals: Vec<i32> = (0..c).map(|_| rng.next() as i32).collect();
ed.write_values("x", &block(&[s], &[c]), &vals).unwrap();
m.write_block(&[s], &[c], &vals);
}
if round % 150 == 149 {
drop(ed);
verify(&path, "x", &m);
check_tools(&path, h5dump);
ed = FileEditor::open(&path).unwrap();
}
}
drop(ed);
verify(&path, "x", &m);
check_tools(&path, h5dump);
// h5py can go on appending to what we wrote.
py(&format!(
"import h5py, numpy as np\n\
with h5py.File({p:?}, 'r+') as f:\n\
\x20 d = f['x']\n\
\x20 n = d.shape[0]\n\
\x20 d.resize((n + 50,))\n\
\x20 d[n:] = np.arange(50, dtype='<i4') + 1000\n\
\x20 d[0] = -1\n",
p = path.to_str().unwrap()
));
let n = m.shape[0];
m.resize(&[n + 50], 0);
let vals: Vec<i32> = (0..50).map(|k| 1000 + k).collect();
m.write_block(&[n], &[50], &vals);
m.data[0] = -1;
verify(&path, "x", &m);
check_tools(&path, h5dump);
}
#[test]
fn append_many_unfiltered() {
if !tools_ok() {
return;
}
for (i, (lv, dump)) in LIBVERS.iter().enumerate() {
append_many(lv, *dump, "", &format!("plain{i}"));
}
}
#[test]
fn append_many_gzip() {
if !tools_ok() {
return;
}
for (i, (lv, dump)) in LIBVERS.iter().enumerate() {
append_many(
lv,
*dump,
", compression='gzip', shuffle=True",
&format!("gzip{i}"),
);
}
}
/// Random operations — grow, hyperslab writes, point writes, attributes —
/// on a 2-D dataset with one unlimited dimension, checked against a model
/// after every few operations.
fn random_ops(libver: &str, h5dump: bool, extra: &str, tag: &str, seed: u64) {
let dir = tmpdir();
let path = dir.path().join(format!("rand_{tag}.h5"));
py(&format!(
"import h5py, numpy as np\n\
with h5py.File({p:?}, 'w', libver={libver}) as f:\n\
\x20 f.create_dataset('m', shape=(4, 7), maxshape=(None, 7), chunks=(3, 4), \
dtype='<i4', fillvalue=-9{extra})\n\
\x20 f['m'][1:3, 2:6] = 5\n",
p = path.to_str().unwrap()
));
let mut m = Model::new(&[4, 7], |_| -9);
m.write_block(&[1, 2], &[2, 4], &[5; 8]);
let mut attrs: Vec<(String, i64)> = Vec::new();
let mut rng = Rng(seed);
let mut ed = FileEditor::open(&path).unwrap();
for step in 0..120 {
match rng.below(10) {
0..=1 => {
let rows = m.shape[0] + 1 + rng.below(5);
ed.resize("m", &[rows, 7]).unwrap();
m.resize(&[rows, 7], -9);
}
2..=6 => {
let r0 = rng.below(m.shape[0]);
let c0 = rng.below(7);
let cnt = [
1 + rng.below((m.shape[0] - r0).min(6)),
1 + rng.below(7 - c0),
];
let n = cnt[0] * cnt[1];
let vals: Vec<i32> = (0..n).map(|_| (rng.next() % 100_000) as i32).collect();
ed.write_values("m", &block(&[r0, c0], &cnt), &vals)
.unwrap();
m.write_block(&[r0, c0], &cnt, &vals);
}
7 => {
let pts: Vec<Vec<u64>> = (0..1 + rng.below(4))
.map(|_| vec![rng.below(m.shape[0]), rng.below(7)])
.collect();
let vals: Vec<i32> = pts.iter().map(|_| rng.next() as i32).collect();
ed.write_values("m", &Selection::Points(pts.clone()), &vals)
.unwrap();
for (p, v) in pts.iter().zip(&vals) {
let i = m.index(p);
m.data[i] = *v;
}
}
_ => {
let k = rng.below(6);
let name = format!("a{k}");
let v = rng.next() as i64;
match ed.set_attr("m", &name, &AttrValue::I64(v)) {
Ok(()) => {
attrs.retain(|(n, _)| *n != name);
attrs.push((name, v));
}
Err(e) => panic!("set_attr {name}: {e}"),
}
}
}
if step % 30 == 29 {
drop(ed);
verify(&path, "m", &m);
check_tools(&path, h5dump);
check_attrs(&path, "m", &attrs);
ed = FileEditor::open(&path).unwrap();
}
}
drop(ed);
verify(&path, "m", &m);
check_attrs(&path, "m", &attrs);
py(&format!(
"import h5py, numpy as np\n\
with h5py.File({p:?}, 'r+') as f:\n\
\x20 d = f['m']\n\
\x20 n = d.shape[0]\n\
\x20 d.resize((n + 3, 7))\n\
\x20 d[n:, :] = 42\n\
\x20 d.attrs['from_h5py'] = 1.5\n",
p = path.to_str().unwrap()
));
let n = m.shape[0];
m.resize(&[n + 3, 7], -9);
m.write_block(&[n, 0], &[3, 7], &[42; 21]);
verify(&path, "m", &m);
check_tools(&path, h5dump);
}
fn check_attrs(path: &Path, obj: &str, attrs: &[(String, i64)]) {
let f = File::open(path).unwrap();
let got = f.dataset(obj).unwrap().attrs().unwrap();
for (n, v) in attrs {
match got.get(n) {
Some(AttrValue::I64(g)) => assert_eq!(g, v, "attribute {n}"),
other => panic!("attribute {n}: {other:?}"),
}
}
let want: Vec<String> = attrs.iter().map(|(n, v)| format!("{n:?}: {v}")).collect();
py(&format!(
"import h5py\n\
f = h5py.File({p:?}, 'r')\n\
want = {{{w}}}\n\
got = {{k: int(v) for k, v in f[{obj:?}].attrs.items() if k in want}}\n\
assert got == want, (got, want)\n",
p = path.to_str().unwrap(),
w = want.join(", ")
));
}
#[test]
fn random_operations_match_a_model() {
if !tools_ok() {
return;
}
let mut seed = 1;
for (i, (lv, dump)) in LIBVERS.iter().enumerate() {
// h5dump has no LZF decoder (h5py's own filter).
for (j, (extra, lzf)) in [
("", false),
(", compression='gzip', shuffle=True, fletcher32=True", false),
(", compression='lzf'", true),
]
.iter()
.enumerate()
{
seed += 1;
random_ops(lv, *dump && !lzf, extra, &format!("{i}_{j}"), seed);
}
}
}
fn unsupported<T: std::fmt::Debug>(r: Result<T, Error>) {
match r {
Err(Error::Unsupported(_)) => {}
other => panic!("expected Error::Unsupported, got {other:?}"),
}
}
/// The statistics an Extensible Array header keeps (super blocks and their
/// bytes, data blocks and their bytes, one past the highest index set,
/// elements realised), for the file's only Extensible Array.
fn ea_stats(path: &Path) -> [u64; 6] {
let b = std::fs::read(path).unwrap();
let at = b
.windows(4)
.position(|w| w == b"EAHD")
.expect("an EA header");
let mut s = [0u64; 6];
for (k, v) in s.iter_mut().enumerate() {
let o = at + 12 + 8 * k;
*v = u64::from_le_bytes(b[o..o + 8].try_into().unwrap());
}
s
}
/// Version-1 B-tree nodes in the file: (leaves, internal nodes).
fn btree1_nodes(path: &Path) -> (usize, usize) {
let b = std::fs::read(path).unwrap();
let mut out = (0, 0);
for i in 0..b.len().saturating_sub(6) {
if &b[i..i + 4] == b"TREE" && b[i + 4] == 1 {
if b[i + 5] == 0 {
out.0 += 1;
} else {
out.1 += 1;
}
}
}
out
}
/// Growth one chunk at a time and in large steps: the editor's version-1
/// B-tree must split the way libhdf5's does (the same number of leaves and
/// internal nodes for the same insertions).
#[test]
fn btree1_splits_match_libhdf5() {
if !tools_ok() {
return;
}
let dir = tmpdir();
let mut steps: Vec<u64> = (1..=300).collect();
steps.extend([1000, 1001, 5000, 9000]);
let a = dir.path().join("bt_h5py.h5");
let b = dir.path().join("bt_edit.h5");
let create = |p: &Path, write: bool| {
py(&format!(
"import h5py, numpy as np\n\
with h5py.File({p:?}, 'w') as f:\n\
\x20 d = f.create_dataset('x', shape=(0,), maxshape=(None,), chunks=(2,), dtype='<i4')\n\
\x20 if {write}:\n\
\x20 n = 0\n\
\x20 for s in {steps:?}:\n\
\x20 d.resize((s,))\n\
\x20 d[n:s] = np.arange(n, s, dtype='<i4')\n\
\x20 n = s\n",
p = p.to_str().unwrap(),
write = if write { "True" } else { "False" },
))
};
create(&a, true);
create(&b, false);
let mut ed = FileEditor::open(&b).unwrap();
let mut n = 0u64;
for &s in &steps {
ed.resize("x", &[s]).unwrap();
let vals: Vec<i32> = (n..s).map(|v| v as i32).collect();
ed.write_values("x", &block(&[n], &[s - n]), &vals).unwrap();
n = s;
}
drop(ed);
verify(&b, "x", &Model::new(&[n], |i| i as i32));
check_tools(&b, true);
assert_eq!(
btree1_nodes(&b),
btree1_nodes(&a),
"B-tree shape differs from libhdf5's"
);
}
/// Enough chunks that the Extensible Array reaches its paged data blocks
/// (the first one holds element 131060 with h5py's parameters): the same
/// growth done by libhdf5 and by the editor must create the same blocks.
#[test]
fn extensible_array_paged_blocks_match_libhdf5() {
if !tools_ok() {
return;
}
let dir = tmpdir();
let steps = [5u64, 131_000, 131_070, 133_000, 140_000];
let a = dir.path().join("paged_h5py.h5");
let b = dir.path().join("paged_edit.h5");
let create = |p: &Path, write: bool| {
py(&format!(
"import h5py, numpy as np\n\
with h5py.File({p:?}, 'w', libver='v114') as f:\n\
\x20 d = f.create_dataset('x', shape=(0,), maxshape=(None,), chunks=(1,), dtype='<i4')\n\
\x20 if {write}:\n\
\x20 n = 0\n\
\x20 for s in {steps:?}:\n\
\x20 d.resize((s,))\n\
\x20 d[n:s] = np.arange(n, s, dtype='<i4')\n\
\x20 n = s\n",
p = p.to_str().unwrap(),
write = if write { "True" } else { "False" },
))
};
create(&a, true);
create(&b, false);
let mut ed = FileEditor::open(&b).unwrap();
let mut n = 0u64;
for &s in &steps {
ed.resize("x", &[s]).unwrap();
let vals: Vec<i32> = (n..s).map(|v| v as i32).collect();
ed.write_values("x", &block(&[n], &[s - n]), &vals).unwrap();
n = s;
}
drop(ed);
let m = Model::new(&[n], |i| i as i32);
verify(&b, "x", &m);
check_tools(&b, true);
assert_eq!(
ea_stats(&b),
ea_stats(&a),
"EA statistics differ from libhdf5's"
);
}
/// Every element of `sel` over `shape`, in selection order.
/// Every index of the box `ext`, row-major.
fn odometer(ext: &[u64]) -> Vec<Vec<u64>> {
let n: u64 = ext.iter().product();
(0..n)
.map(|flat| {
let mut c = vec![0u64; ext.len()];
let mut r = flat;
for d in (0..c.len()).rev() {
c[d] = r % ext[d];
r /= ext[d];
}
c
})
.collect()
}
fn sel_coords(sel: &Selection, shape: &[u64]) -> Vec<Vec<u64>> {
match sel {
Selection::All => odometer(shape),
Selection::None => Vec::new(),
Selection::Points(p) => p.clone(),
Selection::Hyperslab {
start,
stride,
count,
block,
} => {
let ext: Vec<u64> = count.iter().zip(block).map(|(c, b)| c * b).collect();
odometer(&ext)
.into_iter()
.map(|j| {
(0..j.len())
.map(|d| start[d] + (j[d] / block[d]) * stride[d] + j[d] % block[d])
.collect()
})
.collect()
}
}
}
impl Model {
fn write_sel(&mut self, sel: &Selection, vals: &[i32]) {
let cs = sel_coords(sel, &self.shape);
assert_eq!(cs.len(), vals.len());
for (c, v) in cs.iter().zip(vals) {
let i = self.index(c);
self.data[i] = *v;
}
}
}
/// A random selection of `shape`: a block, a strided hyperslab, points,
/// or everything.
fn random_sel(rng: &mut Rng, shape: &[u64]) -> Selection {
let rank = shape.len();
match rng.below(4) {
0 => Selection::All,
1 => {
let pts = (0..1 + rng.below(5))
.map(|_| shape.iter().map(|&d| rng.below(d)).collect())
.collect();
Selection::Points(pts)
}
_ => {
let mut start = vec![0; rank];
let mut stride = vec![1; rank];
let mut count = vec![1; rank];
let mut block = vec![1; rank];
for d in 0..rank {
start[d] = rng.below(shape[d]);
let room = shape[d] - start[d];
block[d] = 1 + rng.below(room.min(3));
stride[d] = block[d] + rng.below(3);
// count * stride may overshoot as long as the last block fits.
count[d] = 1 + (room - block[d]) / stride[d];
count[d] = 1 + rng.below(count[d]);
}
Selection::Hyperslab {
start,
stride,
count,
block,
}
}
}
}
/// Datasets of every layout and chunk index h5py writes, overwritten under
/// random selections: contiguous (allocated, and never written — late
/// allocation), compact, chunked with a Fixed Array (paged, too), single
/// chunk (unfiltered and filtered), implicit (early allocation), version-1
/// B-tree (every chunked layout of the default libver), and version-2
/// B-tree (existing unfiltered chunks only).
#[test]
fn overwrite_every_layout() {
if !tools_ok() {
return;
}
let names: &[(&str, &[u64])] = &[
("contig", &[10, 4]),
("contig_late", &[6]),
("compact", &[3, 4]),
("fixed", &[10, 7]),
("fixed_gz", &[10, 7]),
("fa_paged", &[3000]),
("single", &[5, 6]),
("single_gz", &[5, 6]),
("implicit", &[8, 8]),
("fa_empty", &[10, 7]),
("fa_gz_empty", &[3000]),
("ea_gz_empty", &[20]),
("bt2", &[6, 6]),
];
for (li, (lv, dump)) in LIBVERS.iter().enumerate() {
let dir = tmpdir();
let path = dir.path().join(format!("overwrite_{li}.h5"));
py(&format!(
"import h5py, numpy as np\n\
from h5py import h5p, h5d, h5s, h5t\n\
def ar(*s): return np.arange(int(np.prod(s)), dtype='<i4').reshape(s)\n\
with h5py.File({p:?}, 'w', libver={lv}) as f:\n\
\x20 f.create_dataset('contig', data=ar(10, 4))\n\
\x20 f.create_dataset('contig_late', shape=(6,), dtype='<i4', fillvalue=7)\n\
\x20 dcpl = h5p.create(h5p.DATASET_CREATE)\n\
\x20 dcpl.set_layout(h5d.COMPACT)\n\
\x20 h5d.create(f.id, b'compact', h5t.STD_I32LE, h5s.create_simple((3, 4)), dcpl=dcpl)\n\
\x20 f['compact'][...] = ar(3, 4)\n\
\x20 d = f.create_dataset('fixed', shape=(10, 7), chunks=(3, 4), dtype='<i4', fillvalue=-1)\n\
\x20 d[0:3, 0:4] = 1\n\
\x20 d = f.create_dataset('fixed_gz', shape=(10, 7), chunks=(3, 4), dtype='<i4', compression='gzip')\n\
\x20 d[4:10, 3:7] = ar(6, 4)\n\
\x20 d = f.create_dataset('fa_paged', shape=(3000,), chunks=(1,), dtype='<i4')\n\
\x20 d[::7] = 3\n\
\x20 f.create_dataset('single', shape=(5, 6), chunks=(5, 6), dtype='<i4')\n\
\x20 f.create_dataset('single_gz', data=ar(5, 6), chunks=(5, 6), compression='gzip')\n\
\x20 dcpl = h5p.create(h5p.DATASET_CREATE)\n\
\x20 dcpl.set_chunk((4, 4))\n\
\x20 dcpl.set_alloc_time(h5d.ALLOC_TIME_EARLY)\n\
\x20 h5d.create(f.id, b'implicit', h5t.STD_I32LE, h5s.create_simple((8, 8)), dcpl=dcpl)\n\
\x20 f.create_dataset('fa_empty', shape=(10, 7), chunks=(3, 4), dtype='<i4', fillvalue=4)\n\
\x20 f.create_dataset('fa_gz_empty', shape=(3000,), chunks=(1,), dtype='<i4', compression='gzip')\n\
\x20 f.create_dataset('ea_gz_empty', shape=(20,), maxshape=(None,), chunks=(2,), dtype='<i4', compression='gzip')\n\
\x20 f.create_dataset('bt2', data=ar(6, 6), maxshape=(None, None), chunks=(2, 2))\n",
p = path.to_str().unwrap()
));
let mut models: Vec<Model> = vec![
Model::new(&[10, 4], |i| i as i32),
Model::new(&[6], |_| 7),
Model::new(&[3, 4], |i| i as i32),
{
let mut m = Model::new(&[10, 7], |_| -1);
m.write_block(&[0, 0], &[3, 4], &[1; 12]);
m
},
{
let mut m = Model::new(&[10, 7], |_| 0);
m.write_block(&[4, 3], &[6, 4], &(0..24).collect::<Vec<_>>());
m
},
Model::new(&[3000], |i| if i % 7 == 0 { 3 } else { 0 }),
Model::new(&[5, 6], |_| 0),
Model::new(&[5, 6], |i| i as i32),
Model::new(&[8, 8], |_| 0),
Model::new(&[10, 7], |_| 4),
Model::new(&[3000], |_| 0),
Model::new(&[20], |_| 0),
Model::new(&[6, 6], |i| i as i32),
];
let mut rng = Rng(77 + li as u64);
for round in 0..3 {
let mut ed = FileEditor::open(&path).unwrap();
for ((name, shape), m) in names.iter().zip(models.iter_mut()) {
for _ in 0..6 {
let sel = random_sel(&mut rng, shape);
let n = sel_coords(&sel, shape).len();
let vals: Vec<i32> = (0..n).map(|_| rng.next() as i32).collect();
ed.write_values(name, &sel, &vals)
.unwrap_or_else(|e| panic!("{name} round {round} {sel:?}: {e}"));
m.write_sel(&sel, &vals);
}
}
drop(ed);
for ((name, _), m) in names.iter().zip(&models) {
verify(&path, name, m);
}
check_tools(&path, *dump);
}
// A version-2 B-tree index takes new chunks too.
if *lv != "'earliest'" {
let mut ed = FileEditor::open(&path).unwrap();
ed.resize("bt2", &[8, 7]).unwrap();
models[12].resize(&[8, 7], 0);
let vals: Vec<i32> = (0..23).collect();
ed.write_values("bt2", &block(&[6, 0], &[2, 7]), &vals[..14])
.unwrap();
models[12].write_block(&[6, 0], &[2, 7], &vals[..14]);
ed.write_values("bt2", &block(&[0, 6], &[6, 1]), &vals[14..20])
.unwrap();
models[12].write_block(&[0, 6], &[6, 1], &vals[14..20]);
drop(ed);
verify(&path, "bt2", &models[12]);
}
// libhdf5 goes on modifying what we wrote.
py(&format!(
"import h5py, numpy as np\n\
with h5py.File({p:?}, 'r+') as f:\n\
\x20 for n in {names:?}:\n\
\x20 d = f[n]\n\
\x20 d[(0,) * d.ndim] = 123\n",
p = path.to_str().unwrap(),
names = names.iter().map(|(n, _)| *n).collect::<Vec<_>>()
));
for ((name, _), m) in names.iter().zip(models.iter_mut()) {
m.data[0] = 123;
verify(&path, name, m);
}
check_tools(&path, *dump);
}
}
/// Attributes added to and replaced on the root group, a group and a
/// dataset, until the headers need continuation chunks; then h5py adds,
/// changes and deletes attributes in the same headers.
#[test]
fn attributes_in_place() {
if !tools_ok() {
return;
}
for (li, (lv, dump)) in LIBVERS.iter().enumerate() {
let dir = tmpdir();
let path = dir.path().join(format!("attrs_{li}.h5"));
let p = path.to_str().unwrap();
py(&format!(
"import h5py, numpy as np\n\
with h5py.File({p:?}, 'w', libver={lv}) as f:\n\
\x20 f.attrs['title'] = 'root'\n\
\x20 g = f.create_group('g')\n\
\x20 g.attrs['n'] = 3\n\
\x20 d = f.create_dataset('d', data=np.arange(4, dtype='<i4'))\n\
\x20 d.attrs['units'] = 'm'\n\
\x20 d.attrs['scale'] = 0.5\n\
\x20 t = f.create_group('tracked', track_order=True)\n\
\x20 t.attrs['a'] = 1\n\
\x20 dense = f.create_group('dense')\n\
\x20 for i in range(12): dense.attrs[f'k{{i}}'] = i\n"
));
let mut want: Vec<(&str, String, AttrValue)> = Vec::new();
let mut ed = FileEditor::open(&path).unwrap();
// v2 headers hold 8 compact attributes by default.
let many = if *lv == "'earliest'" { 20 } else { 6 };
for obj in ["/", "g", "d"] {
for i in 0..many {
let name = format!("x{i}");
let v = AttrValue::I64Array((0..=i as i64).collect());
ed.set_attr(obj, &name, &v).unwrap();
want.push((obj, name, v));
}
// Replace one with something much larger (moves it).
let big = AttrValue::String("z".repeat(700));
ed.set_attr(obj, "x1", &big).unwrap();
want.retain(|(o, n, _)| !(*o == obj && n == "x1"));
want.push((obj, "x1".into(), big));
// And one smaller, in place.
ed.set_attr(obj, "x0", &AttrValue::F64(2.5)).unwrap();
want.retain(|(o, n, _)| !(*o == obj && n == "x0"));
want.push((obj, "x0".into(), AttrValue::F64(2.5)));
}
ed.set_attr("d", "units", &AttrValue::String("km".into()))
.unwrap();
want.push(("d", "units".into(), AttrValue::String("km".into())));
if *lv != "'earliest'" {
// Up to the compact limit (8), then into dense storage; objects
// already in dense storage and ones tracking creation order.
ed.set_attr("g", "eighth", &AttrValue::I64(8)).unwrap();
want.push(("g", "eighth".into(), AttrValue::I64(8)));
ed.set_attr("g", "ninth", &AttrValue::I64(9)).unwrap();
want.push(("g", "ninth".into(), AttrValue::I64(9)));
ed.set_attr("dense", "k0", &AttrValue::I64(1)).unwrap();
want.push(("dense", "k0".into(), AttrValue::I64(1)));
ed.set_attr("tracked", "b", &AttrValue::I64(1)).unwrap();
want.push(("tracked", "b".into(), AttrValue::I64(1)));
}
drop(ed);
check_tools(&path, *dump);
let f = File::open(&path).unwrap();
let expect_py: Vec<String> = want
.iter()
.map(|(o, n, v)| {
let got = if *o == "/" {
f.root().attrs().unwrap()
} else if *o == "d" {
f.dataset(o).unwrap().attrs().unwrap()
} else {
f.group(o).unwrap().attrs().unwrap()
};
let g = got.get(n).unwrap_or_else(|| panic!("{o}/{n} missing"));
assert_eq!(format!("{g:?}"), format!("{v:?}"), "{o}/{n}");
let pv = match v {
AttrValue::I64Array(a) => format!("{a:?}"),
AttrValue::String(s) => format!("{s:?}"),
AttrValue::F64(x) => format!("{x:?}"),
AttrValue::I64(x) => format!("{x}"),
other => panic!("{other:?}"),
};
format!("({o:?}, {n:?}, {pv})")
})
.collect();
py(&format!(
"import h5py, numpy as np\n\
f = h5py.File({p:?}, 'r')\n\
for o, n, v in [{w}]:\n\
\x20 a = f[o].attrs[n]\n\
\x20 a = a.decode() if isinstance(a, bytes) else a\n\
\x20 a = a.tolist() if hasattr(a, 'tolist') else a\n\
\x20 assert a == v, (o, n, a, v)\n\
assert f['d'][()].tolist() == [0, 1, 2, 3]\n\
assert f['g'].attrs['n'] == 3 and f.attrs['title'] == 'root'\n",
w = expect_py.join(", ")
));
// libhdf5 modifies the headers we changed.
py(&format!(
"import h5py\n\
with h5py.File({p:?}, 'r+') as f:\n\
\x20 for o in ['/', 'd']:\n\
\x20 f[o].attrs['from_h5py'] = 'yes'\n\
\x20 f[o].attrs['x0'] = 9\n\
\x20 del f[o].attrs['x1']\n"
));
check_tools(&path, *dump);
let f = File::open(&path).unwrap();
for attrs in [
f.root().attrs().unwrap(),
f.dataset("d").unwrap().attrs().unwrap(),
] {
assert!(matches!(attrs.get("from_h5py"), Some(AttrValue::String(s)) if s == "yes"));
assert!(matches!(attrs.get("x0"), Some(AttrValue::I64(9))));
assert!(!attrs.contains_key("x1"));
assert!(attrs.contains_key("x2"));
}
}
}
/// Files written by clawhdf5's own writer: appended to, overwritten and
/// given attributes, then read by h5py.
#[test]
fn edit_clawhdf5_written_files() {
if !tools_ok() {
return;
}
let dir = tmpdir();
let path = dir.path().join("ours.h5");
let mut b = FileBuilder::new();
b.create_dataset("ext")
.with_i32_data(&(0..10).collect::<Vec<_>>())
.with_shape(&[10])
.with_maxshape(&[u64::MAX])
.with_chunks(&[4])
.with_deflate(4);
b.create_dataset("plain").with_i32_data(&[1, 2, 3, 4, 5, 6]);
b.create_dataset("grid")
.with_i32_data(&(0..24).collect::<Vec<_>>())
.with_shape(&[4, 6])
.with_maxshape(&[u64::MAX, 6])
.with_chunks(&[2, 3]);
b.set_attr("version", AttrValue::I64(1));
b.write(&path).unwrap();
let mut ext = Model::new(&[10], |i| i as i32);
let mut plain = Model::new(&[6], |i| i as i32 + 1);
let mut grid = Model::new(&[4, 6], |i| i as i32);
let mut ed = FileEditor::open(&path).unwrap();
let mut rng = Rng(9);
for _ in 0..200 {
let n = ext.shape[0];
let add = 1 + rng.below(9);
ed.resize("ext", &[n + add]).unwrap();
ext.resize(&[n + add], 0);
let vals: Vec<i32> = (0..add).map(|_| rng.next() as i32).collect();
ed.write_values("ext", &block(&[n], &[add]), &vals).unwrap();
ext.write_block(&[n], &[add], &vals);
}
for _ in 0..30 {
let rows = grid.shape[0] + rng.below(3);
ed.resize("grid", &[rows, 6]).unwrap();
grid.resize(&[rows, 6], 0);
let sel = random_sel(&mut rng, &grid.shape.clone());
let vals: Vec<i32> = (0..sel_coords(&sel, &grid.shape).len())
.map(|_| rng.next() as i32)
.collect();
ed.write_values("grid", &sel, &vals).unwrap();
grid.write_sel(&sel, &vals);
}
ed.write_values(
"plain",
&Selection::Points(vec![vec![5], vec![0]]),
&[60, 10],
)
.unwrap();
plain.write_sel(&Selection::Points(vec![vec![5], vec![0]]), &[60, 10]);
ed.set_attr("/", "version", &AttrValue::I64(2)).unwrap();
ed.set_attr("ext", "note", &AttrValue::String("appended".into()))
.unwrap();
drop(ed);
verify(&path, "ext", &ext);
verify(&path, "plain", &plain);
verify(&path, "grid", &grid);
check_tools(&path, true);
py(&format!(
"import h5py\n\
f = h5py.File({p:?}, 'r')\n\
assert f.attrs['version'] == 2\n\
assert f['ext'].attrs['note'] in ('appended', b'appended')\n",
p = path.to_str().unwrap()
));
}
/// One writer at a time: a second editor (or libhdf5 with file locking)
/// is refused until the first is dropped.
#[test]
fn editor_locks_the_file() {
let dir = tmpdir();
let path = dir.path().join("lock.h5");
let mut b = FileBuilder::new();
b.create_dataset("x").with_i32_data(&[1, 2, 3]);
b.write(&path).unwrap();
let ed = FileEditor::open(&path).unwrap();
assert!(matches!(FileEditor::open(&path), Err(Error::Locked(_))));
if available(&python(), &["-c", "import h5py"]) {
let o = Command::new(python())
.args([
"-c",
&format!("import h5py; h5py.File({:?}, 'r+')", path.to_str().unwrap()),
])
.env_remove("HDF5_USE_FILE_LOCKING")
.output()
.unwrap();
assert!(!o.status.success(), "h5py opened a locked file r+");
}
drop(ed);
FileEditor::open(&path).unwrap();
}
/// Refused edits leave the file byte for byte as it was.
#[test]
fn refused_edits_change_nothing() {
if !tools_ok() {
return;
}
let dir = tmpdir();
let path = dir.path().join("refuse.h5");
py(&format!(
"import h5py, numpy as np\n\
with h5py.File({p:?}, 'w') as f:\n\
\x20 f.create_dataset('s', data=['a', 'bb'], dtype=h5py.string_dtype())\n\
\x20 f.create_dataset('x', data=np.arange(6, dtype='<i4'), maxshape=(8,), chunks=(2,))\n\
\x20 f.create_dataset('c', data=np.arange(6, dtype='<i4'))\n",
p = path.to_str().unwrap()
));
let before = std::fs::read(&path).unwrap();
let mut ed = FileEditor::open(&path).unwrap();
unsupported(ed.write_all("s", &[0u8; 32]));
unsupported(ed.resize("c", &[4]));
unsupported(
ed.resize("c", &[6, 1])
.map_err(|_| Error::Unsupported(String::new())),
);
assert!(matches!(
ed.resize("x", &[9]),
Err(Error::InvalidArgument(_))
));
assert!(matches!(
ed.write_all("x", &[0u8; 5]),
Err(Error::InvalidArgument(_))
));
assert!(matches!(
ed.write_values("x", &block(&[5], &[2]), &[1i32, 2]),
Err(Error::InvalidArgument(_))
));
assert!(matches!(
ed.write_values("x", &Selection::All, &[1.0f64; 6]),
Err(Error::InvalidArgument(_))
));
drop(ed);
assert!(std::fs::read(&path).unwrap() == before);
}
/// Chunks created out of order, one edit each — descending (every insertion
/// below the tree's first key) and then shuffled — by libhdf5 and by the
/// editor: the same values, and for the version-1 B-tree (default libver)
/// the same node counts, for the Extensible Array (`v114`) the same
/// statistics.
#[test]
fn out_of_order_chunk_creation_matches_libhdf5() {
if !tools_ok() {
return;
}
let len = 1200u64;
let mut order: Vec<u64> = (900..len).rev().collect();
let mut rest: Vec<u64> = (0..900).collect();
let mut rng = Rng(3);
for i in (1..rest.len()).rev() {
rest.swap(i, rng.below(i as u64 + 1) as usize);
}
order.extend(rest.iter().take(500));
for (lv, tag) in [("'earliest'", "bt"), ("'v114'", "ea")] {
let dir = tmpdir();
let a = dir.path().join(format!("ooo_{tag}_h5py.h5"));
let b = dir.path().join(format!("ooo_{tag}_edit.h5"));
let create = |p: &Path, write: bool| {
py(&format!(
"import h5py, numpy as np\n\
with h5py.File({p:?}, 'w', libver={lv}) as f:\n\
\x20 d = f.create_dataset('x', shape=({len},), maxshape=(None,), chunks=(1,), dtype='<i4', fillvalue=-5)\n\
\x20 if {write}:\n\
\x20 for i in {order:?}:\n\
\x20 d[i] = i * 3\n",
p = p.to_str().unwrap(),
write = if write { "True" } else { "False" },
))
};
create(&a, true);
create(&b, false);
let mut ed = FileEditor::open(&b).unwrap();
for &i in &order {
ed.write_values("x", &block(&[i], &[1]), &[i as i32 * 3])
.unwrap();
}
drop(ed);
let mut m = Model::new(&[len], |_| -5);
for &i in &order {
m.data[i as usize] = i as i32 * 3;
}
verify(&b, "x", &m);
check_tools(&b, true);
if tag == "bt" {
assert_eq!(
btree1_nodes(&b),
btree1_nodes(&a),
"B-tree shape differs from libhdf5's"
);
} else {
assert_eq!(
ea_stats(&b),
ea_stats(&a),
"EA statistics differ from libhdf5's"
);
}
py(&format!(
"import h5py, numpy as np\n\
with h5py.File({p:?}, 'r+') as f:\n\
\x20 f['x'][:] = np.arange({len}, dtype='<i4')\n",
p = b.to_str().unwrap()
));
verify(&b, "x", &Model::new(&[len], |i| i as i32));
check_tools(&b, true);
}
}
/// Not a check: prints how much space an append workload leaks (one
/// editor for the whole workload, which reuses the space it frees but not
/// space it cannot fit a grown chunk into), against libhdf5 doing the same
/// appends and against `h5repack` of each. Run with `--ignored --nocapture`.
#[test]
#[ignore]
fn measure_append_waste() {
if !tools_ok() {
return;
}
let dir = tmpdir();
for (comp, chunk, add, rounds) in [
("", 1024u64, 100u64, 1000u64),
(", compression='gzip'", 1024, 100, 1000),
(", compression='gzip'", 4096, 10, 2000),
] {
let a = dir.path().join("waste_h5py.h5");
let b = dir.path().join("waste_edit.h5");
for (p, write) in [(&a, true), (&b, false)] {
py(&format!(
"import h5py, numpy as np\n\
rng = np.random.default_rng(1)\n\
with h5py.File({p:?}, 'w') as f:\n\
\x20 d = f.create_dataset('x', shape=(0,), maxshape=(None,), chunks=({chunk},), dtype='<f8'{comp})\n\
\x20 if {write}:\n\
\x20 for r in range({rounds}):\n\
\x20 n = d.shape[0]\n\
\x20 d.resize((n + {add},))\n\
\x20 d[n:] = np.round(np.sin(np.arange(n, n + {add}) / 50.0), 3)\n",
p = p.to_str().unwrap(),
write = if write { "True" } else { "False" },
));
}
let mut ed = FileEditor::open(&b).unwrap();
for r in 0..rounds {
let n = r * add;
ed.resize("x", &[n + add]).unwrap();
let vals: Vec<f64> = (n..n + add)
.map(|i| ((i as f64 / 50.0).sin() * 1000.0).round() / 1000.0)
.collect();
ed.write_values("x", &block(&[n], &[add]), &vals).unwrap();
}
drop(ed);
let size = |p: &Path| std::fs::metadata(p).unwrap().len();
let repacked = |p: &Path| {
let out = p.with_extension("repacked.h5");
let _ = std::fs::remove_file(&out);
let o = Command::new("h5repack")
.args([p.to_str().unwrap(), out.to_str().unwrap()])
.output()
.unwrap();
assert!(o.status.success(), "{}", text(&o));
size(&out)
};
println!(
"chunks {chunk}{comp}, {rounds} appends of {add}: editor {} bytes \
(repacked {}), libhdf5 {} bytes (repacked {})",
size(&b),
repacked(&b),
size(&a),
repacked(&a)
);
}
}
/// Optional filters that fail are skipped as libhdf5 skips them: an LZF
/// that does not shrink a chunk leaves it stored unfiltered with its mask
/// bit set, exactly as h5py stores it. Storing the LZF stream with a clear
/// mask at the raw chunk's size once let a later libhdf5 rewrite of the
/// chunk (raw, same size) keep the stale mask, and h5py then failed to read
/// the dataset. After the editor, h5py r+ rewrites and extends the datasets
/// and h5py, h5dump (on the chunks it can decode: LZF is not in h5dump) and
/// our reader read every value.
#[test]
fn optional_filters_that_fail_are_skipped() {
if !tools_ok() {
return;
}
for (li, (lv, _)) in LIBVERS.iter().enumerate() {
let dir = tmpdir();
let path = dir.path().join(format!("optional_{li}.h5"));
let p = path.to_str().unwrap();
py(&format!(
"import h5py, numpy as np\n\
with h5py.File({p:?}, 'w', libver={lv}) as f:\n\
\x20 f.create_dataset('u8', shape=(5,), dtype='u1', chunks=(5,), maxshape=(None,), compression='lzf')\n\
\x20 f.create_dataset('mix', shape=(32,), dtype='<i4', chunks=(8,), maxshape=(None,), compression='lzf', shuffle=True, fletcher32=True)\n\
\x20 f.create_dataset('ref', shape=(5,), dtype='u1', chunks=(5,), maxshape=(None,), compression='lzf')\n\
\x20 f['ref'][...] = [182, 0, 0, 0, 0]\n"
));
let mut rng = Rng(li as u64 + 17);
// Chunks 0 and 2 incompressible, 1 and 3 compressible.
let mix: Vec<i32> = (0..32)
.map(|i| {
if (i / 8) % 2 == 0 {
rng.next() as i32
} else {
7
}
})
.collect();
let mut ed = FileEditor::open(&path).unwrap();
ed.write_all("u8", &[182, 0, 0, 0, 0]).unwrap();
ed.write_values("mix", &Selection::All, &mix).unwrap();
drop(ed);
// Stored as h5py stores it: raw, LZF's bit (0; 1 behind shuffle) set.
let masks = py(&format!(
"import h5py\n\
f = h5py.File({p:?}, 'r')\n\
i = lambda d, k: f[d].id.get_chunk_info(k)\n\
print(i('u8', 0).filter_mask, i('u8', 0).size, i('ref', 0).filter_mask, i('ref', 0).size,\n\
\x20 *[(i('mix', k).filter_mask, i('mix', k).size < 36) for k in range(4)])\n"
));
assert_eq!(
masks, "1 5 1 5 (2, False) (0, True) (2, False) (0, True)",
"{lv}"
);
let dump = |ds: &str| {
let o = Command::new("h5dump")
.args(["-d", ds, "-y", "-w", "0", p])
.output()
.unwrap();
assert!(o.status.success(), "h5dump -d {ds} {p}:\n{}", text(&o));
let s = String::from_utf8_lossy(&o.stdout).into_owned();
s.split_once("DATA {")
.and_then(|(_, r)| r.split_once('}'))
.map(|(d, _)| {
d.split(|c: char| c == ',' || c.is_whitespace())
.filter(|t| !t.is_empty())
.map(|t| t.parse::<i64>().unwrap())
.collect::<Vec<_>>()
})
.unwrap_or_default()
};
if *lv != "'latest'" {
assert_eq!(dump("/u8"), [182, 0, 0, 0, 0]);
}
// libhdf5 rewrites the chunk raw at the same size, then extends the
// datasets with more incompressible data.
py(&format!(
"import h5py, numpy as np\n\
with h5py.File({p:?}, 'r+') as f:\n\
\x20 f['u8'][...] = [182, 0, 0, 0, 1]\n\
\x20 f['u8'].resize((12,))\n\
\x20 f['u8'][5:] = [9, 200, 3, 77, 1, 250, 42]\n\
\x20 m = f['mix']\n\
\x20 m[8:16] = np.arange(8, dtype='<i4') * 104729 + 12345\n\
\x20 m[0:8] = 5\n\
\x20 m.resize((40,))\n\
\x20 m[32:] = 11\n"
));
let u8_want: Vec<u8> = vec![182, 0, 0, 0, 1, 9, 200, 3, 77, 1, 250, 42];
let mut mix_want = mix.clone();
mix_want[0..8].fill(5);
for (k, v) in mix_want[8..16].iter_mut().enumerate() {
*v = k as i32 * 104_729 + 12_345;
}
mix_want.extend([11; 8]);
let f = File::open(&path).unwrap();
assert_eq!(
f.dataset("u8")
.unwrap()
.read_selection(&Selection::All)
.unwrap(),
u8_want
);
assert_eq!(f.dataset("mix").unwrap().read_i32().unwrap(), mix_want);
drop(f);
verify(
&path,
"mix",
&Model {
shape: vec![40],
data: mix_want,
},
);
py(&format!(
"import h5py\n\
f = h5py.File({p:?}, 'r')\n\
assert f['u8'][()].tolist() == {u8_want:?}, f['u8'][()]\n"
));
if *lv != "'latest'" {
let got: Vec<u8> = dump("/u8").into_iter().map(|v| v as u8).collect();
assert_eq!(got, u8_want);
}
let o = Command::new(env!("CARGO_BIN_EXE_h5rs"))
.args(["check", "--data", "-q", p])
.output()
.unwrap();
assert!(o.status.success(), "h5rs check --data {p}:\n{}", text(&o));
}
}
/// libhdf5 counts a version-2 object header's attributes through its
/// Attribute Info message and reports none without one. The editor adds
/// that message, as `H5O__attr_create` does, when it gives a version-2
/// header (h5py `libver='v110'`/`'latest'`) its first attribute: afterwards
/// h5py lists, counts and reads every attribute, new and existing, h5dump
/// agrees, and h5py `r+` can add more (past the compact limit, into dense
/// storage) with the count still right.
#[test]
fn attribute_count_in_version_2_headers() {
if !tools_ok() {
return;
}
for (lv, dump) in [("'v110'", true), ("'latest'", false)] {
let dir = tmpdir();
let path = dir.path().join("acount.h5");
let p = path.to_str().unwrap();
py(&format!(
"import h5py, numpy as np\n\
with h5py.File({p:?}, 'w', libver={lv}) as f:\n\
\x20 f.create_group('g')\n\
\x20 f.create_group('has').attrs['old'] = 7\n\
\x20 f.create_dataset('d', data=np.arange(3, dtype='<i4'))\n\
\x20 f.create_dataset('e', data=np.arange(3, dtype='<i4')).attrs['old'] = 7\n"
));
let mut ed = FileEditor::open(&path).unwrap();
for obj in ["/", "g", "has", "d", "e"] {
for i in 0..3 {
ed.set_attr(obj, &format!("a{i}"), &AttrValue::I64(i))
.unwrap();
}
// Replacing one keeps the count.
ed.set_attr(obj, "a1", &AttrValue::F64(1.5)).unwrap();
}
drop(ed);
check_tools(&path, dump);
let check = |extra: usize| {
py(&format!(
"import h5py\n\
f = h5py.File({p:?}, 'r')\n\
for o in ['/', 'g', 'has', 'd', 'e']:\n\
\x20 a = f[o].attrs\n\
\x20 want = {{'a0': 0, 'a1': 1.5, 'a2': 2}}\n\
\x20 if o in ('has', 'e'): want['old'] = 7\n\
\x20 for i in range({extra}): want[f'h{{i}}'] = i\n\
\x20 assert len(a) == len(want), (o, len(a), list(a))\n\
\x20 assert h5py.h5o.get_info(f[o].id).num_attrs == len(want), o\n\
\x20 assert sorted(a) == sorted(want), (o, list(a))\n\
\x20 assert {{k: a[k] for k in a}} == want, (o, dict(a))\n"
));
let f = File::open(&path).unwrap();
for (o, n) in [("/", 3), ("g", 3), ("has", 4), ("d", 3), ("e", 4)] {
let attrs = match o {
"/" => f.root().attrs().unwrap(),
"d" | "e" => f.dataset(o).unwrap().attrs().unwrap(),
_ => f.group(o).unwrap().attrs().unwrap(),
};
assert_eq!(attrs.len(), n + extra, "{o}");
assert!(matches!(attrs.get("a1"), Some(AttrValue::F64(x)) if *x == 1.5));
}
if dump {
// Every object's attributes: 3 set by the editor on each of
// the five, 2 existing, `extra` from h5py on each.
let out = Command::new("h5dump").args(["-A", p]).output().unwrap();
assert!(out.status.success(), "h5dump -A:\n{}", text(&out));
let s = String::from_utf8_lossy(&out.stdout);
assert_eq!(
s.matches("ATTRIBUTE \"").count(),
5 * (3 + extra) + 2,
"h5dump -A:\n{s}"
);
}
};
check(0);
// libhdf5 adds more: to 7 or 8 (compact), then past its limit.
for extra in [4usize, 10] {
py(&format!(
"import h5py\n\
with h5py.File({p:?}, 'r+') as f:\n\
\x20 for o in ['/', 'g', 'has', 'd', 'e']:\n\
\x20 for i in range({extra}): f[o].attrs[f'h{{i}}'] = i\n"
));
check(extra);
check_tools(&path, dump);
}
}
}
/// A dataset whose filters this build cannot encode (scale-offset)
/// is `Error::Unsupported`, as the editor documents, not a format error,
/// and the file is left byte for byte as it was.
#[test]
fn unencodable_filters_are_unsupported() {
if !tools_ok() {
return;
}
let dir = tmpdir();
let path = dir.path().join("unencodable.h5");
py(&format!(
"import h5py, numpy as np\n\
with h5py.File({p:?}, 'w') as f:\n\
\x20 f.create_dataset('so', data=np.arange(16, dtype='<i4'), chunks=(8,), maxshape=(None,), scaleoffset=0)\n\
\x20 f.create_dataset('so_new', shape=(16,), dtype='<i4', chunks=(8,), maxshape=(None,), scaleoffset=0)\n\
\x20 f.create_dataset('sof', data=np.arange(16, dtype='<f8'), chunks=(8,), scaleoffset=2)\n",
p = path.to_str().unwrap()
));
let before = std::fs::read(&path).unwrap();
let mut ed = FileEditor::open(&path).unwrap();
for (ds, es) in [("so", 4), ("so_new", 4), ("sof", 8)] {
match ed.write_all(ds, &vec![0u8; 16 * es]) {
Err(Error::Unsupported(msg)) => assert!(msg.contains("filter"), "{ds}: {msg}"),
other => panic!("{ds}: {other:?}"),
}
}
drop(ed);
assert!(
std::fs::read(&path).unwrap() == before,
"a refused edit changed the file"
);
}