//! `clawhdf5::FileEditor` against libhdf5: files written by h5py (default //! `libver`, `v114`, and HDF5 2.0's `latest`) and by clawhdf5 are modified //! in place — values overwritten, datasets grown and appended to many //! times (crossing Extensible Array super-block / data-block boundaries and //! version-1 B-tree node splits), attributes added and replaced — and after //! each round h5py must read exactly the expected values, h5dump must read //! the file, `h5rs check --data` must find nothing, and our reader must //! agree. At the end h5py opens the file `r+` and modifies it further. //! //! Needs python3 with h5py and numpy (`CLAWHDF5_PYTHON`) and h5dump; skips //! when they are missing, unless `CLAWHDF5_REQUIRE_INTEROP=1`. use std::path::Path; use std::process::{Command, Output}; use clawhdf5::{AttrValue, Error, File, FileBuilder, FileEditor, Selection}; fn python() -> String { std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string()) } fn interop_required() -> bool { std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1") } fn available(cmd: &str, args: &[&str]) -> bool { Command::new(cmd) .args(args) .output() .map(|o| o.status.success()) .unwrap_or(false) } /// Whether the tools are here; false (skip) or a panic when interop is /// required. fn tools_ok() -> bool { let ok = available(&python(), &["-c", "import h5py, numpy"]) && available("h5dump", &["--version"]); if !ok { assert!( !interop_required(), "CLAWHDF5_REQUIRE_INTEROP=1 but h5py/numpy or h5dump is not available" ); eprintln!("SKIP: h5py/numpy or h5dump not available"); } ok } fn py(script: &str) -> String { let o = Command::new(python()) .args(["-c", script]) .output() .expect("run python"); assert!( o.status.success(), "python failed:\n{script}\nSTDOUT: {}\nSTDERR: {}", String::from_utf8_lossy(&o.stdout), String::from_utf8_lossy(&o.stderr) ); String::from_utf8_lossy(&o.stdout).trim().to_string() } fn text(o: &Output) -> String { format!( "{}{}", String::from_utf8_lossy(&o.stdout), String::from_utf8_lossy(&o.stderr) ) } /// The file is valid for libhdf5's tools and for `h5rs check --data`. /// HDF5 2.0 `latest` files are beyond h5dump 1.14. fn check_tools(path: &Path, h5dump: bool) { let p = path.to_str().unwrap(); let o = Command::new(env!("CARGO_BIN_EXE_h5rs")) .args(["check", "--data", "-q", p]) .output() .unwrap(); assert!(o.status.success(), "h5rs check --data {p}:\n{}", text(&o)); if h5dump { let o = Command::new("h5dump") .args(["-o", "/dev/null", p]) .output() .unwrap(); assert!( o.status.success() && o.stderr.is_empty(), "h5dump {p}:\n{}", text(&o) ); } } /// A dataset's expected contents. #[derive(Clone, Debug)] struct Model { shape: Vec, /// Row-major values. data: Vec, } impl Model { fn new(shape: &[u64], f: impl Fn(u64) -> i32) -> Self { let n: u64 = shape.iter().product(); Self { shape: shape.to_vec(), data: (0..n).map(f).collect(), } } fn index(&self, c: &[u64]) -> usize { c.iter() .zip(&self.shape) .fold(0u64, |a, (&x, &d)| a * d + x) as usize } /// Grow to `shape`, new elements `fill`. fn resize(&mut self, shape: &[u64], fill: i32) { let old = self.clone(); *self = Self::new(shape, |_| fill); let n: u64 = old.shape.iter().product(); for flat in 0..n { let mut c = vec![0u64; old.shape.len()]; let mut r = flat; for d in (0..c.len()).rev() { c[d] = r % old.shape[d]; r /= old.shape[d]; } let i = self.index(&c); self.data[i] = old.data[flat as usize]; } } /// Apply a hyperslab write of `vals` (row-major over the block). fn write_block(&mut self, start: &[u64], count: &[u64], vals: &[i32]) { let n: u64 = count.iter().product(); for flat in 0..n { let mut c = vec![0u64; count.len()]; let mut r = flat; for d in (0..c.len()).rev() { c[d] = start[d] + r % count[d]; r /= count[d]; } let i = self.index(&c); self.data[i] = vals[flat as usize]; } } } fn block(start: &[u64], count: &[u64]) -> Selection { Selection::Hyperslab { start: start.to_vec(), stride: vec![1; start.len()], count: count.to_vec(), block: vec![1; start.len()], } } /// h5py and our reader both read `m` from dataset `name`. fn verify(path: &Path, name: &str, m: &Model) { let f = File::open(path).unwrap(); let ds = f.dataset(name).unwrap(); assert_eq!(ds.shape().unwrap(), m.shape, "our shape of {name}"); assert!(ds.read_i32().unwrap() == m.data, "our values of {name}"); let exp = path.with_extension("expect"); let bytes: Vec = m.data.iter().flat_map(|v| v.to_le_bytes()).collect(); std::fs::write(&exp, bytes).unwrap(); let shape: Vec = m.shape.iter().map(|d| d.to_string()).collect(); py(&format!( "import h5py, numpy as np\n\ f = h5py.File({p:?}, 'r')\n\ a = f[{name:?}][()]\n\ e = np.fromfile({e:?}, dtype=' u64 { self.0 = self.0.wrapping_add(0x9E37_79B9_7F4A_7C15); let mut z = self.0; z = (z ^ (z >> 30)).wrapping_mul(0xBF58_476D_1CE4_E5B9); z = (z ^ (z >> 27)).wrapping_mul(0x94D0_49BB_1331_11EB); z ^ (z >> 31) } fn below(&mut self, n: u64) -> u64 { self.next() % n } } /// `h5py.File(..., libver=...)` argument for each flavour tested; the last /// is HDF5 2.0's own format, which h5dump 1.14 cannot read. const LIBVERS: &[(&str, bool)] = &[("'earliest'", true), ("'v114'", true), ("'latest'", false)]; fn tmpdir() -> tempfile::TempDir { let base = std::path::PathBuf::from(env!("CARGO_TARGET_TMPDIR")); tempfile::TempDir::new_in(base).unwrap() } /// Append to a 1-D unlimited dataset hundreds of times, in runs of random /// length: enough chunks to fill an Extensible Array's index block, direct /// data blocks and several super blocks (paged data blocks included), and /// to split version-1 B-tree nodes several levels deep. fn append_many(libver: &str, h5dump: bool, compression: &str, tag: &str) { let dir = tmpdir(); let path = dir.path().join(format!("append_{tag}.h5")); py(&format!( "import h5py, numpy as np\n\ with h5py.File({p:?}, 'w', libver={libver}) as f:\n\ \x20 f.create_dataset('x', shape=(5,), maxshape=(None,), chunks=(3,), dtype=' = (0..add).map(|k| (n + k) as i32 * 7 - 3).collect(); ed.write_values("x", &block(&[n], &[add]), &vals).unwrap(); m.write_block(&[n], &[add], &vals); // Occasionally rewrite something earlier (in-place, or a filtered // chunk that has to move). if round % 7 == 0 { let s = rng.below(m.shape[0]); let c = 1 + rng.below((m.shape[0] - s).min(9)); let vals: Vec = (0..c).map(|_| rng.next() as i32).collect(); ed.write_values("x", &block(&[s], &[c]), &vals).unwrap(); m.write_block(&[s], &[c], &vals); } if round % 150 == 149 { drop(ed); verify(&path, "x", &m); check_tools(&path, h5dump); ed = FileEditor::open(&path).unwrap(); } } drop(ed); verify(&path, "x", &m); check_tools(&path, h5dump); // h5py can go on appending to what we wrote. py(&format!( "import h5py, numpy as np\n\ with h5py.File({p:?}, 'r+') as f:\n\ \x20 d = f['x']\n\ \x20 n = d.shape[0]\n\ \x20 d.resize((n + 50,))\n\ \x20 d[n:] = np.arange(50, dtype=' = (0..50).map(|k| 1000 + k).collect(); m.write_block(&[n], &[50], &vals); m.data[0] = -1; verify(&path, "x", &m); check_tools(&path, h5dump); } #[test] fn append_many_unfiltered() { if !tools_ok() { return; } for (i, (lv, dump)) in LIBVERS.iter().enumerate() { append_many(lv, *dump, "", &format!("plain{i}")); } } #[test] fn append_many_gzip() { if !tools_ok() { return; } for (i, (lv, dump)) in LIBVERS.iter().enumerate() { append_many( lv, *dump, ", compression='gzip', shuffle=True", &format!("gzip{i}"), ); } } /// Random operations — grow, hyperslab writes, point writes, attributes — /// on a 2-D dataset with one unlimited dimension, checked against a model /// after every few operations. fn random_ops(libver: &str, h5dump: bool, extra: &str, tag: &str, seed: u64) { let dir = tmpdir(); let path = dir.path().join(format!("rand_{tag}.h5")); py(&format!( "import h5py, numpy as np\n\ with h5py.File({p:?}, 'w', libver={libver}) as f:\n\ \x20 f.create_dataset('m', shape=(4, 7), maxshape=(None, 7), chunks=(3, 4), \ dtype=' = Vec::new(); let mut rng = Rng(seed); let mut ed = FileEditor::open(&path).unwrap(); for step in 0..120 { match rng.below(10) { 0..=1 => { let rows = m.shape[0] + 1 + rng.below(5); ed.resize("m", &[rows, 7]).unwrap(); m.resize(&[rows, 7], -9); } 2..=6 => { let r0 = rng.below(m.shape[0]); let c0 = rng.below(7); let cnt = [ 1 + rng.below((m.shape[0] - r0).min(6)), 1 + rng.below(7 - c0), ]; let n = cnt[0] * cnt[1]; let vals: Vec = (0..n).map(|_| (rng.next() % 100_000) as i32).collect(); ed.write_values("m", &block(&[r0, c0], &cnt), &vals) .unwrap(); m.write_block(&[r0, c0], &cnt, &vals); } 7 => { let pts: Vec> = (0..1 + rng.below(4)) .map(|_| vec![rng.below(m.shape[0]), rng.below(7)]) .collect(); let vals: Vec = pts.iter().map(|_| rng.next() as i32).collect(); ed.write_values("m", &Selection::Points(pts.clone()), &vals) .unwrap(); for (p, v) in pts.iter().zip(&vals) { let i = m.index(p); m.data[i] = *v; } } _ => { let k = rng.below(6); let name = format!("a{k}"); let v = rng.next() as i64; match ed.set_attr("m", &name, &AttrValue::I64(v)) { Ok(()) => { attrs.retain(|(n, _)| *n != name); attrs.push((name, v)); } Err(e) => panic!("set_attr {name}: {e}"), } } } if step % 30 == 29 { drop(ed); verify(&path, "m", &m); check_tools(&path, h5dump); check_attrs(&path, "m", &attrs); ed = FileEditor::open(&path).unwrap(); } } drop(ed); verify(&path, "m", &m); check_attrs(&path, "m", &attrs); py(&format!( "import h5py, numpy as np\n\ with h5py.File({p:?}, 'r+') as f:\n\ \x20 d = f['m']\n\ \x20 n = d.shape[0]\n\ \x20 d.resize((n + 3, 7))\n\ \x20 d[n:, :] = 42\n\ \x20 d.attrs['from_h5py'] = 1.5\n", p = path.to_str().unwrap() )); let n = m.shape[0]; m.resize(&[n + 3, 7], -9); m.write_block(&[n, 0], &[3, 7], &[42; 21]); verify(&path, "m", &m); check_tools(&path, h5dump); } fn check_attrs(path: &Path, obj: &str, attrs: &[(String, i64)]) { let f = File::open(path).unwrap(); let got = f.dataset(obj).unwrap().attrs().unwrap(); for (n, v) in attrs { match got.get(n) { Some(AttrValue::I64(g)) => assert_eq!(g, v, "attribute {n}"), other => panic!("attribute {n}: {other:?}"), } } let want: Vec = attrs.iter().map(|(n, v)| format!("{n:?}: {v}")).collect(); py(&format!( "import h5py\n\ f = h5py.File({p:?}, 'r')\n\ want = {{{w}}}\n\ got = {{k: int(v) for k, v in f[{obj:?}].attrs.items() if k in want}}\n\ assert got == want, (got, want)\n", p = path.to_str().unwrap(), w = want.join(", ") )); } #[test] fn random_operations_match_a_model() { if !tools_ok() { return; } let mut seed = 1; for (i, (lv, dump)) in LIBVERS.iter().enumerate() { // h5dump has no LZF decoder (h5py's own filter). for (j, (extra, lzf)) in [ ("", false), (", compression='gzip', shuffle=True, fletcher32=True", false), (", compression='lzf'", true), ] .iter() .enumerate() { seed += 1; random_ops(lv, *dump && !lzf, extra, &format!("{i}_{j}"), seed); } } } fn unsupported(r: Result) { match r { Err(Error::Unsupported(_)) => {} other => panic!("expected Error::Unsupported, got {other:?}"), } } /// The statistics an Extensible Array header keeps (super blocks and their /// bytes, data blocks and their bytes, one past the highest index set, /// elements realised), for the file's only Extensible Array. fn ea_stats(path: &Path) -> [u64; 6] { let b = std::fs::read(path).unwrap(); let at = b .windows(4) .position(|w| w == b"EAHD") .expect("an EA header"); let mut s = [0u64; 6]; for (k, v) in s.iter_mut().enumerate() { let o = at + 12 + 8 * k; *v = u64::from_le_bytes(b[o..o + 8].try_into().unwrap()); } s } /// Version-1 B-tree nodes in the file: (leaves, internal nodes). fn btree1_nodes(path: &Path) -> (usize, usize) { let b = std::fs::read(path).unwrap(); let mut out = (0, 0); for i in 0..b.len().saturating_sub(6) { if &b[i..i + 4] == b"TREE" && b[i + 4] == 1 { if b[i + 5] == 0 { out.0 += 1; } else { out.1 += 1; } } } out } /// Growth one chunk at a time and in large steps: the editor's version-1 /// B-tree must split the way libhdf5's does (the same number of leaves and /// internal nodes for the same insertions). #[test] fn btree1_splits_match_libhdf5() { if !tools_ok() { return; } let dir = tmpdir(); let mut steps: Vec = (1..=300).collect(); steps.extend([1000, 1001, 5000, 9000]); let a = dir.path().join("bt_h5py.h5"); let b = dir.path().join("bt_edit.h5"); let create = |p: &Path, write: bool| { py(&format!( "import h5py, numpy as np\n\ with h5py.File({p:?}, 'w') as f:\n\ \x20 d = f.create_dataset('x', shape=(0,), maxshape=(None,), chunks=(2,), dtype=' = (n..s).map(|v| v as i32).collect(); ed.write_values("x", &block(&[n], &[s - n]), &vals).unwrap(); n = s; } drop(ed); verify(&b, "x", &Model::new(&[n], |i| i as i32)); check_tools(&b, true); assert_eq!( btree1_nodes(&b), btree1_nodes(&a), "B-tree shape differs from libhdf5's" ); } /// Enough chunks that the Extensible Array reaches its paged data blocks /// (the first one holds element 131060 with h5py's parameters): the same /// growth done by libhdf5 and by the editor must create the same blocks. #[test] fn extensible_array_paged_blocks_match_libhdf5() { if !tools_ok() { return; } let dir = tmpdir(); let steps = [5u64, 131_000, 131_070, 133_000, 140_000]; let a = dir.path().join("paged_h5py.h5"); let b = dir.path().join("paged_edit.h5"); let create = |p: &Path, write: bool| { py(&format!( "import h5py, numpy as np\n\ with h5py.File({p:?}, 'w', libver='v114') as f:\n\ \x20 d = f.create_dataset('x', shape=(0,), maxshape=(None,), chunks=(1,), dtype=' = (n..s).map(|v| v as i32).collect(); ed.write_values("x", &block(&[n], &[s - n]), &vals).unwrap(); n = s; } drop(ed); let m = Model::new(&[n], |i| i as i32); verify(&b, "x", &m); check_tools(&b, true); assert_eq!( ea_stats(&b), ea_stats(&a), "EA statistics differ from libhdf5's" ); } /// Every element of `sel` over `shape`, in selection order. /// Every index of the box `ext`, row-major. fn odometer(ext: &[u64]) -> Vec> { let n: u64 = ext.iter().product(); (0..n) .map(|flat| { let mut c = vec![0u64; ext.len()]; let mut r = flat; for d in (0..c.len()).rev() { c[d] = r % ext[d]; r /= ext[d]; } c }) .collect() } fn sel_coords(sel: &Selection, shape: &[u64]) -> Vec> { match sel { Selection::All => odometer(shape), Selection::None => Vec::new(), Selection::Points(p) => p.clone(), Selection::Hyperslab { start, stride, count, block, } => { let ext: Vec = count.iter().zip(block).map(|(c, b)| c * b).collect(); odometer(&ext) .into_iter() .map(|j| { (0..j.len()) .map(|d| start[d] + (j[d] / block[d]) * stride[d] + j[d] % block[d]) .collect() }) .collect() } } } impl Model { fn write_sel(&mut self, sel: &Selection, vals: &[i32]) { let cs = sel_coords(sel, &self.shape); assert_eq!(cs.len(), vals.len()); for (c, v) in cs.iter().zip(vals) { let i = self.index(c); self.data[i] = *v; } } } /// A random selection of `shape`: a block, a strided hyperslab, points, /// or everything. fn random_sel(rng: &mut Rng, shape: &[u64]) -> Selection { let rank = shape.len(); match rng.below(4) { 0 => Selection::All, 1 => { let pts = (0..1 + rng.below(5)) .map(|_| shape.iter().map(|&d| rng.below(d)).collect()) .collect(); Selection::Points(pts) } _ => { let mut start = vec![0; rank]; let mut stride = vec![1; rank]; let mut count = vec![1; rank]; let mut block = vec![1; rank]; for d in 0..rank { start[d] = rng.below(shape[d]); let room = shape[d] - start[d]; block[d] = 1 + rng.below(room.min(3)); stride[d] = block[d] + rng.below(3); // count * stride may overshoot as long as the last block fits. count[d] = 1 + (room - block[d]) / stride[d]; count[d] = 1 + rng.below(count[d]); } Selection::Hyperslab { start, stride, count, block, } } } } /// Datasets of every layout and chunk index h5py writes, overwritten under /// random selections: contiguous (allocated, and never written — late /// allocation), compact, chunked with a Fixed Array (paged, too), single /// chunk (unfiltered and filtered), implicit (early allocation), version-1 /// B-tree (every chunked layout of the default libver), and version-2 /// B-tree (existing unfiltered chunks only). #[test] fn overwrite_every_layout() { if !tools_ok() { return; } let names: &[(&str, &[u64])] = &[ ("contig", &[10, 4]), ("contig_late", &[6]), ("compact", &[3, 4]), ("fixed", &[10, 7]), ("fixed_gz", &[10, 7]), ("fa_paged", &[3000]), ("single", &[5, 6]), ("single_gz", &[5, 6]), ("implicit", &[8, 8]), ("fa_empty", &[10, 7]), ("fa_gz_empty", &[3000]), ("ea_gz_empty", &[20]), ("bt2", &[6, 6]), ]; for (li, (lv, dump)) in LIBVERS.iter().enumerate() { let dir = tmpdir(); let path = dir.path().join(format!("overwrite_{li}.h5")); py(&format!( "import h5py, numpy as np\n\ from h5py import h5p, h5d, h5s, h5t\n\ def ar(*s): return np.arange(int(np.prod(s)), dtype=' = vec![ Model::new(&[10, 4], |i| i as i32), Model::new(&[6], |_| 7), Model::new(&[3, 4], |i| i as i32), { let mut m = Model::new(&[10, 7], |_| -1); m.write_block(&[0, 0], &[3, 4], &[1; 12]); m }, { let mut m = Model::new(&[10, 7], |_| 0); m.write_block(&[4, 3], &[6, 4], &(0..24).collect::>()); m }, Model::new(&[3000], |i| if i % 7 == 0 { 3 } else { 0 }), Model::new(&[5, 6], |_| 0), Model::new(&[5, 6], |i| i as i32), Model::new(&[8, 8], |_| 0), Model::new(&[10, 7], |_| 4), Model::new(&[3000], |_| 0), Model::new(&[20], |_| 0), Model::new(&[6, 6], |i| i as i32), ]; let mut rng = Rng(77 + li as u64); for round in 0..3 { let mut ed = FileEditor::open(&path).unwrap(); for ((name, shape), m) in names.iter().zip(models.iter_mut()) { for _ in 0..6 { let sel = random_sel(&mut rng, shape); let n = sel_coords(&sel, shape).len(); let vals: Vec = (0..n).map(|_| rng.next() as i32).collect(); ed.write_values(name, &sel, &vals) .unwrap_or_else(|e| panic!("{name} round {round} {sel:?}: {e}")); m.write_sel(&sel, &vals); } } drop(ed); for ((name, _), m) in names.iter().zip(&models) { verify(&path, name, m); } check_tools(&path, *dump); } // A version-2 B-tree index can take new chunks only from libhdf5 for // now: growing works, writing the new chunks is refused and changes // nothing. if *lv != "'earliest'" { let mut ed = FileEditor::open(&path).unwrap(); ed.resize("bt2", &[8, 6]).unwrap(); let before = std::fs::read(&path).unwrap(); unsupported(ed.write_values("bt2", &block(&[6, 0], &[2, 6]), &[5; 12])); assert!( std::fs::read(&path).unwrap() == before, "a refused edit changed the file" ); models[12].resize(&[8, 6], 0); } // libhdf5 goes on modifying what we wrote. py(&format!( "import h5py, numpy as np\n\ with h5py.File({p:?}, 'r+') as f:\n\ \x20 for n in {names:?}:\n\ \x20 d = f[n]\n\ \x20 d[(0,) * d.ndim] = 123\n", p = path.to_str().unwrap(), names = names.iter().map(|(n, _)| *n).collect::>() )); for ((name, _), m) in names.iter().zip(models.iter_mut()) { m.data[0] = 123; verify(&path, name, m); } check_tools(&path, *dump); } } /// Attributes added to and replaced on the root group, a group and a /// dataset, until the headers need continuation chunks; then h5py adds, /// changes and deletes attributes in the same headers. #[test] fn attributes_in_place() { if !tools_ok() { return; } for (li, (lv, dump)) in LIBVERS.iter().enumerate() { let dir = tmpdir(); let path = dir.path().join(format!("attrs_{li}.h5")); let p = path.to_str().unwrap(); py(&format!( "import h5py, numpy as np\n\ with h5py.File({p:?}, 'w', libver={lv}) as f:\n\ \x20 f.attrs['title'] = 'root'\n\ \x20 g = f.create_group('g')\n\ \x20 g.attrs['n'] = 3\n\ \x20 d = f.create_dataset('d', data=np.arange(4, dtype=' = Vec::new(); let mut ed = FileEditor::open(&path).unwrap(); // v2 headers hold 8 compact attributes by default. let many = if *lv == "'earliest'" { 20 } else { 6 }; for obj in ["/", "g", "d"] { for i in 0..many { let name = format!("x{i}"); let v = AttrValue::I64Array((0..=i as i64).collect()); ed.set_attr(obj, &name, &v).unwrap(); want.push((obj, name, v)); } // Replace one with something much larger (moves it). let big = AttrValue::String("z".repeat(700)); ed.set_attr(obj, "x1", &big).unwrap(); want.retain(|(o, n, _)| !(*o == obj && n == "x1")); want.push((obj, "x1".into(), big)); // And one smaller, in place. ed.set_attr(obj, "x0", &AttrValue::F64(2.5)).unwrap(); want.retain(|(o, n, _)| !(*o == obj && n == "x0")); want.push((obj, "x0".into(), AttrValue::F64(2.5))); } ed.set_attr("d", "units", &AttrValue::String("km".into())) .unwrap(); want.push(("d", "units".into(), AttrValue::String("km".into()))); if *lv != "'earliest'" { // Up to the compact limit (8) and no further; attributes in // dense storage and tracked creation order are refused, and a // refused edit writes nothing. ed.set_attr("g", "eighth", &AttrValue::I64(8)).unwrap(); want.push(("g", "eighth".into(), AttrValue::I64(8))); let before = std::fs::read(&path).unwrap(); unsupported(ed.set_attr("g", "ninth", &AttrValue::I64(9))); unsupported(ed.set_attr("dense", "k0", &AttrValue::I64(1))); unsupported(ed.set_attr("tracked", "b", &AttrValue::I64(1))); assert!( std::fs::read(&path).unwrap() == before, "a refused edit changed the file" ); } drop(ed); check_tools(&path, *dump); let f = File::open(&path).unwrap(); let expect_py: Vec = want .iter() .map(|(o, n, v)| { let got = if *o == "/" { f.root().attrs().unwrap() } else if *o == "d" { f.dataset(o).unwrap().attrs().unwrap() } else { f.group(o).unwrap().attrs().unwrap() }; let g = got.get(n).unwrap_or_else(|| panic!("{o}/{n} missing")); assert_eq!(format!("{g:?}"), format!("{v:?}"), "{o}/{n}"); let pv = match v { AttrValue::I64Array(a) => format!("{a:?}"), AttrValue::String(s) => format!("{s:?}"), AttrValue::F64(x) => format!("{x:?}"), AttrValue::I64(x) => format!("{x}"), other => panic!("{other:?}"), }; format!("({o:?}, {n:?}, {pv})") }) .collect(); py(&format!( "import h5py, numpy as np\n\ f = h5py.File({p:?}, 'r')\n\ for o, n, v in [{w}]:\n\ \x20 a = f[o].attrs[n]\n\ \x20 a = a.decode() if isinstance(a, bytes) else a\n\ \x20 a = a.tolist() if hasattr(a, 'tolist') else a\n\ \x20 assert a == v, (o, n, a, v)\n\ assert f['d'][()].tolist() == [0, 1, 2, 3]\n\ assert f['g'].attrs['n'] == 3 and f.attrs['title'] == 'root'\n", w = expect_py.join(", ") )); // libhdf5 modifies the headers we changed. py(&format!( "import h5py\n\ with h5py.File({p:?}, 'r+') as f:\n\ \x20 for o in ['/', 'd']:\n\ \x20 f[o].attrs['from_h5py'] = 'yes'\n\ \x20 f[o].attrs['x0'] = 9\n\ \x20 del f[o].attrs['x1']\n" )); check_tools(&path, *dump); let f = File::open(&path).unwrap(); for attrs in [ f.root().attrs().unwrap(), f.dataset("d").unwrap().attrs().unwrap(), ] { assert!(matches!(attrs.get("from_h5py"), Some(AttrValue::String(s)) if s == "yes")); assert!(matches!(attrs.get("x0"), Some(AttrValue::I64(9)))); assert!(!attrs.contains_key("x1")); assert!(attrs.contains_key("x2")); } } } /// Files written by clawhdf5's own writer: appended to, overwritten and /// given attributes, then read by h5py. #[test] fn edit_clawhdf5_written_files() { if !tools_ok() { return; } let dir = tmpdir(); let path = dir.path().join("ours.h5"); let mut b = FileBuilder::new(); b.create_dataset("ext") .with_i32_data(&(0..10).collect::>()) .with_shape(&[10]) .with_maxshape(&[u64::MAX]) .with_chunks(&[4]) .with_deflate(4); b.create_dataset("plain").with_i32_data(&[1, 2, 3, 4, 5, 6]); b.create_dataset("grid") .with_i32_data(&(0..24).collect::>()) .with_shape(&[4, 6]) .with_maxshape(&[u64::MAX, 6]) .with_chunks(&[2, 3]); b.set_attr("version", AttrValue::I64(1)); b.write(&path).unwrap(); let mut ext = Model::new(&[10], |i| i as i32); let mut plain = Model::new(&[6], |i| i as i32 + 1); let mut grid = Model::new(&[4, 6], |i| i as i32); let mut ed = FileEditor::open(&path).unwrap(); let mut rng = Rng(9); for _ in 0..200 { let n = ext.shape[0]; let add = 1 + rng.below(9); ed.resize("ext", &[n + add]).unwrap(); ext.resize(&[n + add], 0); let vals: Vec = (0..add).map(|_| rng.next() as i32).collect(); ed.write_values("ext", &block(&[n], &[add]), &vals).unwrap(); ext.write_block(&[n], &[add], &vals); } for _ in 0..30 { let rows = grid.shape[0] + rng.below(3); ed.resize("grid", &[rows, 6]).unwrap(); grid.resize(&[rows, 6], 0); let sel = random_sel(&mut rng, &grid.shape.clone()); let vals: Vec = (0..sel_coords(&sel, &grid.shape).len()) .map(|_| rng.next() as i32) .collect(); ed.write_values("grid", &sel, &vals).unwrap(); grid.write_sel(&sel, &vals); } ed.write_values( "plain", &Selection::Points(vec![vec![5], vec![0]]), &[60, 10], ) .unwrap(); plain.write_sel(&Selection::Points(vec![vec![5], vec![0]]), &[60, 10]); ed.set_attr("/", "version", &AttrValue::I64(2)).unwrap(); ed.set_attr("ext", "note", &AttrValue::String("appended".into())) .unwrap(); drop(ed); verify(&path, "ext", &ext); verify(&path, "plain", &plain); verify(&path, "grid", &grid); check_tools(&path, true); py(&format!( "import h5py\n\ f = h5py.File({p:?}, 'r')\n\ assert f.attrs['version'] == 2\n\ assert f['ext'].attrs['note'] in ('appended', b'appended')\n", p = path.to_str().unwrap() )); } /// One writer at a time: a second editor (or libhdf5 with file locking) /// is refused until the first is dropped. #[test] fn editor_locks_the_file() { let dir = tmpdir(); let path = dir.path().join("lock.h5"); let mut b = FileBuilder::new(); b.create_dataset("x").with_i32_data(&[1, 2, 3]); b.write(&path).unwrap(); let ed = FileEditor::open(&path).unwrap(); assert!(matches!(FileEditor::open(&path), Err(Error::Locked(_)))); if available(&python(), &["-c", "import h5py"]) { let o = Command::new(python()) .args([ "-c", &format!("import h5py; h5py.File({:?}, 'r+')", path.to_str().unwrap()), ]) .env_remove("HDF5_USE_FILE_LOCKING") .output() .unwrap(); assert!(!o.status.success(), "h5py opened a locked file r+"); } drop(ed); FileEditor::open(&path).unwrap(); } /// Refused edits leave the file byte for byte as it was. #[test] fn refused_edits_change_nothing() { if !tools_ok() { return; } let dir = tmpdir(); let path = dir.path().join("refuse.h5"); py(&format!( "import h5py, numpy as np\n\ with h5py.File({p:?}, 'w') as f:\n\ \x20 f.create_dataset('s', data=['a', 'bb'], dtype=h5py.string_dtype())\n\ \x20 f.create_dataset('x', data=np.arange(6, dtype=' = (900..len).rev().collect(); let mut rest: Vec = (0..900).collect(); let mut rng = Rng(3); for i in (1..rest.len()).rev() { rest.swap(i, rng.below(i as u64 + 1) as usize); } order.extend(rest.iter().take(500)); for (lv, tag) in [("'earliest'", "bt"), ("'v114'", "ea")] { let dir = tmpdir(); let a = dir.path().join(format!("ooo_{tag}_h5py.h5")); let b = dir.path().join(format!("ooo_{tag}_edit.h5")); let create = |p: &Path, write: bool| { py(&format!( "import h5py, numpy as np\n\ with h5py.File({p:?}, 'w', libver={lv}) as f:\n\ \x20 d = f.create_dataset('x', shape=({len},), maxshape=(None,), chunks=(1,), dtype=' = (n..n + add) .map(|i| ((i as f64 / 50.0).sin() * 1000.0).round() / 1000.0) .collect(); ed.write_values("x", &block(&[n], &[add]), &vals).unwrap(); } drop(ed); let size = |p: &Path| std::fs::metadata(p).unwrap().len(); let repacked = |p: &Path| { let out = p.with_extension("repacked.h5"); let _ = std::fs::remove_file(&out); let o = Command::new("h5repack") .args([p.to_str().unwrap(), out.to_str().unwrap()]) .output() .unwrap(); assert!(o.status.success(), "{}", text(&o)); size(&out) }; println!( "chunks {chunk}{comp}, {rounds} appends of {add}: editor {} bytes \ (repacked {}), libhdf5 {} bytes (repacked {})", size(&b), repacked(&b), size(&a), repacked(&a) ); } } /// Optional filters that fail are skipped as libhdf5 skips them: an LZF /// that does not shrink a chunk leaves it stored unfiltered with its mask /// bit set, exactly as h5py stores it. Storing the LZF stream with a clear /// mask at the raw chunk's size once let a later libhdf5 rewrite of the /// chunk (raw, same size) keep the stale mask, and h5py then failed to read /// the dataset. After the editor, h5py r+ rewrites and extends the datasets /// and h5py, h5dump (on the chunks it can decode: LZF is not in h5dump) and /// our reader read every value. #[test] fn optional_filters_that_fail_are_skipped() { if !tools_ok() { return; } for (li, (lv, _)) in LIBVERS.iter().enumerate() { let dir = tmpdir(); let path = dir.path().join(format!("optional_{li}.h5")); let p = path.to_str().unwrap(); py(&format!( "import h5py, numpy as np\n\ with h5py.File({p:?}, 'w', libver={lv}) as f:\n\ \x20 f.create_dataset('u8', shape=(5,), dtype='u1', chunks=(5,), maxshape=(None,), compression='lzf')\n\ \x20 f.create_dataset('mix', shape=(32,), dtype=' = (0..32) .map(|i| { if (i / 8) % 2 == 0 { rng.next() as i32 } else { 7 } }) .collect(); let mut ed = FileEditor::open(&path).unwrap(); ed.write_all("u8", &[182, 0, 0, 0, 0]).unwrap(); ed.write_values("mix", &Selection::All, &mix).unwrap(); drop(ed); // Stored as h5py stores it: raw, LZF's bit (0; 1 behind shuffle) set. let masks = py(&format!( "import h5py\n\ f = h5py.File({p:?}, 'r')\n\ i = lambda d, k: f[d].id.get_chunk_info(k)\n\ print(i('u8', 0).filter_mask, i('u8', 0).size, i('ref', 0).filter_mask, i('ref', 0).size,\n\ \x20 *[(i('mix', k).filter_mask, i('mix', k).size < 36) for k in range(4)])\n" )); assert_eq!( masks, "1 5 1 5 (2, False) (0, True) (2, False) (0, True)", "{lv}" ); let dump = |ds: &str| { let o = Command::new("h5dump") .args(["-d", ds, "-y", "-w", "0", p]) .output() .unwrap(); assert!(o.status.success(), "h5dump -d {ds} {p}:\n{}", text(&o)); let s = String::from_utf8_lossy(&o.stdout).into_owned(); s.split_once("DATA {") .and_then(|(_, r)| r.split_once('}')) .map(|(d, _)| { d.split(|c: char| c == ',' || c.is_whitespace()) .filter(|t| !t.is_empty()) .map(|t| t.parse::().unwrap()) .collect::>() }) .unwrap_or_default() }; if *lv != "'latest'" { assert_eq!(dump("/u8"), [182, 0, 0, 0, 0]); } // libhdf5 rewrites the chunk raw at the same size, then extends the // datasets with more incompressible data. py(&format!( "import h5py, numpy as np\n\ with h5py.File({p:?}, 'r+') as f:\n\ \x20 f['u8'][...] = [182, 0, 0, 0, 1]\n\ \x20 f['u8'].resize((12,))\n\ \x20 f['u8'][5:] = [9, 200, 3, 77, 1, 250, 42]\n\ \x20 m = f['mix']\n\ \x20 m[8:16] = np.arange(8, dtype=' = vec![182, 0, 0, 0, 1, 9, 200, 3, 77, 1, 250, 42]; let mut mix_want = mix.clone(); mix_want[0..8].fill(5); for (k, v) in mix_want[8..16].iter_mut().enumerate() { *v = k as i32 * 104_729 + 12_345; } mix_want.extend([11; 8]); let f = File::open(&path).unwrap(); assert_eq!( f.dataset("u8") .unwrap() .read_selection(&Selection::All) .unwrap(), u8_want ); assert_eq!(f.dataset("mix").unwrap().read_i32().unwrap(), mix_want); drop(f); verify( &path, "mix", &Model { shape: vec![40], data: mix_want, }, ); py(&format!( "import h5py\n\ f = h5py.File({p:?}, 'r')\n\ assert f['u8'][()].tolist() == {u8_want:?}, f['u8'][()]\n" )); if *lv != "'latest'" { let got: Vec = dump("/u8").into_iter().map(|v| v as u8).collect(); assert_eq!(got, u8_want); } let o = Command::new(env!("CARGO_BIN_EXE_h5rs")) .args(["check", "--data", "-q", p]) .output() .unwrap(); assert!(o.status.success(), "h5rs check --data {p}:\n{}", text(&o)); } } /// libhdf5 counts a version-2 object header's attributes through its /// Attribute Info message and reports none without one. The editor adds /// that message, as `H5O__attr_create` does, when it gives a version-2 /// header (h5py `libver='v110'`/`'latest'`) its first attribute: afterwards /// h5py lists, counts and reads every attribute, new and existing, h5dump /// agrees, and h5py `r+` can add more (past the compact limit, into dense /// storage) with the count still right. #[test] fn attribute_count_in_version_2_headers() { if !tools_ok() { return; } for (lv, dump) in [("'v110'", true), ("'latest'", false)] { let dir = tmpdir(); let path = dir.path().join("acount.h5"); let p = path.to_str().unwrap(); py(&format!( "import h5py, numpy as np\n\ with h5py.File({p:?}, 'w', libver={lv}) as f:\n\ \x20 f.create_group('g')\n\ \x20 f.create_group('has').attrs['old'] = 7\n\ \x20 f.create_dataset('d', data=np.arange(3, dtype=' f.root().attrs().unwrap(), "d" | "e" => f.dataset(o).unwrap().attrs().unwrap(), _ => f.group(o).unwrap().attrs().unwrap(), }; assert_eq!(attrs.len(), n + extra, "{o}"); assert!(matches!(attrs.get("a1"), Some(AttrValue::F64(x)) if *x == 1.5)); } if dump { // Every object's attributes: 3 set by the editor on each of // the five, 2 existing, `extra` from h5py on each. let out = Command::new("h5dump").args(["-A", p]).output().unwrap(); assert!(out.status.success(), "h5dump -A:\n{}", text(&out)); let s = String::from_utf8_lossy(&out.stdout); assert_eq!( s.matches("ATTRIBUTE \"").count(), 5 * (3 + extra) + 2, "h5dump -A:\n{s}" ); } }; check(0); // libhdf5 adds more: to 7 or 8 (compact), then past its limit. for extra in [4usize, 10] { py(&format!( "import h5py\n\ with h5py.File({p:?}, 'r+') as f:\n\ \x20 for o in ['/', 'g', 'has', 'd', 'e']:\n\ \x20 for i in range({extra}): f[o].attrs[f'h{{i}}'] = i\n" )); check(extra); check_tools(&path, dump); } } }