Files
clawhdf5/crates/clawhdf5-tools/tests/libver_v18.rs
T
osobhandClaude Opus 5.5 b5a5041655 Write files HDF5 1.8 can read: FileWriter/FileBuilder::libver_bounds
New `LibVer` (V18, V110, V112, V114, V200, Latest) and
`libver_bounds(low, high)` on the format crate's `FileWriter` and the
facade's `FileBuilder`, as libhdf5's H5Pset_libver_bounds / h5py's
libver=(low, high). The default stays (V110, Latest), byte for byte what
was written before.

With a low bound of 1.8: superblock version 2, layout message version 3
(contiguous, compact, chunked) and a version-1 B-tree chunk index for
every chunked dataset, resizable ones included -- what libhdf5 2.x writes
under libver=('v108', 'latest'). The new chunk B-tree writer
(btree_v1_write.rs) replays H5B_insert with the H5Dbtree.c callbacks for
row-major insertion (split ratios 0.1/0.5/0.9, right keys moved as
H5D__btree_cmp3 moves them, root kept in place): its trees equal
libhdf5's node for node for 1-D/2-D/3-D, 2- and 3-level, filtered and
unfiltered datasets (libhdf5 writing without a chunk cache).

The high bound refuses, with FormatError::LibverBound before anything is
written, what needs a newer format: virtual datasets and the paged
file-space strategy (1.10), the 1.12 reference types (datatype v4),
native complex (datatype v5, HDF5 2.0), and a low bound above the high.

Tests: tools/tests/libver_v18.rs writes every writer feature under
(V18, V18), and HDF5 1.8.23's h5dump (scripts/build-hdf5-1.8.sh; skipped
when absent) dumps it exactly as h5dump 1.14 does and returns our bytes
for every numeric dataset; h5py, clawhdf5 and h5rs check --data agree;
then FileEditor grows/appends/annotates it and h5py appends, and every
reader checks again. read_harness gains --v18 and --chunk N.

Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
2026-09-28 23:44:48 -05:00

737 lines
26 KiB
Rust

//! Files written with `FileBuilder::libver_bounds(LibVer::V18, LibVer::V18)`
//! must be readable by HDF5 1.8. Every writer feature is written under that
//! bound and read back by HDF5 1.8.23's h5dump (values dumped in binary and
//! compared), by h5py (libhdf5 2.x), by h5dump 1.14, by clawhdf5 and by
//! `h5rs check --data`; then `FileEditor` grows, appends to and annotates
//! the file (splitting version-1 B-tree nodes) and every reader checks it
//! again. The version-1 B-trees we write are compared node by node with
//! the ones libhdf5 writes for the same data under h5py's
//! `libver=('v108', 'latest')`.
//!
//! HDF5 1.8 is found through `CLAWHDF5_H5DUMP18` (the path to its h5dump)
//! or at `~/.cache/hdf5-1.8.23/bin/h5dump`, where
//! `scripts/build-hdf5-1.8.sh` builds it; without it the 1.8 checks are
//! skipped (CI has no HDF5 1.8), even with `CLAWHDF5_REQUIRE_INTEROP=1`.
//! h5py/numpy (`CLAWHDF5_PYTHON`) and h5dump are needed otherwise; they
//! skip when missing unless `CLAWHDF5_REQUIRE_INTEROP=1`.
use std::path::{Path, PathBuf};
use std::process::{Command, Output};
use clawhdf5::{AttrValue, File, FileBuilder, FileEditor, LibVer, Selection};
use clawhdf5_format::datatype::{CharacterSet, Datatype, StringPadding};
use clawhdf5_format::file_writer::{CompoundTypeBuilder, EnumTypeBuilder};
use clawhdf5_format::type_builders::make_i32_type;
fn python() -> String {
std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string())
}
fn interop_required() -> bool {
std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1")
}
fn available(cmd: &str, args: &[&str]) -> bool {
Command::new(cmd)
.args(args)
.output()
.map(|o| o.status.success())
.unwrap_or(false)
}
fn tools_ok() -> bool {
let ok =
available(&python(), &["-c", "import h5py, numpy"]) && available("h5dump", &["--version"]);
if !ok {
assert!(
!interop_required(),
"CLAWHDF5_REQUIRE_INTEROP=1 but h5py/numpy or h5dump is not available"
);
eprintln!("SKIP: h5py/numpy or h5dump not available");
}
ok
}
/// HDF5 1.8's h5dump, when there is one.
fn h5dump18() -> Option<PathBuf> {
let p = match std::env::var_os("CLAWHDF5_H5DUMP18") {
Some(p) => PathBuf::from(p),
None => PathBuf::from(std::env::var_os("HOME")?).join(".cache/hdf5-1.8.23/bin/h5dump"),
};
let o = Command::new(&p).arg("--version").output().ok()?;
let v = String::from_utf8_lossy(&o.stdout).to_string();
if !v.contains("1.8.") {
eprintln!("SKIP 1.8 checks: {} is not HDF5 1.8 ({v})", p.display());
return None;
}
Some(p)
}
fn py(script: &str) -> String {
let o = Command::new(python())
.args(["-c", script])
.output()
.expect("run python");
assert!(
o.status.success(),
"python failed:\n{script}\nSTDOUT: {}\nSTDERR: {}",
String::from_utf8_lossy(&o.stdout),
String::from_utf8_lossy(&o.stderr)
);
String::from_utf8_lossy(&o.stdout).trim().to_string()
}
fn text(o: &Output) -> String {
format!(
"{}{}",
String::from_utf8_lossy(&o.stdout),
String::from_utf8_lossy(&o.stderr)
)
}
fn tmpdir() -> tempfile::TempDir {
tempfile::TempDir::new_in(env!("CARGO_TARGET_TMPDIR")).unwrap()
}
/// The hyperslab of `count` elements from `start`.
fn block(start: &[u64], count: &[u64]) -> Selection {
Selection::Hyperslab {
start: start.to_vec(),
stride: vec![1; start.len()],
count: count.to_vec(),
block: vec![1; start.len()],
}
}
fn le<T: Copy, const N: usize>(v: &[T], f: impl Fn(T) -> [u8; N]) -> Vec<u8> {
v.iter().flat_map(|&x| f(x)).collect()
}
/// The datasets of the test file and the bytes each holds (little-endian,
/// row-major, as h5py's `tobytes()` and h5dump's `-b LE` give them).
struct Expect {
datasets: Vec<(String, Vec<u8>)>,
}
impl Expect {
fn set(&mut self, name: &str, bytes: Vec<u8>) {
match self.datasets.iter_mut().find(|(n, _)| n == name) {
Some(e) => e.1 = bytes,
None => self.datasets.push((name.to_string(), bytes)),
}
}
}
const MANY: usize = 100_000;
/// Write every feature under the 1.8 bound.
fn write_file(path: &Path) -> Expect {
let mut e = Expect {
datasets: Vec::new(),
};
let mut b = FileBuilder::new();
b.libver_bounds(LibVer::V18, LibVer::V18);
// Contiguous, with dense attributes (more than 8).
let v: Vec<f64> = (0..1000).map(|i| i as f64 * 0.5).collect();
let d = b.create_dataset("contig").with_f64_data(&v);
for i in 0..12 {
d.set_attr(&format!("a{i:02}"), AttrValue::I64(i));
}
e.set("/contig", le(&v, f64::to_le_bytes));
b.create_dataset("empty").with_f64_data(&[]);
e.set("/empty", vec![]);
b.create_dataset("scalar")
.with_f64_data(&[2.5])
.with_shape(&[]);
e.set("/scalar", 2.5f64.to_le_bytes().to_vec());
let v: Vec<i32> = (0..16).map(|i| i * 3 - 7).collect();
b.create_dataset("compact").with_i32_data(&v).compact();
e.set("/compact", le(&v, i32::to_le_bytes));
let v: Vec<f32> = (0..20).map(|i| i as f32 / 3.0).collect();
b.create_dataset("f32").with_f32_data(&v);
e.set("/f32", le(&v, f32::to_le_bytes));
let v: Vec<f32> = vec![0.5, -2.0, 1024.0, 0.0];
b.create_dataset("f16").with_f16_data(&v);
e.set(
"/f16",
le(&v, |x| {
clawhdf5_format::float16::f32_to_f16_bits(x).to_le_bytes()
}),
);
// Chunked with every built-in filter HDF5 1.8 has, and edge chunks.
let v: Vec<f32> = (0..60 * 70).map(|i| (i % 97) as f32 * 1.25).collect();
b.create_dataset("chunked")
.with_f32_data(&v)
.with_shape(&[60, 70])
.with_chunks(&[16, 16])
.with_deflate(6)
.with_shuffle()
.with_fletcher32();
e.set("/chunked", le(&v, f32::to_le_bytes));
// What the 1.10 indexes would be: Extensible Array, version-2 B-tree,
// Fixed Array, single chunk. All become version-1 B-trees.
let v: Vec<i32> = (0..25).collect();
b.create_dataset("resizable")
.with_i32_data(&v)
.with_maxshape(&[u64::MAX])
.with_chunks(&[4]);
e.set("/resizable", le(&v, i32::to_le_bytes));
let v: Vec<i64> = (0..30).map(|i| i * 1_000_000_007).collect();
b.create_dataset("resizable2")
.with_i64_data(&v)
.with_shape(&[5, 6])
.with_maxshape(&[u64::MAX, u64::MAX])
.with_chunks(&[2, 4]);
e.set("/resizable2", le(&v, i64::to_le_bytes));
let v: Vec<i32> = (0..10).map(|i| -i).collect();
b.create_dataset("fixedmax")
.with_i32_data(&v)
.with_maxshape(&[100])
.with_chunks(&[3])
.with_deflate(1);
e.set("/fixedmax", le(&v, i32::to_le_bytes));
let v: Vec<i32> = (0..8).map(|i| i * i).collect();
b.create_dataset("single")
.with_i32_data(&v)
.with_chunks(&[8]);
e.set("/single", le(&v, i32::to_le_bytes));
// Enough chunks for a three-level tree, and a 2-D two-level one.
let v: Vec<i32> = (0..MANY as i32).map(|i| i ^ 0x5a5a).collect();
b.create_dataset("many")
.with_i32_data(&v)
.with_maxshape(&[u64::MAX])
.with_chunks(&[1]);
e.set("/many", le(&v, i32::to_le_bytes));
let v: Vec<f64> = (0..100 * 100).map(|i| i as f64).collect();
b.create_dataset("grid")
.with_f64_data(&v)
.with_shape(&[100, 100])
.with_chunks(&[1, 1]);
e.set("/grid", le(&v, f64::to_le_bytes));
b.create_dataset("empty_chunked")
.with_f64_data(&[])
.with_maxshape(&[u64::MAX])
.with_chunks(&[10]);
e.set("/empty_chunked", vec![]);
let v: Vec<i32> = (0..12).collect();
b.create_dataset("filled")
.with_i32_data(&v)
.with_maxshape(&[u64::MAX])
.with_chunks(&[5])
.with_fill_value(&(-9i32).to_le_bytes());
e.set("/filled", le(&v, i32::to_le_bytes));
// Datatypes.
let raw = b"abcdehello\0\0\0\0\0".to_vec();
b.create_dataset("strings").with_compound_data(
Datatype::String {
size: 5,
padding: StringPadding::NullPad,
charset: CharacterSet::Ascii,
},
raw.clone(),
3,
);
e.set("/strings", raw);
let ct = CompoundTypeBuilder::new()
.i32_field("a")
.f64_field("b")
.build();
let mut raw = Vec::new();
for i in 0..5i32 {
raw.extend_from_slice(&i.to_le_bytes());
raw.extend_from_slice(&(f64::from(i) * 1.5).to_le_bytes());
}
b.create_dataset("compound")
.with_compound_data(ct, raw.clone(), 5);
e.set("/compound", raw);
let et = EnumTypeBuilder::i32_based()
.value("RED", 0)
.value("GREEN", 1)
.value("BLUE", 7)
.build();
let v = [0, 7, 1, 1, 0];
b.create_dataset("enum").with_enum_i32_data(et, &v);
e.set("/enum", le(&v, i32::to_le_bytes));
let v: Vec<i32> = (0..24).collect();
b.create_dataset("array").with_array_data(
make_i32_type(),
&[2, 3],
le(&v, i32::to_le_bytes),
4,
);
e.set("/array", le(&v, i32::to_le_bytes));
let v: Vec<i64> = vec![-1, 0, i64::MAX];
b.create_dataset("i64").with_i64_data(&v);
e.set("/i64", le(&v, i64::to_le_bytes));
// Groups: compact with links of every kind, dense (links and
// attributes), and tracking creation order.
let mut g = b.create_group("g_compact");
g.create_dataset("x").with_i32_data(&[1, 2, 3]);
e.set("/g_compact/x", le(&[1i32, 2, 3], i32::to_le_bytes));
g.add_soft_link("soft", "/contig");
g.add_external_link("ext", "other.h5", "/data");
g.set_attr("title", AttrValue::String("compact group".into()));
b.add_group(g.finish());
b.add_hard_link("hard", "/g_compact/x");
let mut g = b.create_group("g_dense");
for i in 0..20 {
let v = [i, i + 1];
g.create_dataset(&format!("d{i:02}")).with_i32_data(&v);
e.set(&format!("/g_dense/d{i:02}"), le(&v, i32::to_le_bytes));
}
for i in 0..12 {
g.set_attr(&format!("attr{i:02}"), AttrValue::F64(f64::from(i) / 4.0));
}
b.add_group(g.finish());
let mut g = b.create_group("g_order");
g.track_order(true);
for i in 0..10 {
let n = format!("z{}", 9 - i);
g.create_dataset(&n).with_i32_data(&[i]);
e.set(&format!("/g_order/{n}"), i.to_le_bytes().to_vec());
}
for i in 0..10 {
g.set_attr(&format!("b{}", 9 - i), AttrValue::I64(i));
}
b.add_group(g.finish());
b.create_dataset("deep/er/path").with_i32_data(&[42]);
e.set("/deep/er/path", 42i32.to_le_bytes().to_vec());
b.set_attr("version", AttrValue::F64(1.8));
b.set_attr("ints", AttrValue::I64Array(vec![1, -2, 3]));
b.set_attr("text", AttrValue::String("readable by 1.8".into()));
b.set_attr(
"texts",
AttrValue::StringArray(vec!["a".into(), "bb".into(), "ccc".into()]),
);
b.set_attr("big", AttrValue::U64(u64::MAX));
b.write(path).unwrap();
e
}
/// Structural facts of a file: superblock and layout message versions.
fn check_versions(path: &Path) {
use clawhdf5_format::message_type::MessageType;
use clawhdf5_format::object_header::ObjectHeader;
let bytes = std::fs::read(path).unwrap();
assert_eq!(bytes[8], 2, "superblock version");
let f = File::open(path).unwrap();
let sb = f.superblock().clone();
for name in [
"contig",
"compact",
"chunked",
"resizable",
"resizable2",
"many",
] {
let addr = clawhdf5_format::group_v2::resolve_path_any(&bytes, &sb, name).unwrap();
let oh = ObjectHeader::parse(&bytes, addr as usize, 8, 8).unwrap();
let layout = oh
.messages
.iter()
.find(|m| m.msg_type == MessageType::DataLayout)
.unwrap();
assert_eq!(layout.data[0], 3, "layout version of {name}");
}
}
/// clawhdf5 reads every dataset's bytes back.
fn check_ours(path: &Path, e: &Expect) {
let f = File::open(path).unwrap();
for (name, want) in &e.datasets {
let got = f
.dataset(name)
.unwrap()
.read_selection(&Selection::All)
.unwrap();
assert!(got == *want, "our read of {name} differs");
}
let attrs = f.dataset("contig").unwrap().attrs().unwrap();
assert!((12..=13).contains(&attrs.len()), "{attrs:?}");
}
/// h5py (libhdf5 2.x) reads every dataset's bytes back.
fn check_h5py(path: &Path, e: &Expect, dir: &Path) {
let mut script = format!(
"import h5py, numpy as np\nf = h5py.File({p:?}, 'r')\n",
p = path.to_str().unwrap()
);
for (i, (name, want)) in e.datasets.iter().enumerate() {
let exp = dir.join(format!("expect{i}.bin"));
std::fs::write(&exp, want).unwrap();
script.push_str(&format!(
"a = np.ascontiguousarray(f[{name:?}][()]).tobytes()\n\
assert a == open({x:?}, 'rb').read(), {name:?}\n",
x = exp.to_str().unwrap()
));
}
script.push_str(
"assert f.attrs['text'] in (b'readable by 1.8', 'readable by 1.8')\n\
assert list(f.attrs['ints']) == [1, -2, 3]\n\
assert len(f['g_dense'].attrs) == 12 and len(f['g_dense']) == 20\n\
assert list(f['g_order']) == ['z9', 'z8', 'z7', 'z6', 'z5', 'z4', 'z3', 'z2', 'z1', 'z0']\n\
assert f['g_compact/soft'].shape == (1000,)\n\
assert f['resizable'].maxshape == (None,)\n\
assert f['filled'].fillvalue == -9\n\
print('ok')\n",
);
assert_eq!(py(&script), "ok");
}
/// h5dump (1.14) and `h5rs check --data` accept the file.
fn check_tools(path: &Path) {
let p = path.to_str().unwrap();
let o = Command::new(env!("CARGO_BIN_EXE_h5rs"))
.args(["check", "--data", "-q", p])
.output()
.unwrap();
assert!(o.status.success(), "h5rs check --data {p}:\n{}", text(&o));
let o = Command::new("h5dump")
.args(["-o", "/dev/null", p])
.output()
.unwrap();
assert!(
o.status.success() && o.stderr.is_empty(),
"h5dump {p}:\n{}",
text(&o)
);
}
/// HDF5 1.8's h5dump reads the whole file exactly as h5dump 1.14 does, and
/// dumps each numeric dataset's values as the bytes we wrote.
fn check_18(h5dump: &Path, path: &Path, e: &Expect, dir: &Path) {
let p = path.to_str().unwrap();
let o = Command::new(h5dump).arg(p).output().unwrap();
assert!(
o.status.success() && o.stderr.is_empty(),
"h5dump 1.8 {p}:\n{}",
text(&o)
);
let out18 = String::from_utf8_lossy(&o.stdout).to_string();
assert!(out18.contains("EXTERNAL_LINK \"ext\""), "{out18}");
let o = Command::new("h5dump").arg(p).output().unwrap();
assert!(o.status.success(), "h5dump {p}:\n{}", text(&o));
// HDF5 1.8 has no name for IEEE half floats.
let out = String::from_utf8_lossy(&o.stdout).replace(
"H5T_IEEE_F16LE",
"16-bit little-endian floating-point 16-bit precision",
);
if let Some((n, (a, b))) = out18
.lines()
.zip(out.lines())
.enumerate()
.find(|(_, (a, b))| a != b)
{
panic!("h5dump 1.8 and h5dump differ at line {}:\n{a}\n{b}", n + 1);
}
assert_eq!(out18.lines().count(), out.lines().count());
// `-b` writes nothing for compounds, enums, arrays, strings and half
// floats (HDF5 1.8
// has no native half float), for h5py's files too: those are covered
// by the text above.
let skip = ["/f16", "/strings", "/compound", "/enum", "/array"];
for (i, (name, want)) in e.datasets.iter().enumerate() {
if want.is_empty() || skip.contains(&name.as_str()) {
continue;
}
let bin = dir.join(format!("dump18_{i}.bin"));
let o = Command::new(h5dump)
.args(["-d", name, "-b", "LE", "-o"])
.arg(&bin)
.arg(p)
.output()
.unwrap();
assert!(o.status.success(), "h5dump 1.8 -d {name}:\n{}", text(&o));
let got = std::fs::read(&bin).unwrap();
assert!(got == *want, "HDF5 1.8 read of {name} differs");
}
}
fn check_all(path: &Path, e: &Expect, dir: &Path, h5dump: Option<&Path>) {
check_ours(path, e);
check_h5py(path, e, dir);
check_tools(path);
if let Some(h) = h5dump {
check_18(h, path, e, dir);
}
}
#[test]
fn every_feature_reads_in_hdf5_1_8() {
if !tools_ok() {
return;
}
let h18 = h5dump18();
let dir = tmpdir();
let path = dir.path().join("v18.h5");
let mut e = write_file(&path);
check_versions(&path);
check_all(&path, &e, dir.path(), h18.as_deref());
// FileEditor: grow the unlimited datasets (splitting B-tree nodes),
// overwrite values in filtered and unfiltered chunks, set attributes
// (compact and dense).
let mut ed = FileEditor::open(&path).unwrap();
let mut res: Vec<i32> = (0..25).collect();
for round in 0..300usize {
let n = res.len() as u64;
let add = 1 + (round % 9) as u64;
ed.resize("resizable", &[n + add]).unwrap();
let vals: Vec<i32> = (0..add).map(|k| (n + k) as i32 * 7 - 3).collect();
ed.write_values("resizable", &block(&[n], &[add]), &vals)
.unwrap();
res.extend(&vals);
}
e.set("/resizable", le(&res, i32::to_le_bytes));
let mut many: Vec<i32> = (0..MANY as i32).map(|i| i ^ 0x5a5a).collect();
ed.resize("many", &[MANY as u64 + 500]).unwrap();
let vals: Vec<i32> = (0..500).collect();
ed.write_values("many", &block(&[MANY as u64], &[500]), &vals)
.unwrap();
many.extend(&vals);
many[12_345] = -1;
ed.write_values("many", &block(&[12_345], &[1]), &[-1i32])
.unwrap();
e.set("/many", le(&many, i32::to_le_bytes));
let mut chunked: Vec<f32> = (0..60 * 70).map(|i| (i % 97) as f32 * 1.25).collect();
let row: Vec<f32> = (0..70).map(|i| -(i as f32)).collect();
ed.write_values("chunked", &block(&[31, 0], &[1, 70]), &row)
.unwrap();
chunked[31 * 70..32 * 70].copy_from_slice(&row);
e.set("/chunked", le(&chunked, f32::to_le_bytes));
ed.resize("resizable2", &[9, 6]).unwrap();
let mut r2: Vec<i64> = (0..30).map(|i| i * 1_000_000_007).collect();
r2.extend(std::iter::repeat_n(0, 24));
e.set("/resizable2", le(&r2, i64::to_le_bytes));
ed.set_attr("contig", "added", &AttrValue::F64(3.25))
.unwrap();
ed.set_attr("g_dense", "attr03", &AttrValue::F64(-1.0))
.unwrap();
ed.set_attr("/", "text", &AttrValue::String("edited".into()))
.unwrap();
drop(ed);
check_ours(&path, &e);
let f = File::open(&path).unwrap();
assert!(matches!(
f.dataset("contig").unwrap().attr("added").unwrap(),
Some(AttrValue::F64(v)) if v == 3.25
));
drop(f);
// Everything but the root's "text" attribute is as check_h5py expects.
py(&format!(
"import h5py\nf = h5py.File({p:?}, 'r')\n\
assert f.attrs['text'] in (b'edited', 'edited')\n\
assert f['contig'].attrs['added'] == 3.25\n\
assert f['g_dense'].attrs['attr03'] == -1.0\n",
p = path.to_str().unwrap()
));
let o = Command::new(env!("CARGO_BIN_EXE_h5rs"))
.args(["check", "--data", "-q", path.to_str().unwrap()])
.output()
.unwrap();
assert!(o.status.success(), "h5rs check after edits:\n{}", text(&o));
if let Some(h) = h18.as_deref() {
check_18(h, &path, &e, dir.path());
}
// h5py (libhdf5 2.x) goes on appending to the 1.8-format file, and 1.8
// still reads it.
py(&format!(
"import h5py, numpy as np\n\
with h5py.File({p:?}, 'r+') as f:\n\
\x20 d = f['resizable']\n\
\x20 n = d.shape[0]\n\
\x20 d.resize((n + 100,))\n\
\x20 d[n:] = np.arange(100, dtype='<i4') + 5000\n",
p = path.to_str().unwrap()
));
res.extend((0..100).map(|k| 5000 + k));
e.set("/resizable", le(&res, i32::to_le_bytes));
check_ours(&path, &e);
if let Some(h) = h18.as_deref() {
check_18(h, &path, &e, dir.path());
}
}
// ---- The trees against libhdf5's ----
/// One node of a version-1 chunk B-tree: level, and per key its stored
/// size and offsets; children below.
#[derive(Debug, PartialEq)]
struct TreeNode {
level: u8,
keys: Vec<(u32, Vec<u64>)>,
children: Vec<TreeNode>,
}
fn read_tree(bytes: &[u8], addr: u64, ndims: usize, sizes: bool) -> TreeNode {
let a = addr as usize;
assert_eq!(&bytes[a..a + 4], b"TREE");
let level = bytes[a + 5];
let n = u16::from_le_bytes([bytes[a + 6], bytes[a + 7]]) as usize;
let mut p = a + 24;
let mut keys = Vec::new();
let mut kids = Vec::new();
for i in 0..=n {
let size = u32::from_le_bytes(bytes[p..p + 4].try_into().unwrap());
let offs = (0..ndims)
.map(|d| u64::from_le_bytes(bytes[p + 8 + 8 * d..p + 16 + 8 * d].try_into().unwrap()))
.collect();
keys.push((if sizes { size } else { 0 }, offs));
p += 8 + 8 * ndims;
if i < n {
kids.push(u64::from_le_bytes(bytes[p..p + 8].try_into().unwrap()));
p += 8;
}
}
let children = if level > 0 {
kids.iter()
.map(|&c| read_tree(bytes, c, ndims, sizes))
.collect()
} else {
Vec::new()
};
TreeNode {
level,
keys,
children,
}
}
/// Where two trees first differ (libhdf5's first).
fn first_difference(a: &TreeNode, b: &TreeNode, at: &str) -> Option<String> {
if a.level != b.level || a.keys.len() != b.keys.len() {
return Some(format!(
"{at}: level {} with {} keys vs level {} with {} keys",
a.level,
a.keys.len(),
b.level,
b.keys.len()
));
}
if let Some(i) = (0..a.keys.len()).find(|&i| a.keys[i] != b.keys[i]) {
return Some(format!("{at} key {i}: {:?} vs {:?}", a.keys[i], b.keys[i]));
}
a.children
.iter()
.zip(&b.children)
.enumerate()
.find_map(|(i, (x, y))| first_difference(x, y, &format!("{at}/{i}")))
}
/// The chunk B-tree of dataset `name`: (tree, ndims) from its layout.
fn tree_of(path: &Path, name: &str, sizes: bool) -> TreeNode {
use clawhdf5_format::message_type::MessageType;
use clawhdf5_format::object_header::ObjectHeader;
let bytes = std::fs::read(path).unwrap();
let f = File::open(path).unwrap();
let sb = f.superblock().clone();
let addr = clawhdf5_format::group_v2::resolve_path_any(&bytes, &sb, name).unwrap();
let oh = ObjectHeader::parse(&bytes, addr as usize, 8, 8).unwrap();
let l = &oh
.messages
.iter()
.find(|m| m.msg_type == MessageType::DataLayout)
.unwrap()
.data;
assert_eq!((l[0], l[1]), (3, 2), "{name}: layout v3, chunked");
let ndims = l[2] as usize;
let root = u64::from_le_bytes(l[3..11].try_into().unwrap());
read_tree(&bytes, root, ndims, sizes)
}
/// Our version-1 chunk B-trees are libhdf5's, node for node (levels, child
/// counts, every key's offsets, and its chunk size where the chunks are
/// the same bytes), for 1-D, 2-D and 3-D datasets with two- and three-level
/// trees, filtered or not.
///
/// libhdf5 inserts each chunk into the tree when it leaves its chunk cache.
/// A whole-dataset write with no cache (`rdcc_nbytes=0`, or chunks larger
/// than the cache) inserts them in row-major order, as we build the tree;
/// with the default cache small chunks of a 1-D dataset still arrive in
/// order, but those of a multi-dimensional one arrive in the order the
/// cache's hash evicts them, which gives the same keys in differently
/// filled nodes. Both trees index the same chunks; we do not model the
/// cache.
#[test]
fn chunk_btrees_match_libhdf5() {
if !tools_ok() {
return;
}
let dir = tmpdir();
let theirs = dir.path().join("libhdf5.h5");
py(&format!(
"import h5py, numpy as np\n\
with h5py.File({p:?}, 'w', libver=('v108', 'latest'), rdcc_nbytes=0) as f:\n\
\x20 f.create_dataset('d1000', data=np.arange(10000.0), chunks=(10,))\n\
\x20 f.create_dataset('d999', data=np.arange(9990.0), chunks=(10,))\n\
\x20 f.create_dataset('big', data=np.arange(100000, dtype='<i4'), chunks=(1,), maxshape=(None,))\n\
\x20 f.create_dataset('grid', data=np.arange(10000.0).reshape(100, 100), chunks=(1, 1))\n\
\x20 f.create_dataset('cube', data=np.arange(27000, dtype='<i2').reshape(30, 30, 30), chunks=(2, 3, 5))\n\
\x20 f.create_dataset('gz', data=np.arange(20000, dtype='<i8') % 13, chunks=(7,), compression='gzip')\n",
p = theirs.to_str().unwrap()
));
let ours = dir.path().join("ours.h5");
let mut b = FileBuilder::new();
b.libver_bounds(LibVer::V18, LibVer::V18);
let v: Vec<f64> = (0..10000).map(f64::from).collect();
b.create_dataset("d1000")
.with_f64_data(&v)
.with_chunks(&[10]);
b.create_dataset("d999")
.with_f64_data(&v[..9990])
.with_chunks(&[10]);
let v: Vec<i32> = (0..100_000).collect();
b.create_dataset("big")
.with_i32_data(&v)
.with_chunks(&[1])
.with_maxshape(&[u64::MAX]);
let v: Vec<f64> = (0..10000).map(f64::from).collect();
b.create_dataset("grid")
.with_f64_data(&v)
.with_shape(&[100, 100])
.with_chunks(&[1, 1]);
let raw: Vec<u8> = (0..27000i16).flat_map(|x| x.to_le_bytes()).collect();
b.create_dataset("cube")
.with_compound_data(
clawhdf5_format::datatype::Datatype::FixedPoint {
size: 2,
byte_order: clawhdf5_format::datatype::DatatypeByteOrder::LittleEndian,
signed: true,
bit_offset: 0,
bit_precision: 16,
},
raw,
27000,
)
.with_shape(&[30, 30, 30])
.with_chunks(&[2, 3, 5]);
let v: Vec<i64> = (0..20000).map(|i| i % 13).collect();
b.create_dataset("gz")
.with_i64_data(&v)
.with_chunks(&[7])
.with_deflate(4);
b.write(&ours).unwrap();
for (name, sizes) in [
("d1000", true),
("d999", true),
("big", true),
("grid", true),
("cube", true),
// Compressed sizes differ between zlib-rs and zlib.
("gz", false),
] {
let a = tree_of(&theirs, name, sizes);
let b = tree_of(&ours, name, sizes);
if let Some(d) = first_difference(&a, &b, "root") {
panic!("{name}: our chunk B-tree differs from libhdf5's: {d}");
}
}
}