New `LibVer` (V18, V110, V112, V114, V200, Latest) and
`libver_bounds(low, high)` on the format crate's `FileWriter` and the
facade's `FileBuilder`, as libhdf5's H5Pset_libver_bounds / h5py's
libver=(low, high). The default stays (V110, Latest), byte for byte what
was written before.
With a low bound of 1.8: superblock version 2, layout message version 3
(contiguous, compact, chunked) and a version-1 B-tree chunk index for
every chunked dataset, resizable ones included -- what libhdf5 2.x writes
under libver=('v108', 'latest'). The new chunk B-tree writer
(btree_v1_write.rs) replays H5B_insert with the H5Dbtree.c callbacks for
row-major insertion (split ratios 0.1/0.5/0.9, right keys moved as
H5D__btree_cmp3 moves them, root kept in place): its trees equal
libhdf5's node for node for 1-D/2-D/3-D, 2- and 3-level, filtered and
unfiltered datasets (libhdf5 writing without a chunk cache).
The high bound refuses, with FormatError::LibverBound before anything is
written, what needs a newer format: virtual datasets and the paged
file-space strategy (1.10), the 1.12 reference types (datatype v4),
native complex (datatype v5, HDF5 2.0), and a low bound above the high.
Tests: tools/tests/libver_v18.rs writes every writer feature under
(V18, V18), and HDF5 1.8.23's h5dump (scripts/build-hdf5-1.8.sh; skipped
when absent) dumps it exactly as h5dump 1.14 does and returns our bytes
for every numeric dataset; h5py, clawhdf5 and h5rs check --data agree;
then FileEditor grows/appends/annotates it and h5py appends, and every
reader checks again. read_harness gains --v18 and --chunk N.
Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
737 lines
26 KiB
Rust
737 lines
26 KiB
Rust
//! Files written with `FileBuilder::libver_bounds(LibVer::V18, LibVer::V18)`
|
|
//! must be readable by HDF5 1.8. Every writer feature is written under that
|
|
//! bound and read back by HDF5 1.8.23's h5dump (values dumped in binary and
|
|
//! compared), by h5py (libhdf5 2.x), by h5dump 1.14, by clawhdf5 and by
|
|
//! `h5rs check --data`; then `FileEditor` grows, appends to and annotates
|
|
//! the file (splitting version-1 B-tree nodes) and every reader checks it
|
|
//! again. The version-1 B-trees we write are compared node by node with
|
|
//! the ones libhdf5 writes for the same data under h5py's
|
|
//! `libver=('v108', 'latest')`.
|
|
//!
|
|
//! HDF5 1.8 is found through `CLAWHDF5_H5DUMP18` (the path to its h5dump)
|
|
//! or at `~/.cache/hdf5-1.8.23/bin/h5dump`, where
|
|
//! `scripts/build-hdf5-1.8.sh` builds it; without it the 1.8 checks are
|
|
//! skipped (CI has no HDF5 1.8), even with `CLAWHDF5_REQUIRE_INTEROP=1`.
|
|
//! h5py/numpy (`CLAWHDF5_PYTHON`) and h5dump are needed otherwise; they
|
|
//! skip when missing unless `CLAWHDF5_REQUIRE_INTEROP=1`.
|
|
|
|
use std::path::{Path, PathBuf};
|
|
use std::process::{Command, Output};
|
|
|
|
use clawhdf5::{AttrValue, File, FileBuilder, FileEditor, LibVer, Selection};
|
|
use clawhdf5_format::datatype::{CharacterSet, Datatype, StringPadding};
|
|
use clawhdf5_format::file_writer::{CompoundTypeBuilder, EnumTypeBuilder};
|
|
use clawhdf5_format::type_builders::make_i32_type;
|
|
|
|
fn python() -> String {
|
|
std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string())
|
|
}
|
|
|
|
fn interop_required() -> bool {
|
|
std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1")
|
|
}
|
|
|
|
fn available(cmd: &str, args: &[&str]) -> bool {
|
|
Command::new(cmd)
|
|
.args(args)
|
|
.output()
|
|
.map(|o| o.status.success())
|
|
.unwrap_or(false)
|
|
}
|
|
|
|
fn tools_ok() -> bool {
|
|
let ok =
|
|
available(&python(), &["-c", "import h5py, numpy"]) && available("h5dump", &["--version"]);
|
|
if !ok {
|
|
assert!(
|
|
!interop_required(),
|
|
"CLAWHDF5_REQUIRE_INTEROP=1 but h5py/numpy or h5dump is not available"
|
|
);
|
|
eprintln!("SKIP: h5py/numpy or h5dump not available");
|
|
}
|
|
ok
|
|
}
|
|
|
|
/// HDF5 1.8's h5dump, when there is one.
|
|
fn h5dump18() -> Option<PathBuf> {
|
|
let p = match std::env::var_os("CLAWHDF5_H5DUMP18") {
|
|
Some(p) => PathBuf::from(p),
|
|
None => PathBuf::from(std::env::var_os("HOME")?).join(".cache/hdf5-1.8.23/bin/h5dump"),
|
|
};
|
|
let o = Command::new(&p).arg("--version").output().ok()?;
|
|
let v = String::from_utf8_lossy(&o.stdout).to_string();
|
|
if !v.contains("1.8.") {
|
|
eprintln!("SKIP 1.8 checks: {} is not HDF5 1.8 ({v})", p.display());
|
|
return None;
|
|
}
|
|
Some(p)
|
|
}
|
|
|
|
fn py(script: &str) -> String {
|
|
let o = Command::new(python())
|
|
.args(["-c", script])
|
|
.output()
|
|
.expect("run python");
|
|
assert!(
|
|
o.status.success(),
|
|
"python failed:\n{script}\nSTDOUT: {}\nSTDERR: {}",
|
|
String::from_utf8_lossy(&o.stdout),
|
|
String::from_utf8_lossy(&o.stderr)
|
|
);
|
|
String::from_utf8_lossy(&o.stdout).trim().to_string()
|
|
}
|
|
|
|
fn text(o: &Output) -> String {
|
|
format!(
|
|
"{}{}",
|
|
String::from_utf8_lossy(&o.stdout),
|
|
String::from_utf8_lossy(&o.stderr)
|
|
)
|
|
}
|
|
|
|
fn tmpdir() -> tempfile::TempDir {
|
|
tempfile::TempDir::new_in(env!("CARGO_TARGET_TMPDIR")).unwrap()
|
|
}
|
|
|
|
/// The hyperslab of `count` elements from `start`.
|
|
fn block(start: &[u64], count: &[u64]) -> Selection {
|
|
Selection::Hyperslab {
|
|
start: start.to_vec(),
|
|
stride: vec![1; start.len()],
|
|
count: count.to_vec(),
|
|
block: vec![1; start.len()],
|
|
}
|
|
}
|
|
|
|
fn le<T: Copy, const N: usize>(v: &[T], f: impl Fn(T) -> [u8; N]) -> Vec<u8> {
|
|
v.iter().flat_map(|&x| f(x)).collect()
|
|
}
|
|
|
|
/// The datasets of the test file and the bytes each holds (little-endian,
|
|
/// row-major, as h5py's `tobytes()` and h5dump's `-b LE` give them).
|
|
struct Expect {
|
|
datasets: Vec<(String, Vec<u8>)>,
|
|
}
|
|
|
|
impl Expect {
|
|
fn set(&mut self, name: &str, bytes: Vec<u8>) {
|
|
match self.datasets.iter_mut().find(|(n, _)| n == name) {
|
|
Some(e) => e.1 = bytes,
|
|
None => self.datasets.push((name.to_string(), bytes)),
|
|
}
|
|
}
|
|
}
|
|
|
|
const MANY: usize = 100_000;
|
|
|
|
/// Write every feature under the 1.8 bound.
|
|
fn write_file(path: &Path) -> Expect {
|
|
let mut e = Expect {
|
|
datasets: Vec::new(),
|
|
};
|
|
let mut b = FileBuilder::new();
|
|
b.libver_bounds(LibVer::V18, LibVer::V18);
|
|
|
|
// Contiguous, with dense attributes (more than 8).
|
|
let v: Vec<f64> = (0..1000).map(|i| i as f64 * 0.5).collect();
|
|
let d = b.create_dataset("contig").with_f64_data(&v);
|
|
for i in 0..12 {
|
|
d.set_attr(&format!("a{i:02}"), AttrValue::I64(i));
|
|
}
|
|
e.set("/contig", le(&v, f64::to_le_bytes));
|
|
b.create_dataset("empty").with_f64_data(&[]);
|
|
e.set("/empty", vec![]);
|
|
b.create_dataset("scalar")
|
|
.with_f64_data(&[2.5])
|
|
.with_shape(&[]);
|
|
e.set("/scalar", 2.5f64.to_le_bytes().to_vec());
|
|
let v: Vec<i32> = (0..16).map(|i| i * 3 - 7).collect();
|
|
b.create_dataset("compact").with_i32_data(&v).compact();
|
|
e.set("/compact", le(&v, i32::to_le_bytes));
|
|
let v: Vec<f32> = (0..20).map(|i| i as f32 / 3.0).collect();
|
|
b.create_dataset("f32").with_f32_data(&v);
|
|
e.set("/f32", le(&v, f32::to_le_bytes));
|
|
let v: Vec<f32> = vec![0.5, -2.0, 1024.0, 0.0];
|
|
b.create_dataset("f16").with_f16_data(&v);
|
|
e.set(
|
|
"/f16",
|
|
le(&v, |x| {
|
|
clawhdf5_format::float16::f32_to_f16_bits(x).to_le_bytes()
|
|
}),
|
|
);
|
|
|
|
// Chunked with every built-in filter HDF5 1.8 has, and edge chunks.
|
|
let v: Vec<f32> = (0..60 * 70).map(|i| (i % 97) as f32 * 1.25).collect();
|
|
b.create_dataset("chunked")
|
|
.with_f32_data(&v)
|
|
.with_shape(&[60, 70])
|
|
.with_chunks(&[16, 16])
|
|
.with_deflate(6)
|
|
.with_shuffle()
|
|
.with_fletcher32();
|
|
e.set("/chunked", le(&v, f32::to_le_bytes));
|
|
// What the 1.10 indexes would be: Extensible Array, version-2 B-tree,
|
|
// Fixed Array, single chunk. All become version-1 B-trees.
|
|
let v: Vec<i32> = (0..25).collect();
|
|
b.create_dataset("resizable")
|
|
.with_i32_data(&v)
|
|
.with_maxshape(&[u64::MAX])
|
|
.with_chunks(&[4]);
|
|
e.set("/resizable", le(&v, i32::to_le_bytes));
|
|
let v: Vec<i64> = (0..30).map(|i| i * 1_000_000_007).collect();
|
|
b.create_dataset("resizable2")
|
|
.with_i64_data(&v)
|
|
.with_shape(&[5, 6])
|
|
.with_maxshape(&[u64::MAX, u64::MAX])
|
|
.with_chunks(&[2, 4]);
|
|
e.set("/resizable2", le(&v, i64::to_le_bytes));
|
|
let v: Vec<i32> = (0..10).map(|i| -i).collect();
|
|
b.create_dataset("fixedmax")
|
|
.with_i32_data(&v)
|
|
.with_maxshape(&[100])
|
|
.with_chunks(&[3])
|
|
.with_deflate(1);
|
|
e.set("/fixedmax", le(&v, i32::to_le_bytes));
|
|
let v: Vec<i32> = (0..8).map(|i| i * i).collect();
|
|
b.create_dataset("single")
|
|
.with_i32_data(&v)
|
|
.with_chunks(&[8]);
|
|
e.set("/single", le(&v, i32::to_le_bytes));
|
|
// Enough chunks for a three-level tree, and a 2-D two-level one.
|
|
let v: Vec<i32> = (0..MANY as i32).map(|i| i ^ 0x5a5a).collect();
|
|
b.create_dataset("many")
|
|
.with_i32_data(&v)
|
|
.with_maxshape(&[u64::MAX])
|
|
.with_chunks(&[1]);
|
|
e.set("/many", le(&v, i32::to_le_bytes));
|
|
let v: Vec<f64> = (0..100 * 100).map(|i| i as f64).collect();
|
|
b.create_dataset("grid")
|
|
.with_f64_data(&v)
|
|
.with_shape(&[100, 100])
|
|
.with_chunks(&[1, 1]);
|
|
e.set("/grid", le(&v, f64::to_le_bytes));
|
|
b.create_dataset("empty_chunked")
|
|
.with_f64_data(&[])
|
|
.with_maxshape(&[u64::MAX])
|
|
.with_chunks(&[10]);
|
|
e.set("/empty_chunked", vec![]);
|
|
let v: Vec<i32> = (0..12).collect();
|
|
b.create_dataset("filled")
|
|
.with_i32_data(&v)
|
|
.with_maxshape(&[u64::MAX])
|
|
.with_chunks(&[5])
|
|
.with_fill_value(&(-9i32).to_le_bytes());
|
|
e.set("/filled", le(&v, i32::to_le_bytes));
|
|
|
|
// Datatypes.
|
|
let raw = b"abcdehello\0\0\0\0\0".to_vec();
|
|
b.create_dataset("strings").with_compound_data(
|
|
Datatype::String {
|
|
size: 5,
|
|
padding: StringPadding::NullPad,
|
|
charset: CharacterSet::Ascii,
|
|
},
|
|
raw.clone(),
|
|
3,
|
|
);
|
|
e.set("/strings", raw);
|
|
let ct = CompoundTypeBuilder::new()
|
|
.i32_field("a")
|
|
.f64_field("b")
|
|
.build();
|
|
let mut raw = Vec::new();
|
|
for i in 0..5i32 {
|
|
raw.extend_from_slice(&i.to_le_bytes());
|
|
raw.extend_from_slice(&(f64::from(i) * 1.5).to_le_bytes());
|
|
}
|
|
b.create_dataset("compound")
|
|
.with_compound_data(ct, raw.clone(), 5);
|
|
e.set("/compound", raw);
|
|
let et = EnumTypeBuilder::i32_based()
|
|
.value("RED", 0)
|
|
.value("GREEN", 1)
|
|
.value("BLUE", 7)
|
|
.build();
|
|
let v = [0, 7, 1, 1, 0];
|
|
b.create_dataset("enum").with_enum_i32_data(et, &v);
|
|
e.set("/enum", le(&v, i32::to_le_bytes));
|
|
let v: Vec<i32> = (0..24).collect();
|
|
b.create_dataset("array").with_array_data(
|
|
make_i32_type(),
|
|
&[2, 3],
|
|
le(&v, i32::to_le_bytes),
|
|
4,
|
|
);
|
|
e.set("/array", le(&v, i32::to_le_bytes));
|
|
let v: Vec<i64> = vec![-1, 0, i64::MAX];
|
|
b.create_dataset("i64").with_i64_data(&v);
|
|
e.set("/i64", le(&v, i64::to_le_bytes));
|
|
|
|
// Groups: compact with links of every kind, dense (links and
|
|
// attributes), and tracking creation order.
|
|
let mut g = b.create_group("g_compact");
|
|
g.create_dataset("x").with_i32_data(&[1, 2, 3]);
|
|
e.set("/g_compact/x", le(&[1i32, 2, 3], i32::to_le_bytes));
|
|
g.add_soft_link("soft", "/contig");
|
|
g.add_external_link("ext", "other.h5", "/data");
|
|
g.set_attr("title", AttrValue::String("compact group".into()));
|
|
b.add_group(g.finish());
|
|
b.add_hard_link("hard", "/g_compact/x");
|
|
let mut g = b.create_group("g_dense");
|
|
for i in 0..20 {
|
|
let v = [i, i + 1];
|
|
g.create_dataset(&format!("d{i:02}")).with_i32_data(&v);
|
|
e.set(&format!("/g_dense/d{i:02}"), le(&v, i32::to_le_bytes));
|
|
}
|
|
for i in 0..12 {
|
|
g.set_attr(&format!("attr{i:02}"), AttrValue::F64(f64::from(i) / 4.0));
|
|
}
|
|
b.add_group(g.finish());
|
|
let mut g = b.create_group("g_order");
|
|
g.track_order(true);
|
|
for i in 0..10 {
|
|
let n = format!("z{}", 9 - i);
|
|
g.create_dataset(&n).with_i32_data(&[i]);
|
|
e.set(&format!("/g_order/{n}"), i.to_le_bytes().to_vec());
|
|
}
|
|
for i in 0..10 {
|
|
g.set_attr(&format!("b{}", 9 - i), AttrValue::I64(i));
|
|
}
|
|
b.add_group(g.finish());
|
|
b.create_dataset("deep/er/path").with_i32_data(&[42]);
|
|
e.set("/deep/er/path", 42i32.to_le_bytes().to_vec());
|
|
|
|
b.set_attr("version", AttrValue::F64(1.8));
|
|
b.set_attr("ints", AttrValue::I64Array(vec![1, -2, 3]));
|
|
b.set_attr("text", AttrValue::String("readable by 1.8".into()));
|
|
b.set_attr(
|
|
"texts",
|
|
AttrValue::StringArray(vec!["a".into(), "bb".into(), "ccc".into()]),
|
|
);
|
|
b.set_attr("big", AttrValue::U64(u64::MAX));
|
|
b.write(path).unwrap();
|
|
e
|
|
}
|
|
|
|
/// Structural facts of a file: superblock and layout message versions.
|
|
fn check_versions(path: &Path) {
|
|
use clawhdf5_format::message_type::MessageType;
|
|
use clawhdf5_format::object_header::ObjectHeader;
|
|
let bytes = std::fs::read(path).unwrap();
|
|
assert_eq!(bytes[8], 2, "superblock version");
|
|
let f = File::open(path).unwrap();
|
|
let sb = f.superblock().clone();
|
|
for name in [
|
|
"contig",
|
|
"compact",
|
|
"chunked",
|
|
"resizable",
|
|
"resizable2",
|
|
"many",
|
|
] {
|
|
let addr = clawhdf5_format::group_v2::resolve_path_any(&bytes, &sb, name).unwrap();
|
|
let oh = ObjectHeader::parse(&bytes, addr as usize, 8, 8).unwrap();
|
|
let layout = oh
|
|
.messages
|
|
.iter()
|
|
.find(|m| m.msg_type == MessageType::DataLayout)
|
|
.unwrap();
|
|
assert_eq!(layout.data[0], 3, "layout version of {name}");
|
|
}
|
|
}
|
|
|
|
/// clawhdf5 reads every dataset's bytes back.
|
|
fn check_ours(path: &Path, e: &Expect) {
|
|
let f = File::open(path).unwrap();
|
|
for (name, want) in &e.datasets {
|
|
let got = f
|
|
.dataset(name)
|
|
.unwrap()
|
|
.read_selection(&Selection::All)
|
|
.unwrap();
|
|
assert!(got == *want, "our read of {name} differs");
|
|
}
|
|
let attrs = f.dataset("contig").unwrap().attrs().unwrap();
|
|
assert!((12..=13).contains(&attrs.len()), "{attrs:?}");
|
|
}
|
|
|
|
/// h5py (libhdf5 2.x) reads every dataset's bytes back.
|
|
fn check_h5py(path: &Path, e: &Expect, dir: &Path) {
|
|
let mut script = format!(
|
|
"import h5py, numpy as np\nf = h5py.File({p:?}, 'r')\n",
|
|
p = path.to_str().unwrap()
|
|
);
|
|
for (i, (name, want)) in e.datasets.iter().enumerate() {
|
|
let exp = dir.join(format!("expect{i}.bin"));
|
|
std::fs::write(&exp, want).unwrap();
|
|
script.push_str(&format!(
|
|
"a = np.ascontiguousarray(f[{name:?}][()]).tobytes()\n\
|
|
assert a == open({x:?}, 'rb').read(), {name:?}\n",
|
|
x = exp.to_str().unwrap()
|
|
));
|
|
}
|
|
script.push_str(
|
|
"assert f.attrs['text'] in (b'readable by 1.8', 'readable by 1.8')\n\
|
|
assert list(f.attrs['ints']) == [1, -2, 3]\n\
|
|
assert len(f['g_dense'].attrs) == 12 and len(f['g_dense']) == 20\n\
|
|
assert list(f['g_order']) == ['z9', 'z8', 'z7', 'z6', 'z5', 'z4', 'z3', 'z2', 'z1', 'z0']\n\
|
|
assert f['g_compact/soft'].shape == (1000,)\n\
|
|
assert f['resizable'].maxshape == (None,)\n\
|
|
assert f['filled'].fillvalue == -9\n\
|
|
print('ok')\n",
|
|
);
|
|
assert_eq!(py(&script), "ok");
|
|
}
|
|
|
|
/// h5dump (1.14) and `h5rs check --data` accept the file.
|
|
fn check_tools(path: &Path) {
|
|
let p = path.to_str().unwrap();
|
|
let o = Command::new(env!("CARGO_BIN_EXE_h5rs"))
|
|
.args(["check", "--data", "-q", p])
|
|
.output()
|
|
.unwrap();
|
|
assert!(o.status.success(), "h5rs check --data {p}:\n{}", text(&o));
|
|
let o = Command::new("h5dump")
|
|
.args(["-o", "/dev/null", p])
|
|
.output()
|
|
.unwrap();
|
|
assert!(
|
|
o.status.success() && o.stderr.is_empty(),
|
|
"h5dump {p}:\n{}",
|
|
text(&o)
|
|
);
|
|
}
|
|
|
|
/// HDF5 1.8's h5dump reads the whole file exactly as h5dump 1.14 does, and
|
|
/// dumps each numeric dataset's values as the bytes we wrote.
|
|
fn check_18(h5dump: &Path, path: &Path, e: &Expect, dir: &Path) {
|
|
let p = path.to_str().unwrap();
|
|
let o = Command::new(h5dump).arg(p).output().unwrap();
|
|
assert!(
|
|
o.status.success() && o.stderr.is_empty(),
|
|
"h5dump 1.8 {p}:\n{}",
|
|
text(&o)
|
|
);
|
|
let out18 = String::from_utf8_lossy(&o.stdout).to_string();
|
|
assert!(out18.contains("EXTERNAL_LINK \"ext\""), "{out18}");
|
|
let o = Command::new("h5dump").arg(p).output().unwrap();
|
|
assert!(o.status.success(), "h5dump {p}:\n{}", text(&o));
|
|
// HDF5 1.8 has no name for IEEE half floats.
|
|
let out = String::from_utf8_lossy(&o.stdout).replace(
|
|
"H5T_IEEE_F16LE",
|
|
"16-bit little-endian floating-point 16-bit precision",
|
|
);
|
|
if let Some((n, (a, b))) = out18
|
|
.lines()
|
|
.zip(out.lines())
|
|
.enumerate()
|
|
.find(|(_, (a, b))| a != b)
|
|
{
|
|
panic!("h5dump 1.8 and h5dump differ at line {}:\n{a}\n{b}", n + 1);
|
|
}
|
|
assert_eq!(out18.lines().count(), out.lines().count());
|
|
// `-b` writes nothing for compounds, enums, arrays, strings and half
|
|
// floats (HDF5 1.8
|
|
// has no native half float), for h5py's files too: those are covered
|
|
// by the text above.
|
|
let skip = ["/f16", "/strings", "/compound", "/enum", "/array"];
|
|
for (i, (name, want)) in e.datasets.iter().enumerate() {
|
|
if want.is_empty() || skip.contains(&name.as_str()) {
|
|
continue;
|
|
}
|
|
let bin = dir.join(format!("dump18_{i}.bin"));
|
|
let o = Command::new(h5dump)
|
|
.args(["-d", name, "-b", "LE", "-o"])
|
|
.arg(&bin)
|
|
.arg(p)
|
|
.output()
|
|
.unwrap();
|
|
assert!(o.status.success(), "h5dump 1.8 -d {name}:\n{}", text(&o));
|
|
let got = std::fs::read(&bin).unwrap();
|
|
assert!(got == *want, "HDF5 1.8 read of {name} differs");
|
|
}
|
|
}
|
|
|
|
fn check_all(path: &Path, e: &Expect, dir: &Path, h5dump: Option<&Path>) {
|
|
check_ours(path, e);
|
|
check_h5py(path, e, dir);
|
|
check_tools(path);
|
|
if let Some(h) = h5dump {
|
|
check_18(h, path, e, dir);
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn every_feature_reads_in_hdf5_1_8() {
|
|
if !tools_ok() {
|
|
return;
|
|
}
|
|
let h18 = h5dump18();
|
|
let dir = tmpdir();
|
|
let path = dir.path().join("v18.h5");
|
|
let mut e = write_file(&path);
|
|
check_versions(&path);
|
|
check_all(&path, &e, dir.path(), h18.as_deref());
|
|
|
|
// FileEditor: grow the unlimited datasets (splitting B-tree nodes),
|
|
// overwrite values in filtered and unfiltered chunks, set attributes
|
|
// (compact and dense).
|
|
let mut ed = FileEditor::open(&path).unwrap();
|
|
let mut res: Vec<i32> = (0..25).collect();
|
|
for round in 0..300usize {
|
|
let n = res.len() as u64;
|
|
let add = 1 + (round % 9) as u64;
|
|
ed.resize("resizable", &[n + add]).unwrap();
|
|
let vals: Vec<i32> = (0..add).map(|k| (n + k) as i32 * 7 - 3).collect();
|
|
ed.write_values("resizable", &block(&[n], &[add]), &vals)
|
|
.unwrap();
|
|
res.extend(&vals);
|
|
}
|
|
e.set("/resizable", le(&res, i32::to_le_bytes));
|
|
let mut many: Vec<i32> = (0..MANY as i32).map(|i| i ^ 0x5a5a).collect();
|
|
ed.resize("many", &[MANY as u64 + 500]).unwrap();
|
|
let vals: Vec<i32> = (0..500).collect();
|
|
ed.write_values("many", &block(&[MANY as u64], &[500]), &vals)
|
|
.unwrap();
|
|
many.extend(&vals);
|
|
many[12_345] = -1;
|
|
ed.write_values("many", &block(&[12_345], &[1]), &[-1i32])
|
|
.unwrap();
|
|
e.set("/many", le(&many, i32::to_le_bytes));
|
|
let mut chunked: Vec<f32> = (0..60 * 70).map(|i| (i % 97) as f32 * 1.25).collect();
|
|
let row: Vec<f32> = (0..70).map(|i| -(i as f32)).collect();
|
|
ed.write_values("chunked", &block(&[31, 0], &[1, 70]), &row)
|
|
.unwrap();
|
|
chunked[31 * 70..32 * 70].copy_from_slice(&row);
|
|
e.set("/chunked", le(&chunked, f32::to_le_bytes));
|
|
ed.resize("resizable2", &[9, 6]).unwrap();
|
|
let mut r2: Vec<i64> = (0..30).map(|i| i * 1_000_000_007).collect();
|
|
r2.extend(std::iter::repeat_n(0, 24));
|
|
e.set("/resizable2", le(&r2, i64::to_le_bytes));
|
|
ed.set_attr("contig", "added", &AttrValue::F64(3.25))
|
|
.unwrap();
|
|
ed.set_attr("g_dense", "attr03", &AttrValue::F64(-1.0))
|
|
.unwrap();
|
|
ed.set_attr("/", "text", &AttrValue::String("edited".into()))
|
|
.unwrap();
|
|
drop(ed);
|
|
|
|
check_ours(&path, &e);
|
|
let f = File::open(&path).unwrap();
|
|
assert!(matches!(
|
|
f.dataset("contig").unwrap().attr("added").unwrap(),
|
|
Some(AttrValue::F64(v)) if v == 3.25
|
|
));
|
|
drop(f);
|
|
// Everything but the root's "text" attribute is as check_h5py expects.
|
|
py(&format!(
|
|
"import h5py\nf = h5py.File({p:?}, 'r')\n\
|
|
assert f.attrs['text'] in (b'edited', 'edited')\n\
|
|
assert f['contig'].attrs['added'] == 3.25\n\
|
|
assert f['g_dense'].attrs['attr03'] == -1.0\n",
|
|
p = path.to_str().unwrap()
|
|
));
|
|
let o = Command::new(env!("CARGO_BIN_EXE_h5rs"))
|
|
.args(["check", "--data", "-q", path.to_str().unwrap()])
|
|
.output()
|
|
.unwrap();
|
|
assert!(o.status.success(), "h5rs check after edits:\n{}", text(&o));
|
|
if let Some(h) = h18.as_deref() {
|
|
check_18(h, &path, &e, dir.path());
|
|
}
|
|
|
|
// h5py (libhdf5 2.x) goes on appending to the 1.8-format file, and 1.8
|
|
// still reads it.
|
|
py(&format!(
|
|
"import h5py, numpy as np\n\
|
|
with h5py.File({p:?}, 'r+') as f:\n\
|
|
\x20 d = f['resizable']\n\
|
|
\x20 n = d.shape[0]\n\
|
|
\x20 d.resize((n + 100,))\n\
|
|
\x20 d[n:] = np.arange(100, dtype='<i4') + 5000\n",
|
|
p = path.to_str().unwrap()
|
|
));
|
|
res.extend((0..100).map(|k| 5000 + k));
|
|
e.set("/resizable", le(&res, i32::to_le_bytes));
|
|
check_ours(&path, &e);
|
|
if let Some(h) = h18.as_deref() {
|
|
check_18(h, &path, &e, dir.path());
|
|
}
|
|
}
|
|
|
|
// ---- The trees against libhdf5's ----
|
|
|
|
/// One node of a version-1 chunk B-tree: level, and per key its stored
|
|
/// size and offsets; children below.
|
|
#[derive(Debug, PartialEq)]
|
|
struct TreeNode {
|
|
level: u8,
|
|
keys: Vec<(u32, Vec<u64>)>,
|
|
children: Vec<TreeNode>,
|
|
}
|
|
|
|
fn read_tree(bytes: &[u8], addr: u64, ndims: usize, sizes: bool) -> TreeNode {
|
|
let a = addr as usize;
|
|
assert_eq!(&bytes[a..a + 4], b"TREE");
|
|
let level = bytes[a + 5];
|
|
let n = u16::from_le_bytes([bytes[a + 6], bytes[a + 7]]) as usize;
|
|
let mut p = a + 24;
|
|
let mut keys = Vec::new();
|
|
let mut kids = Vec::new();
|
|
for i in 0..=n {
|
|
let size = u32::from_le_bytes(bytes[p..p + 4].try_into().unwrap());
|
|
let offs = (0..ndims)
|
|
.map(|d| u64::from_le_bytes(bytes[p + 8 + 8 * d..p + 16 + 8 * d].try_into().unwrap()))
|
|
.collect();
|
|
keys.push((if sizes { size } else { 0 }, offs));
|
|
p += 8 + 8 * ndims;
|
|
if i < n {
|
|
kids.push(u64::from_le_bytes(bytes[p..p + 8].try_into().unwrap()));
|
|
p += 8;
|
|
}
|
|
}
|
|
let children = if level > 0 {
|
|
kids.iter()
|
|
.map(|&c| read_tree(bytes, c, ndims, sizes))
|
|
.collect()
|
|
} else {
|
|
Vec::new()
|
|
};
|
|
TreeNode {
|
|
level,
|
|
keys,
|
|
children,
|
|
}
|
|
}
|
|
|
|
/// Where two trees first differ (libhdf5's first).
|
|
fn first_difference(a: &TreeNode, b: &TreeNode, at: &str) -> Option<String> {
|
|
if a.level != b.level || a.keys.len() != b.keys.len() {
|
|
return Some(format!(
|
|
"{at}: level {} with {} keys vs level {} with {} keys",
|
|
a.level,
|
|
a.keys.len(),
|
|
b.level,
|
|
b.keys.len()
|
|
));
|
|
}
|
|
if let Some(i) = (0..a.keys.len()).find(|&i| a.keys[i] != b.keys[i]) {
|
|
return Some(format!("{at} key {i}: {:?} vs {:?}", a.keys[i], b.keys[i]));
|
|
}
|
|
a.children
|
|
.iter()
|
|
.zip(&b.children)
|
|
.enumerate()
|
|
.find_map(|(i, (x, y))| first_difference(x, y, &format!("{at}/{i}")))
|
|
}
|
|
|
|
/// The chunk B-tree of dataset `name`: (tree, ndims) from its layout.
|
|
fn tree_of(path: &Path, name: &str, sizes: bool) -> TreeNode {
|
|
use clawhdf5_format::message_type::MessageType;
|
|
use clawhdf5_format::object_header::ObjectHeader;
|
|
let bytes = std::fs::read(path).unwrap();
|
|
let f = File::open(path).unwrap();
|
|
let sb = f.superblock().clone();
|
|
let addr = clawhdf5_format::group_v2::resolve_path_any(&bytes, &sb, name).unwrap();
|
|
let oh = ObjectHeader::parse(&bytes, addr as usize, 8, 8).unwrap();
|
|
let l = &oh
|
|
.messages
|
|
.iter()
|
|
.find(|m| m.msg_type == MessageType::DataLayout)
|
|
.unwrap()
|
|
.data;
|
|
assert_eq!((l[0], l[1]), (3, 2), "{name}: layout v3, chunked");
|
|
let ndims = l[2] as usize;
|
|
let root = u64::from_le_bytes(l[3..11].try_into().unwrap());
|
|
read_tree(&bytes, root, ndims, sizes)
|
|
}
|
|
|
|
/// Our version-1 chunk B-trees are libhdf5's, node for node (levels, child
|
|
/// counts, every key's offsets, and its chunk size where the chunks are
|
|
/// the same bytes), for 1-D, 2-D and 3-D datasets with two- and three-level
|
|
/// trees, filtered or not.
|
|
///
|
|
/// libhdf5 inserts each chunk into the tree when it leaves its chunk cache.
|
|
/// A whole-dataset write with no cache (`rdcc_nbytes=0`, or chunks larger
|
|
/// than the cache) inserts them in row-major order, as we build the tree;
|
|
/// with the default cache small chunks of a 1-D dataset still arrive in
|
|
/// order, but those of a multi-dimensional one arrive in the order the
|
|
/// cache's hash evicts them, which gives the same keys in differently
|
|
/// filled nodes. Both trees index the same chunks; we do not model the
|
|
/// cache.
|
|
#[test]
|
|
fn chunk_btrees_match_libhdf5() {
|
|
if !tools_ok() {
|
|
return;
|
|
}
|
|
let dir = tmpdir();
|
|
let theirs = dir.path().join("libhdf5.h5");
|
|
py(&format!(
|
|
"import h5py, numpy as np\n\
|
|
with h5py.File({p:?}, 'w', libver=('v108', 'latest'), rdcc_nbytes=0) as f:\n\
|
|
\x20 f.create_dataset('d1000', data=np.arange(10000.0), chunks=(10,))\n\
|
|
\x20 f.create_dataset('d999', data=np.arange(9990.0), chunks=(10,))\n\
|
|
\x20 f.create_dataset('big', data=np.arange(100000, dtype='<i4'), chunks=(1,), maxshape=(None,))\n\
|
|
\x20 f.create_dataset('grid', data=np.arange(10000.0).reshape(100, 100), chunks=(1, 1))\n\
|
|
\x20 f.create_dataset('cube', data=np.arange(27000, dtype='<i2').reshape(30, 30, 30), chunks=(2, 3, 5))\n\
|
|
\x20 f.create_dataset('gz', data=np.arange(20000, dtype='<i8') % 13, chunks=(7,), compression='gzip')\n",
|
|
p = theirs.to_str().unwrap()
|
|
));
|
|
let ours = dir.path().join("ours.h5");
|
|
let mut b = FileBuilder::new();
|
|
b.libver_bounds(LibVer::V18, LibVer::V18);
|
|
let v: Vec<f64> = (0..10000).map(f64::from).collect();
|
|
b.create_dataset("d1000")
|
|
.with_f64_data(&v)
|
|
.with_chunks(&[10]);
|
|
b.create_dataset("d999")
|
|
.with_f64_data(&v[..9990])
|
|
.with_chunks(&[10]);
|
|
let v: Vec<i32> = (0..100_000).collect();
|
|
b.create_dataset("big")
|
|
.with_i32_data(&v)
|
|
.with_chunks(&[1])
|
|
.with_maxshape(&[u64::MAX]);
|
|
let v: Vec<f64> = (0..10000).map(f64::from).collect();
|
|
b.create_dataset("grid")
|
|
.with_f64_data(&v)
|
|
.with_shape(&[100, 100])
|
|
.with_chunks(&[1, 1]);
|
|
let raw: Vec<u8> = (0..27000i16).flat_map(|x| x.to_le_bytes()).collect();
|
|
b.create_dataset("cube")
|
|
.with_compound_data(
|
|
clawhdf5_format::datatype::Datatype::FixedPoint {
|
|
size: 2,
|
|
byte_order: clawhdf5_format::datatype::DatatypeByteOrder::LittleEndian,
|
|
signed: true,
|
|
bit_offset: 0,
|
|
bit_precision: 16,
|
|
},
|
|
raw,
|
|
27000,
|
|
)
|
|
.with_shape(&[30, 30, 30])
|
|
.with_chunks(&[2, 3, 5]);
|
|
let v: Vec<i64> = (0..20000).map(|i| i % 13).collect();
|
|
b.create_dataset("gz")
|
|
.with_i64_data(&v)
|
|
.with_chunks(&[7])
|
|
.with_deflate(4);
|
|
b.write(&ours).unwrap();
|
|
for (name, sizes) in [
|
|
("d1000", true),
|
|
("d999", true),
|
|
("big", true),
|
|
("grid", true),
|
|
("cube", true),
|
|
// Compressed sizes differ between zlib-rs and zlib.
|
|
("gz", false),
|
|
] {
|
|
let a = tree_of(&theirs, name, sizes);
|
|
let b = tree_of(&ours, name, sizes);
|
|
if let Some(d) = first_difference(&a, &b, "root") {
|
|
panic!("{name}: our chunk B-tree differs from libhdf5's: {d}");
|
|
}
|
|
}
|
|
}
|