Write files HDF5 1.8 can read: FileWriter/FileBuilder::libver_bounds

New `LibVer` (V18, V110, V112, V114, V200, Latest) and
`libver_bounds(low, high)` on the format crate's `FileWriter` and the
facade's `FileBuilder`, as libhdf5's H5Pset_libver_bounds / h5py's
libver=(low, high). The default stays (V110, Latest), byte for byte what
was written before.

With a low bound of 1.8: superblock version 2, layout message version 3
(contiguous, compact, chunked) and a version-1 B-tree chunk index for
every chunked dataset, resizable ones included -- what libhdf5 2.x writes
under libver=('v108', 'latest'). The new chunk B-tree writer
(btree_v1_write.rs) replays H5B_insert with the H5Dbtree.c callbacks for
row-major insertion (split ratios 0.1/0.5/0.9, right keys moved as
H5D__btree_cmp3 moves them, root kept in place): its trees equal
libhdf5's node for node for 1-D/2-D/3-D, 2- and 3-level, filtered and
unfiltered datasets (libhdf5 writing without a chunk cache).

The high bound refuses, with FormatError::LibverBound before anything is
written, what needs a newer format: virtual datasets and the paged
file-space strategy (1.10), the 1.12 reference types (datatype v4),
native complex (datatype v5, HDF5 2.0), and a low bound above the high.

Tests: tools/tests/libver_v18.rs writes every writer feature under
(V18, V18), and HDF5 1.8.23's h5dump (scripts/build-hdf5-1.8.sh; skipped
when absent) dumps it exactly as h5dump 1.14 does and returns our bytes
for every numeric dataset; h5py, clawhdf5 and h5rs check --data agree;
then FileEditor grows/appends/annotates it and h5py appends, and every
reader checks again. read_harness gains --v18 and --chunk N.

Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
osobh
2026-09-28 23:44:48 -05:00
co-authored by Claude Opus 5.5
parent 00b6f76ee0
commit b5a5041655
12 changed files with 1787 additions and 17 deletions
+32 -9
View File
@@ -7,15 +7,19 @@
//! ```text
//! cargo run --release -p clawhdf5-bench --bin read_harness
//! cargo run --release -p clawhdf5-bench --bin read_harness -- --large # 512 MB
//! cargo run --release -p clawhdf5-bench --bin read_harness -- --v18 # HDF5 1.8 format
//! cargo run --release -p clawhdf5-bench --bin read_harness -- --chunk 32 # 32 x 32 chunks
//! ```
//!
//! `--v18` writes the file with `libver_bounds(V18, V18)` (version-1 B-tree
//! chunk indexes) instead of the default 1.10 format (Fixed Array indexes
//! here), to compare the two.
use std::time::{Duration, Instant};
use clawhdf5::{File, FileBuilder};
use clawhdf5::{File, FileBuilder, LibVer};
use clawhdf5_format::selection::Selection;
const CHUNK: u64 = 256;
struct Layout {
name: &'static str,
chunked: bool,
@@ -46,16 +50,19 @@ fn value(row: u64, col: u64) -> f64 {
(row * 100_003 + col) as f64 * 0.5
}
fn write_file(path: &std::path::Path, rows: u64, cols: u64) {
fn write_file(path: &std::path::Path, rows: u64, cols: u64, chunk: u64, v18: bool) {
let data: Vec<f64> = (0..rows)
.flat_map(|r| (0..cols).map(move |c| value(r, c)))
.collect();
let mut builder = FileBuilder::new();
if v18 {
builder.libver_bounds(LibVer::V18, LibVer::V18);
}
for (i, layout) in LAYOUTS.iter().enumerate() {
let ds = builder.create_dataset(&format!("d{i}"));
ds.with_f64_data(&data).with_shape(&[rows, cols]);
if layout.chunked {
ds.with_chunks(&[CHUNK, CHUNK]);
ds.with_chunks(&[chunk, chunk]);
}
if layout.deflate {
ds.with_deflate(4);
@@ -91,7 +98,14 @@ fn slab(start: [u64; 2], count: [u64; 2]) -> Selection {
}
fn main() {
let large = std::env::args().any(|a| a == "--large");
let args: Vec<String> = std::env::args().collect();
let large = args.iter().any(|a| a == "--large");
let v18 = args.iter().any(|a| a == "--v18");
let chunk: u64 = args
.iter()
.position(|a| a == "--chunk")
.and_then(|i| args.get(i + 1))
.map_or(256, |c| c.parse().expect("--chunk N"));
let (rows, cols) = if large { (8192, 8192) } else { (4096, 2048) };
let total_mb = (rows * cols * 8) as f64 / (1 << 20) as f64;
if cfg!(debug_assertions) {
@@ -100,12 +114,21 @@ fn main() {
let dir = tempfile::TempDir::new().unwrap();
let path = dir.path().join("read_harness.h5");
write_file(&path, rows, cols);
let file_mb = std::fs::metadata(&path).unwrap().len() as f64 / (1 << 20) as f64;
let t = Instant::now();
write_file(&path, rows, cols, chunk, v18);
let write_ms = t.elapsed().as_secs_f64() * 1e3;
let file_bytes = std::fs::metadata(&path).unwrap().len();
let file_mb = file_bytes as f64 / (1 << 20) as f64;
println!("## Read harness");
println!(
"\n{rows} x {cols} f64 ({total_mb:.0} MB per dataset), chunks {CHUNK} x {CHUNK}, file {file_mb:.0} MB\n"
"\n{rows} x {cols} f64 ({total_mb:.0} MB per dataset), chunks {chunk} x {chunk}, \
format {}, file {file_mb:.0} MB ({file_bytes} bytes), written in {write_ms:.0} ms\n",
if v18 {
"1.8 (v1 B-tree)"
} else {
"1.10 (default)"
}
);
// (label, selection, elements selected)