Write files HDF5 1.8 can read: FileWriter/FileBuilder::libver_bounds

New `LibVer` (V18, V110, V112, V114, V200, Latest) and
`libver_bounds(low, high)` on the format crate's `FileWriter` and the
facade's `FileBuilder`, as libhdf5's H5Pset_libver_bounds / h5py's
libver=(low, high). The default stays (V110, Latest), byte for byte what
was written before.

With a low bound of 1.8: superblock version 2, layout message version 3
(contiguous, compact, chunked) and a version-1 B-tree chunk index for
every chunked dataset, resizable ones included -- what libhdf5 2.x writes
under libver=('v108', 'latest'). The new chunk B-tree writer
(btree_v1_write.rs) replays H5B_insert with the H5Dbtree.c callbacks for
row-major insertion (split ratios 0.1/0.5/0.9, right keys moved as
H5D__btree_cmp3 moves them, root kept in place): its trees equal
libhdf5's node for node for 1-D/2-D/3-D, 2- and 3-level, filtered and
unfiltered datasets (libhdf5 writing without a chunk cache).

The high bound refuses, with FormatError::LibverBound before anything is
written, what needs a newer format: virtual datasets and the paged
file-space strategy (1.10), the 1.12 reference types (datatype v4),
native complex (datatype v5, HDF5 2.0), and a low bound above the high.

Tests: tools/tests/libver_v18.rs writes every writer feature under
(V18, V18), and HDF5 1.8.23's h5dump (scripts/build-hdf5-1.8.sh; skipped
when absent) dumps it exactly as h5dump 1.14 does and returns our bytes
for every numeric dataset; h5py, clawhdf5 and h5rs check --data agree;
then FileEditor grows/appends/annotates it and h5py appends, and every
reader checks again. read_harness gains --v18 and --chunk N.

Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
osobh
2026-09-28 23:44:48 -05:00
co-authored by Claude Opus 5.5
parent 00b6f76ee0
commit b5a5041655
12 changed files with 1787 additions and 17 deletions
+105
View File
@@ -7,6 +7,7 @@ use crate::addr::saturating_usize;
#[cfg(not(feature = "std"))]
use alloc::{format, vec, vec::Vec};
use crate::btree_v1_write;
use crate::btree_v2_write::{BTreeV2Params, build_btree_v2};
use crate::checksum::jenkins_lookup3;
use crate::chunk_cache::{CACHE_LINE_SIZE, align_to_cache_line};
@@ -19,6 +20,7 @@ use crate::filter_pipeline::{
FilterPipeline,
};
use crate::filters::compress_chunk_masked;
use crate::libver::LibVer;
/// Round a file offset up to the next cache-line boundary.
///
/// This ensures chunk data starts at an address that is a multiple of the
@@ -866,6 +868,23 @@ pub fn build_chunked_data_from_precompressed(
base_address: u64,
maxshape: Option<&[u64]>,
) -> Result<ChunkedDataResult, FormatError> {
build_chunked_data_from_precompressed_libver(pre, base_address, maxshape, LibVer::Latest)
}
/// [`build_chunked_data_from_precompressed`] for a file whose low library
/// version bound is `low`: below [`LibVer::V110`] (that is, for HDF5 1.8)
/// every chunked dataset gets a version-3 layout message and a version-1
/// B-tree chunk index, whatever its maximum shape, as libhdf5 writes it;
/// otherwise the version-4 layout and the index libhdf5 picks for it.
pub fn build_chunked_data_from_precompressed_libver(
pre: &PrecompressedChunks,
base_address: u64,
maxshape: Option<&[u64]>,
low: LibVer,
) -> Result<ChunkedDataResult, FormatError> {
if low < LibVer::V110 {
return build_btree_v1_chunked_data(pre, base_address, maxshape);
}
let index = ChunkIndexPlan::new(&pre.shape, maxshape, &pre.chunk_dims)?;
let offset_size: u8 = 8;
let length_size: u8 = 8;
@@ -992,6 +1011,92 @@ pub fn build_chunked_data_from_precompressed(
})
}
/// Lay out precompressed chunks at `base_address` followed by a version-1
/// B-tree chunk index, with a version-3 layout message: what libhdf5 writes
/// for a chunked dataset under a low bound of 1.8.
fn build_btree_v1_chunked_data(
pre: &PrecompressedChunks,
base_address: u64,
maxshape: Option<&[u64]>,
) -> Result<ChunkedDataResult, FormatError> {
if let Some(ms) = maxshape {
let bad = |what: &str| FormatError::ChunkedReadError(format!("maxshape: {what}"));
if ms.len() != pre.shape.len() {
return Err(bad("rank differs from the shape"));
}
if ms.iter().zip(&pre.shape).any(|(&m, &s)| m < s) {
return Err(bad("smaller than the shape"));
}
}
let offset_size: u8 = 8;
let mut data_buf = Vec::new();
let mut entries = Vec::with_capacity(pre.chunks.len());
for (i, (_raw_size, stored, filter_mask)) in pre.chunks.iter().enumerate() {
let aligned_offset = align_to_cache_line(data_buf.len());
if aligned_offset > data_buf.len() {
data_buf.resize(aligned_offset, 0u8);
}
entries.push(btree_v1_write::ChunkEntry {
scaled: scaled_coords(&pre.shape, &pre.chunk_dims, i),
nbytes: stored.len() as u64,
filter_mask: *filter_mask,
address: base_address + data_buf.len() as u64,
});
data_buf.extend_from_slice(stored);
}
let element_size = u32::try_from(pre.element_size)
.map_err(|_| FormatError::Overflow("element size".into()))?;
// A dataset with no chunks has no tree: its address is undefined, as
// libhdf5 leaves it until the first chunk is written.
let btree_address = if entries.is_empty() {
u64::MAX
} else {
let aligned_idx = align_to_cache_line(data_buf.len());
if aligned_idx > data_buf.len() {
data_buf.resize(aligned_idx, 0u8);
}
let addr = base_address + data_buf.len() as u64;
let tree = btree_v1_write::build_chunk_btree_v1_at(
&entries,
&pre.chunk_dims,
element_size,
addr,
offset_size,
)?;
data_buf.extend_from_slice(&tree);
addr
};
let layout_message =
serialize_v3_chunked(&pre.chunk_dims, btree_address, offset_size, element_size)?;
Ok(ChunkedDataResult {
data_bytes: data_buf,
layout_message,
pipeline_message: pre.pipeline_message.clone(),
})
}
/// A version-3 layout message for a chunked dataset: dimensionality (the
/// rank plus one), the B-tree's address, then each chunk dimension and the
/// element size, four bytes each.
fn serialize_v3_chunked(
chunk_dims: &[u64],
btree_address: u64,
offset_size: u8,
element_size: u32,
) -> Result<Vec<u8>, FormatError> {
let ndims = u8::try_from(chunk_dims.len() + 1)
.map_err(|_| FormatError::Overflow("chunked layout rank".into()))?;
let mut buf = vec![3u8, 2, ndims];
push_addr(&mut buf, btree_address, offset_size);
for &d in chunk_dims {
let d =
u32::try_from(d).map_err(|_| FormatError::Overflow(format!("chunk dimension {d}")))?;
buf.extend_from_slice(&d.to_le_bytes());
}
buf.extend_from_slice(&element_size.to_le_bytes());
Ok(buf)
}
/// Most slots a Fixed Array index may have before we refuse to build it: its
/// data block holds one element per chunk of the *maximum* extent, so a huge
/// finite maxshape with small chunks would otherwise exhaust memory.