Write files HDF5 1.8 can read: FileWriter/FileBuilder::libver_bounds
New `LibVer` (V18, V110, V112, V114, V200, Latest) and
`libver_bounds(low, high)` on the format crate's `FileWriter` and the
facade's `FileBuilder`, as libhdf5's H5Pset_libver_bounds / h5py's
libver=(low, high). The default stays (V110, Latest), byte for byte what
was written before.
With a low bound of 1.8: superblock version 2, layout message version 3
(contiguous, compact, chunked) and a version-1 B-tree chunk index for
every chunked dataset, resizable ones included -- what libhdf5 2.x writes
under libver=('v108', 'latest'). The new chunk B-tree writer
(btree_v1_write.rs) replays H5B_insert with the H5Dbtree.c callbacks for
row-major insertion (split ratios 0.1/0.5/0.9, right keys moved as
H5D__btree_cmp3 moves them, root kept in place): its trees equal
libhdf5's node for node for 1-D/2-D/3-D, 2- and 3-level, filtered and
unfiltered datasets (libhdf5 writing without a chunk cache).
The high bound refuses, with FormatError::LibverBound before anything is
written, what needs a newer format: virtual datasets and the paged
file-space strategy (1.10), the 1.12 reference types (datatype v4),
native complex (datatype v5, HDF5 2.0), and a low bound above the high.
Tests: tools/tests/libver_v18.rs writes every writer feature under
(V18, V18), and HDF5 1.8.23's h5dump (scripts/build-hdf5-1.8.sh; skipped
when absent) dumps it exactly as h5dump 1.14 does and returns our bytes
for every numeric dataset; h5py, clawhdf5 and h5rs check --data agree;
then FileEditor grows/appends/annotates it and h5py appends, and every
reader checks again. read_harness gains --v18 and --chunk N.
Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
@@ -7,6 +7,7 @@ use crate::addr::saturating_usize;
|
||||
#[cfg(not(feature = "std"))]
|
||||
use alloc::{format, vec, vec::Vec};
|
||||
|
||||
use crate::btree_v1_write;
|
||||
use crate::btree_v2_write::{BTreeV2Params, build_btree_v2};
|
||||
use crate::checksum::jenkins_lookup3;
|
||||
use crate::chunk_cache::{CACHE_LINE_SIZE, align_to_cache_line};
|
||||
@@ -19,6 +20,7 @@ use crate::filter_pipeline::{
|
||||
FilterPipeline,
|
||||
};
|
||||
use crate::filters::compress_chunk_masked;
|
||||
use crate::libver::LibVer;
|
||||
/// Round a file offset up to the next cache-line boundary.
|
||||
///
|
||||
/// This ensures chunk data starts at an address that is a multiple of the
|
||||
@@ -866,6 +868,23 @@ pub fn build_chunked_data_from_precompressed(
|
||||
base_address: u64,
|
||||
maxshape: Option<&[u64]>,
|
||||
) -> Result<ChunkedDataResult, FormatError> {
|
||||
build_chunked_data_from_precompressed_libver(pre, base_address, maxshape, LibVer::Latest)
|
||||
}
|
||||
|
||||
/// [`build_chunked_data_from_precompressed`] for a file whose low library
|
||||
/// version bound is `low`: below [`LibVer::V110`] (that is, for HDF5 1.8)
|
||||
/// every chunked dataset gets a version-3 layout message and a version-1
|
||||
/// B-tree chunk index, whatever its maximum shape, as libhdf5 writes it;
|
||||
/// otherwise the version-4 layout and the index libhdf5 picks for it.
|
||||
pub fn build_chunked_data_from_precompressed_libver(
|
||||
pre: &PrecompressedChunks,
|
||||
base_address: u64,
|
||||
maxshape: Option<&[u64]>,
|
||||
low: LibVer,
|
||||
) -> Result<ChunkedDataResult, FormatError> {
|
||||
if low < LibVer::V110 {
|
||||
return build_btree_v1_chunked_data(pre, base_address, maxshape);
|
||||
}
|
||||
let index = ChunkIndexPlan::new(&pre.shape, maxshape, &pre.chunk_dims)?;
|
||||
let offset_size: u8 = 8;
|
||||
let length_size: u8 = 8;
|
||||
@@ -992,6 +1011,92 @@ pub fn build_chunked_data_from_precompressed(
|
||||
})
|
||||
}
|
||||
|
||||
/// Lay out precompressed chunks at `base_address` followed by a version-1
|
||||
/// B-tree chunk index, with a version-3 layout message: what libhdf5 writes
|
||||
/// for a chunked dataset under a low bound of 1.8.
|
||||
fn build_btree_v1_chunked_data(
|
||||
pre: &PrecompressedChunks,
|
||||
base_address: u64,
|
||||
maxshape: Option<&[u64]>,
|
||||
) -> Result<ChunkedDataResult, FormatError> {
|
||||
if let Some(ms) = maxshape {
|
||||
let bad = |what: &str| FormatError::ChunkedReadError(format!("maxshape: {what}"));
|
||||
if ms.len() != pre.shape.len() {
|
||||
return Err(bad("rank differs from the shape"));
|
||||
}
|
||||
if ms.iter().zip(&pre.shape).any(|(&m, &s)| m < s) {
|
||||
return Err(bad("smaller than the shape"));
|
||||
}
|
||||
}
|
||||
let offset_size: u8 = 8;
|
||||
let mut data_buf = Vec::new();
|
||||
let mut entries = Vec::with_capacity(pre.chunks.len());
|
||||
for (i, (_raw_size, stored, filter_mask)) in pre.chunks.iter().enumerate() {
|
||||
let aligned_offset = align_to_cache_line(data_buf.len());
|
||||
if aligned_offset > data_buf.len() {
|
||||
data_buf.resize(aligned_offset, 0u8);
|
||||
}
|
||||
entries.push(btree_v1_write::ChunkEntry {
|
||||
scaled: scaled_coords(&pre.shape, &pre.chunk_dims, i),
|
||||
nbytes: stored.len() as u64,
|
||||
filter_mask: *filter_mask,
|
||||
address: base_address + data_buf.len() as u64,
|
||||
});
|
||||
data_buf.extend_from_slice(stored);
|
||||
}
|
||||
let element_size = u32::try_from(pre.element_size)
|
||||
.map_err(|_| FormatError::Overflow("element size".into()))?;
|
||||
// A dataset with no chunks has no tree: its address is undefined, as
|
||||
// libhdf5 leaves it until the first chunk is written.
|
||||
let btree_address = if entries.is_empty() {
|
||||
u64::MAX
|
||||
} else {
|
||||
let aligned_idx = align_to_cache_line(data_buf.len());
|
||||
if aligned_idx > data_buf.len() {
|
||||
data_buf.resize(aligned_idx, 0u8);
|
||||
}
|
||||
let addr = base_address + data_buf.len() as u64;
|
||||
let tree = btree_v1_write::build_chunk_btree_v1_at(
|
||||
&entries,
|
||||
&pre.chunk_dims,
|
||||
element_size,
|
||||
addr,
|
||||
offset_size,
|
||||
)?;
|
||||
data_buf.extend_from_slice(&tree);
|
||||
addr
|
||||
};
|
||||
let layout_message =
|
||||
serialize_v3_chunked(&pre.chunk_dims, btree_address, offset_size, element_size)?;
|
||||
Ok(ChunkedDataResult {
|
||||
data_bytes: data_buf,
|
||||
layout_message,
|
||||
pipeline_message: pre.pipeline_message.clone(),
|
||||
})
|
||||
}
|
||||
|
||||
/// A version-3 layout message for a chunked dataset: dimensionality (the
|
||||
/// rank plus one), the B-tree's address, then each chunk dimension and the
|
||||
/// element size, four bytes each.
|
||||
fn serialize_v3_chunked(
|
||||
chunk_dims: &[u64],
|
||||
btree_address: u64,
|
||||
offset_size: u8,
|
||||
element_size: u32,
|
||||
) -> Result<Vec<u8>, FormatError> {
|
||||
let ndims = u8::try_from(chunk_dims.len() + 1)
|
||||
.map_err(|_| FormatError::Overflow("chunked layout rank".into()))?;
|
||||
let mut buf = vec![3u8, 2, ndims];
|
||||
push_addr(&mut buf, btree_address, offset_size);
|
||||
for &d in chunk_dims {
|
||||
let d =
|
||||
u32::try_from(d).map_err(|_| FormatError::Overflow(format!("chunk dimension {d}")))?;
|
||||
buf.extend_from_slice(&d.to_le_bytes());
|
||||
}
|
||||
buf.extend_from_slice(&element_size.to_le_bytes());
|
||||
Ok(buf)
|
||||
}
|
||||
|
||||
/// Most slots a Fixed Array index may have before we refuse to build it: its
|
||||
/// data block holds one element per chunk of the *maximum* extent, so a huge
|
||||
/// finite maxshape with small chunks would otherwise exhaust memory.
|
||||
|
||||
Reference in New Issue
Block a user