Write files HDF5 1.8 can read: FileWriter/FileBuilder::libver_bounds

New `LibVer` (V18, V110, V112, V114, V200, Latest) and
`libver_bounds(low, high)` on the format crate's `FileWriter` and the
facade's `FileBuilder`, as libhdf5's H5Pset_libver_bounds / h5py's
libver=(low, high). The default stays (V110, Latest), byte for byte what
was written before.

With a low bound of 1.8: superblock version 2, layout message version 3
(contiguous, compact, chunked) and a version-1 B-tree chunk index for
every chunked dataset, resizable ones included -- what libhdf5 2.x writes
under libver=('v108', 'latest'). The new chunk B-tree writer
(btree_v1_write.rs) replays H5B_insert with the H5Dbtree.c callbacks for
row-major insertion (split ratios 0.1/0.5/0.9, right keys moved as
H5D__btree_cmp3 moves them, root kept in place): its trees equal
libhdf5's node for node for 1-D/2-D/3-D, 2- and 3-level, filtered and
unfiltered datasets (libhdf5 writing without a chunk cache).

The high bound refuses, with FormatError::LibverBound before anything is
written, what needs a newer format: virtual datasets and the paged
file-space strategy (1.10), the 1.12 reference types (datatype v4),
native complex (datatype v5, HDF5 2.0), and a low bound above the high.

Tests: tools/tests/libver_v18.rs writes every writer feature under
(V18, V18), and HDF5 1.8.23's h5dump (scripts/build-hdf5-1.8.sh; skipped
when absent) dumps it exactly as h5dump 1.14 does and returns our bytes
for every numeric dataset; h5py, clawhdf5 and h5rs check --data agree;
then FileEditor grows/appends/annotates it and h5py appends, and every
reader checks again. read_harness gains --v18 and --chunk N.

Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
osobh
2026-09-28 23:44:48 -05:00
co-authored by Claude Opus 5.5
parent 00b6f76ee0
commit b5a5041655
12 changed files with 1787 additions and 17 deletions
+236 -8
View File
@@ -5,12 +5,13 @@
use crate::addr::saturating_usize;
#[cfg(not(feature = "std"))]
use alloc::{format, vec, vec::Vec};
use alloc::{format, string::String, vec, vec::Vec};
use crate::attribute::AttributeMessage;
use crate::btree_v2_write::{BTreeV2Params, build_btree_v2};
use crate::chunked_write::{
ChunkOptions, PrecompressedChunks, build_chunked_data_from_precompressed, precompress_chunks,
ChunkOptions, PrecompressedChunks, build_chunked_data_from_precompressed_libver,
precompress_chunks,
};
use crate::data_layout::VdsMapping;
use crate::dataspace::{Dataspace, DataspaceType};
@@ -31,6 +32,7 @@ pub use crate::type_builders::ProvenanceConfig;
pub use crate::type_builders::{AttrValue, CompoundTypeBuilder, EnumTypeBuilder};
use crate::datatype::{CharacterSet, Datatype};
use crate::libver::LibVer;
pub(crate) const OFFSET_SIZE: u8 = 8;
pub(crate) const LENGTH_SIZE: u8 = 8;
@@ -168,13 +170,15 @@ pub(crate) fn build_dataset_oh(
attrs: AttrStorage<'_>,
fill_message: &[u8],
refcount: u32,
layout_version: u8,
) -> Result<Vec<u8>, FormatError> {
let mut w = ObjectHeaderWriter::new();
w.add_message_with_flags(MessageType::Datatype, dt.serialize(), 0x01);
w.add_message(MessageType::Dataspace, ds.serialize(LENGTH_SIZE));
w.add_message_with_flags(MessageType::FillValue, fill_message.to_vec(), 0x01);
// Versions 3 and 4 encode a contiguous layout the same way.
let mut dl = Vec::new();
dl.push(4); // version
dl.push(layout_version);
dl.push(1); // class = contiguous
// An empty dataset has no storage: its address must be the undefined
// address, as libhdf5 writes it. A real address with size 0 trips
@@ -198,14 +202,16 @@ pub(crate) fn build_compact_dataset_oh(
attrs: AttrStorage<'_>,
fill_message: &[u8],
refcount: u32,
layout_version: u8,
) -> Result<Vec<u8>, FormatError> {
let mut w = ObjectHeaderWriter::new();
w.add_message_with_flags(MessageType::Datatype, dt.serialize(), 0x01);
w.add_message(MessageType::Dataspace, ds.serialize(LENGTH_SIZE));
w.add_message_with_flags(MessageType::FillValue, fill_message.to_vec(), 0x01);
// Compact layout message: version=4, class=0, u16 size, inline data
// Compact layout message: version (3 and 4 are the same here), class=0,
// u16 size, inline data
let mut dl = Vec::new();
dl.push(4); // version
dl.push(layout_version);
dl.push(0); // class = compact
dl.extend_from_slice(&(data.len() as u16).to_le_bytes());
dl.extend_from_slice(data);
@@ -1371,6 +1377,10 @@ pub struct FileWriter {
/// file-space strategy (a File Space Info message in the superblock
/// extension).
page_size: Option<u32>,
/// Library version bounds: the low bound picks the format versions
/// written, the high bound limits the features allowed.
low: LibVer,
high: LibVer,
}
impl Default for FileWriter {
@@ -1485,9 +1495,37 @@ impl FileWriter {
alignment_threshold: 0,
alignment_bytes: 0,
page_size: None,
low: LibVer::V110,
high: LibVer::Latest,
}
}
/// Set the library version bounds, as libhdf5's `H5Pset_libver_bounds`
/// (h5py's `libver=(low, high)`): the oldest HDF5 release whose format
/// the file uses (`low`), and the newest whose features it may use
/// (`high`). See [`crate::libver`] for what each bound changes.
///
/// The default, `(LibVer::V110, LibVer::Latest)`, is what clawhdf5 has
/// always written: the HDF5 1.10 format (version-3 superblock, version-4
/// layouts with the 1.10 chunk indexes), readable by HDF5 1.10 and later.
///
/// `(LibVer::V18, LibVer::V18)` writes a file HDF5 1.8 can read — the
/// low bound libhdf5 2.0 uses by default: a version-2 superblock,
/// version-3 layouts, and a version-1 B-tree for every chunked dataset,
/// resizable ones included; [`Self::finish`] then fails with
/// [`FormatError::LibverBound`] for anything HDF5 1.8 cannot read
/// (virtual datasets, a paged file, the 1.12 reference types, native
/// complex numbers). With a low bound of 1.8 and a later high bound
/// such objects are written in the newer format, as libhdf5 writes them;
/// the rest of the file stays readable by 1.8.
///
/// A low bound above the high bound makes [`Self::finish`] fail.
pub fn libver_bounds(&mut self, low: LibVer, high: LibVer) -> &mut Self {
self.low = low;
self.high = high;
self
}
/// Set global file alignment: datasets with raw data >= `threshold` bytes
/// will have their data aligned to `bytes` boundary.
///
@@ -1583,6 +1621,33 @@ impl FileWriter {
)));
}
let (low, high) = (self.low, self.high);
let within_bounds = |what: &dyn Fn() -> String, needs: LibVer| {
if needs > high {
Err(FormatError::LibverBound {
what: what(),
needs,
high,
})
} else {
Ok(())
}
};
within_bounds(&|| format!("a low library version bound of {low}"), low)?;
if page_size.is_some() {
within_bounds(&|| "the paged file-space strategy".into(), LibVer::V110)?;
}
// Versions 3 of the layout message and 2 of the superblock are what
// HDF5 1.8 reads; 1.10 added version 4 (with its chunk indexes) and
// version 3. A paged file needs the version-3 superblock whatever
// the low bound (libhdf5 raises it as far as the high bound allows).
let layout_version: u8 = if low < LibVer::V110 { 3 } else { 4 };
let superblock_version: u8 = if low < LibVer::V110 && page_size.is_none() {
2
} else {
3
};
// The group tree, in layout order: groups depth-first from the root,
// then every group's datasets in the same order.
let tree = writer_tree::build(self.root, self.track_order)?;
@@ -1621,9 +1686,20 @@ impl FileWriter {
let ds_attrs = all_ds.iter().flat_map(|d| &d.attrs);
for a in group_attrs.chain(ds_attrs) {
a.datatype.check_encodable()?;
within_bounds(
&|| format!("the datatype of attribute {:?}", a.name),
LibVer::for_datatype_version(a.datatype.max_encoded_version()),
)?;
}
for d in &all_ds {
d.dt.check_encodable()?;
within_bounds(
&|| "a dataset's datatype".into(),
LibVer::for_datatype_version(d.dt.max_encoded_version()),
)?;
if d.virtual_sources.is_some() {
within_bounds(&|| "a virtual dataset".into(), LibVer::V110)?;
}
}
let is_vds: Vec<bool> = all_ds.iter().map(|d| d.virtual_sources.is_some()).collect();
@@ -1749,10 +1825,11 @@ impl FileWriter {
elem_size,
&d.chunk_options,
)?;
let result = build_chunked_data_from_precompressed(
let result = build_chunked_data_from_precompressed_libver(
&pre,
dummy_cursor,
d.maxshape.as_deref(),
low,
)?;
dummy_cursor += result.data_bytes.len() as u64;
let oh = build_chunked_dataset_oh(
@@ -1785,6 +1862,7 @@ impl FileWriter {
},
&d.fill_message,
d.refcount,
layout_version,
)?;
dummy_blobs.push(DataBlob {
data: vec![],
@@ -1804,6 +1882,7 @@ impl FileWriter {
},
&d.fill_message,
d.refcount,
layout_version,
)?;
dummy_blobs.push(DataBlob {
data: vec![],
@@ -1904,13 +1983,14 @@ impl FileWriter {
let base_address = cursor2 as u64;
// Reuse precompressed chunks from Pass 1 — avoids re-compressing
// the same data a second time.
let result = build_chunked_data_from_precompressed(
let result = build_chunked_data_from_precompressed_libver(
dummy_blobs[i]
.precompressed
.as_ref()
.expect("chunked dataset missing precompressed cache"),
base_address,
d.maxshape.as_deref(),
low,
)?;
cursor2 += result.data_bytes.len();
let oh = build_chunked_dataset_oh(
@@ -1944,6 +2024,7 @@ impl FileWriter {
},
&d.fill_message,
d.refcount,
layout_version,
)?;
ds_blobs2.push(DataBlob {
data: vec![],
@@ -1973,6 +2054,7 @@ impl FileWriter {
},
&d.fill_message,
d.refcount,
layout_version,
)?;
let mut data = vec![0u8; padding];
data.extend_from_slice(&d.raw);
@@ -1997,7 +2079,7 @@ impl FileWriter {
let mut buf = Vec::with_capacity(cursor2);
let sb = Superblock {
version: 3,
version: superblock_version,
offset_size: OFFSET_SIZE,
length_size: LENGTH_SIZE,
base_address: 0,
@@ -2783,4 +2865,150 @@ mod tests {
assert_eq!(sb.version, 3);
assert_eq!(sb.page_size, None);
}
fn layout_of(bytes: &[u8], name: &str) -> Vec<u8> {
let sb = Superblock::parse(bytes, 0).unwrap();
let addr = resolve_path_any(bytes, &sb, name).unwrap();
let hdr = ObjectHeader::parse(bytes, addr as usize, 8, 8).unwrap();
hdr.messages
.iter()
.find(|m| m.msg_type == MessageType::DataLayout)
.unwrap()
.data
.clone()
}
#[test]
fn libver_v18_writes_the_1_8_format() {
let mut fw = FileWriter::new();
fw.libver_bounds(LibVer::V18, LibVer::V18);
fw.create_dataset("contig").with_f64_data(&[1.0, 2.0]);
fw.create_dataset("compact").with_f64_data(&[3.0]).compact();
fw.create_dataset("grow")
.with_f64_data(&[1.0, 2.0, 3.0])
.with_maxshape(&[u64::MAX])
.with_chunks(&[2]);
fw.create_dataset("none")
.with_f64_data(&[])
.with_maxshape(&[u64::MAX])
.with_chunks(&[2]);
let bytes = fw.finish().unwrap();
assert_eq!(Superblock::parse(&bytes, 0).unwrap().version, 2);
assert_eq!(layout_of(&bytes, "contig")[..2], [3, 1]);
assert_eq!(layout_of(&bytes, "compact")[..2], [3, 0]);
let grow = layout_of(&bytes, "grow");
// Version 3, chunked, 2 dimensions (the element size is the last),
// B-tree address, chunk dims 2 and 8.
assert_eq!(grow[..3], [3, 2, 2]);
assert_eq!(grow[11..], [2, 0, 0, 0, 8, 0, 0, 0]);
let root = u64::from_le_bytes(grow[3..11].try_into().unwrap()) as usize;
assert_eq!(&bytes[root..root + 5], b"TREE\x01");
// No chunks, no tree.
assert_eq!(layout_of(&bytes, "none")[3..11], [0xff; 8]);
assert_eq!(read_dataset_f64(&bytes, "grow"), vec![1.0, 2.0, 3.0]);
assert_eq!(read_dataset_f64(&bytes, "contig"), vec![1.0, 2.0]);
assert_eq!(read_dataset_f64(&bytes, "compact"), vec![3.0]);
}
#[test]
fn default_libver_bounds_keep_the_1_10_format() {
let mut fw = FileWriter::new();
fw.create_dataset("contig").with_f64_data(&[1.0, 2.0]);
fw.create_dataset("grow")
.with_f64_data(&[1.0, 2.0, 3.0])
.with_maxshape(&[u64::MAX])
.with_chunks(&[2]);
let default = fw.finish().unwrap();
let mut fw = FileWriter::new();
fw.libver_bounds(LibVer::V110, LibVer::Latest);
fw.create_dataset("contig").with_f64_data(&[1.0, 2.0]);
fw.create_dataset("grow")
.with_f64_data(&[1.0, 2.0, 3.0])
.with_maxshape(&[u64::MAX])
.with_chunks(&[2]);
assert_eq!(fw.finish().unwrap(), default);
assert_eq!(layout_of(&default, "contig")[0], 4);
assert_eq!(layout_of(&default, "grow")[..2], [4, 2]);
}
#[test]
fn libver_high_bound_refuses_newer_features() {
let bound = |r: Result<Vec<u8>, FormatError>, needs: LibVer| match r {
Err(FormatError::LibverBound { needs: n, high, .. }) => {
assert_eq!((n, high), (needs, LibVer::V18));
}
other => panic!("expected a bound error, got {other:?}"),
};
let mut fw = FileWriter::new();
fw.libver_bounds(LibVer::V18, LibVer::V18);
fw.create_dataset("z")
.with_native_complex_f64_data(&[[1.0, 2.0]]);
bound(fw.finish(), LibVer::V200);
let mut fw = FileWriter::new();
fw.libver_bounds(LibVer::V18, LibVer::V18);
fw.create_dataset("x").with_f64_data(&[1.0]).set_attr(
"z",
AttrValue::Raw {
datatype: crate::type_builders::make_native_complex_f64_type(),
shape: vec![],
data: vec![0; 16],
},
);
bound(fw.finish(), LibVer::V200);
let mut fw = FileWriter::new();
fw.libver_bounds(LibVer::V18, LibVer::V18);
fw.create_dataset("r").with_compound_data(
Datatype::Reference {
size: 16,
ref_type: crate::datatype::ReferenceType::Object2,
},
vec![0; 16],
1,
);
bound(fw.finish(), LibVer::V112);
let mut fw = FileWriter::new();
fw.libver_bounds(LibVer::V18, LibVer::V18);
fw.create_dataset("src").with_f64_data(&[1.0, 2.0]);
fw.create_dataset("vds")
.with_shape(&[2])
.with_f64_data(&[])
.with_virtual_sources(vec![VdsMapping {
source_file: ".".into(),
source_dataset: "src".into(),
source_selection: sel_all(),
virtual_selection: sel_hyper_1d(0, 2),
}]);
bound(fw.finish(), LibVer::V110);
let mut fw = FileWriter::new();
fw.libver_bounds(LibVer::V18, LibVer::V18)
.with_page_size(4096);
bound(fw.finish(), LibVer::V110);
let mut fw = FileWriter::new();
fw.libver_bounds(LibVer::V110, LibVer::V18);
bound(fw.finish(), LibVer::V110);
}
#[test]
fn libver_low_v18_high_latest_allows_newer_objects() {
// As libhdf5 does: the object that needs a newer format gets it,
// the rest of the file keeps the 1.8 format.
let mut fw = FileWriter::new();
fw.libver_bounds(LibVer::V18, LibVer::Latest);
fw.create_dataset("z")
.with_native_complex_f64_data(&[[1.0, 2.0]]);
let bytes = fw.finish().unwrap();
assert_eq!(Superblock::parse(&bytes, 0).unwrap().version, 2);
let mut fw = FileWriter::new();
fw.libver_bounds(LibVer::V18, LibVer::Latest)
.with_page_size(4096);
fw.create_dataset("x").with_f64_data(&[1.0]);
let bytes = fw.finish().unwrap();
assert_eq!(Superblock::parse(&bytes, 0).unwrap().version, 3);
assert_eq!(layout_of(&bytes, "x")[0], 3);
}
}