diff --git a/crates/clawhdf5-bench/src/bin/read_harness.rs b/crates/clawhdf5-bench/src/bin/read_harness.rs index b09879a..0a517d6 100644 --- a/crates/clawhdf5-bench/src/bin/read_harness.rs +++ b/crates/clawhdf5-bench/src/bin/read_harness.rs @@ -7,15 +7,19 @@ //! ```text //! cargo run --release -p clawhdf5-bench --bin read_harness //! cargo run --release -p clawhdf5-bench --bin read_harness -- --large # 512 MB +//! cargo run --release -p clawhdf5-bench --bin read_harness -- --v18 # HDF5 1.8 format +//! cargo run --release -p clawhdf5-bench --bin read_harness -- --chunk 32 # 32 x 32 chunks //! ``` +//! +//! `--v18` writes the file with `libver_bounds(V18, V18)` (version-1 B-tree +//! chunk indexes) instead of the default 1.10 format (Fixed Array indexes +//! here), to compare the two. use std::time::{Duration, Instant}; -use clawhdf5::{File, FileBuilder}; +use clawhdf5::{File, FileBuilder, LibVer}; use clawhdf5_format::selection::Selection; -const CHUNK: u64 = 256; - struct Layout { name: &'static str, chunked: bool, @@ -46,16 +50,19 @@ fn value(row: u64, col: u64) -> f64 { (row * 100_003 + col) as f64 * 0.5 } -fn write_file(path: &std::path::Path, rows: u64, cols: u64) { +fn write_file(path: &std::path::Path, rows: u64, cols: u64, chunk: u64, v18: bool) { let data: Vec = (0..rows) .flat_map(|r| (0..cols).map(move |c| value(r, c))) .collect(); let mut builder = FileBuilder::new(); + if v18 { + builder.libver_bounds(LibVer::V18, LibVer::V18); + } for (i, layout) in LAYOUTS.iter().enumerate() { let ds = builder.create_dataset(&format!("d{i}")); ds.with_f64_data(&data).with_shape(&[rows, cols]); if layout.chunked { - ds.with_chunks(&[CHUNK, CHUNK]); + ds.with_chunks(&[chunk, chunk]); } if layout.deflate { ds.with_deflate(4); @@ -91,7 +98,14 @@ fn slab(start: [u64; 2], count: [u64; 2]) -> Selection { } fn main() { - let large = std::env::args().any(|a| a == "--large"); + let args: Vec = std::env::args().collect(); + let large = args.iter().any(|a| a == "--large"); + let v18 = args.iter().any(|a| a == "--v18"); + let chunk: u64 = args + .iter() + .position(|a| a == "--chunk") + .and_then(|i| args.get(i + 1)) + .map_or(256, |c| c.parse().expect("--chunk N")); let (rows, cols) = if large { (8192, 8192) } else { (4096, 2048) }; let total_mb = (rows * cols * 8) as f64 / (1 << 20) as f64; if cfg!(debug_assertions) { @@ -100,12 +114,21 @@ fn main() { let dir = tempfile::TempDir::new().unwrap(); let path = dir.path().join("read_harness.h5"); - write_file(&path, rows, cols); - let file_mb = std::fs::metadata(&path).unwrap().len() as f64 / (1 << 20) as f64; + let t = Instant::now(); + write_file(&path, rows, cols, chunk, v18); + let write_ms = t.elapsed().as_secs_f64() * 1e3; + let file_bytes = std::fs::metadata(&path).unwrap().len(); + let file_mb = file_bytes as f64 / (1 << 20) as f64; println!("## Read harness"); println!( - "\n{rows} x {cols} f64 ({total_mb:.0} MB per dataset), chunks {CHUNK} x {CHUNK}, file {file_mb:.0} MB\n" + "\n{rows} x {cols} f64 ({total_mb:.0} MB per dataset), chunks {chunk} x {chunk}, \ + format {}, file {file_mb:.0} MB ({file_bytes} bytes), written in {write_ms:.0} ms\n", + if v18 { + "1.8 (v1 B-tree)" + } else { + "1.10 (default)" + } ); // (label, selection, elements selected) diff --git a/crates/clawhdf5-format/src/btree_v1_write.rs b/crates/clawhdf5-format/src/btree_v1_write.rs new file mode 100644 index 0000000..c0524d8 --- /dev/null +++ b/crates/clawhdf5-format/src/btree_v1_write.rs @@ -0,0 +1,470 @@ +//! Writing a version-1 B-tree chunk index (node type 1): the chunk index of +//! layout message versions 1-3, and the only one HDF5 1.8 reads. +//! +//! The tree is built the way libhdf5 builds it when the chunks reach it one +//! after another in row-major order (a whole-dataset `H5Dwrite` of a 1-D +//! dataset, or of any dataset without a chunk cache; with one, libhdf5 +//! inserts the small chunks of a multi-dimensional dataset in the order its +//! cache evicts them, which fills the nodes differently): each +//! chunk goes through the same steps as `H5B_insert` (`H5B.c`) with the +//! chunk callbacks of `H5Dbtree.c`, so nodes split where libhdf5's split, +//! with its default split ratios (a full right-most node keeps 90% of its +//! children, a left-most one 10%, any other half), and keys hold what +//! libhdf5's hold: +//! +//! - a chunk's key is its size in the file, its filter mask and its offsets +//! (the element-size coordinate 0); +//! - a node's final key is the zero-size key one chunk past the chunk that +//! last moved it (every scaled coordinate plus one, `H5D__btree_new_node`), +//! which libhdf5 moves only when a new chunk is not below it +//! (`H5D__btree_cmp3`) — so after an even number of appends in one +//! dimension it lies on the last chunk itself; +//! - a full root is copied to a new node and becomes the parent of the copy +//! and its new sibling, so the root's address (the layout message's) never +//! changes. +//! +//! Nodes are laid out in the order libhdf5 allocates them (the root first, +//! then each new node as a split creates it), all of the full node size, the +//! unused slots zero. + +#[cfg(not(feature = "std"))] +use alloc::{format, vec, vec::Vec}; + +use core::cmp::Ordering; + +use crate::error::FormatError; + +/// libhdf5's default chunk B-tree K (`HDF5_BTREE_CHUNK_IK_DEF`): nodes hold +/// up to 2K = 64 children. Superblocks of version 2 cannot record another +/// value without a superblock extension, which this writer does not emit. +pub(crate) const CHUNK_BTREE_K: u16 = 32; + +/// libhdf5's default split ratios (`H5D_XFER_BTREE_SPLIT_RATIO_DEF`) for a +/// left-most, middle and right-most node. +const SPLIT_RATIOS: [f64; 3] = [0.1, 0.5, 0.9]; + +/// A chunk to index: scaled coordinates (offset / chunk dimension) in each +/// dataset dimension, stored size, filter mask and address. +pub(crate) struct ChunkEntry { + pub(crate) scaled: Vec, + pub(crate) nbytes: u64, + pub(crate) filter_mask: u32, + pub(crate) address: u64, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +struct Key { + nbytes: u32, + mask: u32, + /// Scaled coordinates, the element-size one (0 or 1) last. + scaled: Vec, +} + +impl Key { + /// `H5D__btree_new_node`'s right key: one chunk past `self` in every + /// dimension, with no storage. + fn right_of(&self) -> Key { + Key { + nbytes: 0, + mask: 0, + scaled: self.scaled.iter().map(|s| s + 1).collect(), + } + } + + fn cmp_scaled(&self, other: &Key) -> Ordering { + self.scaled.cmp(&other.scaled) + } +} + +#[derive(Debug, Clone)] +struct Node { + level: u8, + left: Option, + right: Option, + /// `children.len() + 1` keys once the node holds a child. + keys: Vec, + /// Chunk addresses in a leaf, node indexes above. + children: Vec, +} + +/// What an insertion below a node did (`H5B__insert_helper`'s outputs). +#[derive(Default)] +struct Ret { + /// The node's new left key (`lt_key_changed`). + lt: Option, + /// The node's new right key (`rt_key_changed`). + rt: Option, + /// The node split: the key shared by the halves and the new right node. + split: Option<(Key, usize)>, +} + +struct Tree { + nodes: Vec, + two_k: usize, +} + +fn bad(why: &str) -> FormatError { + FormatError::SerializationError(format!("version-1 B-tree chunk index: {why}")) +} + +impl Tree { + fn new(k: u16) -> Self { + Self { + nodes: vec![Node { + level: 0, + left: None, + right: None, + keys: Vec::new(), + children: Vec::new(), + }], + two_k: 2 * usize::from(k), + } + } + + /// `H5B_insert` of `key` (a chunk after every chunk already inserted). + fn insert(&mut self, key: &Key, addr: u64) -> Result<(), FormatError> { + let r = self.insert_helper(0, key, addr, 64)?; + let Some((md, split)) = r.split else { + return Ok(()); + }; + // The root split: copy it to a new node and make the root the + // parent of the copy and its new right sibling. + let lt = r.lt.unwrap_or_else(|| self.nodes[0].keys[0].clone()); + let rt = match r.rt { + Some(rt) => rt, + None => self.nodes[split] + .keys + .last() + .cloned() + .ok_or_else(|| bad("empty node"))?, + }; + let moved = self.nodes[0].clone(); + let level = moved.level; + let moved_id = self.nodes.len(); + self.nodes.push(moved); + self.nodes[split].left = Some(moved_id); + self.nodes[0] = Node { + level: level + 1, + left: None, + right: None, + keys: vec![lt, md, rt], + children: vec![moved_id as u64, split as u64], + }; + Ok(()) + } + + fn insert_helper( + &mut self, + id: usize, + key: &Key, + addr: u64, + depth: u8, + ) -> Result { + if depth == 0 { + return Err(bad("tree too deep")); + } + let n = self.nodes[id].children.len(); + let level = self.nodes[id].level; + let mut ret = Ret::default(); + if n == 0 { + // The first chunk (H5B_INS_FIRST): its key and the right key. + let node = &mut self.nodes[id]; + node.keys = vec![key.clone(), key.right_of()]; + node.children = vec![addr]; + return Ok(ret); + } + // Binary search with H5D__btree_cmp3: 1 when the chunk is not below + // the right key, -1 when below the left key, else 0. + let (mut lo, mut hi, mut idx) = (0usize, n, 0usize); + let mut cmp = Ordering::Less; + while lo < hi && cmp != Ordering::Equal { + idx = (lo + hi) / 2; + let node = &self.nodes[id]; + cmp = if key.cmp_scaled(&node.keys[idx + 1]) != Ordering::Less { + Ordering::Greater + } else if key.cmp_scaled(&node.keys[idx]) == Ordering::Less { + Ordering::Less + } else { + Ordering::Equal + }; + if cmp == Ordering::Less { + hi = idx; + } else { + lo = idx + 1; + } + } + let (mut lt_changed, mut rt_changed) = (false, false); + // The child to add after child `idx`, with its left key. + let mut new_child: Option<(Key, u64)> = None; + match cmp { + Ordering::Less => return Err(bad("chunks out of order")), + Ordering::Greater if idx + 1 < n => { + return Err(bad("cannot place chunk")); + } + Ordering::Greater if level == 0 => { + // Past every chunk of the right-most leaf: a new maximum + // (H5B_INS_RIGHT through `new_node`), which moves the right + // key one chunk past it. + idx = n - 1; + self.nodes[id].keys[idx + 1] = key.right_of(); + rt_changed = true; + new_child = Some((key.clone(), addr)); + } + Ordering::Equal if level == 0 => { + // Inside the last chunk's range: H5D__btree_insert adds it + // to the right of that chunk; the right key stays. + if key.scaled == self.nodes[id].keys[idx].scaled { + return Err(bad("duplicate chunk")); + } + new_child = Some((key.clone(), addr)); + } + _ => { + if cmp == Ordering::Greater { + idx = n - 1; + } + let child = usize::try_from(self.nodes[id].children[idx]) + .map_err(|_| bad("bad node index"))?; + let r = self.insert_helper(child, key, addr, depth - 1)?; + if let Some(lt) = r.lt { + self.nodes[id].keys[idx] = lt; + lt_changed = true; + } + if let Some(rt) = r.rt { + self.nodes[id].keys[idx + 1] = rt; + rt_changed = true; + } + if let Some((md, split)) = r.split { + new_child = Some((md, split as u64)); + } + } + } + // Pass the node's changed end keys up, as H5B__insert_helper does. + if lt_changed && idx == 0 { + ret.lt = Some(self.nodes[id].keys[0].clone()); + } + if rt_changed && idx + 1 >= n { + ret.rt = Some(self.nodes[id].keys[idx + 1].clone()); + } + if let Some((md, child)) = new_child { + // A full node splits first; the child goes to the half that + // holds child `idx`. + let (mut target, mut split) = (id, None); + if n == self.two_k { + let s = self.split(id, idx); + let nleft = self.nodes[id].children.len(); + if idx >= nleft { + idx -= nleft; + target = s; + } + split = Some(s); + } + // H5B__insert_child (H5B_INS_RIGHT): the new child after child + // `idx`, its left key after that child's. + let node = &mut self.nodes[target]; + node.keys.insert(idx + 1, md); + node.children.insert(idx + 1, child); + ret.split = split.map(|s| (self.nodes[s].keys[0].clone(), s)); + } + Ok(ret) + } + + /// `H5B__split` of the full node `id`, the insertion going after child + /// `idx`; returns the new right node. + fn split(&mut self, id: usize, idx: usize) -> usize { + let node = &self.nodes[id]; + let ratio = if node.right.is_none() { + SPLIT_RATIOS[2] + } else if node.left.is_none() { + SPLIT_RATIOS[0] + } else { + SPLIT_RATIOS[1] + }; + let mut nleft = (self.two_k as f64 * ratio) as usize; + if idx < nleft && nleft == self.two_k { + nleft -= 1; + } else if idx >= nleft && nleft == 0 { + nleft += 1; + } + let new_id = self.nodes.len(); + let right = Node { + level: node.level, + left: Some(id), + right: node.right, + keys: node.keys[nleft..].to_vec(), + children: node.children[nleft..].to_vec(), + }; + let old_right = node.right; + self.nodes.push(right); + if let Some(r) = old_right { + self.nodes[r].left = Some(new_id); + } + let node = &mut self.nodes[id]; + node.keys.truncate(nleft + 1); + node.children.truncate(nleft); + node.right = Some(new_id); + new_id + } +} + +/// Bytes of one node of a chunk B-tree with `ndims` key dimensions (the +/// dataset's rank plus the element-size one). +fn node_size(two_k: usize, ndims: usize, offset_size: usize) -> usize { + let key = 8 + 8 * ndims; + 8 + 2 * offset_size + (two_k + 1) * key + two_k * offset_size +} + +/// Build the chunk B-tree for `chunks`, given in row-major order of their +/// scaled coordinates, with nodes laid out from `base_address`. `chunk_dims` +/// are the chunk's dimensions (the dataset's rank of them) and `elem_size` +/// the element size, the key's last dimension. Returns the nodes' bytes; the +/// root is at `base_address`. `chunks` must not be empty: an index without +/// chunks has no tree (its address is undefined). +pub(crate) fn build_chunk_btree_v1_at( + chunks: &[ChunkEntry], + chunk_dims: &[u64], + elem_size: u32, + base_address: u64, + offset_size: u8, +) -> Result, FormatError> { + if chunks.is_empty() { + return Err(bad("no chunks")); + } + let rank = chunk_dims.len(); + let mut tree = Tree::new(CHUNK_BTREE_K); + for c in chunks { + if c.scaled.len() != rank { + return Err(bad("chunk rank differs from the dataset's")); + } + let nbytes = u32::try_from(c.nbytes).map_err(|_| { + FormatError::SerializationError(format!( + "a chunk of {} bytes cannot be indexed by a version-1 B-tree \ + (HDF5 1.8 chunks are under 4 GiB)", + c.nbytes + )) + })?; + let mut scaled = c.scaled.clone(); + scaled.push(0); + let key = Key { + nbytes, + mask: c.filter_mask, + scaled, + }; + tree.insert(&key, c.address)?; + } + + let os = usize::from(offset_size); + let ndims = rank + 1; + let nsize = node_size(tree.two_k, ndims, os); + let addr_of = |id: usize| base_address + (id * nsize) as u64; + let mut dims: Vec = chunk_dims.to_vec(); + dims.push(u64::from(elem_size)); + let mut out = vec![0u8; tree.nodes.len() * nsize]; + for (i, node) in tree.nodes.iter().enumerate() { + let d = &mut out[i * nsize..(i + 1) * nsize]; + d[0..4].copy_from_slice(b"TREE"); + d[4] = 1; // node type: raw data chunks + d[5] = node.level; + let n = u16::try_from(node.children.len()).map_err(|_| bad("node too large"))?; + d[6..8].copy_from_slice(&n.to_le_bytes()); + let undef = u64::MAX; + put_addr(&mut d[8..], node.left.map_or(undef, addr_of), os); + put_addr(&mut d[8 + os..], node.right.map_or(undef, addr_of), os); + let mut p = 8 + 2 * os; + for (k, key) in node.keys.iter().enumerate() { + d[p..p + 4].copy_from_slice(&key.nbytes.to_le_bytes()); + d[p + 4..p + 8].copy_from_slice(&key.mask.to_le_bytes()); + for (j, (&s, &dim)) in key.scaled.iter().zip(&dims).enumerate() { + let off = s + .checked_mul(dim) + .ok_or_else(|| FormatError::Overflow("chunk key offset".into()))?; + d[p + 8 + 8 * j..p + 16 + 8 * j].copy_from_slice(&off.to_le_bytes()); + } + p += 8 + 8 * ndims; + if let Some(&child) = node.children.get(k) { + let a = if node.level == 0 { + child + } else { + addr_of(usize::try_from(child).map_err(|_| bad("bad node index"))?) + }; + put_addr(&mut d[p..], a, os); + p += os; + } + } + } + Ok(out) +} + +fn put_addr(d: &mut [u8], v: u64, os: usize) { + d[..os].copy_from_slice(&v.to_le_bytes()[..os]); +} + +#[cfg(test)] +mod tests { + use super::*; + + fn build(n: u64) -> Tree { + let mut t = Tree::new(CHUNK_BTREE_K); + for i in 0..n { + let key = Key { + nbytes: 80, + mask: 0, + scaled: vec![i, 0], + }; + t.insert(&key, 1000 + i).unwrap(); + } + t + } + + /// Leaves in order from the root, with their child counts. + fn leaves(t: &Tree, id: usize, out: &mut Vec) { + let n = &t.nodes[id]; + if n.level == 0 { + out.push(n.children.len()); + } else { + for &c in &n.children { + leaves(t, c as usize, out); + } + } + } + + #[test] + fn sequential_appends_split_as_libhdf5_does() { + // libhdf5 2.0 (h5py, libver=('v108', 'latest')) writes 1000 chunks + // as a root over 17 leaves of 57 chunks and one of 31, with the + // root's right key on the last chunk (9990, 8 for 10-element f8 + // chunks). + let t = build(1000); + assert_eq!(t.nodes[0].level, 1); + let mut l = Vec::new(); + leaves(&t, 0, &mut l); + let mut want = vec![57; 17]; + want.push(31); + assert_eq!(l, want); + assert_eq!(t.nodes[0].keys.last().unwrap().scaled, vec![999, 1]); + // 100 000 chunks: three levels, a root of 31 children. + let t = build(100_000); + assert_eq!(t.nodes[0].level, 2); + assert_eq!(t.nodes[0].children.len(), 31); + } + + #[test] + fn right_key_moves_every_other_append() { + let t = build(5); + assert_eq!(t.nodes[0].keys.last().unwrap().scaled, vec![5, 1]); + let t = build(6); + assert_eq!(t.nodes[0].keys.last().unwrap().scaled, vec![5, 1]); + } + + #[test] + fn keys_and_siblings_are_consistent() { + let t = build(5000); + for (i, n) in t.nodes.iter().enumerate() { + assert!(n.children.len() <= t.two_k); + assert_eq!(n.keys.len(), n.children.len() + 1); + if let Some(r) = n.right { + assert_eq!(t.nodes[r].left, Some(i)); + assert_eq!(n.keys.last(), t.nodes[r].keys.first()); + } + } + } +} diff --git a/crates/clawhdf5-format/src/chunked_write.rs b/crates/clawhdf5-format/src/chunked_write.rs index 78c4161..10f1099 100644 --- a/crates/clawhdf5-format/src/chunked_write.rs +++ b/crates/clawhdf5-format/src/chunked_write.rs @@ -7,6 +7,7 @@ use crate::addr::saturating_usize; #[cfg(not(feature = "std"))] use alloc::{format, vec, vec::Vec}; +use crate::btree_v1_write; use crate::btree_v2_write::{BTreeV2Params, build_btree_v2}; use crate::checksum::jenkins_lookup3; use crate::chunk_cache::{CACHE_LINE_SIZE, align_to_cache_line}; @@ -19,6 +20,7 @@ use crate::filter_pipeline::{ FilterPipeline, }; use crate::filters::compress_chunk_masked; +use crate::libver::LibVer; /// Round a file offset up to the next cache-line boundary. /// /// This ensures chunk data starts at an address that is a multiple of the @@ -866,6 +868,23 @@ pub fn build_chunked_data_from_precompressed( base_address: u64, maxshape: Option<&[u64]>, ) -> Result { + build_chunked_data_from_precompressed_libver(pre, base_address, maxshape, LibVer::Latest) +} + +/// [`build_chunked_data_from_precompressed`] for a file whose low library +/// version bound is `low`: below [`LibVer::V110`] (that is, for HDF5 1.8) +/// every chunked dataset gets a version-3 layout message and a version-1 +/// B-tree chunk index, whatever its maximum shape, as libhdf5 writes it; +/// otherwise the version-4 layout and the index libhdf5 picks for it. +pub fn build_chunked_data_from_precompressed_libver( + pre: &PrecompressedChunks, + base_address: u64, + maxshape: Option<&[u64]>, + low: LibVer, +) -> Result { + if low < LibVer::V110 { + return build_btree_v1_chunked_data(pre, base_address, maxshape); + } let index = ChunkIndexPlan::new(&pre.shape, maxshape, &pre.chunk_dims)?; let offset_size: u8 = 8; let length_size: u8 = 8; @@ -992,6 +1011,92 @@ pub fn build_chunked_data_from_precompressed( }) } +/// Lay out precompressed chunks at `base_address` followed by a version-1 +/// B-tree chunk index, with a version-3 layout message: what libhdf5 writes +/// for a chunked dataset under a low bound of 1.8. +fn build_btree_v1_chunked_data( + pre: &PrecompressedChunks, + base_address: u64, + maxshape: Option<&[u64]>, +) -> Result { + if let Some(ms) = maxshape { + let bad = |what: &str| FormatError::ChunkedReadError(format!("maxshape: {what}")); + if ms.len() != pre.shape.len() { + return Err(bad("rank differs from the shape")); + } + if ms.iter().zip(&pre.shape).any(|(&m, &s)| m < s) { + return Err(bad("smaller than the shape")); + } + } + let offset_size: u8 = 8; + let mut data_buf = Vec::new(); + let mut entries = Vec::with_capacity(pre.chunks.len()); + for (i, (_raw_size, stored, filter_mask)) in pre.chunks.iter().enumerate() { + let aligned_offset = align_to_cache_line(data_buf.len()); + if aligned_offset > data_buf.len() { + data_buf.resize(aligned_offset, 0u8); + } + entries.push(btree_v1_write::ChunkEntry { + scaled: scaled_coords(&pre.shape, &pre.chunk_dims, i), + nbytes: stored.len() as u64, + filter_mask: *filter_mask, + address: base_address + data_buf.len() as u64, + }); + data_buf.extend_from_slice(stored); + } + let element_size = u32::try_from(pre.element_size) + .map_err(|_| FormatError::Overflow("element size".into()))?; + // A dataset with no chunks has no tree: its address is undefined, as + // libhdf5 leaves it until the first chunk is written. + let btree_address = if entries.is_empty() { + u64::MAX + } else { + let aligned_idx = align_to_cache_line(data_buf.len()); + if aligned_idx > data_buf.len() { + data_buf.resize(aligned_idx, 0u8); + } + let addr = base_address + data_buf.len() as u64; + let tree = btree_v1_write::build_chunk_btree_v1_at( + &entries, + &pre.chunk_dims, + element_size, + addr, + offset_size, + )?; + data_buf.extend_from_slice(&tree); + addr + }; + let layout_message = + serialize_v3_chunked(&pre.chunk_dims, btree_address, offset_size, element_size)?; + Ok(ChunkedDataResult { + data_bytes: data_buf, + layout_message, + pipeline_message: pre.pipeline_message.clone(), + }) +} + +/// A version-3 layout message for a chunked dataset: dimensionality (the +/// rank plus one), the B-tree's address, then each chunk dimension and the +/// element size, four bytes each. +fn serialize_v3_chunked( + chunk_dims: &[u64], + btree_address: u64, + offset_size: u8, + element_size: u32, +) -> Result, FormatError> { + let ndims = u8::try_from(chunk_dims.len() + 1) + .map_err(|_| FormatError::Overflow("chunked layout rank".into()))?; + let mut buf = vec![3u8, 2, ndims]; + push_addr(&mut buf, btree_address, offset_size); + for &d in chunk_dims { + let d = + u32::try_from(d).map_err(|_| FormatError::Overflow(format!("chunk dimension {d}")))?; + buf.extend_from_slice(&d.to_le_bytes()); + } + buf.extend_from_slice(&element_size.to_le_bytes()); + Ok(buf) +} + /// Most slots a Fixed Array index may have before we refuse to build it: its /// data block holds one element per chunk of the *maximum* extent, so a huge /// finite maxshape with small chunks would otherwise exhaust memory. diff --git a/crates/clawhdf5-format/src/datatype.rs b/crates/clawhdf5-format/src/datatype.rs index 8e2120d..5e9a678 100644 --- a/crates/clawhdf5-format/src/datatype.rs +++ b/crates/clawhdf5-format/src/datatype.rs @@ -1216,6 +1216,26 @@ impl Datatype { } } + /// The highest datatype message version in this type's encoding, its + /// members' and base types' included (the version decides which HDF5 + /// releases can read it: 1-3 HDF5 1.8, 4 HDF5 1.12, 5 HDF5 2.0). + pub fn max_encoded_version(&self) -> u8 { + let own = self.serialize().first().map_or(0, |b| b >> 4); + let inner = match self { + Datatype::Compound { members, .. } => members + .iter() + .map(|m| m.datatype.max_encoded_version()) + .max() + .unwrap_or(0), + Datatype::Enumeration { base_type, .. } + | Datatype::VariableLength { base_type, .. } + | Datatype::Array { base_type, .. } + | Datatype::Complex { base_type, .. } => base_type.max_encoded_version(), + _ => 0, + }; + own.max(inner) + } + /// Check that this datatype can be written: every part of it has an /// on-disk encoding, and the encoding is one the reader (and libhdf5) /// accepts. [`Self::serialize`] cannot report errors, so the writer calls diff --git a/crates/clawhdf5-format/src/error.rs b/crates/clawhdf5-format/src/error.rs index 449ff13..9dd22b4 100644 --- a/crates/clawhdf5-format/src/error.rs +++ b/crates/clawhdf5-format/src/error.rs @@ -167,6 +167,19 @@ pub enum FormatError { VlDataError(String), /// Serialization error. SerializationError(String), + /// The file's library version bounds + /// ([`FileWriter::libver_bounds`](crate::file_writer::FileWriter::libver_bounds)) + /// do not allow what was asked for: `what` needs the format of HDF5 + /// `needs` or later, and the high bound is `high` (or the low bound is + /// above the high one, with `needs` the low bound). + LibverBound { + /// What cannot be written. + what: String, + /// The oldest release whose format holds it. + needs: crate::libver::LibVer, + /// The file's high bound. + high: crate::libver::LibVer, + }, /// Dataset is missing data. DatasetMissingData, /// Dataset is missing shape. @@ -450,6 +463,13 @@ impl fmt::Display for FormatError { FormatError::SerializationError(msg) => { write!(f, "serialization error: {msg}") } + FormatError::LibverBound { what, needs, high } => { + write!( + f, + "{what} needs the HDF5 {needs} file format, above the high \ + library version bound ({high})" + ) + } FormatError::DatasetMissingData => { write!(f, "dataset is missing data") } diff --git a/crates/clawhdf5-format/src/file_writer.rs b/crates/clawhdf5-format/src/file_writer.rs index c64e1cf..6dcfe47 100644 --- a/crates/clawhdf5-format/src/file_writer.rs +++ b/crates/clawhdf5-format/src/file_writer.rs @@ -5,12 +5,13 @@ use crate::addr::saturating_usize; #[cfg(not(feature = "std"))] -use alloc::{format, vec, vec::Vec}; +use alloc::{format, string::String, vec, vec::Vec}; use crate::attribute::AttributeMessage; use crate::btree_v2_write::{BTreeV2Params, build_btree_v2}; use crate::chunked_write::{ - ChunkOptions, PrecompressedChunks, build_chunked_data_from_precompressed, precompress_chunks, + ChunkOptions, PrecompressedChunks, build_chunked_data_from_precompressed_libver, + precompress_chunks, }; use crate::data_layout::VdsMapping; use crate::dataspace::{Dataspace, DataspaceType}; @@ -31,6 +32,7 @@ pub use crate::type_builders::ProvenanceConfig; pub use crate::type_builders::{AttrValue, CompoundTypeBuilder, EnumTypeBuilder}; use crate::datatype::{CharacterSet, Datatype}; +use crate::libver::LibVer; pub(crate) const OFFSET_SIZE: u8 = 8; pub(crate) const LENGTH_SIZE: u8 = 8; @@ -168,13 +170,15 @@ pub(crate) fn build_dataset_oh( attrs: AttrStorage<'_>, fill_message: &[u8], refcount: u32, + layout_version: u8, ) -> Result, FormatError> { let mut w = ObjectHeaderWriter::new(); w.add_message_with_flags(MessageType::Datatype, dt.serialize(), 0x01); w.add_message(MessageType::Dataspace, ds.serialize(LENGTH_SIZE)); w.add_message_with_flags(MessageType::FillValue, fill_message.to_vec(), 0x01); + // Versions 3 and 4 encode a contiguous layout the same way. let mut dl = Vec::new(); - dl.push(4); // version + dl.push(layout_version); dl.push(1); // class = contiguous // An empty dataset has no storage: its address must be the undefined // address, as libhdf5 writes it. A real address with size 0 trips @@ -198,14 +202,16 @@ pub(crate) fn build_compact_dataset_oh( attrs: AttrStorage<'_>, fill_message: &[u8], refcount: u32, + layout_version: u8, ) -> Result, FormatError> { let mut w = ObjectHeaderWriter::new(); w.add_message_with_flags(MessageType::Datatype, dt.serialize(), 0x01); w.add_message(MessageType::Dataspace, ds.serialize(LENGTH_SIZE)); w.add_message_with_flags(MessageType::FillValue, fill_message.to_vec(), 0x01); - // Compact layout message: version=4, class=0, u16 size, inline data + // Compact layout message: version (3 and 4 are the same here), class=0, + // u16 size, inline data let mut dl = Vec::new(); - dl.push(4); // version + dl.push(layout_version); dl.push(0); // class = compact dl.extend_from_slice(&(data.len() as u16).to_le_bytes()); dl.extend_from_slice(data); @@ -1371,6 +1377,10 @@ pub struct FileWriter { /// file-space strategy (a File Space Info message in the superblock /// extension). page_size: Option, + /// Library version bounds: the low bound picks the format versions + /// written, the high bound limits the features allowed. + low: LibVer, + high: LibVer, } impl Default for FileWriter { @@ -1485,9 +1495,37 @@ impl FileWriter { alignment_threshold: 0, alignment_bytes: 0, page_size: None, + low: LibVer::V110, + high: LibVer::Latest, } } + /// Set the library version bounds, as libhdf5's `H5Pset_libver_bounds` + /// (h5py's `libver=(low, high)`): the oldest HDF5 release whose format + /// the file uses (`low`), and the newest whose features it may use + /// (`high`). See [`crate::libver`] for what each bound changes. + /// + /// The default, `(LibVer::V110, LibVer::Latest)`, is what clawhdf5 has + /// always written: the HDF5 1.10 format (version-3 superblock, version-4 + /// layouts with the 1.10 chunk indexes), readable by HDF5 1.10 and later. + /// + /// `(LibVer::V18, LibVer::V18)` writes a file HDF5 1.8 can read — the + /// low bound libhdf5 2.0 uses by default: a version-2 superblock, + /// version-3 layouts, and a version-1 B-tree for every chunked dataset, + /// resizable ones included; [`Self::finish`] then fails with + /// [`FormatError::LibverBound`] for anything HDF5 1.8 cannot read + /// (virtual datasets, a paged file, the 1.12 reference types, native + /// complex numbers). With a low bound of 1.8 and a later high bound + /// such objects are written in the newer format, as libhdf5 writes them; + /// the rest of the file stays readable by 1.8. + /// + /// A low bound above the high bound makes [`Self::finish`] fail. + pub fn libver_bounds(&mut self, low: LibVer, high: LibVer) -> &mut Self { + self.low = low; + self.high = high; + self + } + /// Set global file alignment: datasets with raw data >= `threshold` bytes /// will have their data aligned to `bytes` boundary. /// @@ -1583,6 +1621,33 @@ impl FileWriter { ))); } + let (low, high) = (self.low, self.high); + let within_bounds = |what: &dyn Fn() -> String, needs: LibVer| { + if needs > high { + Err(FormatError::LibverBound { + what: what(), + needs, + high, + }) + } else { + Ok(()) + } + }; + within_bounds(&|| format!("a low library version bound of {low}"), low)?; + if page_size.is_some() { + within_bounds(&|| "the paged file-space strategy".into(), LibVer::V110)?; + } + // Versions 3 of the layout message and 2 of the superblock are what + // HDF5 1.8 reads; 1.10 added version 4 (with its chunk indexes) and + // version 3. A paged file needs the version-3 superblock whatever + // the low bound (libhdf5 raises it as far as the high bound allows). + let layout_version: u8 = if low < LibVer::V110 { 3 } else { 4 }; + let superblock_version: u8 = if low < LibVer::V110 && page_size.is_none() { + 2 + } else { + 3 + }; + // The group tree, in layout order: groups depth-first from the root, // then every group's datasets in the same order. let tree = writer_tree::build(self.root, self.track_order)?; @@ -1621,9 +1686,20 @@ impl FileWriter { let ds_attrs = all_ds.iter().flat_map(|d| &d.attrs); for a in group_attrs.chain(ds_attrs) { a.datatype.check_encodable()?; + within_bounds( + &|| format!("the datatype of attribute {:?}", a.name), + LibVer::for_datatype_version(a.datatype.max_encoded_version()), + )?; } for d in &all_ds { d.dt.check_encodable()?; + within_bounds( + &|| "a dataset's datatype".into(), + LibVer::for_datatype_version(d.dt.max_encoded_version()), + )?; + if d.virtual_sources.is_some() { + within_bounds(&|| "a virtual dataset".into(), LibVer::V110)?; + } } let is_vds: Vec = all_ds.iter().map(|d| d.virtual_sources.is_some()).collect(); @@ -1749,10 +1825,11 @@ impl FileWriter { elem_size, &d.chunk_options, )?; - let result = build_chunked_data_from_precompressed( + let result = build_chunked_data_from_precompressed_libver( &pre, dummy_cursor, d.maxshape.as_deref(), + low, )?; dummy_cursor += result.data_bytes.len() as u64; let oh = build_chunked_dataset_oh( @@ -1785,6 +1862,7 @@ impl FileWriter { }, &d.fill_message, d.refcount, + layout_version, )?; dummy_blobs.push(DataBlob { data: vec![], @@ -1804,6 +1882,7 @@ impl FileWriter { }, &d.fill_message, d.refcount, + layout_version, )?; dummy_blobs.push(DataBlob { data: vec![], @@ -1904,13 +1983,14 @@ impl FileWriter { let base_address = cursor2 as u64; // Reuse precompressed chunks from Pass 1 — avoids re-compressing // the same data a second time. - let result = build_chunked_data_from_precompressed( + let result = build_chunked_data_from_precompressed_libver( dummy_blobs[i] .precompressed .as_ref() .expect("chunked dataset missing precompressed cache"), base_address, d.maxshape.as_deref(), + low, )?; cursor2 += result.data_bytes.len(); let oh = build_chunked_dataset_oh( @@ -1944,6 +2024,7 @@ impl FileWriter { }, &d.fill_message, d.refcount, + layout_version, )?; ds_blobs2.push(DataBlob { data: vec![], @@ -1973,6 +2054,7 @@ impl FileWriter { }, &d.fill_message, d.refcount, + layout_version, )?; let mut data = vec![0u8; padding]; data.extend_from_slice(&d.raw); @@ -1997,7 +2079,7 @@ impl FileWriter { let mut buf = Vec::with_capacity(cursor2); let sb = Superblock { - version: 3, + version: superblock_version, offset_size: OFFSET_SIZE, length_size: LENGTH_SIZE, base_address: 0, @@ -2783,4 +2865,150 @@ mod tests { assert_eq!(sb.version, 3); assert_eq!(sb.page_size, None); } + + fn layout_of(bytes: &[u8], name: &str) -> Vec { + let sb = Superblock::parse(bytes, 0).unwrap(); + let addr = resolve_path_any(bytes, &sb, name).unwrap(); + let hdr = ObjectHeader::parse(bytes, addr as usize, 8, 8).unwrap(); + hdr.messages + .iter() + .find(|m| m.msg_type == MessageType::DataLayout) + .unwrap() + .data + .clone() + } + + #[test] + fn libver_v18_writes_the_1_8_format() { + let mut fw = FileWriter::new(); + fw.libver_bounds(LibVer::V18, LibVer::V18); + fw.create_dataset("contig").with_f64_data(&[1.0, 2.0]); + fw.create_dataset("compact").with_f64_data(&[3.0]).compact(); + fw.create_dataset("grow") + .with_f64_data(&[1.0, 2.0, 3.0]) + .with_maxshape(&[u64::MAX]) + .with_chunks(&[2]); + fw.create_dataset("none") + .with_f64_data(&[]) + .with_maxshape(&[u64::MAX]) + .with_chunks(&[2]); + let bytes = fw.finish().unwrap(); + assert_eq!(Superblock::parse(&bytes, 0).unwrap().version, 2); + assert_eq!(layout_of(&bytes, "contig")[..2], [3, 1]); + assert_eq!(layout_of(&bytes, "compact")[..2], [3, 0]); + let grow = layout_of(&bytes, "grow"); + // Version 3, chunked, 2 dimensions (the element size is the last), + // B-tree address, chunk dims 2 and 8. + assert_eq!(grow[..3], [3, 2, 2]); + assert_eq!(grow[11..], [2, 0, 0, 0, 8, 0, 0, 0]); + let root = u64::from_le_bytes(grow[3..11].try_into().unwrap()) as usize; + assert_eq!(&bytes[root..root + 5], b"TREE\x01"); + // No chunks, no tree. + assert_eq!(layout_of(&bytes, "none")[3..11], [0xff; 8]); + assert_eq!(read_dataset_f64(&bytes, "grow"), vec![1.0, 2.0, 3.0]); + assert_eq!(read_dataset_f64(&bytes, "contig"), vec![1.0, 2.0]); + assert_eq!(read_dataset_f64(&bytes, "compact"), vec![3.0]); + } + + #[test] + fn default_libver_bounds_keep_the_1_10_format() { + let mut fw = FileWriter::new(); + fw.create_dataset("contig").with_f64_data(&[1.0, 2.0]); + fw.create_dataset("grow") + .with_f64_data(&[1.0, 2.0, 3.0]) + .with_maxshape(&[u64::MAX]) + .with_chunks(&[2]); + let default = fw.finish().unwrap(); + let mut fw = FileWriter::new(); + fw.libver_bounds(LibVer::V110, LibVer::Latest); + fw.create_dataset("contig").with_f64_data(&[1.0, 2.0]); + fw.create_dataset("grow") + .with_f64_data(&[1.0, 2.0, 3.0]) + .with_maxshape(&[u64::MAX]) + .with_chunks(&[2]); + assert_eq!(fw.finish().unwrap(), default); + assert_eq!(layout_of(&default, "contig")[0], 4); + assert_eq!(layout_of(&default, "grow")[..2], [4, 2]); + } + + #[test] + fn libver_high_bound_refuses_newer_features() { + let bound = |r: Result, FormatError>, needs: LibVer| match r { + Err(FormatError::LibverBound { needs: n, high, .. }) => { + assert_eq!((n, high), (needs, LibVer::V18)); + } + other => panic!("expected a bound error, got {other:?}"), + }; + let mut fw = FileWriter::new(); + fw.libver_bounds(LibVer::V18, LibVer::V18); + fw.create_dataset("z") + .with_native_complex_f64_data(&[[1.0, 2.0]]); + bound(fw.finish(), LibVer::V200); + + let mut fw = FileWriter::new(); + fw.libver_bounds(LibVer::V18, LibVer::V18); + fw.create_dataset("x").with_f64_data(&[1.0]).set_attr( + "z", + AttrValue::Raw { + datatype: crate::type_builders::make_native_complex_f64_type(), + shape: vec![], + data: vec![0; 16], + }, + ); + bound(fw.finish(), LibVer::V200); + + let mut fw = FileWriter::new(); + fw.libver_bounds(LibVer::V18, LibVer::V18); + fw.create_dataset("r").with_compound_data( + Datatype::Reference { + size: 16, + ref_type: crate::datatype::ReferenceType::Object2, + }, + vec![0; 16], + 1, + ); + bound(fw.finish(), LibVer::V112); + + let mut fw = FileWriter::new(); + fw.libver_bounds(LibVer::V18, LibVer::V18); + fw.create_dataset("src").with_f64_data(&[1.0, 2.0]); + fw.create_dataset("vds") + .with_shape(&[2]) + .with_f64_data(&[]) + .with_virtual_sources(vec![VdsMapping { + source_file: ".".into(), + source_dataset: "src".into(), + source_selection: sel_all(), + virtual_selection: sel_hyper_1d(0, 2), + }]); + bound(fw.finish(), LibVer::V110); + + let mut fw = FileWriter::new(); + fw.libver_bounds(LibVer::V18, LibVer::V18) + .with_page_size(4096); + bound(fw.finish(), LibVer::V110); + + let mut fw = FileWriter::new(); + fw.libver_bounds(LibVer::V110, LibVer::V18); + bound(fw.finish(), LibVer::V110); + } + + #[test] + fn libver_low_v18_high_latest_allows_newer_objects() { + // As libhdf5 does: the object that needs a newer format gets it, + // the rest of the file keeps the 1.8 format. + let mut fw = FileWriter::new(); + fw.libver_bounds(LibVer::V18, LibVer::Latest); + fw.create_dataset("z") + .with_native_complex_f64_data(&[[1.0, 2.0]]); + let bytes = fw.finish().unwrap(); + assert_eq!(Superblock::parse(&bytes, 0).unwrap().version, 2); + let mut fw = FileWriter::new(); + fw.libver_bounds(LibVer::V18, LibVer::Latest) + .with_page_size(4096); + fw.create_dataset("x").with_f64_data(&[1.0]); + let bytes = fw.finish().unwrap(); + assert_eq!(Superblock::parse(&bytes, 0).unwrap().version, 3); + assert_eq!(layout_of(&bytes, "x")[0], 3); + } } diff --git a/crates/clawhdf5-format/src/lib.rs b/crates/clawhdf5-format/src/lib.rs index 9ce3ff9..32c3834 100644 --- a/crates/clawhdf5-format/src/lib.rs +++ b/crates/clawhdf5-format/src/lib.rs @@ -61,6 +61,7 @@ pub mod addr; pub mod attribute; pub mod attribute_info; pub mod btree_v1; +mod btree_v1_write; pub mod btree_v2; mod btree_v2_write; mod bulk_alloc; @@ -107,6 +108,7 @@ pub mod group_v1; pub mod group_v2; #[cfg(feature = "parallel")] pub mod lane_partition; +pub mod libver; pub mod link_info; pub mod link_message; pub mod local_heap; diff --git a/crates/clawhdf5-format/src/libver.rs b/crates/clawhdf5-format/src/libver.rs new file mode 100644 index 0000000..42fe4fa --- /dev/null +++ b/crates/clawhdf5-format/src/libver.rs @@ -0,0 +1,93 @@ +//! Library version bounds for writing: which HDF5 releases can read a file. +//! +//! libhdf5 picks the version of every object it writes from the file's +//! *low* bound (`H5Pset_libver_bounds`; h5py's `libver=`): the oldest +//! format version that holds the object, but never older than the one the +//! low bound names. The *high* bound caps it: a feature that needs a newer +//! format than the high bound is an error. [`LibVer`] names the same +//! releases, and [`crate::file_writer::FileWriter::libver_bounds`] sets them. +//! +//! What the low bound changes in what clawhdf5 writes: +//! +//! | | low [`LibVer::V18`] | low [`LibVer::V110`] or later (the default) | +//! |---|---|---| +//! | superblock | version 2 | version 3 | +//! | data layout message | version 3 | version 4 | +//! | chunk index | version-1 B-tree (every chunked dataset) | single chunk, Fixed Array, Extensible Array or version-2 B-tree, as libhdf5 picks | +//! +//! Everything else (version-2 object headers, link and group-info messages, +//! dense storage in fractal heaps with version-2 B-trees, filter pipeline +//! version 2, fill value version 3, datatype versions up to 3) is the same +//! and already readable by HDF5 1.8. +//! +//! What the high bound refuses: anything that needs 1.10 (virtual datasets, +//! the paged file-space strategy) above [`LibVer::V18`], the 1.12 reference +//! types (datatype version 4) above [`LibVer::V110`], and HDF5 2.0's native +//! complex numbers (datatype version 5) above [`LibVer::V114`]. +//! `libver_bounds(LibVer::V18, LibVer::V18)` therefore writes a file HDF5 +//! 1.8 can read, or fails. + +use core::fmt; + +/// An HDF5 library release, as a bound on the file format versions a writer +/// may use (libhdf5's `H5F_libver_t`). Ordered oldest first. +/// +/// There is no `Earliest`: clawhdf5 cannot write the pre-1.8 format +/// (symbol-table groups, version-1 object headers). +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash)] +#[non_exhaustive] +pub enum LibVer { + /// HDF5 1.8 (`H5F_LIBVER_V18`, h5py `'v108'`). + V18, + /// HDF5 1.10 (`H5F_LIBVER_V110`, h5py `'v110'`). + V110, + /// HDF5 1.12 (`H5F_LIBVER_V112`, h5py `'v112'`). + V112, + /// HDF5 1.14 (`H5F_LIBVER_V114`, h5py `'v114'`). + V114, + /// HDF5 2.0 (`H5F_LIBVER_V200`). + V200, + /// The newest format this build of clawhdf5 writes + /// (`H5F_LIBVER_LATEST`, h5py `'latest'`). + Latest, +} + +impl LibVer { + /// The release a datatype message of this version first appeared in: + /// versions 1-3 are readable by HDF5 1.8, 4 needs 1.12, 5 needs 2.0. + pub(crate) fn for_datatype_version(version: u8) -> Self { + match version { + 0..=3 => LibVer::V18, + 4 => LibVer::V112, + _ => LibVer::V200, + } + } +} + +impl fmt::Display for LibVer { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.write_str(match self { + LibVer::V18 => "1.8", + LibVer::V110 => "1.10", + LibVer::V112 => "1.12", + LibVer::V114 => "1.14", + LibVer::V200 => "2.0", + LibVer::Latest => "latest", + }) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn ordered_oldest_first() { + assert!(LibVer::V18 < LibVer::V110); + assert!(LibVer::V114 < LibVer::V200); + assert!(LibVer::V200 < LibVer::Latest); + assert_eq!(LibVer::for_datatype_version(3), LibVer::V18); + assert_eq!(LibVer::for_datatype_version(4), LibVer::V112); + assert_eq!(LibVer::for_datatype_version(5), LibVer::V200); + } +} diff --git a/crates/clawhdf5-tools/tests/libver_v18.rs b/crates/clawhdf5-tools/tests/libver_v18.rs new file mode 100644 index 0000000..0d27a39 --- /dev/null +++ b/crates/clawhdf5-tools/tests/libver_v18.rs @@ -0,0 +1,736 @@ +//! Files written with `FileBuilder::libver_bounds(LibVer::V18, LibVer::V18)` +//! must be readable by HDF5 1.8. Every writer feature is written under that +//! bound and read back by HDF5 1.8.23's h5dump (values dumped in binary and +//! compared), by h5py (libhdf5 2.x), by h5dump 1.14, by clawhdf5 and by +//! `h5rs check --data`; then `FileEditor` grows, appends to and annotates +//! the file (splitting version-1 B-tree nodes) and every reader checks it +//! again. The version-1 B-trees we write are compared node by node with +//! the ones libhdf5 writes for the same data under h5py's +//! `libver=('v108', 'latest')`. +//! +//! HDF5 1.8 is found through `CLAWHDF5_H5DUMP18` (the path to its h5dump) +//! or at `~/.cache/hdf5-1.8.23/bin/h5dump`, where +//! `scripts/build-hdf5-1.8.sh` builds it; without it the 1.8 checks are +//! skipped (CI has no HDF5 1.8), even with `CLAWHDF5_REQUIRE_INTEROP=1`. +//! h5py/numpy (`CLAWHDF5_PYTHON`) and h5dump are needed otherwise; they +//! skip when missing unless `CLAWHDF5_REQUIRE_INTEROP=1`. + +use std::path::{Path, PathBuf}; +use std::process::{Command, Output}; + +use clawhdf5::{AttrValue, File, FileBuilder, FileEditor, LibVer, Selection}; +use clawhdf5_format::datatype::{CharacterSet, Datatype, StringPadding}; +use clawhdf5_format::file_writer::{CompoundTypeBuilder, EnumTypeBuilder}; +use clawhdf5_format::type_builders::make_i32_type; + +fn python() -> String { + std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string()) +} + +fn interop_required() -> bool { + std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1") +} + +fn available(cmd: &str, args: &[&str]) -> bool { + Command::new(cmd) + .args(args) + .output() + .map(|o| o.status.success()) + .unwrap_or(false) +} + +fn tools_ok() -> bool { + let ok = + available(&python(), &["-c", "import h5py, numpy"]) && available("h5dump", &["--version"]); + if !ok { + assert!( + !interop_required(), + "CLAWHDF5_REQUIRE_INTEROP=1 but h5py/numpy or h5dump is not available" + ); + eprintln!("SKIP: h5py/numpy or h5dump not available"); + } + ok +} + +/// HDF5 1.8's h5dump, when there is one. +fn h5dump18() -> Option { + let p = match std::env::var_os("CLAWHDF5_H5DUMP18") { + Some(p) => PathBuf::from(p), + None => PathBuf::from(std::env::var_os("HOME")?).join(".cache/hdf5-1.8.23/bin/h5dump"), + }; + let o = Command::new(&p).arg("--version").output().ok()?; + let v = String::from_utf8_lossy(&o.stdout).to_string(); + if !v.contains("1.8.") { + eprintln!("SKIP 1.8 checks: {} is not HDF5 1.8 ({v})", p.display()); + return None; + } + Some(p) +} + +fn py(script: &str) -> String { + let o = Command::new(python()) + .args(["-c", script]) + .output() + .expect("run python"); + assert!( + o.status.success(), + "python failed:\n{script}\nSTDOUT: {}\nSTDERR: {}", + String::from_utf8_lossy(&o.stdout), + String::from_utf8_lossy(&o.stderr) + ); + String::from_utf8_lossy(&o.stdout).trim().to_string() +} + +fn text(o: &Output) -> String { + format!( + "{}{}", + String::from_utf8_lossy(&o.stdout), + String::from_utf8_lossy(&o.stderr) + ) +} + +fn tmpdir() -> tempfile::TempDir { + tempfile::TempDir::new_in(env!("CARGO_TARGET_TMPDIR")).unwrap() +} + +/// The hyperslab of `count` elements from `start`. +fn block(start: &[u64], count: &[u64]) -> Selection { + Selection::Hyperslab { + start: start.to_vec(), + stride: vec![1; start.len()], + count: count.to_vec(), + block: vec![1; start.len()], + } +} + +fn le(v: &[T], f: impl Fn(T) -> [u8; N]) -> Vec { + v.iter().flat_map(|&x| f(x)).collect() +} + +/// The datasets of the test file and the bytes each holds (little-endian, +/// row-major, as h5py's `tobytes()` and h5dump's `-b LE` give them). +struct Expect { + datasets: Vec<(String, Vec)>, +} + +impl Expect { + fn set(&mut self, name: &str, bytes: Vec) { + match self.datasets.iter_mut().find(|(n, _)| n == name) { + Some(e) => e.1 = bytes, + None => self.datasets.push((name.to_string(), bytes)), + } + } +} + +const MANY: usize = 100_000; + +/// Write every feature under the 1.8 bound. +fn write_file(path: &Path) -> Expect { + let mut e = Expect { + datasets: Vec::new(), + }; + let mut b = FileBuilder::new(); + b.libver_bounds(LibVer::V18, LibVer::V18); + + // Contiguous, with dense attributes (more than 8). + let v: Vec = (0..1000).map(|i| i as f64 * 0.5).collect(); + let d = b.create_dataset("contig").with_f64_data(&v); + for i in 0..12 { + d.set_attr(&format!("a{i:02}"), AttrValue::I64(i)); + } + e.set("/contig", le(&v, f64::to_le_bytes)); + b.create_dataset("empty").with_f64_data(&[]); + e.set("/empty", vec![]); + b.create_dataset("scalar") + .with_f64_data(&[2.5]) + .with_shape(&[]); + e.set("/scalar", 2.5f64.to_le_bytes().to_vec()); + let v: Vec = (0..16).map(|i| i * 3 - 7).collect(); + b.create_dataset("compact").with_i32_data(&v).compact(); + e.set("/compact", le(&v, i32::to_le_bytes)); + let v: Vec = (0..20).map(|i| i as f32 / 3.0).collect(); + b.create_dataset("f32").with_f32_data(&v); + e.set("/f32", le(&v, f32::to_le_bytes)); + let v: Vec = vec![0.5, -2.0, 1024.0, 0.0]; + b.create_dataset("f16").with_f16_data(&v); + e.set( + "/f16", + le(&v, |x| { + clawhdf5_format::float16::f32_to_f16_bits(x).to_le_bytes() + }), + ); + + // Chunked with every built-in filter HDF5 1.8 has, and edge chunks. + let v: Vec = (0..60 * 70).map(|i| (i % 97) as f32 * 1.25).collect(); + b.create_dataset("chunked") + .with_f32_data(&v) + .with_shape(&[60, 70]) + .with_chunks(&[16, 16]) + .with_deflate(6) + .with_shuffle() + .with_fletcher32(); + e.set("/chunked", le(&v, f32::to_le_bytes)); + // What the 1.10 indexes would be: Extensible Array, version-2 B-tree, + // Fixed Array, single chunk. All become version-1 B-trees. + let v: Vec = (0..25).collect(); + b.create_dataset("resizable") + .with_i32_data(&v) + .with_maxshape(&[u64::MAX]) + .with_chunks(&[4]); + e.set("/resizable", le(&v, i32::to_le_bytes)); + let v: Vec = (0..30).map(|i| i * 1_000_000_007).collect(); + b.create_dataset("resizable2") + .with_i64_data(&v) + .with_shape(&[5, 6]) + .with_maxshape(&[u64::MAX, u64::MAX]) + .with_chunks(&[2, 4]); + e.set("/resizable2", le(&v, i64::to_le_bytes)); + let v: Vec = (0..10).map(|i| -i).collect(); + b.create_dataset("fixedmax") + .with_i32_data(&v) + .with_maxshape(&[100]) + .with_chunks(&[3]) + .with_deflate(1); + e.set("/fixedmax", le(&v, i32::to_le_bytes)); + let v: Vec = (0..8).map(|i| i * i).collect(); + b.create_dataset("single") + .with_i32_data(&v) + .with_chunks(&[8]); + e.set("/single", le(&v, i32::to_le_bytes)); + // Enough chunks for a three-level tree, and a 2-D two-level one. + let v: Vec = (0..MANY as i32).map(|i| i ^ 0x5a5a).collect(); + b.create_dataset("many") + .with_i32_data(&v) + .with_maxshape(&[u64::MAX]) + .with_chunks(&[1]); + e.set("/many", le(&v, i32::to_le_bytes)); + let v: Vec = (0..100 * 100).map(|i| i as f64).collect(); + b.create_dataset("grid") + .with_f64_data(&v) + .with_shape(&[100, 100]) + .with_chunks(&[1, 1]); + e.set("/grid", le(&v, f64::to_le_bytes)); + b.create_dataset("empty_chunked") + .with_f64_data(&[]) + .with_maxshape(&[u64::MAX]) + .with_chunks(&[10]); + e.set("/empty_chunked", vec![]); + let v: Vec = (0..12).collect(); + b.create_dataset("filled") + .with_i32_data(&v) + .with_maxshape(&[u64::MAX]) + .with_chunks(&[5]) + .with_fill_value(&(-9i32).to_le_bytes()); + e.set("/filled", le(&v, i32::to_le_bytes)); + + // Datatypes. + let raw = b"abcdehello\0\0\0\0\0".to_vec(); + b.create_dataset("strings").with_compound_data( + Datatype::String { + size: 5, + padding: StringPadding::NullPad, + charset: CharacterSet::Ascii, + }, + raw.clone(), + 3, + ); + e.set("/strings", raw); + let ct = CompoundTypeBuilder::new() + .i32_field("a") + .f64_field("b") + .build(); + let mut raw = Vec::new(); + for i in 0..5i32 { + raw.extend_from_slice(&i.to_le_bytes()); + raw.extend_from_slice(&(f64::from(i) * 1.5).to_le_bytes()); + } + b.create_dataset("compound") + .with_compound_data(ct, raw.clone(), 5); + e.set("/compound", raw); + let et = EnumTypeBuilder::i32_based() + .value("RED", 0) + .value("GREEN", 1) + .value("BLUE", 7) + .build(); + let v = [0, 7, 1, 1, 0]; + b.create_dataset("enum").with_enum_i32_data(et, &v); + e.set("/enum", le(&v, i32::to_le_bytes)); + let v: Vec = (0..24).collect(); + b.create_dataset("array").with_array_data( + make_i32_type(), + &[2, 3], + le(&v, i32::to_le_bytes), + 4, + ); + e.set("/array", le(&v, i32::to_le_bytes)); + let v: Vec = vec![-1, 0, i64::MAX]; + b.create_dataset("i64").with_i64_data(&v); + e.set("/i64", le(&v, i64::to_le_bytes)); + + // Groups: compact with links of every kind, dense (links and + // attributes), and tracking creation order. + let mut g = b.create_group("g_compact"); + g.create_dataset("x").with_i32_data(&[1, 2, 3]); + e.set("/g_compact/x", le(&[1i32, 2, 3], i32::to_le_bytes)); + g.add_soft_link("soft", "/contig"); + g.add_external_link("ext", "other.h5", "/data"); + g.set_attr("title", AttrValue::String("compact group".into())); + b.add_group(g.finish()); + b.add_hard_link("hard", "/g_compact/x"); + let mut g = b.create_group("g_dense"); + for i in 0..20 { + let v = [i, i + 1]; + g.create_dataset(&format!("d{i:02}")).with_i32_data(&v); + e.set(&format!("/g_dense/d{i:02}"), le(&v, i32::to_le_bytes)); + } + for i in 0..12 { + g.set_attr(&format!("attr{i:02}"), AttrValue::F64(f64::from(i) / 4.0)); + } + b.add_group(g.finish()); + let mut g = b.create_group("g_order"); + g.track_order(true); + for i in 0..10 { + let n = format!("z{}", 9 - i); + g.create_dataset(&n).with_i32_data(&[i]); + e.set(&format!("/g_order/{n}"), i.to_le_bytes().to_vec()); + } + for i in 0..10 { + g.set_attr(&format!("b{}", 9 - i), AttrValue::I64(i)); + } + b.add_group(g.finish()); + b.create_dataset("deep/er/path").with_i32_data(&[42]); + e.set("/deep/er/path", 42i32.to_le_bytes().to_vec()); + + b.set_attr("version", AttrValue::F64(1.8)); + b.set_attr("ints", AttrValue::I64Array(vec![1, -2, 3])); + b.set_attr("text", AttrValue::String("readable by 1.8".into())); + b.set_attr( + "texts", + AttrValue::StringArray(vec!["a".into(), "bb".into(), "ccc".into()]), + ); + b.set_attr("big", AttrValue::U64(u64::MAX)); + b.write(path).unwrap(); + e +} + +/// Structural facts of a file: superblock and layout message versions. +fn check_versions(path: &Path) { + use clawhdf5_format::message_type::MessageType; + use clawhdf5_format::object_header::ObjectHeader; + let bytes = std::fs::read(path).unwrap(); + assert_eq!(bytes[8], 2, "superblock version"); + let f = File::open(path).unwrap(); + let sb = f.superblock().clone(); + for name in [ + "contig", + "compact", + "chunked", + "resizable", + "resizable2", + "many", + ] { + let addr = clawhdf5_format::group_v2::resolve_path_any(&bytes, &sb, name).unwrap(); + let oh = ObjectHeader::parse(&bytes, addr as usize, 8, 8).unwrap(); + let layout = oh + .messages + .iter() + .find(|m| m.msg_type == MessageType::DataLayout) + .unwrap(); + assert_eq!(layout.data[0], 3, "layout version of {name}"); + } +} + +/// clawhdf5 reads every dataset's bytes back. +fn check_ours(path: &Path, e: &Expect) { + let f = File::open(path).unwrap(); + for (name, want) in &e.datasets { + let got = f + .dataset(name) + .unwrap() + .read_selection(&Selection::All) + .unwrap(); + assert!(got == *want, "our read of {name} differs"); + } + let attrs = f.dataset("contig").unwrap().attrs().unwrap(); + assert!((12..=13).contains(&attrs.len()), "{attrs:?}"); +} + +/// h5py (libhdf5 2.x) reads every dataset's bytes back. +fn check_h5py(path: &Path, e: &Expect, dir: &Path) { + let mut script = format!( + "import h5py, numpy as np\nf = h5py.File({p:?}, 'r')\n", + p = path.to_str().unwrap() + ); + for (i, (name, want)) in e.datasets.iter().enumerate() { + let exp = dir.join(format!("expect{i}.bin")); + std::fs::write(&exp, want).unwrap(); + script.push_str(&format!( + "a = np.ascontiguousarray(f[{name:?}][()]).tobytes()\n\ + assert a == open({x:?}, 'rb').read(), {name:?}\n", + x = exp.to_str().unwrap() + )); + } + script.push_str( + "assert f.attrs['text'] in (b'readable by 1.8', 'readable by 1.8')\n\ + assert list(f.attrs['ints']) == [1, -2, 3]\n\ + assert len(f['g_dense'].attrs) == 12 and len(f['g_dense']) == 20\n\ + assert list(f['g_order']) == ['z9', 'z8', 'z7', 'z6', 'z5', 'z4', 'z3', 'z2', 'z1', 'z0']\n\ + assert f['g_compact/soft'].shape == (1000,)\n\ + assert f['resizable'].maxshape == (None,)\n\ + assert f['filled'].fillvalue == -9\n\ + print('ok')\n", + ); + assert_eq!(py(&script), "ok"); +} + +/// h5dump (1.14) and `h5rs check --data` accept the file. +fn check_tools(path: &Path) { + let p = path.to_str().unwrap(); + let o = Command::new(env!("CARGO_BIN_EXE_h5rs")) + .args(["check", "--data", "-q", p]) + .output() + .unwrap(); + assert!(o.status.success(), "h5rs check --data {p}:\n{}", text(&o)); + let o = Command::new("h5dump") + .args(["-o", "/dev/null", p]) + .output() + .unwrap(); + assert!( + o.status.success() && o.stderr.is_empty(), + "h5dump {p}:\n{}", + text(&o) + ); +} + +/// HDF5 1.8's h5dump reads the whole file exactly as h5dump 1.14 does, and +/// dumps each numeric dataset's values as the bytes we wrote. +fn check_18(h5dump: &Path, path: &Path, e: &Expect, dir: &Path) { + let p = path.to_str().unwrap(); + let o = Command::new(h5dump).arg(p).output().unwrap(); + assert!( + o.status.success() && o.stderr.is_empty(), + "h5dump 1.8 {p}:\n{}", + text(&o) + ); + let out18 = String::from_utf8_lossy(&o.stdout).to_string(); + assert!(out18.contains("EXTERNAL_LINK \"ext\""), "{out18}"); + let o = Command::new("h5dump").arg(p).output().unwrap(); + assert!(o.status.success(), "h5dump {p}:\n{}", text(&o)); + // HDF5 1.8 has no name for IEEE half floats. + let out = String::from_utf8_lossy(&o.stdout).replace( + "H5T_IEEE_F16LE", + "16-bit little-endian floating-point 16-bit precision", + ); + if let Some((n, (a, b))) = out18 + .lines() + .zip(out.lines()) + .enumerate() + .find(|(_, (a, b))| a != b) + { + panic!("h5dump 1.8 and h5dump differ at line {}:\n{a}\n{b}", n + 1); + } + assert_eq!(out18.lines().count(), out.lines().count()); + // `-b` writes nothing for compounds, enums, arrays, strings and half + // floats (HDF5 1.8 + // has no native half float), for h5py's files too: those are covered + // by the text above. + let skip = ["/f16", "/strings", "/compound", "/enum", "/array"]; + for (i, (name, want)) in e.datasets.iter().enumerate() { + if want.is_empty() || skip.contains(&name.as_str()) { + continue; + } + let bin = dir.join(format!("dump18_{i}.bin")); + let o = Command::new(h5dump) + .args(["-d", name, "-b", "LE", "-o"]) + .arg(&bin) + .arg(p) + .output() + .unwrap(); + assert!(o.status.success(), "h5dump 1.8 -d {name}:\n{}", text(&o)); + let got = std::fs::read(&bin).unwrap(); + assert!(got == *want, "HDF5 1.8 read of {name} differs"); + } +} + +fn check_all(path: &Path, e: &Expect, dir: &Path, h5dump: Option<&Path>) { + check_ours(path, e); + check_h5py(path, e, dir); + check_tools(path); + if let Some(h) = h5dump { + check_18(h, path, e, dir); + } +} + +#[test] +fn every_feature_reads_in_hdf5_1_8() { + if !tools_ok() { + return; + } + let h18 = h5dump18(); + let dir = tmpdir(); + let path = dir.path().join("v18.h5"); + let mut e = write_file(&path); + check_versions(&path); + check_all(&path, &e, dir.path(), h18.as_deref()); + + // FileEditor: grow the unlimited datasets (splitting B-tree nodes), + // overwrite values in filtered and unfiltered chunks, set attributes + // (compact and dense). + let mut ed = FileEditor::open(&path).unwrap(); + let mut res: Vec = (0..25).collect(); + for round in 0..300usize { + let n = res.len() as u64; + let add = 1 + (round % 9) as u64; + ed.resize("resizable", &[n + add]).unwrap(); + let vals: Vec = (0..add).map(|k| (n + k) as i32 * 7 - 3).collect(); + ed.write_values("resizable", &block(&[n], &[add]), &vals) + .unwrap(); + res.extend(&vals); + } + e.set("/resizable", le(&res, i32::to_le_bytes)); + let mut many: Vec = (0..MANY as i32).map(|i| i ^ 0x5a5a).collect(); + ed.resize("many", &[MANY as u64 + 500]).unwrap(); + let vals: Vec = (0..500).collect(); + ed.write_values("many", &block(&[MANY as u64], &[500]), &vals) + .unwrap(); + many.extend(&vals); + many[12_345] = -1; + ed.write_values("many", &block(&[12_345], &[1]), &[-1i32]) + .unwrap(); + e.set("/many", le(&many, i32::to_le_bytes)); + let mut chunked: Vec = (0..60 * 70).map(|i| (i % 97) as f32 * 1.25).collect(); + let row: Vec = (0..70).map(|i| -(i as f32)).collect(); + ed.write_values("chunked", &block(&[31, 0], &[1, 70]), &row) + .unwrap(); + chunked[31 * 70..32 * 70].copy_from_slice(&row); + e.set("/chunked", le(&chunked, f32::to_le_bytes)); + ed.resize("resizable2", &[9, 6]).unwrap(); + let mut r2: Vec = (0..30).map(|i| i * 1_000_000_007).collect(); + r2.extend(std::iter::repeat_n(0, 24)); + e.set("/resizable2", le(&r2, i64::to_le_bytes)); + ed.set_attr("contig", "added", &AttrValue::F64(3.25)) + .unwrap(); + ed.set_attr("g_dense", "attr03", &AttrValue::F64(-1.0)) + .unwrap(); + ed.set_attr("/", "text", &AttrValue::String("edited".into())) + .unwrap(); + drop(ed); + + check_ours(&path, &e); + let f = File::open(&path).unwrap(); + assert!(matches!( + f.dataset("contig").unwrap().attr("added").unwrap(), + Some(AttrValue::F64(v)) if v == 3.25 + )); + drop(f); + // Everything but the root's "text" attribute is as check_h5py expects. + py(&format!( + "import h5py\nf = h5py.File({p:?}, 'r')\n\ + assert f.attrs['text'] in (b'edited', 'edited')\n\ + assert f['contig'].attrs['added'] == 3.25\n\ + assert f['g_dense'].attrs['attr03'] == -1.0\n", + p = path.to_str().unwrap() + )); + let o = Command::new(env!("CARGO_BIN_EXE_h5rs")) + .args(["check", "--data", "-q", path.to_str().unwrap()]) + .output() + .unwrap(); + assert!(o.status.success(), "h5rs check after edits:\n{}", text(&o)); + if let Some(h) = h18.as_deref() { + check_18(h, &path, &e, dir.path()); + } + + // h5py (libhdf5 2.x) goes on appending to the 1.8-format file, and 1.8 + // still reads it. + py(&format!( + "import h5py, numpy as np\n\ + with h5py.File({p:?}, 'r+') as f:\n\ + \x20 d = f['resizable']\n\ + \x20 n = d.shape[0]\n\ + \x20 d.resize((n + 100,))\n\ + \x20 d[n:] = np.arange(100, dtype=')>, + children: Vec, +} + +fn read_tree(bytes: &[u8], addr: u64, ndims: usize, sizes: bool) -> TreeNode { + let a = addr as usize; + assert_eq!(&bytes[a..a + 4], b"TREE"); + let level = bytes[a + 5]; + let n = u16::from_le_bytes([bytes[a + 6], bytes[a + 7]]) as usize; + let mut p = a + 24; + let mut keys = Vec::new(); + let mut kids = Vec::new(); + for i in 0..=n { + let size = u32::from_le_bytes(bytes[p..p + 4].try_into().unwrap()); + let offs = (0..ndims) + .map(|d| u64::from_le_bytes(bytes[p + 8 + 8 * d..p + 16 + 8 * d].try_into().unwrap())) + .collect(); + keys.push((if sizes { size } else { 0 }, offs)); + p += 8 + 8 * ndims; + if i < n { + kids.push(u64::from_le_bytes(bytes[p..p + 8].try_into().unwrap())); + p += 8; + } + } + let children = if level > 0 { + kids.iter() + .map(|&c| read_tree(bytes, c, ndims, sizes)) + .collect() + } else { + Vec::new() + }; + TreeNode { + level, + keys, + children, + } +} + +/// Where two trees first differ (libhdf5's first). +fn first_difference(a: &TreeNode, b: &TreeNode, at: &str) -> Option { + if a.level != b.level || a.keys.len() != b.keys.len() { + return Some(format!( + "{at}: level {} with {} keys vs level {} with {} keys", + a.level, + a.keys.len(), + b.level, + b.keys.len() + )); + } + if let Some(i) = (0..a.keys.len()).find(|&i| a.keys[i] != b.keys[i]) { + return Some(format!("{at} key {i}: {:?} vs {:?}", a.keys[i], b.keys[i])); + } + a.children + .iter() + .zip(&b.children) + .enumerate() + .find_map(|(i, (x, y))| first_difference(x, y, &format!("{at}/{i}"))) +} + +/// The chunk B-tree of dataset `name`: (tree, ndims) from its layout. +fn tree_of(path: &Path, name: &str, sizes: bool) -> TreeNode { + use clawhdf5_format::message_type::MessageType; + use clawhdf5_format::object_header::ObjectHeader; + let bytes = std::fs::read(path).unwrap(); + let f = File::open(path).unwrap(); + let sb = f.superblock().clone(); + let addr = clawhdf5_format::group_v2::resolve_path_any(&bytes, &sb, name).unwrap(); + let oh = ObjectHeader::parse(&bytes, addr as usize, 8, 8).unwrap(); + let l = &oh + .messages + .iter() + .find(|m| m.msg_type == MessageType::DataLayout) + .unwrap() + .data; + assert_eq!((l[0], l[1]), (3, 2), "{name}: layout v3, chunked"); + let ndims = l[2] as usize; + let root = u64::from_le_bytes(l[3..11].try_into().unwrap()); + read_tree(&bytes, root, ndims, sizes) +} + +/// Our version-1 chunk B-trees are libhdf5's, node for node (levels, child +/// counts, every key's offsets, and its chunk size where the chunks are +/// the same bytes), for 1-D, 2-D and 3-D datasets with two- and three-level +/// trees, filtered or not. +/// +/// libhdf5 inserts each chunk into the tree when it leaves its chunk cache. +/// A whole-dataset write with no cache (`rdcc_nbytes=0`, or chunks larger +/// than the cache) inserts them in row-major order, as we build the tree; +/// with the default cache small chunks of a 1-D dataset still arrive in +/// order, but those of a multi-dimensional one arrive in the order the +/// cache's hash evicts them, which gives the same keys in differently +/// filled nodes. Both trees index the same chunks; we do not model the +/// cache. +#[test] +fn chunk_btrees_match_libhdf5() { + if !tools_ok() { + return; + } + let dir = tmpdir(); + let theirs = dir.path().join("libhdf5.h5"); + py(&format!( + "import h5py, numpy as np\n\ + with h5py.File({p:?}, 'w', libver=('v108', 'latest'), rdcc_nbytes=0) as f:\n\ + \x20 f.create_dataset('d1000', data=np.arange(10000.0), chunks=(10,))\n\ + \x20 f.create_dataset('d999', data=np.arange(9990.0), chunks=(10,))\n\ + \x20 f.create_dataset('big', data=np.arange(100000, dtype=' = (0..10000).map(f64::from).collect(); + b.create_dataset("d1000") + .with_f64_data(&v) + .with_chunks(&[10]); + b.create_dataset("d999") + .with_f64_data(&v[..9990]) + .with_chunks(&[10]); + let v: Vec = (0..100_000).collect(); + b.create_dataset("big") + .with_i32_data(&v) + .with_chunks(&[1]) + .with_maxshape(&[u64::MAX]); + let v: Vec = (0..10000).map(f64::from).collect(); + b.create_dataset("grid") + .with_f64_data(&v) + .with_shape(&[100, 100]) + .with_chunks(&[1, 1]); + let raw: Vec = (0..27000i16).flat_map(|x| x.to_le_bytes()).collect(); + b.create_dataset("cube") + .with_compound_data( + clawhdf5_format::datatype::Datatype::FixedPoint { + size: 2, + byte_order: clawhdf5_format::datatype::DatatypeByteOrder::LittleEndian, + signed: true, + bit_offset: 0, + bit_precision: 16, + }, + raw, + 27000, + ) + .with_shape(&[30, 30, 30]) + .with_chunks(&[2, 3, 5]); + let v: Vec = (0..20000).map(|i| i % 13).collect(); + b.create_dataset("gz") + .with_i64_data(&v) + .with_chunks(&[7]) + .with_deflate(4); + b.write(&ours).unwrap(); + for (name, sizes) in [ + ("d1000", true), + ("d999", true), + ("big", true), + ("grid", true), + ("cube", true), + // Compressed sizes differ between zlib-rs and zlib. + ("gz", false), + ] { + let a = tree_of(&theirs, name, sizes); + let b = tree_of(&ours, name, sizes); + if let Some(d) = first_difference(&a, &b, "root") { + panic!("{name}: our chunk B-tree differs from libhdf5's: {d}"); + } + } +} diff --git a/crates/clawhdf5/src/lib.rs b/crates/clawhdf5/src/lib.rs index 24bd435..9678c29 100644 --- a/crates/clawhdf5/src/lib.rs +++ b/crates/clawhdf5/src/lib.rs @@ -68,6 +68,7 @@ pub use writer::{DatasetSpec, create_datasets_parallel}; // Re-export useful types from clawhdf5-format for advanced users pub use clawhdf5_format::data_layout::VdsMapping; pub use clawhdf5_format::dict_encoding::{DictEncoded, DictionaryEncoder}; +pub use clawhdf5_format::libver::LibVer; pub use clawhdf5_format::property_list::{ DatasetCreateProps, FileAccessProps, FileCreateProps, lib_version, }; diff --git a/crates/clawhdf5/src/writer.rs b/crates/clawhdf5/src/writer.rs index ea48bc9..9fcf6fc 100644 --- a/crates/clawhdf5/src/writer.rs +++ b/crates/clawhdf5/src/writer.rs @@ -1,6 +1,7 @@ //! Writing API: FileBuilder and GroupBuilder for creating HDF5 files. use clawhdf5_format::file_writer::FileWriter as FormatWriter; +use clawhdf5_format::libver::LibVer; use clawhdf5_format::type_builders::{ AttrValue, DatasetBuilder as FormatDatasetBuilder, FinishedGroup, GroupBuilder as FormatGroupBuilder, @@ -98,6 +99,34 @@ impl FileBuilder { self } + /// Set the library version bounds, as h5py's `libver=(low, high)`: the + /// oldest HDF5 release whose file format is used, and the newest whose + /// features are allowed. The default is `(LibVer::V110, + /// LibVer::Latest)`, the HDF5 1.10 format clawhdf5 has always written. + /// + /// `libver_bounds(LibVer::V18, LibVer::V18)` writes a file HDF5 1.8 can + /// read (version-1 B-tree chunk indexes, version-2 superblock, the low + /// bound libhdf5 2.0 uses by default), or fails with + /// [`Error::Format`] (`FormatError::LibverBound`) for anything 1.8 + /// cannot read: virtual datasets, the 1.12 reference types, native + /// complex numbers. See `clawhdf5_format::libver` for the details. + /// + /// ``` + /// use clawhdf5::{FileBuilder, LibVer}; + /// + /// let mut b = FileBuilder::new(); + /// b.libver_bounds(LibVer::V18, LibVer::V18); + /// b.create_dataset("x") + /// .with_f64_data(&[1.0, 2.0, 3.0]) + /// .with_maxshape(&[u64::MAX]); + /// let bytes = b.finish().unwrap(); + /// assert_eq!(bytes[8], 2); // superblock version 2 + /// ``` + pub fn libver_bounds(&mut self, low: LibVer, high: LibVer) -> &mut Self { + self.writer.libver_bounds(low, high); + self + } + /// Set an attribute on the root group. pub fn set_attr(&mut self, name: &str, value: AttrValue) { self.writer.set_root_attr(name, value); diff --git a/scripts/build-hdf5-1.8.sh b/scripts/build-hdf5-1.8.sh new file mode 100644 index 0000000..cc75fc0 --- /dev/null +++ b/scripts/build-hdf5-1.8.sh @@ -0,0 +1,43 @@ +#!/usr/bin/env bash +# Build libhdf5 1.8.23 (the last 1.8 release) with its command-line tools, as +# the oracle for files written with `LibVer::V18` (the `libver_v18` test in +# clawhdf5-tools finds h5dump through CLAWHDF5_H5DUMP18 or this default prefix). +# +# bash scripts/build-hdf5-1.8.sh [PREFIX] +# +# PREFIX defaults to ~/.cache/hdf5-1.8.23; sources go to PREFIX-src and the +# build tree to PREFIX-build. Reuses an existing install. Needs git, cmake, a C +# compiler and zlib headers. +set -euo pipefail +PREFIX="${1:-$HOME/.cache/hdf5-1.8.23}" +SRC="$PREFIX-src" +BUILD="$PREFIX-build" +if [ -x "$PREFIX/bin/h5dump" ]; then + echo "reusing $PREFIX/bin/h5dump" + "$PREFIX/bin/h5dump" --version + exit 0 +fi +if [ ! -d "$SRC" ]; then + git clone --depth 1 --branch hdf5-1_8_23 https://github.com/HDFGroup/hdf5.git "$SRC" +fi +# 1.8 predates current compilers (GCC 14 turns its pointer-type mismatches in +# the tools into errors): pin gnu99 and demote those errors to warnings. +CFLAGS18="-w -std=gnu99 -Wno-error=incompatible-pointer-types" +CFLAGS18="$CFLAGS18 -Wno-error=implicit-function-declaration -Wno-error=int-conversion" +cmake -S "$SRC" -B "$BUILD" \ + -DCMAKE_BUILD_TYPE=Release \ + -DCMAKE_INSTALL_PREFIX="$PREFIX" \ + -DCMAKE_C_FLAGS="$CFLAGS18" \ + -DBUILD_SHARED_LIBS=ON \ + -DBUILD_TESTING=OFF \ + -DHDF5_BUILD_TOOLS=ON \ + -DHDF5_BUILD_EXAMPLES=OFF \ + -DHDF5_BUILD_CPP_LIB=OFF \ + -DHDF5_BUILD_FORTRAN=OFF \ + -DHDF5_BUILD_HL_LIB=OFF \ + -DHDF5_BUILD_JAVA=OFF \ + -DHDF5_ENABLE_Z_LIB_SUPPORT=ON \ + -DHDF5_ENABLE_SZIP_SUPPORT=OFF +cmake --build "$BUILD" -j "${JOBS:-6}" +cmake --install "$BUILD" +"$PREFIX/bin/h5dump" --version