Write files HDF5 1.8 can read: FileWriter/FileBuilder::libver_bounds
New `LibVer` (V18, V110, V112, V114, V200, Latest) and
`libver_bounds(low, high)` on the format crate's `FileWriter` and the
facade's `FileBuilder`, as libhdf5's H5Pset_libver_bounds / h5py's
libver=(low, high). The default stays (V110, Latest), byte for byte what
was written before.
With a low bound of 1.8: superblock version 2, layout message version 3
(contiguous, compact, chunked) and a version-1 B-tree chunk index for
every chunked dataset, resizable ones included -- what libhdf5 2.x writes
under libver=('v108', 'latest'). The new chunk B-tree writer
(btree_v1_write.rs) replays H5B_insert with the H5Dbtree.c callbacks for
row-major insertion (split ratios 0.1/0.5/0.9, right keys moved as
H5D__btree_cmp3 moves them, root kept in place): its trees equal
libhdf5's node for node for 1-D/2-D/3-D, 2- and 3-level, filtered and
unfiltered datasets (libhdf5 writing without a chunk cache).
The high bound refuses, with FormatError::LibverBound before anything is
written, what needs a newer format: virtual datasets and the paged
file-space strategy (1.10), the 1.12 reference types (datatype v4),
native complex (datatype v5, HDF5 2.0), and a low bound above the high.
Tests: tools/tests/libver_v18.rs writes every writer feature under
(V18, V18), and HDF5 1.8.23's h5dump (scripts/build-hdf5-1.8.sh; skipped
when absent) dumps it exactly as h5dump 1.14 does and returns our bytes
for every numeric dataset; h5py, clawhdf5 and h5rs check --data agree;
then FileEditor grows/appends/annotates it and h5py appends, and every
reader checks again. read_harness gains --v18 and --chunk N.
Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
@@ -0,0 +1,470 @@
|
||||
//! Writing a version-1 B-tree chunk index (node type 1): the chunk index of
|
||||
//! layout message versions 1-3, and the only one HDF5 1.8 reads.
|
||||
//!
|
||||
//! The tree is built the way libhdf5 builds it when the chunks reach it one
|
||||
//! after another in row-major order (a whole-dataset `H5Dwrite` of a 1-D
|
||||
//! dataset, or of any dataset without a chunk cache; with one, libhdf5
|
||||
//! inserts the small chunks of a multi-dimensional dataset in the order its
|
||||
//! cache evicts them, which fills the nodes differently): each
|
||||
//! chunk goes through the same steps as `H5B_insert` (`H5B.c`) with the
|
||||
//! chunk callbacks of `H5Dbtree.c`, so nodes split where libhdf5's split,
|
||||
//! with its default split ratios (a full right-most node keeps 90% of its
|
||||
//! children, a left-most one 10%, any other half), and keys hold what
|
||||
//! libhdf5's hold:
|
||||
//!
|
||||
//! - a chunk's key is its size in the file, its filter mask and its offsets
|
||||
//! (the element-size coordinate 0);
|
||||
//! - a node's final key is the zero-size key one chunk past the chunk that
|
||||
//! last moved it (every scaled coordinate plus one, `H5D__btree_new_node`),
|
||||
//! which libhdf5 moves only when a new chunk is not below it
|
||||
//! (`H5D__btree_cmp3`) — so after an even number of appends in one
|
||||
//! dimension it lies on the last chunk itself;
|
||||
//! - a full root is copied to a new node and becomes the parent of the copy
|
||||
//! and its new sibling, so the root's address (the layout message's) never
|
||||
//! changes.
|
||||
//!
|
||||
//! Nodes are laid out in the order libhdf5 allocates them (the root first,
|
||||
//! then each new node as a split creates it), all of the full node size, the
|
||||
//! unused slots zero.
|
||||
|
||||
#[cfg(not(feature = "std"))]
|
||||
use alloc::{format, vec, vec::Vec};
|
||||
|
||||
use core::cmp::Ordering;
|
||||
|
||||
use crate::error::FormatError;
|
||||
|
||||
/// libhdf5's default chunk B-tree K (`HDF5_BTREE_CHUNK_IK_DEF`): nodes hold
|
||||
/// up to 2K = 64 children. Superblocks of version 2 cannot record another
|
||||
/// value without a superblock extension, which this writer does not emit.
|
||||
pub(crate) const CHUNK_BTREE_K: u16 = 32;
|
||||
|
||||
/// libhdf5's default split ratios (`H5D_XFER_BTREE_SPLIT_RATIO_DEF`) for a
|
||||
/// left-most, middle and right-most node.
|
||||
const SPLIT_RATIOS: [f64; 3] = [0.1, 0.5, 0.9];
|
||||
|
||||
/// A chunk to index: scaled coordinates (offset / chunk dimension) in each
|
||||
/// dataset dimension, stored size, filter mask and address.
|
||||
pub(crate) struct ChunkEntry {
|
||||
pub(crate) scaled: Vec<u64>,
|
||||
pub(crate) nbytes: u64,
|
||||
pub(crate) filter_mask: u32,
|
||||
pub(crate) address: u64,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
struct Key {
|
||||
nbytes: u32,
|
||||
mask: u32,
|
||||
/// Scaled coordinates, the element-size one (0 or 1) last.
|
||||
scaled: Vec<u64>,
|
||||
}
|
||||
|
||||
impl Key {
|
||||
/// `H5D__btree_new_node`'s right key: one chunk past `self` in every
|
||||
/// dimension, with no storage.
|
||||
fn right_of(&self) -> Key {
|
||||
Key {
|
||||
nbytes: 0,
|
||||
mask: 0,
|
||||
scaled: self.scaled.iter().map(|s| s + 1).collect(),
|
||||
}
|
||||
}
|
||||
|
||||
fn cmp_scaled(&self, other: &Key) -> Ordering {
|
||||
self.scaled.cmp(&other.scaled)
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
struct Node {
|
||||
level: u8,
|
||||
left: Option<usize>,
|
||||
right: Option<usize>,
|
||||
/// `children.len() + 1` keys once the node holds a child.
|
||||
keys: Vec<Key>,
|
||||
/// Chunk addresses in a leaf, node indexes above.
|
||||
children: Vec<u64>,
|
||||
}
|
||||
|
||||
/// What an insertion below a node did (`H5B__insert_helper`'s outputs).
|
||||
#[derive(Default)]
|
||||
struct Ret {
|
||||
/// The node's new left key (`lt_key_changed`).
|
||||
lt: Option<Key>,
|
||||
/// The node's new right key (`rt_key_changed`).
|
||||
rt: Option<Key>,
|
||||
/// The node split: the key shared by the halves and the new right node.
|
||||
split: Option<(Key, usize)>,
|
||||
}
|
||||
|
||||
struct Tree {
|
||||
nodes: Vec<Node>,
|
||||
two_k: usize,
|
||||
}
|
||||
|
||||
fn bad(why: &str) -> FormatError {
|
||||
FormatError::SerializationError(format!("version-1 B-tree chunk index: {why}"))
|
||||
}
|
||||
|
||||
impl Tree {
|
||||
fn new(k: u16) -> Self {
|
||||
Self {
|
||||
nodes: vec![Node {
|
||||
level: 0,
|
||||
left: None,
|
||||
right: None,
|
||||
keys: Vec::new(),
|
||||
children: Vec::new(),
|
||||
}],
|
||||
two_k: 2 * usize::from(k),
|
||||
}
|
||||
}
|
||||
|
||||
/// `H5B_insert` of `key` (a chunk after every chunk already inserted).
|
||||
fn insert(&mut self, key: &Key, addr: u64) -> Result<(), FormatError> {
|
||||
let r = self.insert_helper(0, key, addr, 64)?;
|
||||
let Some((md, split)) = r.split else {
|
||||
return Ok(());
|
||||
};
|
||||
// The root split: copy it to a new node and make the root the
|
||||
// parent of the copy and its new right sibling.
|
||||
let lt = r.lt.unwrap_or_else(|| self.nodes[0].keys[0].clone());
|
||||
let rt = match r.rt {
|
||||
Some(rt) => rt,
|
||||
None => self.nodes[split]
|
||||
.keys
|
||||
.last()
|
||||
.cloned()
|
||||
.ok_or_else(|| bad("empty node"))?,
|
||||
};
|
||||
let moved = self.nodes[0].clone();
|
||||
let level = moved.level;
|
||||
let moved_id = self.nodes.len();
|
||||
self.nodes.push(moved);
|
||||
self.nodes[split].left = Some(moved_id);
|
||||
self.nodes[0] = Node {
|
||||
level: level + 1,
|
||||
left: None,
|
||||
right: None,
|
||||
keys: vec![lt, md, rt],
|
||||
children: vec![moved_id as u64, split as u64],
|
||||
};
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn insert_helper(
|
||||
&mut self,
|
||||
id: usize,
|
||||
key: &Key,
|
||||
addr: u64,
|
||||
depth: u8,
|
||||
) -> Result<Ret, FormatError> {
|
||||
if depth == 0 {
|
||||
return Err(bad("tree too deep"));
|
||||
}
|
||||
let n = self.nodes[id].children.len();
|
||||
let level = self.nodes[id].level;
|
||||
let mut ret = Ret::default();
|
||||
if n == 0 {
|
||||
// The first chunk (H5B_INS_FIRST): its key and the right key.
|
||||
let node = &mut self.nodes[id];
|
||||
node.keys = vec![key.clone(), key.right_of()];
|
||||
node.children = vec![addr];
|
||||
return Ok(ret);
|
||||
}
|
||||
// Binary search with H5D__btree_cmp3: 1 when the chunk is not below
|
||||
// the right key, -1 when below the left key, else 0.
|
||||
let (mut lo, mut hi, mut idx) = (0usize, n, 0usize);
|
||||
let mut cmp = Ordering::Less;
|
||||
while lo < hi && cmp != Ordering::Equal {
|
||||
idx = (lo + hi) / 2;
|
||||
let node = &self.nodes[id];
|
||||
cmp = if key.cmp_scaled(&node.keys[idx + 1]) != Ordering::Less {
|
||||
Ordering::Greater
|
||||
} else if key.cmp_scaled(&node.keys[idx]) == Ordering::Less {
|
||||
Ordering::Less
|
||||
} else {
|
||||
Ordering::Equal
|
||||
};
|
||||
if cmp == Ordering::Less {
|
||||
hi = idx;
|
||||
} else {
|
||||
lo = idx + 1;
|
||||
}
|
||||
}
|
||||
let (mut lt_changed, mut rt_changed) = (false, false);
|
||||
// The child to add after child `idx`, with its left key.
|
||||
let mut new_child: Option<(Key, u64)> = None;
|
||||
match cmp {
|
||||
Ordering::Less => return Err(bad("chunks out of order")),
|
||||
Ordering::Greater if idx + 1 < n => {
|
||||
return Err(bad("cannot place chunk"));
|
||||
}
|
||||
Ordering::Greater if level == 0 => {
|
||||
// Past every chunk of the right-most leaf: a new maximum
|
||||
// (H5B_INS_RIGHT through `new_node`), which moves the right
|
||||
// key one chunk past it.
|
||||
idx = n - 1;
|
||||
self.nodes[id].keys[idx + 1] = key.right_of();
|
||||
rt_changed = true;
|
||||
new_child = Some((key.clone(), addr));
|
||||
}
|
||||
Ordering::Equal if level == 0 => {
|
||||
// Inside the last chunk's range: H5D__btree_insert adds it
|
||||
// to the right of that chunk; the right key stays.
|
||||
if key.scaled == self.nodes[id].keys[idx].scaled {
|
||||
return Err(bad("duplicate chunk"));
|
||||
}
|
||||
new_child = Some((key.clone(), addr));
|
||||
}
|
||||
_ => {
|
||||
if cmp == Ordering::Greater {
|
||||
idx = n - 1;
|
||||
}
|
||||
let child = usize::try_from(self.nodes[id].children[idx])
|
||||
.map_err(|_| bad("bad node index"))?;
|
||||
let r = self.insert_helper(child, key, addr, depth - 1)?;
|
||||
if let Some(lt) = r.lt {
|
||||
self.nodes[id].keys[idx] = lt;
|
||||
lt_changed = true;
|
||||
}
|
||||
if let Some(rt) = r.rt {
|
||||
self.nodes[id].keys[idx + 1] = rt;
|
||||
rt_changed = true;
|
||||
}
|
||||
if let Some((md, split)) = r.split {
|
||||
new_child = Some((md, split as u64));
|
||||
}
|
||||
}
|
||||
}
|
||||
// Pass the node's changed end keys up, as H5B__insert_helper does.
|
||||
if lt_changed && idx == 0 {
|
||||
ret.lt = Some(self.nodes[id].keys[0].clone());
|
||||
}
|
||||
if rt_changed && idx + 1 >= n {
|
||||
ret.rt = Some(self.nodes[id].keys[idx + 1].clone());
|
||||
}
|
||||
if let Some((md, child)) = new_child {
|
||||
// A full node splits first; the child goes to the half that
|
||||
// holds child `idx`.
|
||||
let (mut target, mut split) = (id, None);
|
||||
if n == self.two_k {
|
||||
let s = self.split(id, idx);
|
||||
let nleft = self.nodes[id].children.len();
|
||||
if idx >= nleft {
|
||||
idx -= nleft;
|
||||
target = s;
|
||||
}
|
||||
split = Some(s);
|
||||
}
|
||||
// H5B__insert_child (H5B_INS_RIGHT): the new child after child
|
||||
// `idx`, its left key after that child's.
|
||||
let node = &mut self.nodes[target];
|
||||
node.keys.insert(idx + 1, md);
|
||||
node.children.insert(idx + 1, child);
|
||||
ret.split = split.map(|s| (self.nodes[s].keys[0].clone(), s));
|
||||
}
|
||||
Ok(ret)
|
||||
}
|
||||
|
||||
/// `H5B__split` of the full node `id`, the insertion going after child
|
||||
/// `idx`; returns the new right node.
|
||||
fn split(&mut self, id: usize, idx: usize) -> usize {
|
||||
let node = &self.nodes[id];
|
||||
let ratio = if node.right.is_none() {
|
||||
SPLIT_RATIOS[2]
|
||||
} else if node.left.is_none() {
|
||||
SPLIT_RATIOS[0]
|
||||
} else {
|
||||
SPLIT_RATIOS[1]
|
||||
};
|
||||
let mut nleft = (self.two_k as f64 * ratio) as usize;
|
||||
if idx < nleft && nleft == self.two_k {
|
||||
nleft -= 1;
|
||||
} else if idx >= nleft && nleft == 0 {
|
||||
nleft += 1;
|
||||
}
|
||||
let new_id = self.nodes.len();
|
||||
let right = Node {
|
||||
level: node.level,
|
||||
left: Some(id),
|
||||
right: node.right,
|
||||
keys: node.keys[nleft..].to_vec(),
|
||||
children: node.children[nleft..].to_vec(),
|
||||
};
|
||||
let old_right = node.right;
|
||||
self.nodes.push(right);
|
||||
if let Some(r) = old_right {
|
||||
self.nodes[r].left = Some(new_id);
|
||||
}
|
||||
let node = &mut self.nodes[id];
|
||||
node.keys.truncate(nleft + 1);
|
||||
node.children.truncate(nleft);
|
||||
node.right = Some(new_id);
|
||||
new_id
|
||||
}
|
||||
}
|
||||
|
||||
/// Bytes of one node of a chunk B-tree with `ndims` key dimensions (the
|
||||
/// dataset's rank plus the element-size one).
|
||||
fn node_size(two_k: usize, ndims: usize, offset_size: usize) -> usize {
|
||||
let key = 8 + 8 * ndims;
|
||||
8 + 2 * offset_size + (two_k + 1) * key + two_k * offset_size
|
||||
}
|
||||
|
||||
/// Build the chunk B-tree for `chunks`, given in row-major order of their
|
||||
/// scaled coordinates, with nodes laid out from `base_address`. `chunk_dims`
|
||||
/// are the chunk's dimensions (the dataset's rank of them) and `elem_size`
|
||||
/// the element size, the key's last dimension. Returns the nodes' bytes; the
|
||||
/// root is at `base_address`. `chunks` must not be empty: an index without
|
||||
/// chunks has no tree (its address is undefined).
|
||||
pub(crate) fn build_chunk_btree_v1_at(
|
||||
chunks: &[ChunkEntry],
|
||||
chunk_dims: &[u64],
|
||||
elem_size: u32,
|
||||
base_address: u64,
|
||||
offset_size: u8,
|
||||
) -> Result<Vec<u8>, FormatError> {
|
||||
if chunks.is_empty() {
|
||||
return Err(bad("no chunks"));
|
||||
}
|
||||
let rank = chunk_dims.len();
|
||||
let mut tree = Tree::new(CHUNK_BTREE_K);
|
||||
for c in chunks {
|
||||
if c.scaled.len() != rank {
|
||||
return Err(bad("chunk rank differs from the dataset's"));
|
||||
}
|
||||
let nbytes = u32::try_from(c.nbytes).map_err(|_| {
|
||||
FormatError::SerializationError(format!(
|
||||
"a chunk of {} bytes cannot be indexed by a version-1 B-tree \
|
||||
(HDF5 1.8 chunks are under 4 GiB)",
|
||||
c.nbytes
|
||||
))
|
||||
})?;
|
||||
let mut scaled = c.scaled.clone();
|
||||
scaled.push(0);
|
||||
let key = Key {
|
||||
nbytes,
|
||||
mask: c.filter_mask,
|
||||
scaled,
|
||||
};
|
||||
tree.insert(&key, c.address)?;
|
||||
}
|
||||
|
||||
let os = usize::from(offset_size);
|
||||
let ndims = rank + 1;
|
||||
let nsize = node_size(tree.two_k, ndims, os);
|
||||
let addr_of = |id: usize| base_address + (id * nsize) as u64;
|
||||
let mut dims: Vec<u64> = chunk_dims.to_vec();
|
||||
dims.push(u64::from(elem_size));
|
||||
let mut out = vec![0u8; tree.nodes.len() * nsize];
|
||||
for (i, node) in tree.nodes.iter().enumerate() {
|
||||
let d = &mut out[i * nsize..(i + 1) * nsize];
|
||||
d[0..4].copy_from_slice(b"TREE");
|
||||
d[4] = 1; // node type: raw data chunks
|
||||
d[5] = node.level;
|
||||
let n = u16::try_from(node.children.len()).map_err(|_| bad("node too large"))?;
|
||||
d[6..8].copy_from_slice(&n.to_le_bytes());
|
||||
let undef = u64::MAX;
|
||||
put_addr(&mut d[8..], node.left.map_or(undef, addr_of), os);
|
||||
put_addr(&mut d[8 + os..], node.right.map_or(undef, addr_of), os);
|
||||
let mut p = 8 + 2 * os;
|
||||
for (k, key) in node.keys.iter().enumerate() {
|
||||
d[p..p + 4].copy_from_slice(&key.nbytes.to_le_bytes());
|
||||
d[p + 4..p + 8].copy_from_slice(&key.mask.to_le_bytes());
|
||||
for (j, (&s, &dim)) in key.scaled.iter().zip(&dims).enumerate() {
|
||||
let off = s
|
||||
.checked_mul(dim)
|
||||
.ok_or_else(|| FormatError::Overflow("chunk key offset".into()))?;
|
||||
d[p + 8 + 8 * j..p + 16 + 8 * j].copy_from_slice(&off.to_le_bytes());
|
||||
}
|
||||
p += 8 + 8 * ndims;
|
||||
if let Some(&child) = node.children.get(k) {
|
||||
let a = if node.level == 0 {
|
||||
child
|
||||
} else {
|
||||
addr_of(usize::try_from(child).map_err(|_| bad("bad node index"))?)
|
||||
};
|
||||
put_addr(&mut d[p..], a, os);
|
||||
p += os;
|
||||
}
|
||||
}
|
||||
}
|
||||
Ok(out)
|
||||
}
|
||||
|
||||
fn put_addr(d: &mut [u8], v: u64, os: usize) {
|
||||
d[..os].copy_from_slice(&v.to_le_bytes()[..os]);
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn build(n: u64) -> Tree {
|
||||
let mut t = Tree::new(CHUNK_BTREE_K);
|
||||
for i in 0..n {
|
||||
let key = Key {
|
||||
nbytes: 80,
|
||||
mask: 0,
|
||||
scaled: vec![i, 0],
|
||||
};
|
||||
t.insert(&key, 1000 + i).unwrap();
|
||||
}
|
||||
t
|
||||
}
|
||||
|
||||
/// Leaves in order from the root, with their child counts.
|
||||
fn leaves(t: &Tree, id: usize, out: &mut Vec<usize>) {
|
||||
let n = &t.nodes[id];
|
||||
if n.level == 0 {
|
||||
out.push(n.children.len());
|
||||
} else {
|
||||
for &c in &n.children {
|
||||
leaves(t, c as usize, out);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sequential_appends_split_as_libhdf5_does() {
|
||||
// libhdf5 2.0 (h5py, libver=('v108', 'latest')) writes 1000 chunks
|
||||
// as a root over 17 leaves of 57 chunks and one of 31, with the
|
||||
// root's right key on the last chunk (9990, 8 for 10-element f8
|
||||
// chunks).
|
||||
let t = build(1000);
|
||||
assert_eq!(t.nodes[0].level, 1);
|
||||
let mut l = Vec::new();
|
||||
leaves(&t, 0, &mut l);
|
||||
let mut want = vec![57; 17];
|
||||
want.push(31);
|
||||
assert_eq!(l, want);
|
||||
assert_eq!(t.nodes[0].keys.last().unwrap().scaled, vec![999, 1]);
|
||||
// 100 000 chunks: three levels, a root of 31 children.
|
||||
let t = build(100_000);
|
||||
assert_eq!(t.nodes[0].level, 2);
|
||||
assert_eq!(t.nodes[0].children.len(), 31);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn right_key_moves_every_other_append() {
|
||||
let t = build(5);
|
||||
assert_eq!(t.nodes[0].keys.last().unwrap().scaled, vec![5, 1]);
|
||||
let t = build(6);
|
||||
assert_eq!(t.nodes[0].keys.last().unwrap().scaled, vec![5, 1]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn keys_and_siblings_are_consistent() {
|
||||
let t = build(5000);
|
||||
for (i, n) in t.nodes.iter().enumerate() {
|
||||
assert!(n.children.len() <= t.two_k);
|
||||
assert_eq!(n.keys.len(), n.children.len() + 1);
|
||||
if let Some(r) = n.right {
|
||||
assert_eq!(t.nodes[r].left, Some(i));
|
||||
assert_eq!(n.keys.last(), t.nodes[r].keys.first());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -7,6 +7,7 @@ use crate::addr::saturating_usize;
|
||||
#[cfg(not(feature = "std"))]
|
||||
use alloc::{format, vec, vec::Vec};
|
||||
|
||||
use crate::btree_v1_write;
|
||||
use crate::btree_v2_write::{BTreeV2Params, build_btree_v2};
|
||||
use crate::checksum::jenkins_lookup3;
|
||||
use crate::chunk_cache::{CACHE_LINE_SIZE, align_to_cache_line};
|
||||
@@ -19,6 +20,7 @@ use crate::filter_pipeline::{
|
||||
FilterPipeline,
|
||||
};
|
||||
use crate::filters::compress_chunk_masked;
|
||||
use crate::libver::LibVer;
|
||||
/// Round a file offset up to the next cache-line boundary.
|
||||
///
|
||||
/// This ensures chunk data starts at an address that is a multiple of the
|
||||
@@ -866,6 +868,23 @@ pub fn build_chunked_data_from_precompressed(
|
||||
base_address: u64,
|
||||
maxshape: Option<&[u64]>,
|
||||
) -> Result<ChunkedDataResult, FormatError> {
|
||||
build_chunked_data_from_precompressed_libver(pre, base_address, maxshape, LibVer::Latest)
|
||||
}
|
||||
|
||||
/// [`build_chunked_data_from_precompressed`] for a file whose low library
|
||||
/// version bound is `low`: below [`LibVer::V110`] (that is, for HDF5 1.8)
|
||||
/// every chunked dataset gets a version-3 layout message and a version-1
|
||||
/// B-tree chunk index, whatever its maximum shape, as libhdf5 writes it;
|
||||
/// otherwise the version-4 layout and the index libhdf5 picks for it.
|
||||
pub fn build_chunked_data_from_precompressed_libver(
|
||||
pre: &PrecompressedChunks,
|
||||
base_address: u64,
|
||||
maxshape: Option<&[u64]>,
|
||||
low: LibVer,
|
||||
) -> Result<ChunkedDataResult, FormatError> {
|
||||
if low < LibVer::V110 {
|
||||
return build_btree_v1_chunked_data(pre, base_address, maxshape);
|
||||
}
|
||||
let index = ChunkIndexPlan::new(&pre.shape, maxshape, &pre.chunk_dims)?;
|
||||
let offset_size: u8 = 8;
|
||||
let length_size: u8 = 8;
|
||||
@@ -992,6 +1011,92 @@ pub fn build_chunked_data_from_precompressed(
|
||||
})
|
||||
}
|
||||
|
||||
/// Lay out precompressed chunks at `base_address` followed by a version-1
|
||||
/// B-tree chunk index, with a version-3 layout message: what libhdf5 writes
|
||||
/// for a chunked dataset under a low bound of 1.8.
|
||||
fn build_btree_v1_chunked_data(
|
||||
pre: &PrecompressedChunks,
|
||||
base_address: u64,
|
||||
maxshape: Option<&[u64]>,
|
||||
) -> Result<ChunkedDataResult, FormatError> {
|
||||
if let Some(ms) = maxshape {
|
||||
let bad = |what: &str| FormatError::ChunkedReadError(format!("maxshape: {what}"));
|
||||
if ms.len() != pre.shape.len() {
|
||||
return Err(bad("rank differs from the shape"));
|
||||
}
|
||||
if ms.iter().zip(&pre.shape).any(|(&m, &s)| m < s) {
|
||||
return Err(bad("smaller than the shape"));
|
||||
}
|
||||
}
|
||||
let offset_size: u8 = 8;
|
||||
let mut data_buf = Vec::new();
|
||||
let mut entries = Vec::with_capacity(pre.chunks.len());
|
||||
for (i, (_raw_size, stored, filter_mask)) in pre.chunks.iter().enumerate() {
|
||||
let aligned_offset = align_to_cache_line(data_buf.len());
|
||||
if aligned_offset > data_buf.len() {
|
||||
data_buf.resize(aligned_offset, 0u8);
|
||||
}
|
||||
entries.push(btree_v1_write::ChunkEntry {
|
||||
scaled: scaled_coords(&pre.shape, &pre.chunk_dims, i),
|
||||
nbytes: stored.len() as u64,
|
||||
filter_mask: *filter_mask,
|
||||
address: base_address + data_buf.len() as u64,
|
||||
});
|
||||
data_buf.extend_from_slice(stored);
|
||||
}
|
||||
let element_size = u32::try_from(pre.element_size)
|
||||
.map_err(|_| FormatError::Overflow("element size".into()))?;
|
||||
// A dataset with no chunks has no tree: its address is undefined, as
|
||||
// libhdf5 leaves it until the first chunk is written.
|
||||
let btree_address = if entries.is_empty() {
|
||||
u64::MAX
|
||||
} else {
|
||||
let aligned_idx = align_to_cache_line(data_buf.len());
|
||||
if aligned_idx > data_buf.len() {
|
||||
data_buf.resize(aligned_idx, 0u8);
|
||||
}
|
||||
let addr = base_address + data_buf.len() as u64;
|
||||
let tree = btree_v1_write::build_chunk_btree_v1_at(
|
||||
&entries,
|
||||
&pre.chunk_dims,
|
||||
element_size,
|
||||
addr,
|
||||
offset_size,
|
||||
)?;
|
||||
data_buf.extend_from_slice(&tree);
|
||||
addr
|
||||
};
|
||||
let layout_message =
|
||||
serialize_v3_chunked(&pre.chunk_dims, btree_address, offset_size, element_size)?;
|
||||
Ok(ChunkedDataResult {
|
||||
data_bytes: data_buf,
|
||||
layout_message,
|
||||
pipeline_message: pre.pipeline_message.clone(),
|
||||
})
|
||||
}
|
||||
|
||||
/// A version-3 layout message for a chunked dataset: dimensionality (the
|
||||
/// rank plus one), the B-tree's address, then each chunk dimension and the
|
||||
/// element size, four bytes each.
|
||||
fn serialize_v3_chunked(
|
||||
chunk_dims: &[u64],
|
||||
btree_address: u64,
|
||||
offset_size: u8,
|
||||
element_size: u32,
|
||||
) -> Result<Vec<u8>, FormatError> {
|
||||
let ndims = u8::try_from(chunk_dims.len() + 1)
|
||||
.map_err(|_| FormatError::Overflow("chunked layout rank".into()))?;
|
||||
let mut buf = vec![3u8, 2, ndims];
|
||||
push_addr(&mut buf, btree_address, offset_size);
|
||||
for &d in chunk_dims {
|
||||
let d =
|
||||
u32::try_from(d).map_err(|_| FormatError::Overflow(format!("chunk dimension {d}")))?;
|
||||
buf.extend_from_slice(&d.to_le_bytes());
|
||||
}
|
||||
buf.extend_from_slice(&element_size.to_le_bytes());
|
||||
Ok(buf)
|
||||
}
|
||||
|
||||
/// Most slots a Fixed Array index may have before we refuse to build it: its
|
||||
/// data block holds one element per chunk of the *maximum* extent, so a huge
|
||||
/// finite maxshape with small chunks would otherwise exhaust memory.
|
||||
|
||||
@@ -1216,6 +1216,26 @@ impl Datatype {
|
||||
}
|
||||
}
|
||||
|
||||
/// The highest datatype message version in this type's encoding, its
|
||||
/// members' and base types' included (the version decides which HDF5
|
||||
/// releases can read it: 1-3 HDF5 1.8, 4 HDF5 1.12, 5 HDF5 2.0).
|
||||
pub fn max_encoded_version(&self) -> u8 {
|
||||
let own = self.serialize().first().map_or(0, |b| b >> 4);
|
||||
let inner = match self {
|
||||
Datatype::Compound { members, .. } => members
|
||||
.iter()
|
||||
.map(|m| m.datatype.max_encoded_version())
|
||||
.max()
|
||||
.unwrap_or(0),
|
||||
Datatype::Enumeration { base_type, .. }
|
||||
| Datatype::VariableLength { base_type, .. }
|
||||
| Datatype::Array { base_type, .. }
|
||||
| Datatype::Complex { base_type, .. } => base_type.max_encoded_version(),
|
||||
_ => 0,
|
||||
};
|
||||
own.max(inner)
|
||||
}
|
||||
|
||||
/// Check that this datatype can be written: every part of it has an
|
||||
/// on-disk encoding, and the encoding is one the reader (and libhdf5)
|
||||
/// accepts. [`Self::serialize`] cannot report errors, so the writer calls
|
||||
|
||||
@@ -167,6 +167,19 @@ pub enum FormatError {
|
||||
VlDataError(String),
|
||||
/// Serialization error.
|
||||
SerializationError(String),
|
||||
/// The file's library version bounds
|
||||
/// ([`FileWriter::libver_bounds`](crate::file_writer::FileWriter::libver_bounds))
|
||||
/// do not allow what was asked for: `what` needs the format of HDF5
|
||||
/// `needs` or later, and the high bound is `high` (or the low bound is
|
||||
/// above the high one, with `needs` the low bound).
|
||||
LibverBound {
|
||||
/// What cannot be written.
|
||||
what: String,
|
||||
/// The oldest release whose format holds it.
|
||||
needs: crate::libver::LibVer,
|
||||
/// The file's high bound.
|
||||
high: crate::libver::LibVer,
|
||||
},
|
||||
/// Dataset is missing data.
|
||||
DatasetMissingData,
|
||||
/// Dataset is missing shape.
|
||||
@@ -450,6 +463,13 @@ impl fmt::Display for FormatError {
|
||||
FormatError::SerializationError(msg) => {
|
||||
write!(f, "serialization error: {msg}")
|
||||
}
|
||||
FormatError::LibverBound { what, needs, high } => {
|
||||
write!(
|
||||
f,
|
||||
"{what} needs the HDF5 {needs} file format, above the high \
|
||||
library version bound ({high})"
|
||||
)
|
||||
}
|
||||
FormatError::DatasetMissingData => {
|
||||
write!(f, "dataset is missing data")
|
||||
}
|
||||
|
||||
@@ -5,12 +5,13 @@
|
||||
|
||||
use crate::addr::saturating_usize;
|
||||
#[cfg(not(feature = "std"))]
|
||||
use alloc::{format, vec, vec::Vec};
|
||||
use alloc::{format, string::String, vec, vec::Vec};
|
||||
|
||||
use crate::attribute::AttributeMessage;
|
||||
use crate::btree_v2_write::{BTreeV2Params, build_btree_v2};
|
||||
use crate::chunked_write::{
|
||||
ChunkOptions, PrecompressedChunks, build_chunked_data_from_precompressed, precompress_chunks,
|
||||
ChunkOptions, PrecompressedChunks, build_chunked_data_from_precompressed_libver,
|
||||
precompress_chunks,
|
||||
};
|
||||
use crate::data_layout::VdsMapping;
|
||||
use crate::dataspace::{Dataspace, DataspaceType};
|
||||
@@ -31,6 +32,7 @@ pub use crate::type_builders::ProvenanceConfig;
|
||||
pub use crate::type_builders::{AttrValue, CompoundTypeBuilder, EnumTypeBuilder};
|
||||
|
||||
use crate::datatype::{CharacterSet, Datatype};
|
||||
use crate::libver::LibVer;
|
||||
|
||||
pub(crate) const OFFSET_SIZE: u8 = 8;
|
||||
pub(crate) const LENGTH_SIZE: u8 = 8;
|
||||
@@ -168,13 +170,15 @@ pub(crate) fn build_dataset_oh(
|
||||
attrs: AttrStorage<'_>,
|
||||
fill_message: &[u8],
|
||||
refcount: u32,
|
||||
layout_version: u8,
|
||||
) -> Result<Vec<u8>, FormatError> {
|
||||
let mut w = ObjectHeaderWriter::new();
|
||||
w.add_message_with_flags(MessageType::Datatype, dt.serialize(), 0x01);
|
||||
w.add_message(MessageType::Dataspace, ds.serialize(LENGTH_SIZE));
|
||||
w.add_message_with_flags(MessageType::FillValue, fill_message.to_vec(), 0x01);
|
||||
// Versions 3 and 4 encode a contiguous layout the same way.
|
||||
let mut dl = Vec::new();
|
||||
dl.push(4); // version
|
||||
dl.push(layout_version);
|
||||
dl.push(1); // class = contiguous
|
||||
// An empty dataset has no storage: its address must be the undefined
|
||||
// address, as libhdf5 writes it. A real address with size 0 trips
|
||||
@@ -198,14 +202,16 @@ pub(crate) fn build_compact_dataset_oh(
|
||||
attrs: AttrStorage<'_>,
|
||||
fill_message: &[u8],
|
||||
refcount: u32,
|
||||
layout_version: u8,
|
||||
) -> Result<Vec<u8>, FormatError> {
|
||||
let mut w = ObjectHeaderWriter::new();
|
||||
w.add_message_with_flags(MessageType::Datatype, dt.serialize(), 0x01);
|
||||
w.add_message(MessageType::Dataspace, ds.serialize(LENGTH_SIZE));
|
||||
w.add_message_with_flags(MessageType::FillValue, fill_message.to_vec(), 0x01);
|
||||
// Compact layout message: version=4, class=0, u16 size, inline data
|
||||
// Compact layout message: version (3 and 4 are the same here), class=0,
|
||||
// u16 size, inline data
|
||||
let mut dl = Vec::new();
|
||||
dl.push(4); // version
|
||||
dl.push(layout_version);
|
||||
dl.push(0); // class = compact
|
||||
dl.extend_from_slice(&(data.len() as u16).to_le_bytes());
|
||||
dl.extend_from_slice(data);
|
||||
@@ -1371,6 +1377,10 @@ pub struct FileWriter {
|
||||
/// file-space strategy (a File Space Info message in the superblock
|
||||
/// extension).
|
||||
page_size: Option<u32>,
|
||||
/// Library version bounds: the low bound picks the format versions
|
||||
/// written, the high bound limits the features allowed.
|
||||
low: LibVer,
|
||||
high: LibVer,
|
||||
}
|
||||
|
||||
impl Default for FileWriter {
|
||||
@@ -1485,9 +1495,37 @@ impl FileWriter {
|
||||
alignment_threshold: 0,
|
||||
alignment_bytes: 0,
|
||||
page_size: None,
|
||||
low: LibVer::V110,
|
||||
high: LibVer::Latest,
|
||||
}
|
||||
}
|
||||
|
||||
/// Set the library version bounds, as libhdf5's `H5Pset_libver_bounds`
|
||||
/// (h5py's `libver=(low, high)`): the oldest HDF5 release whose format
|
||||
/// the file uses (`low`), and the newest whose features it may use
|
||||
/// (`high`). See [`crate::libver`] for what each bound changes.
|
||||
///
|
||||
/// The default, `(LibVer::V110, LibVer::Latest)`, is what clawhdf5 has
|
||||
/// always written: the HDF5 1.10 format (version-3 superblock, version-4
|
||||
/// layouts with the 1.10 chunk indexes), readable by HDF5 1.10 and later.
|
||||
///
|
||||
/// `(LibVer::V18, LibVer::V18)` writes a file HDF5 1.8 can read — the
|
||||
/// low bound libhdf5 2.0 uses by default: a version-2 superblock,
|
||||
/// version-3 layouts, and a version-1 B-tree for every chunked dataset,
|
||||
/// resizable ones included; [`Self::finish`] then fails with
|
||||
/// [`FormatError::LibverBound`] for anything HDF5 1.8 cannot read
|
||||
/// (virtual datasets, a paged file, the 1.12 reference types, native
|
||||
/// complex numbers). With a low bound of 1.8 and a later high bound
|
||||
/// such objects are written in the newer format, as libhdf5 writes them;
|
||||
/// the rest of the file stays readable by 1.8.
|
||||
///
|
||||
/// A low bound above the high bound makes [`Self::finish`] fail.
|
||||
pub fn libver_bounds(&mut self, low: LibVer, high: LibVer) -> &mut Self {
|
||||
self.low = low;
|
||||
self.high = high;
|
||||
self
|
||||
}
|
||||
|
||||
/// Set global file alignment: datasets with raw data >= `threshold` bytes
|
||||
/// will have their data aligned to `bytes` boundary.
|
||||
///
|
||||
@@ -1583,6 +1621,33 @@ impl FileWriter {
|
||||
)));
|
||||
}
|
||||
|
||||
let (low, high) = (self.low, self.high);
|
||||
let within_bounds = |what: &dyn Fn() -> String, needs: LibVer| {
|
||||
if needs > high {
|
||||
Err(FormatError::LibverBound {
|
||||
what: what(),
|
||||
needs,
|
||||
high,
|
||||
})
|
||||
} else {
|
||||
Ok(())
|
||||
}
|
||||
};
|
||||
within_bounds(&|| format!("a low library version bound of {low}"), low)?;
|
||||
if page_size.is_some() {
|
||||
within_bounds(&|| "the paged file-space strategy".into(), LibVer::V110)?;
|
||||
}
|
||||
// Versions 3 of the layout message and 2 of the superblock are what
|
||||
// HDF5 1.8 reads; 1.10 added version 4 (with its chunk indexes) and
|
||||
// version 3. A paged file needs the version-3 superblock whatever
|
||||
// the low bound (libhdf5 raises it as far as the high bound allows).
|
||||
let layout_version: u8 = if low < LibVer::V110 { 3 } else { 4 };
|
||||
let superblock_version: u8 = if low < LibVer::V110 && page_size.is_none() {
|
||||
2
|
||||
} else {
|
||||
3
|
||||
};
|
||||
|
||||
// The group tree, in layout order: groups depth-first from the root,
|
||||
// then every group's datasets in the same order.
|
||||
let tree = writer_tree::build(self.root, self.track_order)?;
|
||||
@@ -1621,9 +1686,20 @@ impl FileWriter {
|
||||
let ds_attrs = all_ds.iter().flat_map(|d| &d.attrs);
|
||||
for a in group_attrs.chain(ds_attrs) {
|
||||
a.datatype.check_encodable()?;
|
||||
within_bounds(
|
||||
&|| format!("the datatype of attribute {:?}", a.name),
|
||||
LibVer::for_datatype_version(a.datatype.max_encoded_version()),
|
||||
)?;
|
||||
}
|
||||
for d in &all_ds {
|
||||
d.dt.check_encodable()?;
|
||||
within_bounds(
|
||||
&|| "a dataset's datatype".into(),
|
||||
LibVer::for_datatype_version(d.dt.max_encoded_version()),
|
||||
)?;
|
||||
if d.virtual_sources.is_some() {
|
||||
within_bounds(&|| "a virtual dataset".into(), LibVer::V110)?;
|
||||
}
|
||||
}
|
||||
|
||||
let is_vds: Vec<bool> = all_ds.iter().map(|d| d.virtual_sources.is_some()).collect();
|
||||
@@ -1749,10 +1825,11 @@ impl FileWriter {
|
||||
elem_size,
|
||||
&d.chunk_options,
|
||||
)?;
|
||||
let result = build_chunked_data_from_precompressed(
|
||||
let result = build_chunked_data_from_precompressed_libver(
|
||||
&pre,
|
||||
dummy_cursor,
|
||||
d.maxshape.as_deref(),
|
||||
low,
|
||||
)?;
|
||||
dummy_cursor += result.data_bytes.len() as u64;
|
||||
let oh = build_chunked_dataset_oh(
|
||||
@@ -1785,6 +1862,7 @@ impl FileWriter {
|
||||
},
|
||||
&d.fill_message,
|
||||
d.refcount,
|
||||
layout_version,
|
||||
)?;
|
||||
dummy_blobs.push(DataBlob {
|
||||
data: vec![],
|
||||
@@ -1804,6 +1882,7 @@ impl FileWriter {
|
||||
},
|
||||
&d.fill_message,
|
||||
d.refcount,
|
||||
layout_version,
|
||||
)?;
|
||||
dummy_blobs.push(DataBlob {
|
||||
data: vec![],
|
||||
@@ -1904,13 +1983,14 @@ impl FileWriter {
|
||||
let base_address = cursor2 as u64;
|
||||
// Reuse precompressed chunks from Pass 1 — avoids re-compressing
|
||||
// the same data a second time.
|
||||
let result = build_chunked_data_from_precompressed(
|
||||
let result = build_chunked_data_from_precompressed_libver(
|
||||
dummy_blobs[i]
|
||||
.precompressed
|
||||
.as_ref()
|
||||
.expect("chunked dataset missing precompressed cache"),
|
||||
base_address,
|
||||
d.maxshape.as_deref(),
|
||||
low,
|
||||
)?;
|
||||
cursor2 += result.data_bytes.len();
|
||||
let oh = build_chunked_dataset_oh(
|
||||
@@ -1944,6 +2024,7 @@ impl FileWriter {
|
||||
},
|
||||
&d.fill_message,
|
||||
d.refcount,
|
||||
layout_version,
|
||||
)?;
|
||||
ds_blobs2.push(DataBlob {
|
||||
data: vec![],
|
||||
@@ -1973,6 +2054,7 @@ impl FileWriter {
|
||||
},
|
||||
&d.fill_message,
|
||||
d.refcount,
|
||||
layout_version,
|
||||
)?;
|
||||
let mut data = vec![0u8; padding];
|
||||
data.extend_from_slice(&d.raw);
|
||||
@@ -1997,7 +2079,7 @@ impl FileWriter {
|
||||
let mut buf = Vec::with_capacity(cursor2);
|
||||
|
||||
let sb = Superblock {
|
||||
version: 3,
|
||||
version: superblock_version,
|
||||
offset_size: OFFSET_SIZE,
|
||||
length_size: LENGTH_SIZE,
|
||||
base_address: 0,
|
||||
@@ -2783,4 +2865,150 @@ mod tests {
|
||||
assert_eq!(sb.version, 3);
|
||||
assert_eq!(sb.page_size, None);
|
||||
}
|
||||
|
||||
fn layout_of(bytes: &[u8], name: &str) -> Vec<u8> {
|
||||
let sb = Superblock::parse(bytes, 0).unwrap();
|
||||
let addr = resolve_path_any(bytes, &sb, name).unwrap();
|
||||
let hdr = ObjectHeader::parse(bytes, addr as usize, 8, 8).unwrap();
|
||||
hdr.messages
|
||||
.iter()
|
||||
.find(|m| m.msg_type == MessageType::DataLayout)
|
||||
.unwrap()
|
||||
.data
|
||||
.clone()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn libver_v18_writes_the_1_8_format() {
|
||||
let mut fw = FileWriter::new();
|
||||
fw.libver_bounds(LibVer::V18, LibVer::V18);
|
||||
fw.create_dataset("contig").with_f64_data(&[1.0, 2.0]);
|
||||
fw.create_dataset("compact").with_f64_data(&[3.0]).compact();
|
||||
fw.create_dataset("grow")
|
||||
.with_f64_data(&[1.0, 2.0, 3.0])
|
||||
.with_maxshape(&[u64::MAX])
|
||||
.with_chunks(&[2]);
|
||||
fw.create_dataset("none")
|
||||
.with_f64_data(&[])
|
||||
.with_maxshape(&[u64::MAX])
|
||||
.with_chunks(&[2]);
|
||||
let bytes = fw.finish().unwrap();
|
||||
assert_eq!(Superblock::parse(&bytes, 0).unwrap().version, 2);
|
||||
assert_eq!(layout_of(&bytes, "contig")[..2], [3, 1]);
|
||||
assert_eq!(layout_of(&bytes, "compact")[..2], [3, 0]);
|
||||
let grow = layout_of(&bytes, "grow");
|
||||
// Version 3, chunked, 2 dimensions (the element size is the last),
|
||||
// B-tree address, chunk dims 2 and 8.
|
||||
assert_eq!(grow[..3], [3, 2, 2]);
|
||||
assert_eq!(grow[11..], [2, 0, 0, 0, 8, 0, 0, 0]);
|
||||
let root = u64::from_le_bytes(grow[3..11].try_into().unwrap()) as usize;
|
||||
assert_eq!(&bytes[root..root + 5], b"TREE\x01");
|
||||
// No chunks, no tree.
|
||||
assert_eq!(layout_of(&bytes, "none")[3..11], [0xff; 8]);
|
||||
assert_eq!(read_dataset_f64(&bytes, "grow"), vec![1.0, 2.0, 3.0]);
|
||||
assert_eq!(read_dataset_f64(&bytes, "contig"), vec![1.0, 2.0]);
|
||||
assert_eq!(read_dataset_f64(&bytes, "compact"), vec![3.0]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn default_libver_bounds_keep_the_1_10_format() {
|
||||
let mut fw = FileWriter::new();
|
||||
fw.create_dataset("contig").with_f64_data(&[1.0, 2.0]);
|
||||
fw.create_dataset("grow")
|
||||
.with_f64_data(&[1.0, 2.0, 3.0])
|
||||
.with_maxshape(&[u64::MAX])
|
||||
.with_chunks(&[2]);
|
||||
let default = fw.finish().unwrap();
|
||||
let mut fw = FileWriter::new();
|
||||
fw.libver_bounds(LibVer::V110, LibVer::Latest);
|
||||
fw.create_dataset("contig").with_f64_data(&[1.0, 2.0]);
|
||||
fw.create_dataset("grow")
|
||||
.with_f64_data(&[1.0, 2.0, 3.0])
|
||||
.with_maxshape(&[u64::MAX])
|
||||
.with_chunks(&[2]);
|
||||
assert_eq!(fw.finish().unwrap(), default);
|
||||
assert_eq!(layout_of(&default, "contig")[0], 4);
|
||||
assert_eq!(layout_of(&default, "grow")[..2], [4, 2]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn libver_high_bound_refuses_newer_features() {
|
||||
let bound = |r: Result<Vec<u8>, FormatError>, needs: LibVer| match r {
|
||||
Err(FormatError::LibverBound { needs: n, high, .. }) => {
|
||||
assert_eq!((n, high), (needs, LibVer::V18));
|
||||
}
|
||||
other => panic!("expected a bound error, got {other:?}"),
|
||||
};
|
||||
let mut fw = FileWriter::new();
|
||||
fw.libver_bounds(LibVer::V18, LibVer::V18);
|
||||
fw.create_dataset("z")
|
||||
.with_native_complex_f64_data(&[[1.0, 2.0]]);
|
||||
bound(fw.finish(), LibVer::V200);
|
||||
|
||||
let mut fw = FileWriter::new();
|
||||
fw.libver_bounds(LibVer::V18, LibVer::V18);
|
||||
fw.create_dataset("x").with_f64_data(&[1.0]).set_attr(
|
||||
"z",
|
||||
AttrValue::Raw {
|
||||
datatype: crate::type_builders::make_native_complex_f64_type(),
|
||||
shape: vec![],
|
||||
data: vec![0; 16],
|
||||
},
|
||||
);
|
||||
bound(fw.finish(), LibVer::V200);
|
||||
|
||||
let mut fw = FileWriter::new();
|
||||
fw.libver_bounds(LibVer::V18, LibVer::V18);
|
||||
fw.create_dataset("r").with_compound_data(
|
||||
Datatype::Reference {
|
||||
size: 16,
|
||||
ref_type: crate::datatype::ReferenceType::Object2,
|
||||
},
|
||||
vec![0; 16],
|
||||
1,
|
||||
);
|
||||
bound(fw.finish(), LibVer::V112);
|
||||
|
||||
let mut fw = FileWriter::new();
|
||||
fw.libver_bounds(LibVer::V18, LibVer::V18);
|
||||
fw.create_dataset("src").with_f64_data(&[1.0, 2.0]);
|
||||
fw.create_dataset("vds")
|
||||
.with_shape(&[2])
|
||||
.with_f64_data(&[])
|
||||
.with_virtual_sources(vec![VdsMapping {
|
||||
source_file: ".".into(),
|
||||
source_dataset: "src".into(),
|
||||
source_selection: sel_all(),
|
||||
virtual_selection: sel_hyper_1d(0, 2),
|
||||
}]);
|
||||
bound(fw.finish(), LibVer::V110);
|
||||
|
||||
let mut fw = FileWriter::new();
|
||||
fw.libver_bounds(LibVer::V18, LibVer::V18)
|
||||
.with_page_size(4096);
|
||||
bound(fw.finish(), LibVer::V110);
|
||||
|
||||
let mut fw = FileWriter::new();
|
||||
fw.libver_bounds(LibVer::V110, LibVer::V18);
|
||||
bound(fw.finish(), LibVer::V110);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn libver_low_v18_high_latest_allows_newer_objects() {
|
||||
// As libhdf5 does: the object that needs a newer format gets it,
|
||||
// the rest of the file keeps the 1.8 format.
|
||||
let mut fw = FileWriter::new();
|
||||
fw.libver_bounds(LibVer::V18, LibVer::Latest);
|
||||
fw.create_dataset("z")
|
||||
.with_native_complex_f64_data(&[[1.0, 2.0]]);
|
||||
let bytes = fw.finish().unwrap();
|
||||
assert_eq!(Superblock::parse(&bytes, 0).unwrap().version, 2);
|
||||
let mut fw = FileWriter::new();
|
||||
fw.libver_bounds(LibVer::V18, LibVer::Latest)
|
||||
.with_page_size(4096);
|
||||
fw.create_dataset("x").with_f64_data(&[1.0]);
|
||||
let bytes = fw.finish().unwrap();
|
||||
assert_eq!(Superblock::parse(&bytes, 0).unwrap().version, 3);
|
||||
assert_eq!(layout_of(&bytes, "x")[0], 3);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -61,6 +61,7 @@ pub mod addr;
|
||||
pub mod attribute;
|
||||
pub mod attribute_info;
|
||||
pub mod btree_v1;
|
||||
mod btree_v1_write;
|
||||
pub mod btree_v2;
|
||||
mod btree_v2_write;
|
||||
mod bulk_alloc;
|
||||
@@ -107,6 +108,7 @@ pub mod group_v1;
|
||||
pub mod group_v2;
|
||||
#[cfg(feature = "parallel")]
|
||||
pub mod lane_partition;
|
||||
pub mod libver;
|
||||
pub mod link_info;
|
||||
pub mod link_message;
|
||||
pub mod local_heap;
|
||||
|
||||
@@ -0,0 +1,93 @@
|
||||
//! Library version bounds for writing: which HDF5 releases can read a file.
|
||||
//!
|
||||
//! libhdf5 picks the version of every object it writes from the file's
|
||||
//! *low* bound (`H5Pset_libver_bounds`; h5py's `libver=`): the oldest
|
||||
//! format version that holds the object, but never older than the one the
|
||||
//! low bound names. The *high* bound caps it: a feature that needs a newer
|
||||
//! format than the high bound is an error. [`LibVer`] names the same
|
||||
//! releases, and [`crate::file_writer::FileWriter::libver_bounds`] sets them.
|
||||
//!
|
||||
//! What the low bound changes in what clawhdf5 writes:
|
||||
//!
|
||||
//! | | low [`LibVer::V18`] | low [`LibVer::V110`] or later (the default) |
|
||||
//! |---|---|---|
|
||||
//! | superblock | version 2 | version 3 |
|
||||
//! | data layout message | version 3 | version 4 |
|
||||
//! | chunk index | version-1 B-tree (every chunked dataset) | single chunk, Fixed Array, Extensible Array or version-2 B-tree, as libhdf5 picks |
|
||||
//!
|
||||
//! Everything else (version-2 object headers, link and group-info messages,
|
||||
//! dense storage in fractal heaps with version-2 B-trees, filter pipeline
|
||||
//! version 2, fill value version 3, datatype versions up to 3) is the same
|
||||
//! and already readable by HDF5 1.8.
|
||||
//!
|
||||
//! What the high bound refuses: anything that needs 1.10 (virtual datasets,
|
||||
//! the paged file-space strategy) above [`LibVer::V18`], the 1.12 reference
|
||||
//! types (datatype version 4) above [`LibVer::V110`], and HDF5 2.0's native
|
||||
//! complex numbers (datatype version 5) above [`LibVer::V114`].
|
||||
//! `libver_bounds(LibVer::V18, LibVer::V18)` therefore writes a file HDF5
|
||||
//! 1.8 can read, or fails.
|
||||
|
||||
use core::fmt;
|
||||
|
||||
/// An HDF5 library release, as a bound on the file format versions a writer
|
||||
/// may use (libhdf5's `H5F_libver_t`). Ordered oldest first.
|
||||
///
|
||||
/// There is no `Earliest`: clawhdf5 cannot write the pre-1.8 format
|
||||
/// (symbol-table groups, version-1 object headers).
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash)]
|
||||
#[non_exhaustive]
|
||||
pub enum LibVer {
|
||||
/// HDF5 1.8 (`H5F_LIBVER_V18`, h5py `'v108'`).
|
||||
V18,
|
||||
/// HDF5 1.10 (`H5F_LIBVER_V110`, h5py `'v110'`).
|
||||
V110,
|
||||
/// HDF5 1.12 (`H5F_LIBVER_V112`, h5py `'v112'`).
|
||||
V112,
|
||||
/// HDF5 1.14 (`H5F_LIBVER_V114`, h5py `'v114'`).
|
||||
V114,
|
||||
/// HDF5 2.0 (`H5F_LIBVER_V200`).
|
||||
V200,
|
||||
/// The newest format this build of clawhdf5 writes
|
||||
/// (`H5F_LIBVER_LATEST`, h5py `'latest'`).
|
||||
Latest,
|
||||
}
|
||||
|
||||
impl LibVer {
|
||||
/// The release a datatype message of this version first appeared in:
|
||||
/// versions 1-3 are readable by HDF5 1.8, 4 needs 1.12, 5 needs 2.0.
|
||||
pub(crate) fn for_datatype_version(version: u8) -> Self {
|
||||
match version {
|
||||
0..=3 => LibVer::V18,
|
||||
4 => LibVer::V112,
|
||||
_ => LibVer::V200,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl fmt::Display for LibVer {
|
||||
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
||||
f.write_str(match self {
|
||||
LibVer::V18 => "1.8",
|
||||
LibVer::V110 => "1.10",
|
||||
LibVer::V112 => "1.12",
|
||||
LibVer::V114 => "1.14",
|
||||
LibVer::V200 => "2.0",
|
||||
LibVer::Latest => "latest",
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn ordered_oldest_first() {
|
||||
assert!(LibVer::V18 < LibVer::V110);
|
||||
assert!(LibVer::V114 < LibVer::V200);
|
||||
assert!(LibVer::V200 < LibVer::Latest);
|
||||
assert_eq!(LibVer::for_datatype_version(3), LibVer::V18);
|
||||
assert_eq!(LibVer::for_datatype_version(4), LibVer::V112);
|
||||
assert_eq!(LibVer::for_datatype_version(5), LibVer::V200);
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user