Merge branch 'feat/p3-editor-coverage' into feat/p3-remote-editor

# Conflicts:
#	CHANGELOG.md
#	CLAUDE.md
#	docs/design/range-reads.md
This commit is contained in:
osobh
2026-09-26 19:17:46 -05:00
23 changed files with 7218 additions and 533 deletions
+513
View File
@@ -0,0 +1,513 @@
//! Setting attributes, as `H5O__attr_create` / `H5A__dense_insert` do:
//! compact attributes are object header messages (with their creation
//! index in the message header when the object tracks creation order);
//! when an object reaches its compact limit (or an attribute is too large
//! for a header message) its attributes move to dense storage — a fractal
//! heap for the encoded messages, a version-2 B-tree indexing them by name
//! hash (record type 8) and, when creation order is indexed, a second one
//! by creation index (type 9) — and the Attribute Info message points at
//! them.
use std::cmp::Ordering;
use clawhdf5_format::attribute::AttributeMessage;
use clawhdf5_format::dataspace::DataspaceType;
use crate::edit::btree2::Bt2;
use crate::edit::fheap::Heap;
use crate::edit::image::{Image, get_uint, put_uint, undef};
use crate::edit::ohdr::{Header, MSG_ATTRIBUTE};
use crate::edit::{MSG_ATTR_INFO, MSG_FLAG_DONTSHARE, MSG_FLAG_SHARED, check_plain};
use crate::error::Error;
use crate::reader::File;
use crate::types::AttrValue;
/// `H5O_MESG_MAX_SIZE`: a larger attribute goes to dense storage.
const MESG_MAX_SIZE: usize = 65536;
/// `H5O_MAX_CRT_ORDER_IDX`: the creation index of an attribute of an object
/// that does not track creation order.
const NO_CRT_IDX: u16 = u16::MAX;
/// Name and creation-order index B-trees (`H5A_NAME_BT2_*`,
/// `H5A_CORDER_BT2_*`).
const NAME_BT2_TYPE: u8 = 8;
const CORDER_BT2_TYPE: u8 = 9;
const ATTR_BT2_NODE: u32 = 512;
/// Heap IDs in attribute records.
const ID_LEN: usize = 8;
/// An object's Attribute Info message.
#[derive(Debug, Clone)]
struct AInfo {
/// Its message index in the header.
idx: usize,
track: bool,
index: bool,
max_crt: u16,
fheap: u64,
name_bt2: u64,
corder_bt2: u64,
}
impl AInfo {
fn load(img: &Image<'_>, hdr: &Header) -> Result<Option<Self>, Error> {
let Some(idx) = hdr.find(MSG_ATTR_INFO) else {
return Ok(None);
};
if hdr.msgs[idx].flags & MSG_FLAG_SHARED != 0 {
return Err(Error::Unsupported("shared attribute info message".into()));
}
let d = hdr.data(img, idx)?;
let os = img.os as usize;
let short = || Error::Unsupported("short attribute info message".into());
if d.first() != Some(&0) {
return Err(Error::Unsupported("attribute info message version".into()));
}
let flags = *d.get(1).ok_or_else(short)?;
let track = flags & 0x01 != 0;
let index = flags & 0x02 != 0;
let mut p = 2;
let mut max_crt = 0;
if track {
let b = d.get(p..p + 2).ok_or_else(short)?;
max_crt = u16::from_le_bytes([b[0], b[1]]);
p += 2;
}
let n = if index { 3 } else { 2 };
if d.len() < p + n * os {
return Err(short());
}
Ok(Some(Self {
idx,
track,
index,
max_crt,
fheap: get_uint(&d[p..], img.os),
name_bt2: get_uint(&d[p + os..], img.os),
corder_bt2: if index {
get_uint(&d[p + 2 * os..], img.os)
} else {
undef(img.os)
},
}))
}
fn dense(&self, os: u8) -> bool {
self.fheap != undef(os)
}
/// Store the changeable fields back into the message.
fn store(&self, img: &mut Image<'_>, hdr: &mut Header) -> Result<(), Error> {
let os = img.os as usize;
let mut p = 2;
if self.track {
hdr.patch(img, self.idx, p, &self.max_crt.to_le_bytes())?;
p += 2;
}
let mut a = vec![0u8; os];
for (k, v) in [self.fheap, self.name_bt2, self.corder_bt2]
.into_iter()
.enumerate()
.take(if self.index { 3 } else { 2 })
{
put_uint(&mut a, v, img.os);
hdr.patch(img, self.idx, p + k * os, &a)?;
}
Ok(())
}
/// The next creation index (`H5O__attr_create`), or libhdf5's "none".
fn next_crt(&mut self) -> Result<u16, Error> {
if !self.track {
return Ok(NO_CRT_IDX);
}
if self.max_crt == NO_CRT_IDX {
return Err(Error::Unsupported(
"object's attribute creation index is exhausted".into(),
));
}
self.max_crt += 1;
Ok(self.max_crt - 1)
}
}
/// The name bytes of an attribute message body (without the NUL).
pub(super) fn attr_name(d: &[u8]) -> Result<&[u8], Error> {
let bad = || Error::Unsupported("malformed attribute message".into());
let (len, at) = match d.first() {
Some(1) | Some(2) if d.len() >= 8 => (usize::from(u16::from_le_bytes([d[2], d[3]])), 8),
Some(3) if d.len() >= 9 => (usize::from(u16::from_le_bytes([d[2], d[3]])), 9),
_ => return Err(bad()),
};
let name = d.get(at..at + len).ok_or_else(bad)?;
Ok(name.split(|&b| b == 0).next().unwrap_or(name))
}
/// A new Attribute Info message for a version-2 header with flags
/// `hdr_flags`, as `H5O__attr_create` makes it: version 0, creation order
/// tracked / indexed as the header's flags say, the maximum creation index,
/// and no dense storage (undefined fractal heap and B-tree addresses).
fn attr_info_message(hdr_flags: u8, max_crt: u16, os: u8) -> Vec<u8> {
let track = hdr_flags & 0x04 != 0;
let index = hdr_flags & 0x08 != 0;
let mut b = vec![0u8, u8::from(track) | (u8::from(index) << 1)];
if track {
b.extend_from_slice(&max_crt.to_le_bytes());
}
let undef_addr = vec![0xffu8; os as usize];
b.extend_from_slice(&undef_addr);
b.extend_from_slice(&undef_addr);
if index {
b.extend_from_slice(&undef_addr);
}
b
}
/// A version-2 header's limit on compact attributes: stored when its flags
/// say so, else libhdf5's default of 8.
fn max_compact_attrs(img: &Image<'_>, hdr: &Header) -> Result<u16, Error> {
if hdr.flags & 0x10 == 0 {
return Ok(8);
}
let mut p = hdr.addr + 6;
if hdr.flags & 0x20 != 0 {
p += 16;
}
let b = img.read(p, 2)?;
Ok(u16::from_le_bytes([b[0], b[1]]))
}
/// A version-1 attribute message (what libhdf5 writes in a version-1 object
/// header): name, datatype and dataspace each padded to 8 bytes, the
/// dataspace as a version-1 dataspace message.
fn encode_attr_v1(a: &AttributeMessage, ls: u8) -> Vec<u8> {
let mut name = a.name.as_bytes().to_vec();
name.push(0);
let dt = a.datatype.serialize();
let mut ds = vec![1u8, a.dataspace.rank, 0, 0, 0, 0, 0, 0];
if a.dataspace.space_type == DataspaceType::Simple {
let mut b = vec![0u8; ls as usize];
for &d in &a.dataspace.dimensions {
put_uint(&mut b, d, ls);
ds.extend_from_slice(&b);
}
if let Some(max) = &a.dataspace.max_dimensions {
ds[2] = 0x01;
for &d in max {
put_uint(&mut b, d, ls);
ds.extend_from_slice(&b);
}
}
} else {
ds[1] = 0;
}
let mut out = vec![1u8, 0];
out.extend_from_slice(&(name.len() as u16).to_le_bytes());
out.extend_from_slice(&(dt.len() as u16).to_le_bytes());
out.extend_from_slice(&(ds.len() as u16).to_le_bytes());
for part in [&name, &dt, &ds] {
out.extend_from_slice(part);
out.resize(out.len().next_multiple_of(8), 0);
}
out.extend_from_slice(&a.raw_data);
out
}
/// Dense storage opened for changes.
struct Dense {
heap: Heap,
names: Bt2,
order: Option<Bt2>,
}
/// `H5_checksum_lookup3` of a name, as the name index keys it.
fn name_hash(name: &[u8]) -> u32 {
clawhdf5_format::checksum::jenkins_lookup3(name)
}
/// Compare attribute `name` (hash `hash`) with a name-index record
/// (`H5A__dense_btree2_name_compare`: the hash, then the stored name).
fn cmp_name(
heap: &Heap,
img: &Image<'_>,
hash: u32,
name: &[u8],
rec: &[u8],
) -> Result<Ordering, Error> {
let theirs = u32::from_le_bytes([rec[13], rec[14], rec[15], rec[16]]);
match hash.cmp(&theirs) {
Ordering::Equal => {
if rec[ID_LEN] & MSG_FLAG_SHARED != 0 {
return Err(Error::Unsupported(
"shared attribute in dense storage".into(),
));
}
let obj = heap.read(img, &rec[..ID_LEN])?;
Ok(name.cmp(attr_name(&obj)?))
}
o => Ok(o),
}
}
fn corder_of(rec: &[u8]) -> u32 {
u32::from_le_bytes([rec[9], rec[10], rec[11], rec[12]])
}
impl Dense {
fn open(img: &Image<'_>, ai: &AInfo) -> Result<Self, Error> {
let heap = Heap::open(img, ai.fheap)?;
let names = Bt2::open(img, ai.name_bt2)?;
if names.tree_type() != NAME_BT2_TYPE || names.record_size() != ID_LEN + 9 {
return Err(Error::Unsupported("attribute name index layout".into()));
}
let order = if ai.index {
let t = Bt2::open(img, ai.corder_bt2)?;
if t.tree_type() != CORDER_BT2_TYPE || t.record_size() != ID_LEN + 5 {
return Err(Error::Unsupported(
"attribute creation-order index layout".into(),
));
}
Some(t)
} else {
None
};
Ok(Self { heap, names, order })
}
/// `H5A__dense_create`: heap, name index, [creation-order index].
fn create(img: &mut Image<'_>, index: bool) -> Result<Self, Error> {
let heap = Heap::create_attribute_heap(img)?;
let names = Bt2::create(img, NAME_BT2_TYPE, ATTR_BT2_NODE, ID_LEN + 9, 100, 40)?;
let order = if index {
Some(Bt2::create(
img,
CORDER_BT2_TYPE,
ATTR_BT2_NODE,
ID_LEN + 5,
100,
40,
)?)
} else {
None
};
Ok(Self { heap, names, order })
}
/// `H5A__dense_insert` of an encoded attribute message.
fn insert(&mut self, img: &mut Image<'_>, body: &[u8], crt: u16) -> Result<(), Error> {
let name = attr_name(body)?.to_vec();
let id = self.heap.insert(img, body)?;
if id.len() != ID_LEN {
return Err(Error::Unsupported("attribute heap ID length".into()));
}
let hash = name_hash(&name);
let mut rec = id.clone();
rec.push(0);
rec.extend_from_slice(&u32::from(crt).to_le_bytes());
rec.extend_from_slice(&hash.to_le_bytes());
let heap = &self.heap;
self.names
.insert(img, &mut |im, r| cmp_name(heap, im, hash, &name, r), &rec)?;
if let Some(t) = &mut self.order {
let key = u32::from(crt);
t.insert(
img,
&mut |_, r| Ok(key.cmp(&corder_of(r))),
&rec[..ID_LEN + 5],
)?;
}
Ok(())
}
fn finish(&mut self, img: &mut Image<'_>) -> Result<(), Error> {
self.heap.finish(img)?;
self.names.finish(img)?;
if let Some(t) = &mut self.order {
t.finish(img)?;
}
Ok(())
}
}
/// Set attribute `name` of the object at `path` to `value`.
pub(super) fn set_attr(
f: &File,
img: &mut Image<'_>,
path: &str,
name: &str,
value: &AttrValue,
) -> Result<(), Error> {
let addr = clawhdf5_format::group_v2::resolve_path_any(f.as_bytes(), f.superblock(), path)?;
let mut hdr = Header::load(img, addr)?;
let mut msg = clawhdf5_format::type_builders::build_attr_message(name, value);
check_plain(&msg.datatype)?;
// libhdf5 encodes a simple dataspace with its maximum dimensions (the
// current ones when none were given), so an attribute takes the same
// space in a header or heap as when libhdf5 writes it.
if msg.dataspace.space_type == DataspaceType::Simple && msg.dataspace.max_dimensions.is_none() {
msg.dataspace.max_dimensions = Some(msg.dataspace.dimensions.clone());
}
// H5A__set_version: version 1 unless the name is not ASCII (then 3),
// raised to the file's low bound — which is the earliest for a file
// libhdf5 opens without a libver setting (h5py's `r+`).
let body = if hdr.version == 1 || name.is_ascii() {
encode_attr_v1(&msg, img.ls)
} else {
let mut b = msg.serialize_v3(img.ls);
if !name.is_ascii() {
b[8] = 1; // UTF-8 name
}
b
};
let mut ainfo = if hdr.version == 2 {
AInfo::load(img, &hdr)?
} else {
None
};
if let Some(ai) = ainfo.as_mut().filter(|a| a.dense(img.os)) {
let mut ai = ai.clone();
set_dense(img, &mut hdr, &mut ai, name.as_bytes(), &body)?;
return hdr.finish(img);
}
let mut existing = None;
let mut count = 0usize;
for i in 0..hdr.msgs.len() {
if hdr.msgs[i].mtype != MSG_ATTRIBUTE {
continue;
}
if hdr.msgs[i].flags & MSG_FLAG_SHARED != 0 {
return Err(Error::Unsupported("shared attribute message".into()));
}
count += 1;
if attr_name(&hdr.data(img, i)?)? == name.as_bytes() {
existing = Some(i);
}
}
if let Some(i) = existing {
hdr.delete(img, i)?;
count -= 1;
}
if hdr.version == 1 {
hdr.insert(img, MSG_ATTRIBUTE, 0, &body, None)?;
return hdr.finish(img);
}
let tracked = hdr.flags & 0x04 != 0;
// H5O__attr_create: a missing Attribute Info message starts from
// nothing (and is added below, holding the new maximum creation index).
let new_ainfo = ainfo.is_none();
let mut ai = ainfo.take().unwrap_or(AInfo {
idx: usize::MAX,
track: tracked,
index: hdr.flags & 0x08 != 0,
max_crt: 0,
fheap: undef(img.os),
name_bt2: undef(img.os),
corder_bt2: undef(img.os),
});
let max_compact = usize::from(max_compact_attrs(img, &hdr)?);
if count == max_compact || body.len() >= MESG_MAX_SIZE {
if new_ainfo {
return Err(Error::Unsupported(
"dense attribute storage for an object without an Attribute Info message".into(),
));
}
to_dense(img, &mut hdr, &mut ai)?;
set_dense(img, &mut hdr, &mut ai, name.as_bytes(), &body)?;
return hdr.finish(img);
}
let crt = ai.next_crt()?;
let corder = tracked.then_some(crt);
if new_ainfo {
// libhdf5 appends the Attribute Info message before the attribute
// when free space holds both, else after it, so that a new
// continuation chunk made for the attribute has room for it too.
let a = attr_info_message(hdr.flags, ai.max_crt, img.os);
let first = hdr.has_free(a.len() + hdr.hsize() + body.len());
if first {
hdr.insert(img, MSG_ATTR_INFO, MSG_FLAG_DONTSHARE, &a, Some(0))?;
}
hdr.insert(img, MSG_ATTRIBUTE, 0, &body, corder)?;
if !first {
hdr.insert(img, MSG_ATTR_INFO, MSG_FLAG_DONTSHARE, &a, Some(0))?;
}
} else {
hdr.insert(img, MSG_ATTRIBUTE, 0, &body, corder)?;
ai.store(img, &mut hdr)?;
}
hdr.finish(img)
}
/// Move every compact attribute of the object into new dense storage, in
/// header message order (`H5O__attr_to_dense_cb`), leaving free space where
/// the messages were.
fn to_dense(img: &mut Image<'_>, hdr: &mut Header, ai: &mut AInfo) -> Result<(), Error> {
let mut dense = Dense::create(img, ai.index)?;
for i in 0..hdr.msgs.len() {
if hdr.msgs[i].mtype != MSG_ATTRIBUTE {
continue;
}
if hdr.msgs[i].flags & MSG_FLAG_SHARED != 0 {
return Err(Error::Unsupported("shared attribute message".into()));
}
let body = hdr.data(img, i)?;
let crt = if ai.track {
hdr.msgs[i].corder.unwrap_or(0)
} else {
NO_CRT_IDX
};
dense.insert(img, &body, crt)?;
hdr.delete(img, i)?;
}
dense.finish(img)?;
ai.fheap = dense.heap.address();
ai.name_bt2 = dense.names.address();
if let Some(t) = &dense.order {
ai.corder_bt2 = t.address();
}
ai.store(img, hdr)
}
/// Set an attribute of an object whose attributes are in dense storage: an
/// attribute of that name whose new encoding has the old one's size is
/// rewritten in its heap object (`H5A__dense_write`); otherwise the old one
/// is removed (`H5A__dense_remove`: name index, creation-order index, heap
/// object) and the new one inserted with the next creation index.
fn set_dense(
img: &mut Image<'_>,
hdr: &mut Header,
ai: &mut AInfo,
name: &[u8],
body: &[u8],
) -> Result<(), Error> {
let mut dense = Dense::open(img, ai)?;
let hash = name_hash(name);
let found = {
let heap = &dense.heap;
dense
.names
.find(img, &mut |im, r| cmp_name(heap, im, hash, name, r))?
};
if let Some(rec) = found {
if dense.heap.write_in_place(img, &rec[..ID_LEN], body)? {
return dense.finish(img);
}
{
let heap = &dense.heap;
dense
.names
.remove(img, &mut |im, r| cmp_name(heap, im, hash, name, r))?;
}
if let Some(t) = &mut dense.order {
let key = corder_of(&rec);
t.remove(img, &mut |_, r| Ok(key.cmp(&corder_of(r))))?
.ok_or_else(|| {
Error::Unsupported("attribute missing from its creation-order index".into())
})?;
}
dense.heap.remove(img, &rec[..ID_LEN])?;
}
let crt = ai.next_crt()?;
dense.insert(img, body, crt)?;
dense.finish(img)?;
ai.store(img, hdr)
}
+166 -1
View File
@@ -56,6 +56,15 @@ fn bad(why: &str) -> Error {
))
}
/// What a removal did below a node (`H5B_ins_t`), with the removed chunk's
/// address and size.
enum Rm {
NotFound,
Noop((u64, u32)),
/// The child is gone: the parent must drop it.
Remove((u64, u32)),
}
enum Ins {
Done,
/// The node split; the new right sibling and its first key.
@@ -144,7 +153,14 @@ impl BTree1 {
put_uint(&mut d[8 + osz..], node.right, os);
let ks = self.key_size();
let mut p = 8 + 2 * osz;
for (i, k) in node.keys.iter().enumerate() {
// An empty node (a root whose last chunk was removed) stores no
// keys, as libhdf5 writes it.
let nkeys = if node.children.is_empty() {
0
} else {
node.keys.len()
};
for (i, k) in node.keys.iter().take(nkeys).enumerate() {
d[p..p + 4].copy_from_slice(&k.size.to_le_bytes());
d[p + 4..p + 8].copy_from_slice(&k.mask.to_le_bytes());
for (j, o) in k.offs.iter().enumerate() {
@@ -210,6 +226,18 @@ impl BTree1 {
return Err(bad("bad chunk key"));
}
let root = self.read(img, self.root)?;
if root.children.is_empty() {
// Every chunk was removed (H5B__insert_helper's first
// insertion): the root, a leaf again, takes it.
let right = self.right_key_after(&key);
let node = Node {
level: 0,
keys: vec![key, right],
children: vec![addr],
..root
};
return self.write(img, &node);
}
if let Ins::Split(mid, right_addr) = self.insert_at(img, root, &key, addr, 64)? {
// The root split: move its (left) half to a new node so the root
// keeps its address, then make the root the parent of both.
@@ -372,6 +400,143 @@ impl BTree1 {
cmp(&key.offs, &right.keys[0].offs) != Ordering::Less
}
/// Remove the chunk at offsets `offs` (element-size coordinate 0), as
/// `H5B_remove` does for the chunk index (whose critical key is the
/// left one): no rebalancing; a node left without children is deleted
/// and its siblings relinked (the left one takes over its right key),
/// a root left empty becomes an empty leaf. Returns the chunk's address
/// and stored size, or `None` when the tree has no such chunk (nothing
/// changes then). Deleted nodes are freed in `img`.
pub(crate) fn remove(
&mut self,
img: &mut Image<'_>,
offs: &[u64],
) -> Result<Option<(u64, u32)>, Error> {
if offs.len() != self.ndims {
return Err(bad("bad chunk key"));
}
let mut lt = None;
match self.remove_at(img, self.root, 0, offs, &mut lt, 64)? {
Rm::NotFound => Ok(None),
Rm::Noop(c) | Rm::Remove(c) => Ok(Some(c)),
}
}
fn remove_at(
&self,
img: &mut Image<'_>,
addr: u64,
level: usize,
offs: &[u64],
lt_out: &mut Option<Key>,
depth: u8,
) -> Result<Rm, Error> {
if depth == 0 {
return Err(bad("tree too deep"));
}
let mut node = self.read(img, addr)?;
let n = node.children.len();
// H5D__btree_cmp3 over (keys[i], keys[i + 1]), binary search.
let (mut lo, mut hi, mut idx) = (0usize, n, 0usize);
let mut c = 1i32;
while lo < hi && c != 0 {
idx = (lo + hi) / 2;
c = if cmp(offs, &node.keys[idx + 1].offs) != Ordering::Less {
1
} else if cmp(offs, &node.keys[idx].offs) == Ordering::Less {
-1
} else {
0
};
if c < 0 {
hi = idx;
} else {
lo = idx + 1;
}
}
if c != 0 {
return Ok(Rm::NotFound);
}
let mut lt_changed = None;
let res = if node.level > 0 {
let child = self.read(img, node.children[idx])?;
if usize::from(child.level) + 1 != usize::from(node.level) {
return Err(bad("inconsistent node levels"));
}
self.remove_at(
img,
node.children[idx],
level + 1,
offs,
&mut lt_changed,
depth - 1,
)?
} else {
if node.keys[idx].offs != offs {
return Ok(Rm::NotFound);
}
Rm::Remove((node.children[idx], node.keys[idx].size))
};
let chunk = match res {
Rm::NotFound => return Ok(Rm::NotFound),
Rm::Noop(c) | Rm::Remove(c) => c,
};
let mut dirty = false;
if let Some(k) = lt_changed {
node.keys[idx] = k;
dirty = true;
if idx == 0 {
*lt_out = Some(node.keys[0].clone());
}
}
let out = Rm::Noop(chunk);
if let Rm::Remove(_) = res {
let undefined = undef(img.os);
if n == 1 {
if level > 0 {
if node.left != undefined {
let mut sib = self.read(img, node.left)?;
let last = sib.children.len();
sib.keys[last] = node.keys[1].clone();
sib.right = node.right;
self.write(img, &sib)?;
}
if node.right != undefined {
let mut sib = self.read(img, node.right)?;
sib.left = node.left;
self.write(img, &sib)?;
}
img.free(addr, self.node_size(img.os) as u64);
return Ok(Rm::Remove(chunk));
}
node.children.clear();
node.keys.truncate(1);
node.level = 0;
} else if idx == 0 {
node.keys.remove(0);
node.children.remove(0);
*lt_out = Some(node.keys[0].clone());
} else {
// Right-most or middle child: its left key goes, the next
// key becomes the following child's left key.
node.keys.remove(idx);
node.children.remove(idx);
}
dirty = true;
}
if dirty {
self.write(img, &node)?;
}
// The left sibling's right key follows a changed left key.
if lt_out.is_some() && node.left != undef(img.os) && level > 0 {
let mut sib = self.read(img, node.left)?;
let last = sib.children.len();
sib.keys[last] = node.keys[0].clone();
self.write(img, &sib)?;
}
Ok(out)
}
fn insert_child(&self, node: &mut Node, pos: usize, key: Key, addr: u64) {
let n = node.children.len();
if node.level == 0 {
File diff suppressed because it is too large Load Diff
+26 -3
View File
@@ -383,12 +383,23 @@ impl Ea {
}
/// Set element `idx` to `e`.
pub(crate) fn set(&mut self, img: &mut Image<'_>, idx: u64, e: Elem) -> Result<(), Error> {
/// Set element `idx` to `e`, or back to the fill element (`None`: a
/// removed chunk, `H5D__earray_idx_remove`), which creates no block.
pub(crate) fn set(
&mut self,
img: &mut Image<'_>,
idx: u64,
e: Option<Elem>,
) -> Result<(), Error> {
let os = img.os;
let osz = u64::from(os);
let es = self.slot_size(os) as u64;
let enc = encode_elem(Some(e), self.filtered, self.elem_size, os)?;
let enc = encode_elem(e, self.filtered, self.elem_size, os)?;
let clear = e.is_none();
if self.iblock == undef(os) {
if clear {
return Ok(());
}
self.create_iblock(img)?;
}
let ib = self.iblock;
@@ -420,6 +431,9 @@ impl Ea {
let dblk_idx = l.start_dblk + local;
let slot = dblks_at + dblk_idx * osz;
let mut addr = get_uint(&img.read(slot, os as usize)?, os);
if addr == undef(os) && clear {
return Ok(());
}
if addr == undef(os) {
// libhdf5 records start_idx + (global data block index)
// * nelmts here (H5EA__lookup_elmt), not the block's
@@ -449,6 +463,9 @@ impl Ea {
let sb_prefix = self.dblk_prefix_len(os);
let sb_len = sb_prefix + bitmap_len + l.ndblks * osz;
let mut sb = get_uint(&img.read(sslot, os as usize)?, os);
if sb == undef(os) && clear {
return Ok(());
}
if sb == undef(os) {
let mut d = self.block_prefix(b"EASB", l.start_idx, os);
d.resize(d.len() + bitmap_len as usize, 0);
@@ -471,6 +488,9 @@ impl Ea {
let local = (rel - l.start_idx) / l.dblk_nelmts;
let dslot = sb + sb_prefix + bitmap_len + local * osz;
let mut addr = get_uint(&img.read(dslot, os as usize)?, os);
if addr == undef(os) && clear {
return Ok(());
}
if addr == undef(os) {
let off = l.start_idx + local * l.dblk_nelmts;
addr = self.create_dblock(img, l.dblk_nelmts, off)?;
@@ -491,6 +511,9 @@ impl Ea {
let bpos = sb + sb_prefix + bit / 8;
let mut byte = img.read(bpos, 1)?[0];
let mask = 0x80u8 >> (bit % 8);
if byte & mask == 0 && clear {
return Ok(());
}
if byte & mask == 0 {
let fill = self.fill_elems(page, os)?;
img.write(page_at, &fill)?;
@@ -503,7 +526,7 @@ impl Ea {
}
}
}
if idx + 1 > self.stats[4] {
if !clear && idx + 1 > self.stats[4] {
self.stats[4] = idx + 1;
self.dirty_hdr = true;
}
+12 -3
View File
@@ -137,13 +137,19 @@ impl Fa {
Ok((fa, hdr))
}
/// Set element `idx` to `e`.
pub(crate) fn set(&mut self, img: &mut Image<'_>, idx: u64, e: Elem) -> Result<(), Error> {
/// Set element `idx` to `e`, or back to the fill element (`None`: a
/// removed chunk, `H5D__farray_idx_remove`), which creates no page.
pub(crate) fn set(
&mut self,
img: &mut Image<'_>,
idx: u64,
e: Option<Elem>,
) -> Result<(), Error> {
let os = img.os;
if idx >= self.nelmts {
return Err(bad("index beyond the array"));
}
let enc = encode_elem(Some(e), self.filtered, self.elem_size, os)?;
let enc = encode_elem(e, self.filtered, self.elem_size, os)?;
let es = self.slot(os);
let prefix = 6 + u64::from(os);
let page = self.page();
@@ -162,6 +168,9 @@ impl Fa {
let bpos = self.dblk + prefix + p / 8;
let mut byte = img.read(bpos, 1)?[0];
let mask = 0x80u8 >> (p % 8);
if byte & mask == 0 && e.is_none() {
return Ok(());
}
if byte & mask == 0 {
let fill = encode_elem(None, self.filtered, self.elem_size, os)?;
img.write(page_at, &fill.repeat(count as usize))?;
File diff suppressed because it is too large Load Diff
+199 -26
View File
@@ -35,6 +35,61 @@ pub(crate) struct Image<'a> {
/// Width of addresses and lengths in the file.
pub(crate) os: u8,
pub(crate) ls: u8,
/// Space the edit stopped using. Not reused by this edit: until the
/// edit is committed, the file's metadata still points at it.
freed: Vec<(u64, u64)>,
/// Space earlier edits of the session freed, available to this one.
reusable: FreeList,
/// Blocks this edit took from `reusable`: nothing on disk refers to
/// them, so they are written with the new space, before the changes
/// that link them in (see [`Plan::commit`]).
fresh: Vec<(u64, u64)>,
}
/// Free space, address -> length, adjacent blocks merged.
#[derive(Debug, Clone, Default)]
pub(crate) struct FreeList(BTreeMap<u64, u64>);
impl FreeList {
/// Add `[addr, addr + len)`, merged with neighbours it touches.
pub(crate) fn add(&mut self, addr: u64, len: u64) {
if len == 0 {
return;
}
let (mut lo, mut hi) = (addr, addr.saturating_add(len));
if let Some((&a, &l)) = self.0.range(..=lo).next_back()
&& a + l >= lo
{
lo = a;
hi = hi.max(a + l);
self.0.remove(&a);
}
while let Some((&a, &l)) = self.0.range(lo..=hi).next() {
hi = hi.max(a + l);
self.0.remove(&a);
}
self.0.insert(lo, hi - lo);
}
/// Take `size` bytes from the smallest block that holds them (the
/// lowest address among equals), from its start.
fn take(&mut self, size: u64) -> Option<u64> {
let (&a, &l) = self
.0
.iter()
.filter(|&(_, &l)| l >= size)
.min_by_key(|&(&a, &l)| (l, a))?;
self.0.remove(&a);
if l > size {
self.0.insert(a + size, l - size);
}
Some(a)
}
/// Total bytes.
pub(crate) fn total(&self) -> u64 {
self.0.values().sum()
}
}
impl<'a> Image<'a> {
@@ -47,9 +102,24 @@ impl<'a> Image<'a> {
old_eoa: eoa,
os,
ls,
freed: Vec::new(),
reusable: FreeList::default(),
fresh: Vec::new(),
}
}
/// Let the edit allocate from `free` (space earlier edits freed).
pub(crate) fn with_reusable(mut self, free: FreeList) -> Self {
// Only space inside the file as it is now.
self.reusable = FreeList(
free.0
.into_iter()
.filter(|&(a, l)| a.saturating_add(l) <= self.old_eoa)
.collect(),
);
self
}
pub(crate) fn eoa(&self) -> u64 {
self.eoa
}
@@ -63,11 +133,24 @@ impl<'a> Image<'a> {
!self.patches.is_empty() || self.eoa != self.old_eoa
}
/// Allocate `size` bytes at the end of the file. The space reads as
/// zeros until written. Nothing is ever freed: space an edit stops
/// using (a relocated chunk, say) is leaked, as there is no free-space
/// manager.
/// Allocate `size` bytes: from space an earlier edit of this session
/// freed when a block holds them (best fit), else at the end of the
/// file. The space reads as zeros until written.
pub(crate) fn alloc(&mut self, size: u64) -> Result<u64, Error> {
if size > 0
&& let Some(a) = self.reusable.take(size)
{
self.fresh.push((a, size));
let n = usize::try_from(size)
.map_err(|_| Error::Unsupported("allocation too large".into()))?;
self.write(a, &vec![0u8; n])?;
return Ok(a);
}
self.alloc_end(size)
}
/// Allocate `size` bytes at the end of the file.
fn alloc_end(&mut self, size: u64) -> Result<u64, Error> {
let addr = self.eoa;
let end = addr
.checked_add(size)
@@ -77,6 +160,14 @@ impl<'a> Image<'a> {
Ok(addr)
}
/// Note that the edit no longer uses `[addr, addr + len)`; later edits
/// of the session may reuse it.
pub(crate) fn free(&mut self, addr: u64, len: u64) {
if len > 0 {
self.freed.push((addr, len));
}
}
/// If `[addr, addr + old_len)` is the last allocated space, grow it to
/// `new_len` bytes (a structure at the end of the file can grow where
/// it is) and return true.
@@ -91,7 +182,7 @@ impl<'a> Image<'a> {
}
let old_end = self.eoa;
self.eoa = addr;
if let Err(e) = self.alloc(new_len) {
if let Err(e) = self.alloc_end(new_len) {
self.eoa = old_end;
return Err(e);
}
@@ -189,12 +280,24 @@ impl<'a> Image<'a> {
/// The edit's writes, detached from the base bytes (see the module's
/// invariant: the reader that owns them can then be dropped before
/// anything is written).
pub(crate) fn into_plan(self) -> Plan {
Plan {
patches: self.patches,
eoa: self.eoa,
old_eoa: self.old_eoa,
pub(crate) fn into_plan(self) -> (Plan, FreeList) {
// What the session may reuse once this edit is committed: what it
// did not take, and what it freed.
let mut free = self.reusable;
for (a, l) in self.freed {
free.add(a, l);
}
let mut fresh = self.fresh;
fresh.sort_unstable();
(
Plan {
patches: self.patches,
eoa: self.eoa,
old_eoa: self.old_eoa,
fresh,
},
free,
)
}
}
@@ -203,33 +306,57 @@ pub(crate) struct Plan {
patches: BTreeMap<u64, Vec<u8>>,
eoa: u64,
old_eoa: u64,
/// Reused blocks (sorted): written with the new space.
fresh: Vec<(u64, u64)>,
}
impl Plan {
/// Whether `addr` is in space nothing on disk refers to yet (past the
/// old end of file, or in a reused block), and up to where (before
/// `end`) that stays so.
fn new_space(&self, addr: u64, end: u64) -> (bool, u64) {
if addr >= self.old_eoa {
return (true, end);
}
let limit = end.min(self.old_eoa);
// The reused block holding `addr`, or the next one after it.
let i = self.fresh.partition_point(|&(a, l)| a + l <= addr);
match self.fresh.get(i) {
Some(&(a, l)) if a <= addr => (true, limit.min(a + l)),
Some(&(a, _)) => (false, limit.min(a)),
None => (false, limit),
}
}
/// Write the edit to `file`, whose superblock is at `user_block`.
///
/// Order: first everything in newly allocated space (new chunks, new
/// index blocks, relocated structures), which nothing on disk refers to
/// yet, then a sync; then the changes to existing bytes — raw data
/// overwritten in place and the metadata that links the new space in
/// (superblock end of file, chunk index entries, object header
/// index blocks, relocated structures — past the old end of file, or in
/// space an earlier edit of the session freed), which nothing on disk
/// refers to yet, then a sync; then the changes to existing bytes — raw
/// data overwritten in place and the metadata that links the new space
/// in (superblock end of file, chunk index entries, object header
/// messages) — then a sync. A crash during the first phase leaves the
/// file as it was (plus unreferenced bytes past its end of file); a
/// crash during the second can leave it inconsistent, as with libhdf5
/// without SWMR: there is no journal.
/// file as it was (plus unreferenced bytes); a crash during the second
/// can leave it inconsistent, as with libhdf5 without SWMR: there is no
/// journal.
pub(crate) fn commit(self, file: &mut std::fs::File, user_block: u64) -> Result<(), Error> {
let old_eoa = self.old_eoa;
let mut in_place: Vec<(u64, &[u8])> = Vec::new();
for (&addr, bytes) in &self.patches {
// A patch may run from existing bytes into new space (writes
// merge); its new part goes with the new space.
let split = old_eoa.saturating_sub(addr).min(bytes.len() as u64) as usize;
let (old, new) = bytes.split_at(split);
if !new.is_empty() {
write_at(file, user_block + addr + split as u64, new)?;
}
if !old.is_empty() {
in_place.push((addr, old));
// A patch may run across new and existing space (writes
// merge): split it where that changes.
let end = addr + bytes.len() as u64;
let mut at = addr;
while at < end {
let (new, upto) = self.new_space(at, end);
let part = &bytes[(at - addr) as usize..(upto - addr) as usize];
if new {
write_at(file, user_block + at, part)?;
} else {
in_place.push((at, part));
}
at = upto;
}
}
if self.eoa > old_eoa {
@@ -305,6 +432,52 @@ mod tests {
assert!(img.write(42, &[1]).is_err());
}
#[test]
fn free_list_merges_and_takes_best_fit() {
let mut f = FreeList::default();
f.add(100, 10);
f.add(120, 5);
f.add(110, 10); // joins both neighbours
assert_eq!(
f.0.iter().map(|(&a, &l)| (a, l)).collect::<Vec<_>>(),
[(100, 25)]
);
f.add(300, 8);
f.add(200, 40);
// Best fit: the 8-byte block for 6 bytes, from its start.
assert_eq!(f.take(6), Some(300));
assert_eq!(f.take(30), Some(200));
assert_eq!(f.take(26), None);
assert_eq!(f.total(), 25 + 2 + 10);
}
/// An edit allocates from space earlier edits freed (zeroed), never
/// from what it frees itself; the plan writes reused blocks with the
/// new space.
#[test]
fn reuse_across_edits_only() {
let base = vec![7u8; 64];
let mut free = FreeList::default();
free.add(8, 16);
let mut img = Image::new(&base, 8, 8).with_reusable(free);
img.free(32, 16); // freed by this edit: not reusable yet
let a = img.alloc(16).unwrap();
assert_eq!(a, 8);
assert_eq!(img.read(8, 16).unwrap(), vec![0u8; 16]);
let b = img.alloc(8).unwrap();
assert_eq!(b, 64, "the edit's own freed space is not reused");
img.write(4, &[1; 8]).unwrap(); // existing bytes 4..8, reused 8..12
let (plan, next) = img.into_plan();
assert_eq!(plan.new_space(4, 12), (false, 8));
assert_eq!(plan.new_space(8, 12), (true, 12));
assert_eq!(plan.new_space(30, 40), (false, 40));
assert_eq!(plan.new_space(64, 72), (true, 72));
assert_eq!(
next.0.iter().map(|(&a, &l)| (a, l)).collect::<Vec<_>>(),
[(32, 16)]
);
}
/// Random reads and writes against a flat copy of the bytes.
#[test]
fn matches_a_flat_model() {
File diff suppressed because it is too large Load Diff
+6 -2
View File
@@ -59,7 +59,7 @@ pub(crate) struct Header {
added: usize,
}
const MAX_CHUNKS: usize = 1024;
const MAX_CHUNKS: usize = 1 << 16;
fn corrupt(why: &'static str) -> Error {
Error::Format(FormatError::InvalidObjectHeader(why))
@@ -118,7 +118,11 @@ impl Header {
});
h.scan(img, 0, addr + 16, addr + 16 + size, &mut pending)?;
}
while let Some((caddr, clen)) = pending.pop() {
// Continuation chunks in the order their messages are found, as
// H5O_protect loads them (so messages keep libhdf5's order).
let mut next = 0;
while let Some(&(caddr, clen)) = pending.get(next) {
next += 1;
if h.chunks.len() >= MAX_CHUNKS {
return Err(corrupt("too many object header chunks"));
}
+29 -3
View File
@@ -98,8 +98,8 @@ fn errors_leave_the_file_untouched() {
let mut ed = FileEditor::open(&path).unwrap();
assert!(matches!(FileEditor::open(&path), Err(Error::Locked(_))));
assert!(ed.write_all("missing", &[0; 4]).is_err());
// Wrong length, wrong type, outside the extent, beyond maxshape,
// shrinking, a rank change.
// Wrong length, wrong type, outside the extent, beyond maxshape, a
// rank change, resizing a dataset that is not chunked.
assert!(matches!(
ed.write_all("flat", &[0; 7]),
Err(Error::InvalidArgument(_))
@@ -116,7 +116,10 @@ fn errors_leave_the_file_untouched() {
ed.resize("raw", &[3, 5]),
Err(Error::InvalidArgument(_))
));
assert!(matches!(ed.resize("ext", &[4]), Err(Error::Unsupported(_))));
assert!(matches!(
ed.resize("flat", &[2]),
Err(Error::Unsupported(_))
));
assert!(matches!(
ed.resize("ext", &[4, 1]),
Err(Error::InvalidArgument(_))
@@ -136,3 +139,26 @@ fn errors_leave_the_file_untouched() {
drop(ed);
assert!(std::fs::read(&path).unwrap() == before);
}
/// Shrinking and growing again on a file clawhdf5 wrote: elements that come
/// back read as the fill value, the ones kept keep their values.
#[test]
fn shrink_then_grow_reads_fill() {
let dir = tempfile::tempdir().unwrap();
let path = sample(dir.path());
{
let mut ed = FileEditor::open(&path).unwrap();
ed.resize("ext", &[2]).unwrap();
ed.resize("ext", &[9]).unwrap();
ed.resize("raw", &[1, 4]).unwrap();
ed.resize("raw", &[3, 4]).unwrap();
}
let f = File::open(&path).unwrap();
assert_eq!(
f.dataset("ext").unwrap().read_i32().unwrap(),
[0, 1, 0, 0, 0, 0, 0, 0, 0]
);
let mut raw = vec![0.0f64; 12];
raw[..4].fill(0.5);
assert_eq!(f.dataset("raw").unwrap().read_f64().unwrap(), raw);
}
+321
View File
@@ -0,0 +1,321 @@
//! Fletcher-32 against libhdf5.
//!
//! libhdf5's `H5_checksum_fletcher32` reduces its sums with the
//! ones'-complement fold `(s & 0xffff) + (s >> 16)`, which leaves 0xffff
//! where `% 65535` leaves 0. Our checksum once used `% 65535`, so on about
//! one chunk in 32768 (a sum that is a non-zero multiple of 65535) libhdf5
//! rejected the chunks we wrote and we rejected the chunks it wrote.
//!
//! - The checksum is compared with libhdf5's own `H5_checksum_fletcher32`,
//! called through ctypes from the library h5py loads, over every one-byte
//! and two-byte input and a large corpus of random and fold-heavy inputs.
//! - Chunks engineered to hit the fold are written by `FileBuilder` and by
//! `FileEditor` and read by h5py, and written by h5py and read by us.
//!
//! Skipped when python3 with h5py is unavailable, unless
//! `CLAWHDF5_REQUIRE_INTEROP=1`.
use std::process::Command;
use clawhdf5::{File, FileBuilder, FileEditor};
use clawhdf5_format::checksum::fletcher32;
fn python() -> String {
std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string())
}
fn have_h5py() -> bool {
let ok = Command::new(python())
.args(["-c", "import h5py, numpy"])
.output()
.is_ok_and(|o| o.status.success());
if !ok {
assert!(
std::env::var("CLAWHDF5_REQUIRE_INTEROP").as_deref() != Ok("1"),
"CLAWHDF5_REQUIRE_INTEROP=1 but python3 with h5py is not available"
);
eprintln!("SKIP: python3 with h5py not available");
}
ok
}
fn run_python(script: &str, args: &[&str]) -> String {
let out = Command::new(python())
.arg("-c")
.arg(script)
.args(args)
.output()
.expect("failed to run python");
assert!(
out.status.success(),
"python failed:\nSTDOUT: {}\nSTDERR: {}",
String::from_utf8_lossy(&out.stdout),
String::from_utf8_lossy(&out.stderr)
);
String::from_utf8_lossy(&out.stdout).trim().to_string()
}
fn tmp(name: &str) -> std::path::PathBuf {
let dir = std::env::temp_dir().join(format!("clawhdf5_fletcher32_{}", std::process::id()));
std::fs::create_dir_all(&dir).unwrap();
dir.join(name)
}
/// The checksum our code computed before it was fixed: each sum reduced
/// `% 65535`. Only used to show that the test data hits the disagreement.
fn fletcher32_mod(data: &[u8]) -> u32 {
let (mut s1, mut s2) = (0u64, 0u64);
for w in data.chunks(2) {
let v = (u64::from(w[0]) << 8) | w.get(1).map_or(0, |&b| u64::from(b));
s1 = (s1 + v) % 65535;
s2 = (s2 + s1) % 65535;
}
((s2 as u32) << 16) | s1 as u32
}
/// Splitmix64, so the data is the same on every run.
struct Rng(u64);
impl Rng {
fn next(&mut self) -> u64 {
self.0 = self.0.wrapping_add(0x9e37_79b9_7f4a_7c15);
let mut z = self.0;
z = (z ^ (z >> 30)).wrapping_mul(0xbf58_476d_1ce4_e5b9);
z = (z ^ (z >> 27)).wrapping_mul(0x94d0_49bb_1331_11eb);
z ^ (z >> 31)
}
}
/// Writes every case of the file `argv[1]` (u32 LE length + bytes) back to
/// `argv[2]` as libhdf5's checksum of each, u32 LE.
const LIBHDF5_CHECKSUMS: &str = r#"
import ctypes, glob, os, struct, sys
import h5py
cands = glob.glob(os.path.join(os.path.dirname(h5py.__file__), os.pardir, 'h5py.libs', 'libhdf5-*.so*'))
cands += glob.glob(os.path.join(os.path.dirname(h5py.__file__), '.dylibs', 'libhdf5*.dylib'))
if cands:
lib = ctypes.CDLL(cands[0])
else:
# A system h5py links the system libhdf5, already loaded.
import h5py.h5
lib = ctypes.CDLL(h5py.h5.__file__)
f = lib.H5_checksum_fletcher32
f.restype = ctypes.c_uint32
f.argtypes = [ctypes.c_char_p, ctypes.c_size_t]
data = open(sys.argv[1], 'rb').read()
out = bytearray()
i = 0
while i < len(data):
(n,) = struct.unpack_from('<I', data, i)
i += 4
b = data[i:i + n]
i += n
out += struct.pack('<I', f(b, n))
open(sys.argv[2], 'wb').write(out)
"#;
#[test]
fn checksum_matches_libhdf5() {
if !have_h5py() {
return;
}
let mut cases: Vec<Vec<u8>> = Vec::new();
// Every one-byte input (the odd-length path alone) and every one-word
// input (65535 = 0xffff is the smallest fold).
cases.extend((0..=255u8).map(|b| vec![b]));
cases.extend((0..=u16::MAX).map(|w| w.to_be_bytes().to_vec()));
let mut rng = Rng(0x5eed_f1e7);
// Words drawn from values that make multiples of 65535 frequent, at
// lengths around the 360-word block boundaries, odd and even.
const FOLDY: [u16; 6] = [0, 1, 0xfffe, 0xffff, 0x8000, 0x7fff];
for _ in 0..40_000 {
let len = match rng.next() % 4 {
0 => (rng.next() % 16) as usize,
1 => 718 + (rng.next() % 6) as usize,
2 => 1438 + (rng.next() % 6) as usize,
_ => (rng.next() % 3000) as usize,
};
let foldy = rng.next().is_multiple_of(2);
let mut v = Vec::with_capacity(len + 1);
while v.len() < len {
let w = if foldy {
FOLDY[(rng.next() % 6) as usize]
} else {
rng.next() as u16
};
v.extend_from_slice(&w.to_be_bytes());
}
v.truncate(len);
cases.push(v);
}
// Long runs of 0xff: sums are multiples of 65535 at every block.
for len in [720, 721, 1440, 1441, 7200, 65536, 65537] {
cases.push(vec![0xff; len]);
}
let mut blob = Vec::new();
for c in &cases {
blob.extend_from_slice(&(c.len() as u32).to_le_bytes());
blob.extend_from_slice(c);
}
let input = tmp("cases.bin");
let output = tmp("sums.bin");
std::fs::write(&input, &blob).unwrap();
run_python(
LIBHDF5_CHECKSUMS,
&[input.to_str().unwrap(), output.to_str().unwrap()],
);
let sums = std::fs::read(&output).unwrap();
assert_eq!(sums.len(), cases.len() * 4);
let mut folds = 0;
for (c, s) in cases.iter().zip(sums.as_chunks::<4>().0) {
let want = u32::from_le_bytes(*s);
assert_eq!(
fletcher32(c),
want,
"checksum of {} bytes {:02x?}...",
c.len(),
&c[..c.len().min(16)]
);
if fletcher32_mod(c) != want {
folds += 1;
}
}
// The corpus must exercise the case `% 65535` got wrong.
assert!(folds > 500, "only {folds} fold cases");
}
const CHUNK: usize = 8;
/// `n` chunks of `CHUNK` bytes, each one a chunk on which the old
/// `% 65535` checksum and libhdf5's differ (sum1, sum2 or both a non-zero
/// multiple of 65535), with an ordinary chunk between them.
fn fold_chunks(n: usize) -> Vec<u8> {
let mut rng = Rng(42);
let mut out = Vec::new();
let mut found = 0;
while found < n {
// Build a chunk whose sum1 is a multiple of 65535 half the time,
// otherwise search at random for a sum2 fold.
let mut c: Vec<u8> = (0..CHUNK).map(|_| rng.next() as u8).collect();
if found % 2 == 0 {
let words: u64 = c[..CHUNK - 2]
.chunks(2)
.map(|w| (u64::from(w[0]) << 8) | u64::from(w[1]))
.sum();
let last = ((65535 - words % 65535) % 65535) as u16;
c[CHUNK - 2..].copy_from_slice(&last.to_be_bytes());
}
if fletcher32(&c) != fletcher32_mod(&c) {
out.extend_from_slice(&c);
out.extend((0..CHUNK).map(|i| i as u8 + 1));
found += 1;
}
}
out
}
#[test]
fn h5py_reads_fold_case_chunks_we_write() {
if !have_h5py() {
return;
}
let data = fold_chunks(32);
// FileBuilder.
let built = tmp("built.h5");
let mut b = FileBuilder::new();
b.create_dataset("d")
.with_u8_data(&data)
.with_chunks(&[CHUNK as u64])
.with_fletcher32();
b.write(&built).unwrap();
// FileEditor, into a dataset h5py created.
let edited = tmp("edited.h5");
run_python(
"import sys, h5py, numpy as np\n\
with h5py.File(sys.argv[1], 'w') as f:\n\
\x20 f.create_dataset('d', data=np.zeros(int(sys.argv[2]), 'u1'), chunks=(8,), fletcher32=True)",
&[edited.to_str().unwrap(), &data.len().to_string()],
);
FileEditor::open(&edited)
.unwrap()
.write_all("d", &data)
.unwrap();
for path in [&built, &edited] {
let got = run_python(
"import sys, h5py\n\
with h5py.File(sys.argv[1], 'r') as f:\n\
\x20 assert f['d'].fletcher32\n\
\x20 print(f['d'][:].tobytes().hex())",
&[path.to_str().unwrap()],
);
assert_eq!(got, hex(&data), "{}", path.display());
}
}
#[test]
fn we_read_fold_case_chunks_h5py_writes() {
if !have_h5py() {
return;
}
let data = fold_chunks(32);
let path = tmp("h5py.h5");
run_python(
"import sys, h5py, numpy as np\n\
with h5py.File(sys.argv[1], 'w') as f:\n\
\x20 f.create_dataset('d', data=np.frombuffer(bytes.fromhex(sys.argv[2]), 'u1'), chunks=(8,), fletcher32=True)",
&[path.to_str().unwrap(), &hex(&data)],
);
let file = File::open(&path).unwrap();
let ds = file.dataset("d").unwrap();
assert_eq!(
ds.read_selection(&clawhdf5_format::selection::Selection::All)
.unwrap(),
data
);
}
/// A checksum stored with the bytes of each 16-bit half swapped, as
/// libhdf5 1.6.2 and earlier wrote it, is accepted as libhdf5 accepts it;
/// so is the `% 65535` form clawhdf5 v2.7.0 and earlier wrote, so that
/// their files stay readable.
#[test]
fn legacy_checksums_are_accepted() {
use clawhdf5_format::filter_pipeline::{FILTER_FLETCHER32, FilterDescription, FilterPipeline};
let payload = [1u8, 2, 3, 4, 5];
let sum = fletcher32(&payload);
let swapped = ((sum & 0x00ff_00ff) << 8) | ((sum >> 8) & 0x00ff_00ff);
assert_ne!(sum, swapped);
let pipeline = FilterPipeline {
version: 2,
filters: vec![FilterDescription {
filter_id: FILTER_FLETCHER32,
name: None,
client_data: vec![],
flags: 0,
}],
};
for stored in [sum, swapped] {
let mut chunk = payload.to_vec();
chunk.extend_from_slice(&stored.to_le_bytes());
let out = clawhdf5_format::filters::decompress_chunk(&chunk, &pipeline, payload.len(), 1)
.unwrap();
assert_eq!(out, payload);
}
// Our old checksum of a fold-case chunk.
let fold = fold_chunks(1);
let fold = &fold[..CHUNK];
let old = fletcher32_mod(fold);
assert_ne!(old, fletcher32(fold));
let mut chunk = fold.to_vec();
chunk.extend_from_slice(&old.to_le_bytes());
let out = clawhdf5_format::filters::decompress_chunk(&chunk, &pipeline, CHUNK, 1).unwrap();
assert_eq!(out, fold);
let mut chunk = payload.to_vec();
chunk.extend_from_slice(&(sum ^ 1).to_le_bytes());
assert!(
clawhdf5_format::filters::decompress_chunk(&chunk, &pipeline, payload.len(), 1).is_err()
);
}
fn hex(b: &[u8]) -> String {
b.iter().map(|x| format!("{x:02x}")).collect()
}