h5rs tools, browser reader, libhdf5 header checks, plugin filters, concurrency benchmark #14
@@ -359,6 +359,16 @@
|
|||||||
to each check. **Breaking (format crate):** `FormatError` gained
|
to each check. **Breaking (format crate):** `FormatError` gained
|
||||||
`InvalidObjectHeader`, `InvalidDatatype`, `InvalidChunkDimensions` and
|
`InvalidObjectHeader`, `InvalidDatatype`, `InvalidChunkDimensions` and
|
||||||
`TruncatedFile`; an exhaustive `match` on it needs the new arms.
|
`TruncatedFile`; an exhaustive `match` on it needs the new arms.
|
||||||
|
- **Chunked datasets whose chunk dimensions take 3, 5, 6 or 7 bytes did not
|
||||||
|
open.** A version-4 layout (`libver="latest"`) stores each chunk dimension
|
||||||
|
in the fewest bytes that hold the largest one, so a chunk dimension from
|
||||||
|
65 536 to 16 777 215 (e.g. h5py `chunks=(70000,)`) takes 3 bytes; only 1, 2,
|
||||||
|
4 and 8 were read, and the rest failed with `UnexpectedEof`. Widths 1-8 are
|
||||||
|
read now, and 0 or more than 8 is refused as libhdf5 refuses it. A width
|
||||||
|
larger than needed is accepted: HDF5 2.0.0 refuses one ("stored chunk
|
||||||
|
dimension encoding length does not match"), but libhdf5 since
|
||||||
|
HDFGroup/hdf5@e124c36 (2026-06-05) reads it, and clawhdf5 itself wrote such
|
||||||
|
layouts.
|
||||||
- `clawhdf5-format` VDS: variable-length and reference data from a source in
|
- `clawhdf5-format` VDS: variable-length and reference data from a source in
|
||||||
another file is refused. Those elements are global-heap IDs and object
|
another file is refused. Those elements are global-heap IDs and object
|
||||||
addresses in the source file; copied into the virtual dataset they would
|
addresses in the source file; copied into the virtual dataset they would
|
||||||
|
|||||||
@@ -540,43 +540,30 @@ impl DataLayout {
|
|||||||
));
|
));
|
||||||
}
|
}
|
||||||
|
|
||||||
// dimension sizes
|
// Each dimension takes 1 to 8 bytes (libhdf5 writes the
|
||||||
|
// fewest that hold the largest one, so 3, 5, 6 and 7 occur:
|
||||||
|
// a chunk dimension of 70 000 takes 3). libhdf5 refuses 0
|
||||||
|
// and more than 8.
|
||||||
|
if dim_size_encoded_length == 0 || dim_size_encoded_length > 8 {
|
||||||
|
return Err(FormatError::InvalidChunkDimensions(
|
||||||
|
"encoded chunk dimension size is too large".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
ensure_len(data, p, dimensionality * dim_size_encoded_length)?;
|
ensure_len(data, p, dimensionality * dim_size_encoded_length)?;
|
||||||
let mut chunk_dimensions = Vec::with_capacity(dimensionality);
|
let mut chunk_dimensions = Vec::with_capacity(dimensionality);
|
||||||
for _ in 0..dimensionality {
|
for _ in 0..dimensionality {
|
||||||
let val = match dim_size_encoded_length {
|
let val = data[p..p + dim_size_encoded_length]
|
||||||
1 => data[p] as u32,
|
.iter()
|
||||||
2 => u16::from_le_bytes([data[p], data[p + 1]]) as u32,
|
.rev()
|
||||||
4 => u32::from_le_bytes([data[p], data[p + 1], data[p + 2], data[p + 3]]),
|
.fold(0u64, |acc, &b| (acc << 8) | u64::from(b));
|
||||||
8 => {
|
// Chunk dimensions are held as u32; HDF5 2.0 can write
|
||||||
// V4 chunked encodes dimension sizes as 8 bytes, but
|
// larger ones (layout version 5), which are refused
|
||||||
// our ChunkedStorageV4 stores them as u32. We read only
|
// rather than truncated.
|
||||||
// the low 4 bytes (little-endian). This silently
|
let val = u32::try_from(val).map_err(|_| {
|
||||||
// truncates dimensions > 4 GiB, which are not expected
|
FormatError::InvalidChunkDimensions(format!(
|
||||||
// in practice (HDF5 chunk dimensions are always small).
|
"chunk dimension {val} is larger than 2^32 - 1, which is not supported"
|
||||||
// If the high bytes are non-zero, the file is malformed
|
))
|
||||||
// or uses dimensions we cannot represent.
|
})?;
|
||||||
let high = u32::from_le_bytes([
|
|
||||||
data[p + 4],
|
|
||||||
data[p + 5],
|
|
||||||
data[p + 6],
|
|
||||||
data[p + 7],
|
|
||||||
]);
|
|
||||||
if high != 0 {
|
|
||||||
return Err(FormatError::UnexpectedEof {
|
|
||||||
expected: p + 8,
|
|
||||||
available: data.len(),
|
|
||||||
});
|
|
||||||
}
|
|
||||||
u32::from_le_bytes([data[p], data[p + 1], data[p + 2], data[p + 3]])
|
|
||||||
}
|
|
||||||
_ => {
|
|
||||||
return Err(FormatError::UnexpectedEof {
|
|
||||||
expected: p + dim_size_encoded_length,
|
|
||||||
available: data.len(),
|
|
||||||
});
|
|
||||||
}
|
|
||||||
};
|
|
||||||
chunk_dimensions.push(val);
|
chunk_dimensions.push(val);
|
||||||
p += dim_size_encoded_length;
|
p += dim_size_encoded_length;
|
||||||
}
|
}
|
||||||
@@ -838,6 +825,51 @@ mod tests {
|
|||||||
));
|
));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// A v4 chunked layout (fixed array index) whose `dims` are each
|
||||||
|
/// encoded in `width` bytes.
|
||||||
|
fn v4_chunked_msg(width: u8, dims: &[u64]) -> Vec<u8> {
|
||||||
|
let mut m = vec![4u8, 2, 0, dims.len() as u8, width];
|
||||||
|
for &d in dims {
|
||||||
|
m.extend_from_slice(&d.to_le_bytes()[..width.min(8) as usize]);
|
||||||
|
}
|
||||||
|
m.push(3); // fixed array index
|
||||||
|
m.push(0); // page bits
|
||||||
|
m.extend_from_slice(&0x1000u64.to_le_bytes());
|
||||||
|
m
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn v4_chunk_dimensions_take_1_to_8_bytes() {
|
||||||
|
// libhdf5 encodes each dimension in the fewest bytes that hold the
|
||||||
|
// largest: a chunk dimension of 70 000 takes 3, and 3, 5, 6 and 7
|
||||||
|
// were refused ("UnexpectedEof").
|
||||||
|
for width in 1..=8u8 {
|
||||||
|
let dims = [if width >= 3 { 70_000 } else { 200 }, 8];
|
||||||
|
let layout = DataLayout::parse(&v4_chunked_msg(width, &dims), 8, 8)
|
||||||
|
.unwrap_or_else(|e| panic!("width {width}: {e:?}"));
|
||||||
|
assert!(
|
||||||
|
matches!(&layout, DataLayout::Chunked { chunk_dimensions, .. }
|
||||||
|
if chunk_dimensions.iter().map(|&d| u64::from(d)).eq(dims)),
|
||||||
|
"width {width}: {layout:?}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
// libhdf5 refuses 0 and more than 8 bytes.
|
||||||
|
for width in [0u8, 9] {
|
||||||
|
assert_eq!(
|
||||||
|
DataLayout::parse(&v4_chunked_msg(width, &[4, 8]), 8, 8).unwrap_err(),
|
||||||
|
FormatError::InvalidChunkDimensions(
|
||||||
|
"encoded chunk dimension size is too large".into()
|
||||||
|
)
|
||||||
|
);
|
||||||
|
}
|
||||||
|
// A dimension past u32 cannot be represented and is refused, not
|
||||||
|
// truncated.
|
||||||
|
assert!(matches!(
|
||||||
|
DataLayout::parse(&v4_chunked_msg(5, &[1 << 32, 8]), 8, 8),
|
||||||
|
Err(FormatError::InvalidChunkDimensions(m)) if m.contains("2^32")
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn v1v2_rejects_bad_class_dimensionality_and_truncation() {
|
fn v1v2_rejects_bad_class_dimensionality_and_truncation() {
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
|
|||||||
@@ -343,3 +343,41 @@ print("OK")
|
|||||||
let want: Vec<u8> = line[990..].iter().flat_map(|v| v.to_le_bytes()).collect();
|
let want: Vec<u8> = line[990..].iter().flat_map(|v| v.to_le_bytes()).collect();
|
||||||
assert_eq!(tail, want);
|
assert_eq!(tail, want);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Chunk dimension widths in layout version 4
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// A version-4 layout stores every chunk dimension in the fewest bytes that
|
||||||
|
/// hold the largest one (the element size included). A chunk dimension of
|
||||||
|
/// 70 000 takes 3 bytes and 2^32 + 1 elements would take 5; widths other
|
||||||
|
/// than 1, 2, 4 and 8 were refused, so these h5py files did not open.
|
||||||
|
#[test]
|
||||||
|
fn h5py_layout_v4_chunk_dimensions_of_3_bytes_read() {
|
||||||
|
skip_if_no_python!();
|
||||||
|
let dir = tempfile::tempdir().unwrap();
|
||||||
|
let path = dir.path().join("enc3.h5");
|
||||||
|
let p = path.display().to_string();
|
||||||
|
run_python(&format!(
|
||||||
|
r#"
|
||||||
|
import h5py, numpy as np
|
||||||
|
with h5py.File("{p}", "w", libver="latest") as f:
|
||||||
|
f.create_dataset("single", data=np.arange(70000, dtype="<u4"), chunks=(70000,))
|
||||||
|
ds = f.create_dataset("ea", shape=(10,), maxshape=(None,), chunks=(70000,), dtype="<f8")
|
||||||
|
ds[:] = np.arange(10.0)
|
||||||
|
f.create_dataset("fa", data=np.arange(3 * 70000, dtype="<i4").reshape(3, 70000),
|
||||||
|
chunks=(1, 70000), compression="gzip")
|
||||||
|
with h5py.File("{p}", "r") as f:
|
||||||
|
raw = open("{p}", "rb").read()
|
||||||
|
# version 4, class 2 (chunked), flags, 2 or 3 dims, then 3 bytes each
|
||||||
|
assert raw.find(bytes([4, 2, 0, 2, 3])) > 0 or raw.find(bytes([4, 2, 1, 2, 3])) > 0
|
||||||
|
"#
|
||||||
|
));
|
||||||
|
let file = File::open(&path).unwrap();
|
||||||
|
let single = file.dataset("single").unwrap().read_u64().unwrap();
|
||||||
|
assert!(single.iter().copied().eq(0..70000), "single");
|
||||||
|
let ea = file.dataset("ea").unwrap().read_f64().unwrap();
|
||||||
|
assert_eq!(ea, (0..10).map(f64::from).collect::<Vec<_>>());
|
||||||
|
let fa = file.dataset("fa").unwrap().read_i32().unwrap();
|
||||||
|
assert!(fa.iter().copied().eq(0..3 * 70000), "fa");
|
||||||
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user