Chunked reads beat an h5py process pool; unlimited writer B-trees; Blosc2; 599/697 conformance #16

Merged
osobh merged 48 commits from feat/p2b-scale into main 2026-09-26 17:42:16 +00:00
3 changed files with 107 additions and 6 deletions
Showing only changes of commit 6b3d003950 - Show all commits
+60 -4
View File
@@ -461,6 +461,17 @@ fn collect_chunk_info_inner(
file_data[pos + 7],
]);
let offsets = read_key_offsets(file_data, pos, ndims, chunk_dimensions)?;
// A chunk's key carries 0 in the element-size dimension. libhdf5
// compares that coordinate too when it looks a chunk up
// (`H5D__btree_found`), so whether it finds a chunk keyed
// otherwise depends on where the key falls; in `cve-2025-44905`
// `/Shuffle_float_data_le` (offset 4096) it does not, and h5py
// reads fill values there. Such a key is refused here.
if chunk_dimensions.is_some() && offsets.last().is_some_and(|&o| o != 0) {
return Err(FormatError::ChunkedReadError(format!(
"chunk key {offsets:?} has a non-zero element offset"
)));
}
pos += key_size;
// Parse child address
@@ -820,6 +831,47 @@ pub fn list_chunks(
Ok((chunks, chunk_dims))
}
/// [`list_chunks`] for reading the chunks through `pipeline`: a dataset
/// without filters stores every chunk at the chunk's full size, and a chunk
/// the index records at another size is refused, as libhdf5 refuses it
/// ("incorrect chunk size returned from index for unfiltered chunk"). Such
/// a chunk was read at its recorded size, with the rest of the chunk left
/// as zeros or fill values: `cve-2025-44904`'s `Scale_offset_float_data_le`
/// has chunks of 38 and 37 bytes for 48-byte chunks, where HDF5 2.0 reads
/// whatever its buffer held for the missing bytes.
pub fn list_chunks_for_read(
file_data: &[u8],
layout: &DataLayout,
dataspace: &Dataspace,
elem_size: usize,
pipeline: Option<&FilterPipeline>,
offset_size: u8,
length_size: u8,
) -> Result<(Vec<ChunkInfo>, Vec<usize>), FormatError> {
let (chunks, chunk_dims) = list_chunks(
file_data,
layout,
dataspace,
elem_size,
offset_size,
length_size,
)?;
if pipeline.is_none_or(|p| p.filters.is_empty()) {
let chunk_bytes = checked_chunk_byte_len(&chunk_dims, elem_size)?;
if let Some(c) = chunks
.iter()
.find(|c| c.address != u64::MAX && c.chunk_size as usize != chunk_bytes)
{
return Err(FormatError::ChunkedReadError(format!(
"incorrect chunk size returned from index for unfiltered chunk at {:?}: \
{} bytes, expected {chunk_bytes}",
c.offsets, c.chunk_size
)));
}
}
Ok((chunks, chunk_dims))
}
pub fn read_chunked_data(
file_data: &[u8],
layout: &DataLayout,
@@ -831,11 +883,12 @@ pub fn read_chunked_data(
) -> Result<Vec<u8>, FormatError> {
check_chunk_element_size(layout, datatype, offset_size)?;
let elem_size = datatype.type_size() as usize;
let (chunks, chunk_dims) = list_chunks(
let (chunks, chunk_dims) = list_chunks_for_read(
file_data,
layout,
dataspace,
elem_size,
pipeline,
offset_size,
length_size,
)?;
@@ -979,11 +1032,12 @@ pub fn read_chunked_data_cached(
// lookup is keyed by this dataset's chunk-index address, so another
// dataset's index or chunks are never used for this read.
let chunks = cache.chunks_for(addr, rank, || {
list_chunks(
list_chunks_for_read(
file_data,
layout,
dataspace,
elem_size,
pipeline,
offset_size,
length_size,
)
@@ -1290,11 +1344,12 @@ pub fn read_chunked_data_sweep(
// lookup is keyed by this dataset's chunk-index address, so another
// dataset's index or chunks are never used for this read.
let chunks = cache.chunks_for(addr, rank, || {
list_chunks(
list_chunks_for_read(
file_data,
layout,
dataspace,
elem_size,
pipeline,
offset_size,
length_size,
)
@@ -1431,11 +1486,12 @@ pub fn read_chunked_data_indexed(
addr,
rank,
|| {
list_chunks(
list_chunks_for_read(
file_data,
layout,
dataspace,
elem_size,
pipeline,
offset_size,
length_size,
)
+3 -2
View File
@@ -18,7 +18,7 @@ use alloc::{format, vec, vec::Vec};
#[cfg(feature = "std")]
use std::string as alloc_or_std;
use crate::chunked_read::{alloc_output, checked_byte_len, list_chunks};
use crate::chunked_read::{alloc_output, checked_byte_len, list_chunks_for_read};
use crate::data_layout::DataLayout;
use crate::data_read::extract_selection_from_buffer;
use crate::dataspace::Dataspace;
@@ -294,11 +294,12 @@ pub fn read_selection(
btree_address: Some(_),
..
} => {
let (chunks, chunk_dims) = list_chunks(
let (chunks, chunk_dims) = list_chunks_for_read(
file_data,
layout,
dataspace,
elem_size,
pipeline,
offset_size,
length_size,
)?;
@@ -628,3 +628,47 @@ save("mdc_past_eof", bad)
"DSET",
);
}
/// Chunk index entries HDF5 2.0 mis-reads, refused here. An unfiltered
/// chunk the index records at less than the chunk's size (`cve-2025-44904`):
/// HDF5 2.0 fills the rest of the chunk with whatever its buffer held, and
/// later libhdf5 releases refuse it ("incorrect chunk size returned from
/// index for unfiltered chunk"); we read the rest as zeros. A chunk keyed
/// with a non-zero element offset (`cve-2025-44905`): libhdf5's lookup
/// compares that coordinate too, so whether it finds the chunk depends on
/// where the key falls (in `cve-2025-44905` it does not, and h5py reads
/// fill values; in this file it does); we read the chunk.
#[test]
fn chunk_index_entries_libhdf5_misreads_are_refused() {
skip_if_no_python!();
let dir = tempfile::tempdir().unwrap();
run_python(
dir.path(),
r#"
good = os.path.join(d, "good.h5")
with h5py.File(good, "w", libver="earliest") as f:
f.create_dataset("d", data=np.arange(100, dtype="<i4"), chunks=(37,), fillvalue=-1)
data = bytearray(open(good, "rb").read())
tree = data.find(b"TREE")
while data[tree + 4] != 1: # the chunk index, not the root group's B-tree
tree = data.find(b"TREE", tree + 1)
assert tree > 0 and data[tree + 5] == 0
key = tree + 8 + 16 # the first chunk's key: size, filter mask, offsets
second = key + 24 + 8 # a key (4 + 4 + 2 x 8 bytes), then a child address
assert struct.unpack_from("<IIQQ", data, second) == (148, 0, 37, 0)
bad = bytearray(data); struct.pack_into("<I", bad, second, 100)
open(os.path.join(d, "short_chunk.h5"), "wb").write(bad)
bad = bytearray(data); struct.pack_into("<Q", bad, second + 16, 4096)
open(os.path.join(d, "element_offset.h5"), "wb").write(bad)
"#,
);
assert_eq!(
clawhdf5_reads(&dir.path().join("good.h5"), "d"),
Ok(()),
"good"
);
for name in ["short_chunk", "element_offset"] {
let err = clawhdf5_reads(&dir.path().join(format!("{name}.h5")), "d").unwrap_err();
assert!(err.starts_with("read:"), "{name}: {err}");
}
}