Chunked reads beat an h5py process pool; unlimited writer B-trees; Blosc2; 599/697 conformance #16
@@ -461,6 +461,17 @@ fn collect_chunk_info_inner(
|
|||||||
file_data[pos + 7],
|
file_data[pos + 7],
|
||||||
]);
|
]);
|
||||||
let offsets = read_key_offsets(file_data, pos, ndims, chunk_dimensions)?;
|
let offsets = read_key_offsets(file_data, pos, ndims, chunk_dimensions)?;
|
||||||
|
// A chunk's key carries 0 in the element-size dimension. libhdf5
|
||||||
|
// compares that coordinate too when it looks a chunk up
|
||||||
|
// (`H5D__btree_found`), so whether it finds a chunk keyed
|
||||||
|
// otherwise depends on where the key falls; in `cve-2025-44905`
|
||||||
|
// `/Shuffle_float_data_le` (offset 4096) it does not, and h5py
|
||||||
|
// reads fill values there. Such a key is refused here.
|
||||||
|
if chunk_dimensions.is_some() && offsets.last().is_some_and(|&o| o != 0) {
|
||||||
|
return Err(FormatError::ChunkedReadError(format!(
|
||||||
|
"chunk key {offsets:?} has a non-zero element offset"
|
||||||
|
)));
|
||||||
|
}
|
||||||
pos += key_size;
|
pos += key_size;
|
||||||
|
|
||||||
// Parse child address
|
// Parse child address
|
||||||
@@ -820,6 +831,47 @@ pub fn list_chunks(
|
|||||||
Ok((chunks, chunk_dims))
|
Ok((chunks, chunk_dims))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// [`list_chunks`] for reading the chunks through `pipeline`: a dataset
|
||||||
|
/// without filters stores every chunk at the chunk's full size, and a chunk
|
||||||
|
/// the index records at another size is refused, as libhdf5 refuses it
|
||||||
|
/// ("incorrect chunk size returned from index for unfiltered chunk"). Such
|
||||||
|
/// a chunk was read at its recorded size, with the rest of the chunk left
|
||||||
|
/// as zeros or fill values: `cve-2025-44904`'s `Scale_offset_float_data_le`
|
||||||
|
/// has chunks of 38 and 37 bytes for 48-byte chunks, where HDF5 2.0 reads
|
||||||
|
/// whatever its buffer held for the missing bytes.
|
||||||
|
pub fn list_chunks_for_read(
|
||||||
|
file_data: &[u8],
|
||||||
|
layout: &DataLayout,
|
||||||
|
dataspace: &Dataspace,
|
||||||
|
elem_size: usize,
|
||||||
|
pipeline: Option<&FilterPipeline>,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<(Vec<ChunkInfo>, Vec<usize>), FormatError> {
|
||||||
|
let (chunks, chunk_dims) = list_chunks(
|
||||||
|
file_data,
|
||||||
|
layout,
|
||||||
|
dataspace,
|
||||||
|
elem_size,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
)?;
|
||||||
|
if pipeline.is_none_or(|p| p.filters.is_empty()) {
|
||||||
|
let chunk_bytes = checked_chunk_byte_len(&chunk_dims, elem_size)?;
|
||||||
|
if let Some(c) = chunks
|
||||||
|
.iter()
|
||||||
|
.find(|c| c.address != u64::MAX && c.chunk_size as usize != chunk_bytes)
|
||||||
|
{
|
||||||
|
return Err(FormatError::ChunkedReadError(format!(
|
||||||
|
"incorrect chunk size returned from index for unfiltered chunk at {:?}: \
|
||||||
|
{} bytes, expected {chunk_bytes}",
|
||||||
|
c.offsets, c.chunk_size
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok((chunks, chunk_dims))
|
||||||
|
}
|
||||||
|
|
||||||
pub fn read_chunked_data(
|
pub fn read_chunked_data(
|
||||||
file_data: &[u8],
|
file_data: &[u8],
|
||||||
layout: &DataLayout,
|
layout: &DataLayout,
|
||||||
@@ -831,11 +883,12 @@ pub fn read_chunked_data(
|
|||||||
) -> Result<Vec<u8>, FormatError> {
|
) -> Result<Vec<u8>, FormatError> {
|
||||||
check_chunk_element_size(layout, datatype, offset_size)?;
|
check_chunk_element_size(layout, datatype, offset_size)?;
|
||||||
let elem_size = datatype.type_size() as usize;
|
let elem_size = datatype.type_size() as usize;
|
||||||
let (chunks, chunk_dims) = list_chunks(
|
let (chunks, chunk_dims) = list_chunks_for_read(
|
||||||
file_data,
|
file_data,
|
||||||
layout,
|
layout,
|
||||||
dataspace,
|
dataspace,
|
||||||
elem_size,
|
elem_size,
|
||||||
|
pipeline,
|
||||||
offset_size,
|
offset_size,
|
||||||
length_size,
|
length_size,
|
||||||
)?;
|
)?;
|
||||||
@@ -979,11 +1032,12 @@ pub fn read_chunked_data_cached(
|
|||||||
// lookup is keyed by this dataset's chunk-index address, so another
|
// lookup is keyed by this dataset's chunk-index address, so another
|
||||||
// dataset's index or chunks are never used for this read.
|
// dataset's index or chunks are never used for this read.
|
||||||
let chunks = cache.chunks_for(addr, rank, || {
|
let chunks = cache.chunks_for(addr, rank, || {
|
||||||
list_chunks(
|
list_chunks_for_read(
|
||||||
file_data,
|
file_data,
|
||||||
layout,
|
layout,
|
||||||
dataspace,
|
dataspace,
|
||||||
elem_size,
|
elem_size,
|
||||||
|
pipeline,
|
||||||
offset_size,
|
offset_size,
|
||||||
length_size,
|
length_size,
|
||||||
)
|
)
|
||||||
@@ -1290,11 +1344,12 @@ pub fn read_chunked_data_sweep(
|
|||||||
// lookup is keyed by this dataset's chunk-index address, so another
|
// lookup is keyed by this dataset's chunk-index address, so another
|
||||||
// dataset's index or chunks are never used for this read.
|
// dataset's index or chunks are never used for this read.
|
||||||
let chunks = cache.chunks_for(addr, rank, || {
|
let chunks = cache.chunks_for(addr, rank, || {
|
||||||
list_chunks(
|
list_chunks_for_read(
|
||||||
file_data,
|
file_data,
|
||||||
layout,
|
layout,
|
||||||
dataspace,
|
dataspace,
|
||||||
elem_size,
|
elem_size,
|
||||||
|
pipeline,
|
||||||
offset_size,
|
offset_size,
|
||||||
length_size,
|
length_size,
|
||||||
)
|
)
|
||||||
@@ -1431,11 +1486,12 @@ pub fn read_chunked_data_indexed(
|
|||||||
addr,
|
addr,
|
||||||
rank,
|
rank,
|
||||||
|| {
|
|| {
|
||||||
list_chunks(
|
list_chunks_for_read(
|
||||||
file_data,
|
file_data,
|
||||||
layout,
|
layout,
|
||||||
dataspace,
|
dataspace,
|
||||||
elem_size,
|
elem_size,
|
||||||
|
pipeline,
|
||||||
offset_size,
|
offset_size,
|
||||||
length_size,
|
length_size,
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -18,7 +18,7 @@ use alloc::{format, vec, vec::Vec};
|
|||||||
#[cfg(feature = "std")]
|
#[cfg(feature = "std")]
|
||||||
use std::string as alloc_or_std;
|
use std::string as alloc_or_std;
|
||||||
|
|
||||||
use crate::chunked_read::{alloc_output, checked_byte_len, list_chunks};
|
use crate::chunked_read::{alloc_output, checked_byte_len, list_chunks_for_read};
|
||||||
use crate::data_layout::DataLayout;
|
use crate::data_layout::DataLayout;
|
||||||
use crate::data_read::extract_selection_from_buffer;
|
use crate::data_read::extract_selection_from_buffer;
|
||||||
use crate::dataspace::Dataspace;
|
use crate::dataspace::Dataspace;
|
||||||
@@ -294,11 +294,12 @@ pub fn read_selection(
|
|||||||
btree_address: Some(_),
|
btree_address: Some(_),
|
||||||
..
|
..
|
||||||
} => {
|
} => {
|
||||||
let (chunks, chunk_dims) = list_chunks(
|
let (chunks, chunk_dims) = list_chunks_for_read(
|
||||||
file_data,
|
file_data,
|
||||||
layout,
|
layout,
|
||||||
dataspace,
|
dataspace,
|
||||||
elem_size,
|
elem_size,
|
||||||
|
pipeline,
|
||||||
offset_size,
|
offset_size,
|
||||||
length_size,
|
length_size,
|
||||||
)?;
|
)?;
|
||||||
|
|||||||
@@ -628,3 +628,47 @@ save("mdc_past_eof", bad)
|
|||||||
"DSET",
|
"DSET",
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Chunk index entries HDF5 2.0 mis-reads, refused here. An unfiltered
|
||||||
|
/// chunk the index records at less than the chunk's size (`cve-2025-44904`):
|
||||||
|
/// HDF5 2.0 fills the rest of the chunk with whatever its buffer held, and
|
||||||
|
/// later libhdf5 releases refuse it ("incorrect chunk size returned from
|
||||||
|
/// index for unfiltered chunk"); we read the rest as zeros. A chunk keyed
|
||||||
|
/// with a non-zero element offset (`cve-2025-44905`): libhdf5's lookup
|
||||||
|
/// compares that coordinate too, so whether it finds the chunk depends on
|
||||||
|
/// where the key falls (in `cve-2025-44905` it does not, and h5py reads
|
||||||
|
/// fill values; in this file it does); we read the chunk.
|
||||||
|
#[test]
|
||||||
|
fn chunk_index_entries_libhdf5_misreads_are_refused() {
|
||||||
|
skip_if_no_python!();
|
||||||
|
let dir = tempfile::tempdir().unwrap();
|
||||||
|
run_python(
|
||||||
|
dir.path(),
|
||||||
|
r#"
|
||||||
|
good = os.path.join(d, "good.h5")
|
||||||
|
with h5py.File(good, "w", libver="earliest") as f:
|
||||||
|
f.create_dataset("d", data=np.arange(100, dtype="<i4"), chunks=(37,), fillvalue=-1)
|
||||||
|
data = bytearray(open(good, "rb").read())
|
||||||
|
tree = data.find(b"TREE")
|
||||||
|
while data[tree + 4] != 1: # the chunk index, not the root group's B-tree
|
||||||
|
tree = data.find(b"TREE", tree + 1)
|
||||||
|
assert tree > 0 and data[tree + 5] == 0
|
||||||
|
key = tree + 8 + 16 # the first chunk's key: size, filter mask, offsets
|
||||||
|
second = key + 24 + 8 # a key (4 + 4 + 2 x 8 bytes), then a child address
|
||||||
|
assert struct.unpack_from("<IIQQ", data, second) == (148, 0, 37, 0)
|
||||||
|
bad = bytearray(data); struct.pack_into("<I", bad, second, 100)
|
||||||
|
open(os.path.join(d, "short_chunk.h5"), "wb").write(bad)
|
||||||
|
bad = bytearray(data); struct.pack_into("<Q", bad, second + 16, 4096)
|
||||||
|
open(os.path.join(d, "element_offset.h5"), "wb").write(bad)
|
||||||
|
"#,
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
clawhdf5_reads(&dir.path().join("good.h5"), "d"),
|
||||||
|
Ok(()),
|
||||||
|
"good"
|
||||||
|
);
|
||||||
|
for name in ["short_chunk", "element_offset"] {
|
||||||
|
let err = clawhdf5_reads(&dir.path().join(format!("{name}.h5")), "d").unwrap_err();
|
||||||
|
assert!(err.starts_with("read:"), "{name}: {err}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user