fix(format): refuse chunk index entries libhdf5 mis-reads

- A dataset without filters stores every chunk at the chunk's full size.
  A chunk its index records at another size was read at that size, with
  the rest of the chunk left as zeros (cve-2025-44904's
  Scale_offset_float_data_le: 38- and 37-byte chunks for 48-byte chunks,
  where HDF5 2.0 fills the rest with whatever its buffer held). It is now
  refused, as later libhdf5 releases refuse it ("incorrect chunk size
  returned from index for unfiltered chunk"):
  chunked_read::list_chunks_for_read, used by every read path.
- A v1 B-tree chunk key carries 0 in the element-size dimension. libhdf5
  compares that coordinate when it looks a chunk up, so whether it finds a
  chunk keyed otherwise depends on where the key falls (in cve-2025-44905
  /Shuffle_float_data_le, offset 4096, it does not, and h5py reads fill
  values); we read the chunk. Such a key is now refused.

Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
osobh
2026-09-26 10:36:01 -05:00
co-authored by Claude Opus 5.5
parent d110b1d945
commit 6b3d003950
3 changed files with 107 additions and 6 deletions
+60 -4
View File
@@ -461,6 +461,17 @@ fn collect_chunk_info_inner(
file_data[pos + 7],
]);
let offsets = read_key_offsets(file_data, pos, ndims, chunk_dimensions)?;
// A chunk's key carries 0 in the element-size dimension. libhdf5
// compares that coordinate too when it looks a chunk up
// (`H5D__btree_found`), so whether it finds a chunk keyed
// otherwise depends on where the key falls; in `cve-2025-44905`
// `/Shuffle_float_data_le` (offset 4096) it does not, and h5py
// reads fill values there. Such a key is refused here.
if chunk_dimensions.is_some() && offsets.last().is_some_and(|&o| o != 0) {
return Err(FormatError::ChunkedReadError(format!(
"chunk key {offsets:?} has a non-zero element offset"
)));
}
pos += key_size;
// Parse child address
@@ -820,6 +831,47 @@ pub fn list_chunks(
Ok((chunks, chunk_dims))
}
/// [`list_chunks`] for reading the chunks through `pipeline`: a dataset
/// without filters stores every chunk at the chunk's full size, and a chunk
/// the index records at another size is refused, as libhdf5 refuses it
/// ("incorrect chunk size returned from index for unfiltered chunk"). Such
/// a chunk was read at its recorded size, with the rest of the chunk left
/// as zeros or fill values: `cve-2025-44904`'s `Scale_offset_float_data_le`
/// has chunks of 38 and 37 bytes for 48-byte chunks, where HDF5 2.0 reads
/// whatever its buffer held for the missing bytes.
pub fn list_chunks_for_read(
file_data: &[u8],
layout: &DataLayout,
dataspace: &Dataspace,
elem_size: usize,
pipeline: Option<&FilterPipeline>,
offset_size: u8,
length_size: u8,
) -> Result<(Vec<ChunkInfo>, Vec<usize>), FormatError> {
let (chunks, chunk_dims) = list_chunks(
file_data,
layout,
dataspace,
elem_size,
offset_size,
length_size,
)?;
if pipeline.is_none_or(|p| p.filters.is_empty()) {
let chunk_bytes = checked_chunk_byte_len(&chunk_dims, elem_size)?;
if let Some(c) = chunks
.iter()
.find(|c| c.address != u64::MAX && c.chunk_size as usize != chunk_bytes)
{
return Err(FormatError::ChunkedReadError(format!(
"incorrect chunk size returned from index for unfiltered chunk at {:?}: \
{} bytes, expected {chunk_bytes}",
c.offsets, c.chunk_size
)));
}
}
Ok((chunks, chunk_dims))
}
pub fn read_chunked_data(
file_data: &[u8],
layout: &DataLayout,
@@ -831,11 +883,12 @@ pub fn read_chunked_data(
) -> Result<Vec<u8>, FormatError> {
check_chunk_element_size(layout, datatype, offset_size)?;
let elem_size = datatype.type_size() as usize;
let (chunks, chunk_dims) = list_chunks(
let (chunks, chunk_dims) = list_chunks_for_read(
file_data,
layout,
dataspace,
elem_size,
pipeline,
offset_size,
length_size,
)?;
@@ -979,11 +1032,12 @@ pub fn read_chunked_data_cached(
// lookup is keyed by this dataset's chunk-index address, so another
// dataset's index or chunks are never used for this read.
let chunks = cache.chunks_for(addr, rank, || {
list_chunks(
list_chunks_for_read(
file_data,
layout,
dataspace,
elem_size,
pipeline,
offset_size,
length_size,
)
@@ -1290,11 +1344,12 @@ pub fn read_chunked_data_sweep(
// lookup is keyed by this dataset's chunk-index address, so another
// dataset's index or chunks are never used for this read.
let chunks = cache.chunks_for(addr, rank, || {
list_chunks(
list_chunks_for_read(
file_data,
layout,
dataspace,
elem_size,
pipeline,
offset_size,
length_size,
)
@@ -1431,11 +1486,12 @@ pub fn read_chunked_data_indexed(
addr,
rank,
|| {
list_chunks(
list_chunks_for_read(
file_data,
layout,
dataspace,
elem_size,
pipeline,
offset_size,
length_size,
)
+3 -2
View File
@@ -18,7 +18,7 @@ use alloc::{format, vec, vec::Vec};
#[cfg(feature = "std")]
use std::string as alloc_or_std;
use crate::chunked_read::{alloc_output, checked_byte_len, list_chunks};
use crate::chunked_read::{alloc_output, checked_byte_len, list_chunks_for_read};
use crate::data_layout::DataLayout;
use crate::data_read::extract_selection_from_buffer;
use crate::dataspace::Dataspace;
@@ -294,11 +294,12 @@ pub fn read_selection(
btree_address: Some(_),
..
} => {
let (chunks, chunk_dims) = list_chunks(
let (chunks, chunk_dims) = list_chunks_for_read(
file_data,
layout,
dataspace,
elem_size,
pipeline,
offset_size,
length_size,
)?;