Read chunks of 4 GiB or more in every chunk index

ChunkInfo::chunk_size (and ChunkMapping::file_size) are u64: sizes past
u32 were truncated for Single Chunk, Implicit, Fixed and Extensible Array
indexes, and a v2 B-tree index refused them. A selection of a chunked
dataset with a non-default fill value is read over a box of fill values
instead of a full read, an unfiltered chunk of a file that is not in memory
is read row by row, and an intermediate deflate stage no longer reserves the
chunk's whole bound.

Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
osobh
2026-09-28 23:44:48 -05:00
co-authored by Claude Opus 5.5
parent 01d2a5dc5d
commit ac7871fff5
16 changed files with 604 additions and 43 deletions
+1 -1
View File
@@ -899,7 +899,7 @@ mod tests {
fn make_chunk(offsets: Vec<u64>, address: u64, size: u32) -> ChunkInfo {
ChunkInfo {
chunk_size: size,
chunk_size: u64::from(size),
filter_mask: 0,
offsets,
address,
+2 -2
View File
@@ -109,7 +109,7 @@ pub struct ChunkMapping {
/// File byte address of the compressed chunk.
pub file_offset: u64,
/// Size of the compressed chunk in the file.
pub file_size: u32,
pub file_size: u64,
/// Filter mask (0 = all filters applied).
pub filter_mask: u32,
/// Pre-computed row-copy operations for assembling this chunk into output.
@@ -338,7 +338,7 @@ mod tests {
fn make_chunk(offsets: Vec<u64>, address: u64, size: u32) -> ChunkInfo {
ChunkInfo {
chunk_size: size,
chunk_size: u64::from(size),
filter_mask: 0,
offsets,
address,
+24 -21
View File
@@ -314,7 +314,7 @@ pub(crate) fn chunk_req(
chunk_bytes: usize,
wanted: bool,
) -> ExtentReq {
let len = c.chunk_size as usize;
let len = crate::addr::saturating_usize(c.chunk_size);
ExtentReq {
addr: c.address,
len,
@@ -521,7 +521,7 @@ pub fn decompress_all_chunks_with_stats_in<S: Storage + ?Sized>(
#[derive(Debug, Clone)]
pub struct ChunkInfo {
/// Size of chunk data in the file (after compression).
pub chunk_size: u32,
pub chunk_size: u64,
/// Bitmask of filters that were NOT applied (0 = all applied).
pub filter_mask: u32,
/// N-dimensional offset of this chunk in dataset space.
@@ -610,7 +610,7 @@ fn stored_element_size(dt: &Datatype, offset_size: u8) -> u64 {
/// (`H5D__chunk_set_sizes`: "stored datatype size in chunk layout does not
/// match datatype description"). Reading it anyway laid the chunks out with
/// the wrong element size.
pub(crate) fn check_chunk_element_size(
pub fn check_chunk_element_size(
layout: &DataLayout,
datatype: &Datatype,
offset_size: u8,
@@ -1028,7 +1028,7 @@ fn parse_chunk_node<S: Storage + ?Sized>(
if node_level == 0 {
chunks.push(stored.len());
stored.push(ChunkInfo {
chunk_size,
chunk_size: u64::from(chunk_size),
filter_mask,
offsets: keys[k..].to_vec(),
address,
@@ -1131,7 +1131,7 @@ pub fn generate_implicit_chunks_in_grid(
}
chunks.push(ChunkInfo {
chunk_size: chunk_byte_size as u32,
chunk_size: chunk_byte_size,
filter_mask: 0,
offsets,
address: base_address.saturating_add(grid_idx.saturating_mul(chunk_byte_size)),
@@ -1190,9 +1190,15 @@ fn read_btree_v2_chunks<S: Storage + ?Sized>(
}
_ => return Err(bad("tree is not a chunk index")),
};
let unfiltered_bytes = checked_chunk_byte_len(chunk_dims, elem_size)?;
let unfiltered_bytes =
u32::try_from(unfiltered_bytes).map_err(|_| bad("chunk larger than 4 GiB"))?;
// A u64 whatever the platform: an unfiltered chunk's size is only
// recorded here; a chunk this platform cannot address fails when read.
let unfiltered_bytes = u64::try_from(
chunk_dims
.iter()
.try_fold(elem_size as u128, |acc, &c| acc.checked_mul(c as u128))
.ok_or_else(|| bad("chunk size overflows"))?,
)
.map_err(|_| bad("chunk size overflows"))?;
let records = collect_btree_v2_records_in(file_data, &header, offset_size, length_size)?;
let mut chunks = Vec::with_capacity(records.len());
@@ -1213,10 +1219,7 @@ fn read_btree_v2_chunks<S: Storage + ?Sized>(
pos += size_len;
let mask = u32::from_le_bytes([data[pos], data[pos + 1], data[pos + 2], data[pos + 3]]);
pos += 4;
(
u32::try_from(size).map_err(|_| bad("stored chunk larger than 4 GiB"))?,
mask,
)
(size, mask)
};
let mut offsets = Vec::with_capacity(rank);
for &dim in chunk_dims {
@@ -1334,9 +1337,9 @@ pub fn list_chunks_in<S: Storage + ?Sized>(
// Single chunk — one chunk covering the entire dataset
let chunk_byte_size = checked_chunk_byte_len(&chunk_dims, elem_size)?;
let (csize, fmask) = if let Some(fs) = single_filtered_size {
(fs as u32, single_filter_mask.unwrap_or(0))
(fs, single_filter_mask.unwrap_or(0))
} else {
(chunk_byte_size as u32, 0)
(chunk_byte_size as u64, 0)
};
vec![ChunkInfo {
chunk_size: csize,
@@ -2124,7 +2127,7 @@ pub fn read_chunked_data_indexed_in<S: Storage + ?Sized>(
.iter()
.zip(&hits)
.map(|(m, hit)| {
let len = m.file_size as usize;
let len = crate::addr::saturating_usize(m.file_size);
ExtentReq {
addr: m.file_offset,
len,
@@ -2757,7 +2760,7 @@ mod tests {
}
chunk_infos.push(ChunkInfo {
chunk_size: chunk_bytes as u32,
chunk_size: chunk_bytes as u64,
filter_mask: 0,
offsets: vec![start as u64, 0],
address: data_offset as u64,
@@ -2869,7 +2872,7 @@ mod tests {
.collect();
let stored = crate::filters::compress_chunk(&chunk, &pipeline, 4).unwrap();
chunks.push(ChunkInfo {
chunk_size: stored.len() as u32,
chunk_size: stored.len() as u64,
filter_mask: 0,
offsets: vec![r0 as u64, c0 as u64, 0],
address: file.len() as u64,
@@ -2943,7 +2946,7 @@ mod tests {
let short = crate::filters::compress_chunk(&[1u8; 64], &pipeline, 4).unwrap();
for bad in [5usize, 11, 40] {
chunks[bad].address = file.len() as u64;
chunks[bad].chunk_size = short.len() as u32;
chunks[bad].chunk_size = short.len() as u64;
file.extend_from_slice(&short);
}
for _ in 0..20 {
@@ -3133,7 +3136,7 @@ mod tests {
file_data[data_offset..data_offset + compressed.len()].copy_from_slice(&compressed);
chunk_infos.push(ChunkInfo {
chunk_size: compressed.len() as u32,
chunk_size: compressed.len() as u64,
filter_mask: 0,
offsets: vec![start as u64, 0],
address: data_offset as u64,
@@ -3215,7 +3218,7 @@ mod tests {
file_data[data_offset..data_offset + chunk_size].copy_from_slice(&chunk_bytes);
chunk_infos.push(ChunkInfo {
chunk_size: chunk_size as u32,
chunk_size: chunk_size as u64,
filter_mask: 0,
offsets: vec![row_start as u64, col_start as u64, 0],
address: data_offset as u64,
@@ -3340,7 +3343,7 @@ mod tests {
assert_eq!(c.address, 0x1000 + i as u64 * chunk_byte_size as u64);
assert_eq!(c.offsets, vec![i as u64 * 20]);
assert_eq!(c.filter_mask, 0);
assert_eq!(c.chunk_size, chunk_byte_size as u32);
assert_eq!(c.chunk_size, chunk_byte_size as u64);
}
}
+1 -1
View File
@@ -2125,7 +2125,7 @@ mod tests {
for info in &infos {
// Skipped chunks are stored at the chunk's size (shuffled).
assert_eq!(
info.chunk_size == (c * 8) as u32,
info.chunk_size == (c * 8) as u64,
info.filter_mask != 0,
"{info:?}"
);
@@ -224,7 +224,7 @@ fn read_element(
};
Ok((
Some(ChunkInfo {
chunk_size: chunk_byte_size as u32,
chunk_size: chunk_byte_size,
filter_mask: 0,
offsets,
address,
@@ -259,7 +259,7 @@ fn read_element(
};
Ok((
Some(ChunkInfo {
chunk_size: chunk_size as u32,
chunk_size,
filter_mask,
offsets,
address,
@@ -946,7 +946,7 @@ mod tests {
assert_eq!(chunks.len(), 2);
assert_eq!(chunks[0].address, base_addr);
assert_eq!(chunks[0].offsets, vec![0]);
assert_eq!(chunks[0].chunk_size, chunk_byte_size as u32);
assert_eq!(chunks[0].chunk_size, chunk_byte_size as u64);
assert_eq!(chunks[1].address, base_addr + chunk_byte_size);
assert_eq!(chunks[1].offsets, vec![20]);
}
+15 -2
View File
@@ -340,10 +340,23 @@ pub fn decompress_chunk_exact_with<'s>(
} else {
MAX_DECOMPRESS_SIZE
};
let size_hint = if ctx.max_output != 0 {
// The bound is the output's size when only size-preserving
// filters (shuffle, Fletcher32) remain to be undone; before
// any other filter (a second deflate, N-Bit, ...) it is only
// a ceiling, and the output starts smaller and grows: the
// inflater writes every byte it reserves, so reserving a
// 4 GiB chunk's bound for a stage a few MiB long would hold
// twice the chunk's memory.
let exact = pipeline.filters[..i].iter().enumerate().all(|(j, f)| {
filter_skipped(filter_mask, j)
|| matches!(f.filter_id, FILTER_SHUFFLE | FILTER_FLETCHER32)
});
let size_hint = if ctx.max_output == 0 {
input.len().saturating_mul(4).min(1 << 20)
} else if exact {
ctx.max_output
} else {
input.len().saturating_mul(4).min(1 << 20)
input.len().saturating_mul(4).min(ctx.max_output)
};
let inflater = scratch
.inflater
+4 -4
View File
@@ -383,7 +383,7 @@ fn parse_fa_element(
offset_size: u8,
element_size: u8,
chunk_byte_size: u64,
) -> Result<Option<(u64, u32, u32)>, FormatError> {
) -> Result<Option<(u64, u64, u32)>, FormatError> {
let os = offset_size as usize;
if client_id == 0 {
// Non-filtered: element is just the chunk address.
@@ -393,7 +393,7 @@ fn parse_fa_element(
return Ok(None);
}
let address = read_offset(file_data, abs, offset_size)?;
Ok(Some((address, chunk_byte_size as u32, 0)))
Ok(Some((address, chunk_byte_size, 0)))
} else {
// Filtered: address(offset_size) + chunk_size(variable) + filter_mask(4)
let es = element_size as usize;
@@ -418,7 +418,7 @@ fn parse_fa_element(
file_data[fm_off + 2],
file_data[fm_off + 3],
]);
Ok(Some((address, chunk_size as u32, filter_mask)))
Ok(Some((address, chunk_size, filter_mask)))
}
}
@@ -688,7 +688,7 @@ mod tests {
assert_eq!(c.address, base_addr + i as u64 * chunk_byte_size as u64);
assert_eq!(c.offsets, vec![i as u64 * 20]);
assert_eq!(c.filter_mask, 0);
assert_eq!(c.chunk_size, chunk_byte_size as u32);
assert_eq!(c.chunk_size, chunk_byte_size as u64);
}
}
+1 -1
View File
@@ -426,7 +426,7 @@ mod tests {
for i in 0..8u64 {
let len = if short && i == 5 { 16 } else { 32 };
infos.push(ChunkInfo {
chunk_size: len as u32,
chunk_size: len as u64,
filter_mask: 0,
offsets: vec![i * 8],
address: file.len() as u64,
+114
View File
@@ -243,6 +243,58 @@ fn copy_overlap(
}
}
/// Copy the part of unfiltered chunk `chunk` (`chunk_bytes` long, shape
/// `chunk_shape`) that overlaps the box into `out`, reading only the runs of
/// the overlap from the file. The whole chunk must still lie inside the
/// file, as it must when it is fetched whole.
#[allow(clippy::too_many_arguments)]
fn read_unfiltered_overlap<S: Storage + ?Sized>(
file_data: &S,
chunk: &crate::chunked_read::ChunkInfo,
chunk_bytes: usize,
chunk_shape: &[u64],
elem_size: usize,
out: &mut [u8],
box_start: &[u64],
box_extent: &[u64],
) -> Result<(), FormatError> {
let rank = chunk_shape.len();
let origin = &chunk.offsets[..rank];
let (lo, extent): (Vec<u64>, Vec<u64>) = (0..rank)
.map(|d| {
let lo = origin[d].max(box_start[d]);
let hi = origin[d]
.saturating_add(chunk_shape[d])
.min(box_start[d] + box_extent[d]);
(lo, hi.saturating_sub(lo))
})
.unzip();
let file_len = crate::storage::len_usize(file_data);
let base = crate::addr::to_usize(chunk.address)?;
if base > file_len || chunk_bytes > file_len - base {
return Err(FormatError::UnexpectedEof {
expected: base.saturating_add(chunk_bytes),
available: file_len,
});
}
let overlap = Selection::Hyperslab {
start: lo.iter().zip(origin).map(|(l, o)| l - o).collect(),
stride: vec![1; rank],
count: extent.clone(),
block: vec![1; rank],
};
let rows = crate::gather::gather_storage(
file_data,
chunk.address,
chunk_bytes,
chunk_shape,
elem_size,
&overlap,
)?;
copy_overlap(&rows, &lo, &extent, out, box_start, box_extent, elem_size);
Ok(())
}
/// Read `selection` without materialising the whole dataset, when that is
/// possible and worthwhile. `Ok(None)` means "use the full-read path": an
/// `All`/`None`/invalid selection, a layout this doesn't handle (compact,
@@ -281,6 +333,39 @@ pub fn read_selection_in<S: Storage + ?Sized>(
offset_size: u8,
length_size: u8,
selection: &Selection,
) -> Result<Option<Vec<u8>>, FormatError> {
read_selection_filled_in(
file_data,
layout,
dataspace,
elem_size,
pipeline,
offset_size,
length_size,
selection,
None,
)
}
/// [`read_selection_in`] for a dataset whose fill value is `fill` (one
/// element's bytes; `None` or all zeros is the default fill): the elements
/// of a chunked dataset's selection that lie in chunks never written read as
/// `fill`, as they do in a full read. Only the chunks the selection's
/// bounding box overlaps are read, so a selection of a few elements of a
/// dataset whose chunks are 4 GiB or more costs one decoded chunk (a
/// filtered chunk has to be decoded whole) or, unfiltered, only the bytes
/// it selects.
#[allow(clippy::too_many_arguments)]
pub fn read_selection_filled_in<S: Storage + ?Sized>(
file_data: &S,
layout: &DataLayout,
dataspace: &Dataspace,
elem_size: usize,
pipeline: Option<&FilterPipeline>,
offset_size: u8,
length_size: u8,
selection: &Selection,
fill: Option<&[u8]>,
) -> Result<Option<Vec<u8>>, FormatError> {
let dims = &dataspace.dimensions;
if dims.is_empty() || elem_size == 0 {
@@ -341,8 +426,18 @@ pub fn read_selection_in<S: Storage + ?Sized>(
return Ok(None);
}
let mut boxed = alloc_output(checked_byte_len(box_elements, elem_size)?)?;
if let Some(fill) = fill.filter(|f| <[u8]>::len(f) == elem_size && f.iter().any(|&b| b != 0)) {
for element in boxed.chunks_exact_mut(elem_size) {
element.copy_from_slice(fill);
}
}
match layout {
// No chunk was ever written: every element is the fill value.
DataLayout::Chunked {
btree_address: None,
..
} if fill.is_some() => {}
DataLayout::Chunked {
btree_address: Some(_),
..
@@ -373,6 +468,25 @@ pub fn read_selection_in<S: Storage + ?Sized>(
})
})
.collect();
// An unfiltered chunk of a file that is not in memory: fetch
// only the rows the box needs, not the whole chunk (which may be
// 4 GiB or more).
let (direct, wanted): (Vec<_>, Vec<_>) = wanted.into_iter().partition(|c| {
file_data.as_contiguous().is_none()
&& pipeline.is_none_or(|pl| all_filters_skipped(pl, c.filter_mask))
});
for chunk in direct {
read_unfiltered_overlap(
file_data,
chunk,
chunk_bytes,
&chunk_shape,
elem_size,
&mut boxed,
&box_start,
&box_extent,
)?;
}
// Their stored bytes, batch by batch when the file is not in
// memory; each batch's chunks are decoded into this thread's
// reusable buffers before the next batch is fetched.
@@ -173,7 +173,7 @@ fn crafted() -> (Vec<u8>, Chunked, Vec<ChunkInfo>) {
assert!(
chunks
.iter()
.all(|c| c.chunk_size == HUGE && c.address == blob)
.all(|c| c.chunk_size == u64::from(HUGE) && c.address == blob)
);
(bytes, ds, chunks)
}
@@ -313,7 +313,7 @@ fn large_reads_are_fetched_in_batches() {
let data: Vec<u8> = (0..2 * CHUNK).map(|i| (i % 251) as u8).collect();
let chunks: Vec<ChunkInfo> = (0..40u64)
.map(|i| ChunkInfo {
chunk_size: CHUNK as u32,
chunk_size: CHUNK as u64,
filter_mask: 0,
offsets: vec![i * CHUNK as u64],
address: (i % 2) * CHUNK as u64,