Read chunks of 4 GiB or more in every chunk index
ChunkInfo::chunk_size (and ChunkMapping::file_size) are u64: sizes past u32 were truncated for Single Chunk, Implicit, Fixed and Extensible Array indexes, and a v2 B-tree index refused them. A selection of a chunked dataset with a non-default fill value is read over a box of fill values instead of a full read, an unfiltered chunk of a file that is not in memory is read row by row, and an intermediate deflate stage no longer reserves the chunk's whole bound. Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
@@ -314,7 +314,7 @@ pub(crate) fn chunk_req(
|
||||
chunk_bytes: usize,
|
||||
wanted: bool,
|
||||
) -> ExtentReq {
|
||||
let len = c.chunk_size as usize;
|
||||
let len = crate::addr::saturating_usize(c.chunk_size);
|
||||
ExtentReq {
|
||||
addr: c.address,
|
||||
len,
|
||||
@@ -521,7 +521,7 @@ pub fn decompress_all_chunks_with_stats_in<S: Storage + ?Sized>(
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct ChunkInfo {
|
||||
/// Size of chunk data in the file (after compression).
|
||||
pub chunk_size: u32,
|
||||
pub chunk_size: u64,
|
||||
/// Bitmask of filters that were NOT applied (0 = all applied).
|
||||
pub filter_mask: u32,
|
||||
/// N-dimensional offset of this chunk in dataset space.
|
||||
@@ -610,7 +610,7 @@ fn stored_element_size(dt: &Datatype, offset_size: u8) -> u64 {
|
||||
/// (`H5D__chunk_set_sizes`: "stored datatype size in chunk layout does not
|
||||
/// match datatype description"). Reading it anyway laid the chunks out with
|
||||
/// the wrong element size.
|
||||
pub(crate) fn check_chunk_element_size(
|
||||
pub fn check_chunk_element_size(
|
||||
layout: &DataLayout,
|
||||
datatype: &Datatype,
|
||||
offset_size: u8,
|
||||
@@ -1028,7 +1028,7 @@ fn parse_chunk_node<S: Storage + ?Sized>(
|
||||
if node_level == 0 {
|
||||
chunks.push(stored.len());
|
||||
stored.push(ChunkInfo {
|
||||
chunk_size,
|
||||
chunk_size: u64::from(chunk_size),
|
||||
filter_mask,
|
||||
offsets: keys[k..].to_vec(),
|
||||
address,
|
||||
@@ -1131,7 +1131,7 @@ pub fn generate_implicit_chunks_in_grid(
|
||||
}
|
||||
|
||||
chunks.push(ChunkInfo {
|
||||
chunk_size: chunk_byte_size as u32,
|
||||
chunk_size: chunk_byte_size,
|
||||
filter_mask: 0,
|
||||
offsets,
|
||||
address: base_address.saturating_add(grid_idx.saturating_mul(chunk_byte_size)),
|
||||
@@ -1190,9 +1190,15 @@ fn read_btree_v2_chunks<S: Storage + ?Sized>(
|
||||
}
|
||||
_ => return Err(bad("tree is not a chunk index")),
|
||||
};
|
||||
let unfiltered_bytes = checked_chunk_byte_len(chunk_dims, elem_size)?;
|
||||
let unfiltered_bytes =
|
||||
u32::try_from(unfiltered_bytes).map_err(|_| bad("chunk larger than 4 GiB"))?;
|
||||
// A u64 whatever the platform: an unfiltered chunk's size is only
|
||||
// recorded here; a chunk this platform cannot address fails when read.
|
||||
let unfiltered_bytes = u64::try_from(
|
||||
chunk_dims
|
||||
.iter()
|
||||
.try_fold(elem_size as u128, |acc, &c| acc.checked_mul(c as u128))
|
||||
.ok_or_else(|| bad("chunk size overflows"))?,
|
||||
)
|
||||
.map_err(|_| bad("chunk size overflows"))?;
|
||||
|
||||
let records = collect_btree_v2_records_in(file_data, &header, offset_size, length_size)?;
|
||||
let mut chunks = Vec::with_capacity(records.len());
|
||||
@@ -1213,10 +1219,7 @@ fn read_btree_v2_chunks<S: Storage + ?Sized>(
|
||||
pos += size_len;
|
||||
let mask = u32::from_le_bytes([data[pos], data[pos + 1], data[pos + 2], data[pos + 3]]);
|
||||
pos += 4;
|
||||
(
|
||||
u32::try_from(size).map_err(|_| bad("stored chunk larger than 4 GiB"))?,
|
||||
mask,
|
||||
)
|
||||
(size, mask)
|
||||
};
|
||||
let mut offsets = Vec::with_capacity(rank);
|
||||
for &dim in chunk_dims {
|
||||
@@ -1334,9 +1337,9 @@ pub fn list_chunks_in<S: Storage + ?Sized>(
|
||||
// Single chunk — one chunk covering the entire dataset
|
||||
let chunk_byte_size = checked_chunk_byte_len(&chunk_dims, elem_size)?;
|
||||
let (csize, fmask) = if let Some(fs) = single_filtered_size {
|
||||
(fs as u32, single_filter_mask.unwrap_or(0))
|
||||
(fs, single_filter_mask.unwrap_or(0))
|
||||
} else {
|
||||
(chunk_byte_size as u32, 0)
|
||||
(chunk_byte_size as u64, 0)
|
||||
};
|
||||
vec![ChunkInfo {
|
||||
chunk_size: csize,
|
||||
@@ -2124,7 +2127,7 @@ pub fn read_chunked_data_indexed_in<S: Storage + ?Sized>(
|
||||
.iter()
|
||||
.zip(&hits)
|
||||
.map(|(m, hit)| {
|
||||
let len = m.file_size as usize;
|
||||
let len = crate::addr::saturating_usize(m.file_size);
|
||||
ExtentReq {
|
||||
addr: m.file_offset,
|
||||
len,
|
||||
@@ -2757,7 +2760,7 @@ mod tests {
|
||||
}
|
||||
|
||||
chunk_infos.push(ChunkInfo {
|
||||
chunk_size: chunk_bytes as u32,
|
||||
chunk_size: chunk_bytes as u64,
|
||||
filter_mask: 0,
|
||||
offsets: vec![start as u64, 0],
|
||||
address: data_offset as u64,
|
||||
@@ -2869,7 +2872,7 @@ mod tests {
|
||||
.collect();
|
||||
let stored = crate::filters::compress_chunk(&chunk, &pipeline, 4).unwrap();
|
||||
chunks.push(ChunkInfo {
|
||||
chunk_size: stored.len() as u32,
|
||||
chunk_size: stored.len() as u64,
|
||||
filter_mask: 0,
|
||||
offsets: vec![r0 as u64, c0 as u64, 0],
|
||||
address: file.len() as u64,
|
||||
@@ -2943,7 +2946,7 @@ mod tests {
|
||||
let short = crate::filters::compress_chunk(&[1u8; 64], &pipeline, 4).unwrap();
|
||||
for bad in [5usize, 11, 40] {
|
||||
chunks[bad].address = file.len() as u64;
|
||||
chunks[bad].chunk_size = short.len() as u32;
|
||||
chunks[bad].chunk_size = short.len() as u64;
|
||||
file.extend_from_slice(&short);
|
||||
}
|
||||
for _ in 0..20 {
|
||||
@@ -3133,7 +3136,7 @@ mod tests {
|
||||
file_data[data_offset..data_offset + compressed.len()].copy_from_slice(&compressed);
|
||||
|
||||
chunk_infos.push(ChunkInfo {
|
||||
chunk_size: compressed.len() as u32,
|
||||
chunk_size: compressed.len() as u64,
|
||||
filter_mask: 0,
|
||||
offsets: vec![start as u64, 0],
|
||||
address: data_offset as u64,
|
||||
@@ -3215,7 +3218,7 @@ mod tests {
|
||||
file_data[data_offset..data_offset + chunk_size].copy_from_slice(&chunk_bytes);
|
||||
|
||||
chunk_infos.push(ChunkInfo {
|
||||
chunk_size: chunk_size as u32,
|
||||
chunk_size: chunk_size as u64,
|
||||
filter_mask: 0,
|
||||
offsets: vec![row_start as u64, col_start as u64, 0],
|
||||
address: data_offset as u64,
|
||||
@@ -3340,7 +3343,7 @@ mod tests {
|
||||
assert_eq!(c.address, 0x1000 + i as u64 * chunk_byte_size as u64);
|
||||
assert_eq!(c.offsets, vec![i as u64 * 20]);
|
||||
assert_eq!(c.filter_mask, 0);
|
||||
assert_eq!(c.chunk_size, chunk_byte_size as u32);
|
||||
assert_eq!(c.chunk_size, chunk_byte_size as u64);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user