From ac7871fff5684f9934e5c75f7d0528c4dca4a768 Mon Sep 17 00:00:00 2001 From: osobh Date: Mon, 28 Sep 2026 22:53:12 -0500 Subject: [PATCH 1/7] Read chunks of 4 GiB or more in every chunk index ChunkInfo::chunk_size (and ChunkMapping::file_size) are u64: sizes past u32 were truncated for Single Chunk, Implicit, Fixed and Extensible Array indexes, and a v2 B-tree index refused them. A selection of a chunked dataset with a non-default fill value is read over a box of fill values instead of a full read, an unfiltered chunk of a file that is not in memory is read row by row, and an intermediate deflate stage no longer reserves the chunk's whole bound. Co-Authored-By: Claude Opus 5.5 (1M context) --- .gitignore | 3 + crates/clawhdf5-format/src/chunk_cache.rs | 2 +- crates/clawhdf5-format/src/chunk_index.rs | 4 +- crates/clawhdf5-format/src/chunked_read.rs | 45 +-- crates/clawhdf5-format/src/chunked_write.rs | 2 +- .../clawhdf5-format/src/extensible_array.rs | 6 +- crates/clawhdf5-format/src/filters.rs | 17 +- crates/clawhdf5-format/src/fixed_array.rs | 8 +- crates/clawhdf5-format/src/parallel_read.rs | 2 +- crates/clawhdf5-format/src/partial_read.rs | 114 +++++++ .../clawhdf5-format/tests/raw_fetch_bounds.rs | 4 +- crates/clawhdf5/src/reader.rs | 33 +- .../tests/fixtures/gen_huge_chunks.py | 109 +++++++ .../tests/fixtures/huge_chunks_filtered.h5 | Bin 0 -> 188488 bytes crates/clawhdf5/tests/huge_chunks_interop.rs | 296 ++++++++++++++++++ crates/clawhdf5/tests/parallel_integration.rs | 2 +- 16 files changed, 604 insertions(+), 43 deletions(-) create mode 100644 crates/clawhdf5/tests/fixtures/gen_huge_chunks.py create mode 100644 crates/clawhdf5/tests/fixtures/huge_chunks_filtered.h5 create mode 100644 crates/clawhdf5/tests/huge_chunks_interop.rs diff --git a/.gitignore b/.gitignore index cf90c2c..d0a22c0 100644 --- a/.gitignore +++ b/.gitignore @@ -7,3 +7,6 @@ weights/ .venv __pycache__/ .pytest_cache/ + +# Scratch files the heavy tests generate (huge_chunks_interop) +crates/*/tests/scratch/ diff --git a/crates/clawhdf5-format/src/chunk_cache.rs b/crates/clawhdf5-format/src/chunk_cache.rs index 6e3fee8..cca133d 100644 --- a/crates/clawhdf5-format/src/chunk_cache.rs +++ b/crates/clawhdf5-format/src/chunk_cache.rs @@ -899,7 +899,7 @@ mod tests { fn make_chunk(offsets: Vec, address: u64, size: u32) -> ChunkInfo { ChunkInfo { - chunk_size: size, + chunk_size: u64::from(size), filter_mask: 0, offsets, address, diff --git a/crates/clawhdf5-format/src/chunk_index.rs b/crates/clawhdf5-format/src/chunk_index.rs index 706bad0..1e083a0 100644 --- a/crates/clawhdf5-format/src/chunk_index.rs +++ b/crates/clawhdf5-format/src/chunk_index.rs @@ -109,7 +109,7 @@ pub struct ChunkMapping { /// File byte address of the compressed chunk. pub file_offset: u64, /// Size of the compressed chunk in the file. - pub file_size: u32, + pub file_size: u64, /// Filter mask (0 = all filters applied). pub filter_mask: u32, /// Pre-computed row-copy operations for assembling this chunk into output. @@ -338,7 +338,7 @@ mod tests { fn make_chunk(offsets: Vec, address: u64, size: u32) -> ChunkInfo { ChunkInfo { - chunk_size: size, + chunk_size: u64::from(size), filter_mask: 0, offsets, address, diff --git a/crates/clawhdf5-format/src/chunked_read.rs b/crates/clawhdf5-format/src/chunked_read.rs index 4d97adb..fa3f016 100644 --- a/crates/clawhdf5-format/src/chunked_read.rs +++ b/crates/clawhdf5-format/src/chunked_read.rs @@ -314,7 +314,7 @@ pub(crate) fn chunk_req( chunk_bytes: usize, wanted: bool, ) -> ExtentReq { - let len = c.chunk_size as usize; + let len = crate::addr::saturating_usize(c.chunk_size); ExtentReq { addr: c.address, len, @@ -521,7 +521,7 @@ pub fn decompress_all_chunks_with_stats_in( #[derive(Debug, Clone)] pub struct ChunkInfo { /// Size of chunk data in the file (after compression). - pub chunk_size: u32, + pub chunk_size: u64, /// Bitmask of filters that were NOT applied (0 = all applied). pub filter_mask: u32, /// N-dimensional offset of this chunk in dataset space. @@ -610,7 +610,7 @@ fn stored_element_size(dt: &Datatype, offset_size: u8) -> u64 { /// (`H5D__chunk_set_sizes`: "stored datatype size in chunk layout does not /// match datatype description"). Reading it anyway laid the chunks out with /// the wrong element size. -pub(crate) fn check_chunk_element_size( +pub fn check_chunk_element_size( layout: &DataLayout, datatype: &Datatype, offset_size: u8, @@ -1028,7 +1028,7 @@ fn parse_chunk_node( if node_level == 0 { chunks.push(stored.len()); stored.push(ChunkInfo { - chunk_size, + chunk_size: u64::from(chunk_size), filter_mask, offsets: keys[k..].to_vec(), address, @@ -1131,7 +1131,7 @@ pub fn generate_implicit_chunks_in_grid( } chunks.push(ChunkInfo { - chunk_size: chunk_byte_size as u32, + chunk_size: chunk_byte_size, filter_mask: 0, offsets, address: base_address.saturating_add(grid_idx.saturating_mul(chunk_byte_size)), @@ -1190,9 +1190,15 @@ fn read_btree_v2_chunks( } _ => return Err(bad("tree is not a chunk index")), }; - let unfiltered_bytes = checked_chunk_byte_len(chunk_dims, elem_size)?; - let unfiltered_bytes = - u32::try_from(unfiltered_bytes).map_err(|_| bad("chunk larger than 4 GiB"))?; + // A u64 whatever the platform: an unfiltered chunk's size is only + // recorded here; a chunk this platform cannot address fails when read. + let unfiltered_bytes = u64::try_from( + chunk_dims + .iter() + .try_fold(elem_size as u128, |acc, &c| acc.checked_mul(c as u128)) + .ok_or_else(|| bad("chunk size overflows"))?, + ) + .map_err(|_| bad("chunk size overflows"))?; let records = collect_btree_v2_records_in(file_data, &header, offset_size, length_size)?; let mut chunks = Vec::with_capacity(records.len()); @@ -1213,10 +1219,7 @@ fn read_btree_v2_chunks( pos += size_len; let mask = u32::from_le_bytes([data[pos], data[pos + 1], data[pos + 2], data[pos + 3]]); pos += 4; - ( - u32::try_from(size).map_err(|_| bad("stored chunk larger than 4 GiB"))?, - mask, - ) + (size, mask) }; let mut offsets = Vec::with_capacity(rank); for &dim in chunk_dims { @@ -1334,9 +1337,9 @@ pub fn list_chunks_in( // Single chunk — one chunk covering the entire dataset let chunk_byte_size = checked_chunk_byte_len(&chunk_dims, elem_size)?; let (csize, fmask) = if let Some(fs) = single_filtered_size { - (fs as u32, single_filter_mask.unwrap_or(0)) + (fs, single_filter_mask.unwrap_or(0)) } else { - (chunk_byte_size as u32, 0) + (chunk_byte_size as u64, 0) }; vec![ChunkInfo { chunk_size: csize, @@ -2124,7 +2127,7 @@ pub fn read_chunked_data_indexed_in( .iter() .zip(&hits) .map(|(m, hit)| { - let len = m.file_size as usize; + let len = crate::addr::saturating_usize(m.file_size); ExtentReq { addr: m.file_offset, len, @@ -2757,7 +2760,7 @@ mod tests { } chunk_infos.push(ChunkInfo { - chunk_size: chunk_bytes as u32, + chunk_size: chunk_bytes as u64, filter_mask: 0, offsets: vec![start as u64, 0], address: data_offset as u64, @@ -2869,7 +2872,7 @@ mod tests { .collect(); let stored = crate::filters::compress_chunk(&chunk, &pipeline, 4).unwrap(); chunks.push(ChunkInfo { - chunk_size: stored.len() as u32, + chunk_size: stored.len() as u64, filter_mask: 0, offsets: vec![r0 as u64, c0 as u64, 0], address: file.len() as u64, @@ -2943,7 +2946,7 @@ mod tests { let short = crate::filters::compress_chunk(&[1u8; 64], &pipeline, 4).unwrap(); for bad in [5usize, 11, 40] { chunks[bad].address = file.len() as u64; - chunks[bad].chunk_size = short.len() as u32; + chunks[bad].chunk_size = short.len() as u64; file.extend_from_slice(&short); } for _ in 0..20 { @@ -3133,7 +3136,7 @@ mod tests { file_data[data_offset..data_offset + compressed.len()].copy_from_slice(&compressed); chunk_infos.push(ChunkInfo { - chunk_size: compressed.len() as u32, + chunk_size: compressed.len() as u64, filter_mask: 0, offsets: vec![start as u64, 0], address: data_offset as u64, @@ -3215,7 +3218,7 @@ mod tests { file_data[data_offset..data_offset + chunk_size].copy_from_slice(&chunk_bytes); chunk_infos.push(ChunkInfo { - chunk_size: chunk_size as u32, + chunk_size: chunk_size as u64, filter_mask: 0, offsets: vec![row_start as u64, col_start as u64, 0], address: data_offset as u64, @@ -3340,7 +3343,7 @@ mod tests { assert_eq!(c.address, 0x1000 + i as u64 * chunk_byte_size as u64); assert_eq!(c.offsets, vec![i as u64 * 20]); assert_eq!(c.filter_mask, 0); - assert_eq!(c.chunk_size, chunk_byte_size as u32); + assert_eq!(c.chunk_size, chunk_byte_size as u64); } } diff --git a/crates/clawhdf5-format/src/chunked_write.rs b/crates/clawhdf5-format/src/chunked_write.rs index 10f1099..746a492 100644 --- a/crates/clawhdf5-format/src/chunked_write.rs +++ b/crates/clawhdf5-format/src/chunked_write.rs @@ -2125,7 +2125,7 @@ mod tests { for info in &infos { // Skipped chunks are stored at the chunk's size (shuffled). assert_eq!( - info.chunk_size == (c * 8) as u32, + info.chunk_size == (c * 8) as u64, info.filter_mask != 0, "{info:?}" ); diff --git a/crates/clawhdf5-format/src/extensible_array.rs b/crates/clawhdf5-format/src/extensible_array.rs index 42cf595..a17367b 100644 --- a/crates/clawhdf5-format/src/extensible_array.rs +++ b/crates/clawhdf5-format/src/extensible_array.rs @@ -224,7 +224,7 @@ fn read_element( }; Ok(( Some(ChunkInfo { - chunk_size: chunk_byte_size as u32, + chunk_size: chunk_byte_size, filter_mask: 0, offsets, address, @@ -259,7 +259,7 @@ fn read_element( }; Ok(( Some(ChunkInfo { - chunk_size: chunk_size as u32, + chunk_size, filter_mask, offsets, address, @@ -946,7 +946,7 @@ mod tests { assert_eq!(chunks.len(), 2); assert_eq!(chunks[0].address, base_addr); assert_eq!(chunks[0].offsets, vec![0]); - assert_eq!(chunks[0].chunk_size, chunk_byte_size as u32); + assert_eq!(chunks[0].chunk_size, chunk_byte_size as u64); assert_eq!(chunks[1].address, base_addr + chunk_byte_size); assert_eq!(chunks[1].offsets, vec![20]); } diff --git a/crates/clawhdf5-format/src/filters.rs b/crates/clawhdf5-format/src/filters.rs index 28070d4..48553c1 100644 --- a/crates/clawhdf5-format/src/filters.rs +++ b/crates/clawhdf5-format/src/filters.rs @@ -340,10 +340,23 @@ pub fn decompress_chunk_exact_with<'s>( } else { MAX_DECOMPRESS_SIZE }; - let size_hint = if ctx.max_output != 0 { + // The bound is the output's size when only size-preserving + // filters (shuffle, Fletcher32) remain to be undone; before + // any other filter (a second deflate, N-Bit, ...) it is only + // a ceiling, and the output starts smaller and grows: the + // inflater writes every byte it reserves, so reserving a + // 4 GiB chunk's bound for a stage a few MiB long would hold + // twice the chunk's memory. + let exact = pipeline.filters[..i].iter().enumerate().all(|(j, f)| { + filter_skipped(filter_mask, j) + || matches!(f.filter_id, FILTER_SHUFFLE | FILTER_FLETCHER32) + }); + let size_hint = if ctx.max_output == 0 { + input.len().saturating_mul(4).min(1 << 20) + } else if exact { ctx.max_output } else { - input.len().saturating_mul(4).min(1 << 20) + input.len().saturating_mul(4).min(ctx.max_output) }; let inflater = scratch .inflater diff --git a/crates/clawhdf5-format/src/fixed_array.rs b/crates/clawhdf5-format/src/fixed_array.rs index de80b7a..dbe00ae 100644 --- a/crates/clawhdf5-format/src/fixed_array.rs +++ b/crates/clawhdf5-format/src/fixed_array.rs @@ -383,7 +383,7 @@ fn parse_fa_element( offset_size: u8, element_size: u8, chunk_byte_size: u64, -) -> Result, FormatError> { +) -> Result, FormatError> { let os = offset_size as usize; if client_id == 0 { // Non-filtered: element is just the chunk address. @@ -393,7 +393,7 @@ fn parse_fa_element( return Ok(None); } let address = read_offset(file_data, abs, offset_size)?; - Ok(Some((address, chunk_byte_size as u32, 0))) + Ok(Some((address, chunk_byte_size, 0))) } else { // Filtered: address(offset_size) + chunk_size(variable) + filter_mask(4) let es = element_size as usize; @@ -418,7 +418,7 @@ fn parse_fa_element( file_data[fm_off + 2], file_data[fm_off + 3], ]); - Ok(Some((address, chunk_size as u32, filter_mask))) + Ok(Some((address, chunk_size, filter_mask))) } } @@ -688,7 +688,7 @@ mod tests { assert_eq!(c.address, base_addr + i as u64 * chunk_byte_size as u64); assert_eq!(c.offsets, vec![i as u64 * 20]); assert_eq!(c.filter_mask, 0); - assert_eq!(c.chunk_size, chunk_byte_size as u32); + assert_eq!(c.chunk_size, chunk_byte_size as u64); } } diff --git a/crates/clawhdf5-format/src/parallel_read.rs b/crates/clawhdf5-format/src/parallel_read.rs index 606173c..0594e42 100644 --- a/crates/clawhdf5-format/src/parallel_read.rs +++ b/crates/clawhdf5-format/src/parallel_read.rs @@ -426,7 +426,7 @@ mod tests { for i in 0..8u64 { let len = if short && i == 5 { 16 } else { 32 }; infos.push(ChunkInfo { - chunk_size: len as u32, + chunk_size: len as u64, filter_mask: 0, offsets: vec![i * 8], address: file.len() as u64, diff --git a/crates/clawhdf5-format/src/partial_read.rs b/crates/clawhdf5-format/src/partial_read.rs index 3361485..269c45b 100644 --- a/crates/clawhdf5-format/src/partial_read.rs +++ b/crates/clawhdf5-format/src/partial_read.rs @@ -243,6 +243,58 @@ fn copy_overlap( } } +/// Copy the part of unfiltered chunk `chunk` (`chunk_bytes` long, shape +/// `chunk_shape`) that overlaps the box into `out`, reading only the runs of +/// the overlap from the file. The whole chunk must still lie inside the +/// file, as it must when it is fetched whole. +#[allow(clippy::too_many_arguments)] +fn read_unfiltered_overlap( + file_data: &S, + chunk: &crate::chunked_read::ChunkInfo, + chunk_bytes: usize, + chunk_shape: &[u64], + elem_size: usize, + out: &mut [u8], + box_start: &[u64], + box_extent: &[u64], +) -> Result<(), FormatError> { + let rank = chunk_shape.len(); + let origin = &chunk.offsets[..rank]; + let (lo, extent): (Vec, Vec) = (0..rank) + .map(|d| { + let lo = origin[d].max(box_start[d]); + let hi = origin[d] + .saturating_add(chunk_shape[d]) + .min(box_start[d] + box_extent[d]); + (lo, hi.saturating_sub(lo)) + }) + .unzip(); + let file_len = crate::storage::len_usize(file_data); + let base = crate::addr::to_usize(chunk.address)?; + if base > file_len || chunk_bytes > file_len - base { + return Err(FormatError::UnexpectedEof { + expected: base.saturating_add(chunk_bytes), + available: file_len, + }); + } + let overlap = Selection::Hyperslab { + start: lo.iter().zip(origin).map(|(l, o)| l - o).collect(), + stride: vec![1; rank], + count: extent.clone(), + block: vec![1; rank], + }; + let rows = crate::gather::gather_storage( + file_data, + chunk.address, + chunk_bytes, + chunk_shape, + elem_size, + &overlap, + )?; + copy_overlap(&rows, &lo, &extent, out, box_start, box_extent, elem_size); + Ok(()) +} + /// Read `selection` without materialising the whole dataset, when that is /// possible and worthwhile. `Ok(None)` means "use the full-read path": an /// `All`/`None`/invalid selection, a layout this doesn't handle (compact, @@ -281,6 +333,39 @@ pub fn read_selection_in( offset_size: u8, length_size: u8, selection: &Selection, +) -> Result>, FormatError> { + read_selection_filled_in( + file_data, + layout, + dataspace, + elem_size, + pipeline, + offset_size, + length_size, + selection, + None, + ) +} + +/// [`read_selection_in`] for a dataset whose fill value is `fill` (one +/// element's bytes; `None` or all zeros is the default fill): the elements +/// of a chunked dataset's selection that lie in chunks never written read as +/// `fill`, as they do in a full read. Only the chunks the selection's +/// bounding box overlaps are read, so a selection of a few elements of a +/// dataset whose chunks are 4 GiB or more costs one decoded chunk (a +/// filtered chunk has to be decoded whole) or, unfiltered, only the bytes +/// it selects. +#[allow(clippy::too_many_arguments)] +pub fn read_selection_filled_in( + file_data: &S, + layout: &DataLayout, + dataspace: &Dataspace, + elem_size: usize, + pipeline: Option<&FilterPipeline>, + offset_size: u8, + length_size: u8, + selection: &Selection, + fill: Option<&[u8]>, ) -> Result>, FormatError> { let dims = &dataspace.dimensions; if dims.is_empty() || elem_size == 0 { @@ -341,8 +426,18 @@ pub fn read_selection_in( return Ok(None); } let mut boxed = alloc_output(checked_byte_len(box_elements, elem_size)?)?; + if let Some(fill) = fill.filter(|f| <[u8]>::len(f) == elem_size && f.iter().any(|&b| b != 0)) { + for element in boxed.chunks_exact_mut(elem_size) { + element.copy_from_slice(fill); + } + } match layout { + // No chunk was ever written: every element is the fill value. + DataLayout::Chunked { + btree_address: None, + .. + } if fill.is_some() => {} DataLayout::Chunked { btree_address: Some(_), .. @@ -373,6 +468,25 @@ pub fn read_selection_in( }) }) .collect(); + // An unfiltered chunk of a file that is not in memory: fetch + // only the rows the box needs, not the whole chunk (which may be + // 4 GiB or more). + let (direct, wanted): (Vec<_>, Vec<_>) = wanted.into_iter().partition(|c| { + file_data.as_contiguous().is_none() + && pipeline.is_none_or(|pl| all_filters_skipped(pl, c.filter_mask)) + }); + for chunk in direct { + read_unfiltered_overlap( + file_data, + chunk, + chunk_bytes, + &chunk_shape, + elem_size, + &mut boxed, + &box_start, + &box_extent, + )?; + } // Their stored bytes, batch by batch when the file is not in // memory; each batch's chunks are decoded into this thread's // reusable buffers before the next batch is fetched. diff --git a/crates/clawhdf5-format/tests/raw_fetch_bounds.rs b/crates/clawhdf5-format/tests/raw_fetch_bounds.rs index 8217323..a65d611 100644 --- a/crates/clawhdf5-format/tests/raw_fetch_bounds.rs +++ b/crates/clawhdf5-format/tests/raw_fetch_bounds.rs @@ -173,7 +173,7 @@ fn crafted() -> (Vec, Chunked, Vec) { assert!( chunks .iter() - .all(|c| c.chunk_size == HUGE && c.address == blob) + .all(|c| c.chunk_size == u64::from(HUGE) && c.address == blob) ); (bytes, ds, chunks) } @@ -313,7 +313,7 @@ fn large_reads_are_fetched_in_batches() { let data: Vec = (0..2 * CHUNK).map(|i| (i % 251) as u8).collect(); let chunks: Vec = (0..40u64) .map(|i| ChunkInfo { - chunk_size: CHUNK as u32, + chunk_size: CHUNK as u64, filter_mask: 0, offsets: vec![i * CHUNK as u64], address: (i % 2) * CHUNK as u64, diff --git a/crates/clawhdf5/src/reader.rs b/crates/clawhdf5/src/reader.rs index 4456d66..4426d39 100644 --- a/crates/clawhdf5/src/reader.rs +++ b/crates/clawhdf5/src/reader.rs @@ -1528,11 +1528,11 @@ impl<'f> Dataset<'f> { let ds = self.dataspace()?; let dl = self.data_layout()?; let pipeline = self.filter_pipeline()?; - // The selection reader knows nothing about fill values. When they - // matter — no storage at all, or a non-zero fill on a chunked (possibly - // sparse) dataset — select from a fill-aware full read instead. (The - // selection reader currently decodes the full dataset too, so this - // costs nothing extra.) + // Fill values matter when there is no storage at all, or a non-zero + // fill on a chunked (possibly sparse) dataset: a chunked dataset's + // selection is then read over a box of fill values; anything else + // (and a selection whose box is most of the dataset) is selected + // from a fill-aware full read. let fill = clawhdf5_format::fill_value::dataset_fill_value_from_storage( &self.file.data, &self.header.messages, @@ -1545,6 +1545,29 @@ impl<'f> Dataset<'f> { && !clawhdf5_format::fill_value::is_default(fill.as_deref())); if fill_matters { clawhdf5_format::partial_read::validate(selection, &ds.dimensions)?; + // A chunked dataset: read the chunks the selection touches over + // a box of fill values, not the whole dataset (which may be many + // GiB when its chunks are large). + if matches!(dl, DataLayout::Chunked { .. }) { + clawhdf5_format::chunked_read::check_chunk_element_size( + &dl, + &dt, + self.file.offset_size(), + )?; + if let Some(selected) = clawhdf5_format::partial_read::read_selection_filled_in( + &self.file.data, + &dl, + &ds, + dt.type_size() as usize, + pipeline.as_ref(), + self.file.offset_size(), + self.file.length_size(), + selection, + Some(fill.as_deref().unwrap_or(&[])), + )? { + return Ok(selected); + } + } let full = self.read_raw()?; return Ok(data_read::extract_selection_from_buffer( &full, diff --git a/crates/clawhdf5/tests/fixtures/gen_huge_chunks.py b/crates/clawhdf5/tests/fixtures/gen_huge_chunks.py new file mode 100644 index 0000000..16c18c4 --- /dev/null +++ b/crates/clawhdf5/tests/fixtures/gen_huge_chunks.py @@ -0,0 +1,109 @@ +"""Write HDF5 files whose chunks are 4 GiB or more, one dataset per chunk index. + + python gen_huge_chunks.py filtered OUT.h5 # huge_chunks_filtered.h5 + python gen_huge_chunks.py unfiltered OUT.h5 # generated at test time + +libhdf5 2.x writes a chunk of more than 0xFFFFFFFF bytes with layout message +version 5 (`H5D__chunk_construct`: "chunk size > 4GB requires +H5F_LIBVER_V200"), so never with a version-1 B-tree; a filtered chunk index +element of a version-5 layout stores the chunk's size in "size of lengths" +bytes (8). Every dataset is ` N: + ds[N : N + 10] = np.arange(100.0, 110.0) + if index == "implicit": + ds[2 * N - 10 : 2 * N] = np.arange(500.0, 510.0) + dsid.close() + + +def main(): + mode, out = sys.argv[1], sys.argv[2] + filtered = {"filtered": True, "unfiltered": False}[mode] + fapl = h5py.h5p.create(h5py.h5p.FILE_ACCESS) + fapl.set_libver_bounds(h5py.h5f.LIBVER_EARLIEST, h5py.h5f.LIBVER_V200) + fid = h5py.h5f.create(out.encode(), h5py.h5f.ACC_TRUNC, fapl=fapl) + indexes = ["single", "farray", "earray", "btree2"] + if not filtered: + indexes.insert(1, "implicit") + for index in indexes: + make(fid, index, index, filtered) + fid.close() + + +if __name__ == "__main__": + main() diff --git a/crates/clawhdf5/tests/fixtures/huge_chunks_filtered.h5 b/crates/clawhdf5/tests/fixtures/huge_chunks_filtered.h5 new file mode 100644 index 0000000000000000000000000000000000000000..60a8931a889d10a8a6d6839ef56fa2e8c79e6e13 GIT binary patch literal 188488 zcmeHw2~<;8yEcd-Dg|0w0cG;mS<7e<5Q&Jh0uEJ+We`CSL}q19s0i4CUO3Kc{YL{w%%L<|@i!yJJAb#EN>wVyWE;+--f3)!j{*zZp37PTFh(9PPp2?Du z;|I`h4e}EBRSLaNzVo>T4SkaQSuL@AhQxA0O7tIzDQNza?@yANBsnpcdkV?tjy(Ry zW8^1hNEnEYOH$7h*&ay=#{*7o_Pz%srjaM1pYeA7`{Oba<2^NH(xgd~$wrc2Ao;bW z=tJYbCcYB={|j*o$hN%J%XR%@i^$V=T3D`?L{>tB{15t37#e^5N5AoN-z@!)&KN(A zt&abc|M)!ISPgN1{>Nwk?|t__n^!Z?P?4?2umAhz)%eIxnIthWucW4*z{C9S=MkO> ze?KPcO*gKUkocz-kZ<@mA7)we##$LUp@IAOdjE~q2{iIP z#uxiadM^|W$2)L9!Rdm)`&OqzE(=n2XJ#_k2e5iku;euwYjwwLu*nv6$r-pd;C=<`Os ztTvT~n)}meuf&IZM#uaUVkBv7_TYXkaYH_Q^wZf|8d}*i;9V#l!Tr^?7L{_Rb@nMW zTdZds6MgGy--4+tesz>x6zP;!@K$_1Jf@;J#R~CD6_{i=&s`URNC(|kgIgv z;gR!jU=B+>f_GiCQk<=8W3NR_nCr1?M+*i$}0FGod;o%u!Y;A}g+nsUd2dLUQ+G^99OM zqNf#qMf8?@H(hD#?uC|3;dp-S{+4krS~A%t=uF~Aai9G9ZA?_hRQBO>ilexHZE2cI zJCuH#*3H>~4_PqPyI<+FEHrMk{WL}N$T1l{V(JL*hzrDcm7 z%Gxc(bF*ToWkFoDWZTQzZ)))MO0L=JD@)xOic9HQBc5Ltspn~1OF}E|68Aj9{uMtLU8}4t0$cCm90HK#QPc_N~w(XLk*d!=B_n}8*+!PSd6TTtew`-O7RHl zceeF^(omL)V0Sw9iyxA;oFLD*O(mh;2j6}?gs<1SLn|>_vW8!_)9;daex>Ysp}xRj zwR4Sa+jV>oPA6OUkDB^}nFBLNK83%7nFBKiW)8eL@a9N!2;LlcbKuQE&TL1IRA>`&DHwU>na?Oyag4`VB=AZ-tB?u@%pmYM1AfN;Rl~$;Ie3|;fAeg$&OX(WYe^Vdus^GMb7|(Yk4c;}eO+!`6$zE9BjFA-`A(mPs>_Z_Ud?-x=T?%0vlh+mgf=ci_&!8iMRE+0{ zXH=kidkE3nlldA$OZ)9sfo zSMO^(GZEV2tFFW|6q={r$#7$7sMWc`2NZZAkX0Q;CaI3wvqP4SV^`KXS9u^XPYc35 z*bo>pq1?vIKo}Yp!`S#6j8tf}QB$Aq@Tm39kdFbzeeeQ&j-EG>A;Qz+6fw5IBhJ-O zRqCz1^)n1gS$uS58ftDh-{}Ab%v^t9C5PHO>qB}c1%qlH>VJp83)1%A`?JLOy(+8z zIS4OfAfL6TrVyhN){k%y?Bn&2y1eBU{H1;gr5nnWda}6!6${+p;O3;{>+(emHuwEH zjPPl=y4RvK^oUY|H@@1>O%_X4rFOFmtm-f(<>ukDVZO5G!bc7TQAneqRrv-824$)= zt2qmUd1eXLBS=a`ap#+XbUaD63k?ynVn9$@T)hK-X;WM8c3o-Xu0UEO1tDXua8Bt6 z>#-X)3aAo~K1?~7axmq<%8^UPYG1H&u&~0y3Ja^qflLHn4tzNXA0T|d@eYI!5V1nU z3P}>uJAfn!kTH=7a4rS_R}PSe&-R-FH{^1X2PV_}!4=zW;}?v#%2(|4!Y1~SLi zoKEAi(zbja={qy`_Fq-795_#0^`Dl8 zZW{jdtv;p!Io(}Vwp(9XwuoC*p00tOauRQp&f#{t$r{LIarV$K(4|>U&F`L0i>@F$ z=o~Z-`G(MQFz4Wb08{eMFv3sTE9<7R8G>JgtsZH#QDTf+n9QdY5DSPAzMZh2##_75 z3@Gw|A`d9?fFchl@JIRwIEAUysX z)_LHL0`4f_juKf&;En=bcaX0D`3jJ)Ah!zxcNB0(0e2K|M*(*fat?tz3M#EoX@yEF zR9Z!mGE`bYbCJ?q1d2SM$ODQzWEX*z5m*_4l@W+Q@%{)P0tN0_;GPA#V^PQy+75Kb zKz9rb+rY3*nuEJQ>JFstKMrJVb=EdX zcXD`S6ul%8!;YRcS$sW5YSyKY9~!WPbk10^e1=YoUT}Wn;VGi4h1Sk1FRQaPl9nxQ zDC06U#1pTZlPK_&HTWzoimQ&5!MZnCPVG?gywCUdPvVX@r79;~v9CkvWy6Bh#{2^l2I=U(mE%L8)7Z2h^$lUQ7*9U zdO&f|w!L@LQk9BecfKF66m!*94Dnvjjii$z-aZ+K#S7=ql@z3`oLwJcY+Qk5nj33o zjQGm7ACBAQei>_H*BfgdP(QxHe)lJQ3(C=zsdY-f$z6GN317@RtNLk4=(>Zfw?}b9 zG#Q*?H`!rXJ1xH-Ohuifp6v5c`b)O;zU!GvM}Sm8{L3=gkM!{ zY9wvlX`-ga!K2f~4G%Vw?tf6WWCf-W*E_71a8JG{QG4kiz6IG5&KozR$GyF6oA_cD zISnq34v+pV)8o)e+>nGqp1r`mtLO5vBRLulV>D zq+5Be`qH&Jc8!;l9D>FIXu$N zOU@ID=VsLiy`avrG_?BV!)zh$5gxtY_f(AJp4`~T(i}WG{J^>BCh`y8duEPh%eN+$ zUi%c*v*7v9A@~-;dIP?lgT%tWGsGI`JG+j~MguL=*-&qRX^-2wgU_9mWU{$!?qb#Q z-xpX}R1Bh2gHp4FEg_zp?ZGtGrxb@rE(dL_3EU$Av_K;ygPz({THI=lhdLzx6uJq~ z(~7(4`is^bgQos7ot_lnTX;O2cmAX#SNoO<)d4+W4+rKnh$YqUZaN#6k?$+36pNn-W~&NBYf<%-dF2f%rf7d^AqkS7hZOJh(a@%IWTibe@P$A9GE#UbKuQ^H%Gd2 z;LU+I2i_b6t=L(HpcR5v2wEX%h1?t&D?)A#a&wTIgWMcWFG2|dN)S+jfD!~Q$}~d> z0xGRgX@yEF%B21dl~$;<{tE#mkBHJ{)hqK7Yc4VU=!s%|n`m*JrdZ7{PRC z?2bWKEg4bce!o7NrG!p%9-iIUjqyjBXRCQzbrBG1*fH~SF=9rqyC$Pd$#a(4k`;$A zWFl_IkcP69D?irB`4UD}+_iqX&|!5@&EB@_xHFwjw(cL*$oq^Hxe9+Wy}Kr%L#by@ zvt|r#Z&HCV3n5`F{U7Gn&zK}?+})ePZz9lX$_3Ix6vQX3t=~8_LV8UY`(uc=O*qk3mKtmXj74NGs=pd-@;-Lcx@SDF;&ytenVHt@Z^g2Ma4Ktgx_(%)yBY z@a4dlgYW^u2V@R3j39h~h!rAMNRmL3L_*>O%TVyAsu2VYOCq}P$$E@s3HIRP3hI{Xx7z?vwr;Wu_FWF%krl4f+ zICf{Fxu1zq-LsNVQ-8xiTQQMK*~rH?K6OglWNSG0F?Wz*dU#}~n{0M%9;?|>4Dxy3 znAFvYC`6hb^7Hk?Se?^)?rLhk%NG^e-y_7FGSE2aIhb?sK!7PBZT0`i2p1O_&=rI= z45H~1!Ap;$356*>M5Uszk4X~_#uAd{85?h{Gu^mWLh?(E$qJLFix!9f;MX`@Q9@!I zD=8uIVBE4x9Nv4scb(~`H4>8U;v?>Y@ymQ|IdQpPwvIE4E&9VTLi{svmW0HAbPT`u z|I2IYI&Oxfgp7o!=f=Aa{k=%kU6PZ=58zRmfl9FC8-MK^DG4Rf=Ren=p(~L;t0k7t zkXTNP!(}C=NGOQjpCmO&a{PVroSwqT;^TijM!qvc!a#Jsr1*=++b1dEc)-cc-uHmS zH1Z_$^N)_pj5kwa$|MOX5waECh6ed1Q^@8yF47nleO%%TacjxDB=mA!$B#-#toaVj z*cs%T^Uxv3MU7%~@MLOUb_6|pj?acy=mF>f=mF?~zwZJ6`cF-FU5gM;E|>B9<7GSM zx13fmQ}K=n`(fI~O|B8|w=K$<6dw3s>yKuOu3ik&*mb`4m2K!A1(iw8(;kF3-`co#_Ztq<~XIu|nzrF-wh`x`P!ymXYMg!Xej zgj3XNJjilX*_#?mI%@Wm|(p+EJ;c4<&ib1BLZqv};Qz+hflfb8#P6xdMkDfw% zG@PNVI{Z?*5)X<6P+q%@F~YFat4Xp!MGA7|FyQI{hj}eDsg>YN|@ni;6p+ z*ka1nH{Ijgg1BvT%yfD^Kk0%hxo0UaIa-K!w*y{~PlY6l@Bu#uhXX}p#;&fU9 zX#pQK@dldATO-AevhAPhQyag-TUl;0AW>cpNUTZ>A$(;`0D^%Inc_L}ZQC&eaeyHEHsWeskjGT-AiWzT?_n51&qM6J z%pT96yNY7&SD%60@jXj{IV?=*?4p(8JeAx~<+1R}1m$X2BHG6kZ`!wQYAZ__cuty&G}M zE;Mem{e&s5tp%Fn8>(!({Ems*vD&S^yra&P-0js+);_^3-Le2}vTA#I`%MkLUdc6E zePyXTLvblxYsB-*BK16NYe{Iuo&0C!_#U3mE`8FEwxT_sK`{`gt0$cCm90HK#GBYq zRT(SVP^IRsMQy0MLsu+D)&((3Pwe*UZ0rAo_Lw2sYsY?ySvtrwZc|C9_rbRxsjXPCvXAEoIjW^#u;Aooj6WvK7tLAIuz>Idb;@4rUI_9GE%q=Ki!D7`!>~ z=D?dH*T>Ly@a7EiPwxKY z@Rrlxx3|1qvV4(Yd;2ukgKpj~Ucb$i`Qq79-$`Dc@e9i4EIRq_do#KH8uVU5y|pzU zrsoG1dkS(zFZ1+oo~=gG(FN<5YWUc%&^tD!Uq(1 zA&^xaMItMX+p|N&y2Wdqt2_{xrv>32YzRz{tK7y!^tIbq3}fSOFjAq>MooRb!=u(a zLp}x=_rVMBIeOkih6qoOQ^fEPk2qIDRV1Kn{S1Rr79U-ihMF7BcRGLpGuIzjAtm2^ zvp%GEQZT6Iq5gLWydZ7=y+2DV4P0f_KL_Du4CJ%+)D&WuVEqUO!9HFOsmoh#!C&f^ zP`aT^sVAE&P_e)b4sK3LzAj(HU~}KE!w8>-t9vaO%_X4rFOFmtm<$T zk%ZSqPX+TKsufz+l7V*Sur3e zEw0{yzqF~Xce}2%aaSNMl7f&iS2(923SzrqqX0^@8m1gfIhb-_<vCHK)_Kth6nkNBYjpJ^o%U{!63d(N5ZC+RKyd z3^}^=5Vhu|!?#)NWDI_BdGm8PVz6b+i)x~XfxLL0T7*WPT=lCsj2jX1t>&vsowfIW zhO4s{gb!bkp#sZ}l-B$?5K@vfcX9vPImg@^lUKl#~0; z(m7m2n`*REU^<=zzwb||Uk8{Ed9AzJvolxDJaO{ir)8(FKy5}D2 zDfH#==|r@A?9}eUmYjl!A;RPbA&0nM%U37ZOhHNsW0*tq+6nm#f!1uIvXD;`R`xOk zcEU(vM4jknjCv4*tg!^5c{GJF#uV=AvlFt2)KRUZKT+i=7Jnnm3Y<&G6NSzCR&5K$ zdB#q>$PBPXU|2@&ajUy{!E)9%Nk_cwM$tQrdQnN0F{Lp~$u{mSO@)_fV zUxzWz*1YnvI$I+VrxtgD@Qafu@Rc?AEG>$QIh>8UH&{;XQ1ZOb_xDf26lFQ-ihUhQ zFB=x5Hm=2pOVqk&CNdsHw`(zxo1PMLPX@yCu0LajYp8ARDZiYQjCxs+)+y$WtF;zZ zMAj+wC>PjvJ)joYw)bvYswAh@`+*6)Jh{rC#|-JkF+C`VVO)+zlacjeh7d@=8=>Zc{4>khKs z9>opOWN?bzM1)^{KVUk+GH>q(h>Q+V2lkCpgkSs-Cz`61g&FtV=my*){HkhGBWdeS z6E!ss9-S_3c(9Rl|AVq6D{y?IcUUdqo_tZF_R>Lo3$i7gH*N@QyuEFk_+l104K9uj zkNz#wy}opCM0!FlMm;Aw8X7hrGCD*Z7$Bf{4F$m&%oxd(OGmr}SHwNS z5wv$881hiHZcyxva)|jYD1vWUINyL1Dzv{#fenCz{IrPn@%^(=V)GezVU)*JBc z9E61a&JY8yANF|CvrtD89YnLIBM@fv1*C8iUMTei?Ss6#{@m_EU4 z$n+RQG=9r7HSaKRkL0PZQbqKHZ@OC9UU%?N7uY^caaiq9_i{2UW@q}dDAlX zQ+$>Cf^A=vAeidH zh&r%lg4eKe1s_?B*?SdhCcNR;3kZPT;yTpZV}NahkDb>0YMqN&=DTx#!Xxd%%Z?9G zXa+L}W)2mX^uf%5nFBKi-W+&yWTNRZ54<_>=D?eSpmjX$3_&Xdtq`noL+p+2vwb}Cgqg$gpPDpW;@7LX z$xP{uA_yh*N}uDbQ8_Jghs`a9SkUtEgl6O+!&Q}}EMv6g`Z;2Nh!Kyz3K63@voCpP zP|AmkkH{k#$p^&2qq8tlL9IMISXVmJFYPZP5=dOIC#1F-n~G6-$geSyh3w9E6M~lI zJdvPfN<%6|f%lO$t{HjAb3&GJuSKOEseITPSG8V@!Klcyt3=|JR{l2{Z)3p0*3C_b zDep!n?R+vNE?DY4ml-YDX>;f=;(|!OFIat$!aWq&*=<;V zDDXzJ#$_32|3UfiY&8!_Fv=DWJ7#`9j<>k1*Ik3S_&jH+Em=Y7_r>iPLQHwC{8%Sv ztl!7EYyEVg!|I@#y?@c~LrS2m$W<6@k=|XC(4o{br&%)ww>PQ4m?e^Be3)NPQQ!%` z8CO1(o)8yIYwI^g+?&PPCd!8~DCIeS0CDjpg-1SnK`DV=eZ2}X<<$$X)LBtDm1T;3 zNC~w0{_Z~$HzJHy@sPbGMy_!A0|$(x$0D+#CX^4gDn%j^7u?e+)JDeSoq)l=a8W0fZ0#=aD4C9^ElEUkG^yF>y89wrrEpTo)ndo@VM9 zA$KH^@#AMFy~8{+4QjniPwf-V5KO6R>ArsO3%>kxMxuY>XND(K-*JZN!syCXYI(#l z&gkrB(Y%?*>FB(zQX0BW>1>}E3gjKLvNP2{`uQ5}y?bIT%#NKl7E`@shh><8lD*^D zosH&xCPsD7kTRpcVW6#;$fa!L;~SqkrERh`ocowN$S^%T5^>^X*XFUBEyW<8_l-$i zorprD=^;N~PmI+$t>><$_Pcyhq5VBV%qatngPwyq2M+|864F-xkBo3}kpW#nSi>Ni zJ`ue1IGRwH;zLv_3j3He;b1HwS)Q@+mWU_MwFvR#aT&iqUbbU?%V`BO74L|!AEs^G zcudPUFU0G*@o^>P?_XB?LpYL#p?Cn+%%h7c0@}3#)y`%l{PYTjcBzKWVJEs z#~8$LEI|>W89bWwieDMStY??3Qq#u~stSo- zHxtZDLb-zXH*;T#QLELrKFG)ET+BF>?oH_cXOwv9C`$?L=X?mKsJUul%QqWIyA|M?C>;D{{Az7l&;Vm5pW4J5OM}u@12g3Ozff4uVTWn{*hiyRVjK=apx0T zOu6=^dmI+bZKGqRv+Mau7gR~ia$a(@5Tjdz<=u6W47S{hy&-9sz0dmNfjFI(Kw7{@ zO$?yTyfsqnDBJ$2KDCij?Y`}0y_JU0t$8o;_CMaJm(`}yP;-A8Z2}fN<}ZQ;vj_KM zSn!9BemYxAMU-n3m}OgwN>WoC@xiU1pj?}}0!d>pigZdVpa9jH1NRVAIG<^JL=gk3 z_r4jk*Om4?+W8kiwfgZad)s!*Ko}s%zKyur7UVJ2F*mSb2t5z6_c41sgYGJdx!-&S za>p^tfjKNp=;)%A;yjfERC_GEGC{c(mWXi7@n#vj?eJG9Cj{0x%4RoPUU)bIqmk2G zh1$weM!ADaR$xhSy+bOB)d;Skt<{gA*m-p8-G~iuW^-jM?(!rPTC)hVoSk2baV|Nd zHnNC=PA@qB4}j`~>I_owJR&Quim4%TH5HP(ADbiHZqd_*j30hb)-t-H2OmA(B-8gefkq1)Af4 zYTGVKjlI6Sqt28>dpDG|PcTciEI=UZwwJfx)ZputT(i|zmbx<(m(sNcQ!u8Ur)@0> zt+pt>?v1gKVX*P;N`cj$`6$hyed zX-%NLJKOp{A&fbqy>#rSm?eWek+i*Y zjqP8c=%)T)=0+);eD>>dwCP}g-uK!DSR zQPe)FRMr!1RS|*i#3rvG!&NKMJ|D#Ht1O@hbk!Oi(T*;a^Y_8bk;C;ncyr**fj5_> z7=(Gk_Y{h^ize_XdhS8*z@w+o#=@SXx}y|NIG8yIS|MnKpcSp68b&D)wEk`85WG3a z%|UKX6c|;BfN7AMgWQ~kXg(+;LeL5&2q-~72?Ccs-J1Od4@@kTGm%|UJsa&wTI zyNSyvA-hzi)4OyJYzy!}j)Rt_R(`UA%sqEAz#( zrM{EAJmVLX%~^Ev-S=j4`!(pjgnDahKupgMEcO)SieBdF-#lB5q$f+RCE;({pH+?2 zGk0Zs?51k)u8D0(1JTJ|Sl^716&)d#S|sd4AtHPTiz+6sEBpnO;CY@wM+B)DlM>IU zK=t+zqPHg#D-HVi1q8=OOUC9lSF$Nk3g<yQZnKYVX(RH*I}}{hO2umN|8okf;VpGBIW=fX!0 z1yM+&p;h??sBmVgG^;sFtnKMLORye6QYwl&-wdSVNwQsNh>#Tng3{vZ9r#O|+IqL^ zN*i|t(jqAc8FPhm3gUya8#W4{L@Qy+DQ#mhjE%p+BiLx8ras@{QR|%{9|H_E;sy8| zJ#QjIgr~p?6c$#9SRqLQ5i3r)Kwb`#B#@Vb@PYIV z5bO|A!BBmGR4`N@kTF>826;IsctMR73SLlSMdq*yst-`s2PDbgmi39wrn{TX2<1}y zOgX1%YbPtt|5^E7xcjlN%Tx3|On-Mu$M@;G(`*BoV{1;Qaan0wK9BUBnS1=bT>O_t z$D^IJ&9s*%*%@+l=^<*(ONVc>*vS});PU3@aKvE8nithX5d(SgJhcdoJh|#uaX5(~ zGmpW_j{}A)yq%?NE=IQKHt6%ZvJIsOe#8v-kY3QcmPv7c`K`3sgyQ|7}>r2ZP zajVMHHPBN|(!r;5xQI5-KrV~3hlYVJ&2nmf_jFox1=&I8pmE4Igr0*r2M+|8l7EH~ ze$rl9HVG%~zlSgS^)KV+%>SEn@VO7_?B?S0;T3uSdH{L= zdH{L=df=bk10sstEyai;cZB(NuJY`S3fA$9%1(87m;V^)8EABF&84GDROw|JjUG?q z*U9My>&bZcYO~wxALoP_ILbaSJE6Kc;n)}Lr^`Lfbk9B3Q|Qa#(}`&L*s0xxEja}d zLxjl>LJo1imak5*nSzuO#xRHIwG;9g0)!7<}I5oLU z4N<=%!cR`3z*pAbv$QBK=5RLZ-e5VkL&@_#-`_t8Q? zsCCbfnsU+YT1@1sr^MWof%fyPKVyb>4YawZ{Blw<>SaM%r>E*S*O&aTwvSv zfC{kN-n(h3lAN0F2POzVc`xWj(n%3-pA5v}5Ps-N3feZ3T_0j>T!C?mjWshyd}Z4Y z$L(^zj5q$&8*3g=Kfc0#_a}S{%F&gnbxOa$@z&*mRsx~!}w(c}hQ{&*# z>Eeb58%g&+C|j}u=STGpt0ml%FG|#2I*4yUwuJM>4FOcQw`~(&%p#}3#nIu>zh!zH zT8SHyP{^~_mky3dPsqi)o)R4m4I2;{Eh3u>5Kt6Rg5V5hjAY8CBVK|l;vV4$+B?uT zsyj#313T)sNkRylnGxU-9uRNVoD_^`&nY-|!aG{;x%8 zljZkx_I+NTb)VYiQO+9Co+DzyYuYVNrI_DKO6o*pw1_%5Pe?K0(F^JJ$>l&GkIPj;x*3tN=zplw``$vP=|;*IDLZGnCUTy zXndDvYTjYs9?4T*rHbf@-*mOIz3$+lF0g%?;;`By@wWF&Jkr%4ycY8{@}_0%rx*_R z1>3$TLEE%;$>CWTpPRg1UwYXZ{(}eNskGl`WTiT?H_pR>ISt~?J9jsojmyaQ6;TJ* zOz;|4uHYl9F?+9K%|tdV_5#`s)Z#kS+hc%jgpZxp`)ZwwS?0TQe!?T|!pn{iQD_D; z2WAd|OZ36aftdp{2i_cbbEE@$nFrn+cyr**LC{KC1BGM=S|MnKpcR5v$jyE zHwU>n$j#yOB9tJY1W{a54kZYjTZa+^lpvte3YAu|(k@h5q0$PK)_)D4wM1`iposw8EZqb=9Z5d%bwc=T0>7}c45$vcBm zK3;r89tjUVAPyd#g^>zs<>A4)(wTl~e^EY0;=(;4wbj^EjM77Xjgc&5cfOmDKrH8p zBoI>?QYogskF0Uc=tG_pvdnueD)mU^Vp*Sp}@}Wy2B$6|6GF!&FH>hk!Ezqpew~k zI+`^u%RKuJ%ExD`c}Rj;ws_bv^Yd}M#bv$j8blfBIZJKH3QE5>ZpRR!Ep_F`Iyqzg zUd~nlZS&Nd?9%ku3AW{CbK4SNP4i z^0D-UxNurqzcErdF4i_tK8!&r&-nvLf;lNX^4SYYBKPX+Rfs9KUU;R>io&TZQ|v<$ zxy|=?|5+jzVYG^e>@6{Jh07l}U?e>jkrg$ee5_R|Qa--mo=&kY!<36c)WI<2V9JTg zE{R~}_K$* zYOL3B!5YE`DC@JKtp6uHW(XhtFNY7q9^ElEUkG^yF>y89wrrEpTo)ndo@VM9A$KH^ z@#AMFy~8{+4QjniPwf-V5KO6R>ArsO3%>kxMxuY>XND(K-*JZN!syCXYI(#l&gkrB z(#w{Z$LZ+2tx_7gPU&o)*s^-Ztn5rRkbb^~d+(kY3$tUVjm11=Y3;RS0|zn zX?n=d*AruPPV2d=sr@crRA_&X5Oc~vjDTnjSSlkWI(s@{noPo=No=$c%J z9GXt&GW?0o{R4$UEBhYHD59sZhA1l>D<$5ue1%O6xA#QbLC>C8VKmsCiwZsLFzC3en1y{_;=_Me??{uzWk>Q9EAAshFTDJAZ+s zDqECyA#V&*mHkSy=_1*{vfXXRD3s+&x&dMX<2CU+1v@aN(p;d8sLH|)_a#q+@JF_6 zAbHfj|56R3U1b-Z6!EBK@1Us43b=L$93Jgyzdu3pmd0vZgm~17I6lr3RAuIsBvsj$ zwwAU+3Kh!K93t7ktXF8!G0HSgm8OE&z(%?RgY=u24eUDo9HJ_FtUpp=j$1-@9m%7% zYnBLw-gbM(pOjQ*^KvpMsnN(S-PIzhvhbR?1d6JRmr^bw zJ*;^Vg{_T|N2xqXdRP!tm~V$WAm^s~hiaXDznK^K?8Gg}aPL?|@~A}>UBNr7*jKlU zc+~d%jQjGLpCA>H2u{yGN|C%>AIMcfBfVibxXX=VWK8WK85yICxMk%`4biMtNFJ$f z6OkTrKRlt>z_MB+5$T};E1%V@g*%|}wLA|I#MLff{Q>Qp)8YE+bBlHu?}yN^sx8{ z?;S;Y*w9;yNDq^`n?1S{a0j@5j6{r#N^Wu8jo;yx-2TBxL>GCv2S1w51}@)`bde=z zF5|xJvmutukt$e|ZbNZdx(GBkpppLEgt>Bpk85y00zBwf6A^TxK zFGYITIm||+hZTO^SwmZJ2do$t@`S z79)2UXHH7mU8=c5aam^46Rx0M8O?kr< z>EY2aPegji$sKCoU&9@cIBJg=8KoQEuUzp)jKkOCo{b^tBEigonUjz>0cOq$%p90G z@aDjqLoSns5qNXp&4D)uL8~Z!s)C>uf>sDxA!vo%9I}V|;~+N&xjD$qL2eFbjiCeq zB?u@%KnVg<{6Gl;Dy>jyg-R=#Jh=;%R;aZ8>i{K0Z2rpin+#+h8Ek!bG3wN0&2^hU zJACKIFnhC;7gkR#y*?{JTJ_k)M|sBtPklkD5jMN-HADny_DalXx)^WKzRdOxBBk3V*SXh$0tlq_$D5ag zrV6g|vvK9W`sOT>M~yY`b^=*Z`PNG(L_s*iF0vA19BgTtl?Z>cFpsI)F)r9yWf+oO zXkHrH?dj)pS5d6wx09tAiEtV-c&0vB@b7xo`Q37-k@8^4!Yg)|x-Tk$i%99>&2rz? zQc$C%s}95=`8-;Oi6(a7eQ%M}ef@srWVg@;a;B-bng+QSi$~EFIdBcKr-fUUth=e#TAI4J1iey)kx=&!^F#&@baZ^>Sp2{xlIWl}31Cj<>`;1kk z=tcaxUvPV?Bl!)ey>fml6oatZyC=?N0&4F zb1LRJtTxO%bn+y=T-K|d`pQyw2IIO~F-FTGWiLtH$Ef*X0%>IOL+O)#(_yV(RTHcd z8X(9vMnDZ_5B=y>+}?h+P+wK5pt$M96I*eQz?6e22U8BLoY?S#m4k&97FJkTNp}=M zQoxr3Uk<_t2p>pmu$BYi14OJ4u|krBvf_u^=dTLGiW;s~Nvr-Y*UM|= z3-nA1OT&ijdaUh=S8kQh#h4c9R#(G0QfKXbk7BG!%CKk6(-MT}39_L$Wp1WBZ4$8u z(S5$gboP48q^_$DtG|c{Pq?ok8ZL$8^yXcfs#2RT+2mSNh@Ka;Ze-G)40Z}#u48z~ zhp~N#R7Gt?GBf`t3}6|0TZTArhF3u2uxAWC2XhV{2rwm+MAr50|G)?UNJTl5GBr0=gyzs)Y+@UPSnB>AGn_WqIJe@;D_QNdhf~)(^u3r0e|HgT8a@VZF zWq(#aU3IB-?qbJyIZcNru?K~ z(l!*N7?Q2owWr6buvm~ji&afrIJC!y7$&$v{?<#pZoTdthVAge`&$cz9z+LYB%SDF znhTq*4$z#4K6N2aV~oq-5yEuFuq9DR2<3@)4B-aOpE&XaEJh+h%w-BT{B-#Hq`~n{ zQb=T9wcc79dPK>t4O1ss=PmC>oMvCm_Pa(2I6reT+*}%Jbx!`{N{qcs>vmxypyhgY z=hy(n{uNVX<1P{Q066)mu5bv*uOYtCW&-~ z=7=XQln}A!4o$Qc8bzmj$C+Zo>ll4+pe$9N`_|G4vwv}T zbD8Eeq>}tDGe!JSXM@FxsGC$v_84N?Fh#8wd|%n`!%J*jD0S(3RxyZ{Z`OyfQ40(V zKoDQ@CntNoaYG|4u$;_2SXruG(A@`cc@LlYx=PJl_!SBvSjD zyHJlhZ{$?vb@k&}_FnZ^m|HI?J_>b{Tx)6)_7|cOK7<%a-(J&tCzApm*Dc#H4m=K_ zfX4%?zEMG%;fwN;3x$~EaO5LPZGpq11O9n81(^9RfW}pu@38s^(c5T69eZeEXi*Y6 zcxNE)SA1+F_^ER>lKuP5pc7-Dl`Wny+3G*R!X>qq+1%PfNMX%TVWI;v7{?Ss$Yf%fTrm50E{yAC#

(;I zB(^P5m2$E-xzR1|q}t4alyDNV9iGLjp-2wx1P#RqpDi1^;=Uc8!rjhISCHI#7wy7I z3bSolv#?a88cx@xNDi6uBW|On{&QxfP?F}wn{p5so2y;$3=edUze_WsZd|gwYcmDe zKJ&smUIf{GO+mKzvZ{+rbrH#7!w8p2xyHgydtGVkh| z;iYinS{-E%cJHPAN^zf8Mh5E(kd#sDxr8E7~AU8+42NIVdHwU>n$jw1+PGk*M`$7o< zN)S+jfD#0hAS5Id5~0!xl~$;r#}%hlbEN3SrDd0JS-v$gmy|)AmX(k~ z0gId&JUXdY(_O=z5X{rL)AmZ|qI$%c$6~?h6BYe!`;q$KE%x;{wis{_K>OH+2<(D~ z0*K``VglRh!(YzmQ0mDxKg)N(Es3+piC-n-QS_znG~5!sj`%HgNEP(C+y#6Kc_+^g zpdtKBePJZN(-`%xJ4h^YVWaJZE{s4)61?D(YMcdj=~o>gS;H%qDBp{fYqA3 literal 0 HcmV?d00001 diff --git a/crates/clawhdf5/tests/huge_chunks_interop.rs b/crates/clawhdf5/tests/huge_chunks_interop.rs new file mode 100644 index 0000000..bf2ff19 --- /dev/null +++ b/crates/clawhdf5/tests/huge_chunks_interop.rs @@ -0,0 +1,296 @@ +//! Chunks of 4 GiB or more, which HDF5 2.0 writes (layout message version 5, +//! `H5F_LIBVER_V200`), in every chunk index libhdf5 uses for them. +//! +//! `fixtures/huge_chunks_filtered.h5` (written by libhdf5 2.0.0 through h5py +//! 3.16, `fixtures/gen_huge_chunks.py filtered`) holds one dataset per +//! filtered index — Single Chunk, Fixed Array, Extensible Array, v2 B-tree — +//! whose chunks are 2^29 + 1 `f64` (4 GiB + 8 bytes), stored deflated twice +//! so each takes about 20 KiB. Listing their chunks is cheap and always runs; +//! decoding one inflates 4 GiB, so those tests are opt-in: +//! `CLAWHDF5_HUGE_CHUNKS=1` (run them one at a time, `--test-threads=1`: +//! each needs about 4.5 GiB of memory). The same variable enables the +//! unfiltered tests, which have h5py write a sparse file (about 44 GiB long, +//! a few blocks on disk) under `tests/scratch/`, and the writer tests. +//! +//! libhdf5 never writes such a chunk with a version-1 B-tree (a chunk of more +//! than 0xFFFFFFFF bytes forces layout version 5 and with it the newer +//! indexes) and refuses to open one; `chunked_read`'s unit tests cover that. + +use std::path::{Path, PathBuf}; +use std::process::Command; + +use clawhdf5::File; +use clawhdf5_format::chunked_read::list_chunks; +use clawhdf5_format::data_layout::DataLayout; +use clawhdf5_format::dataspace::Dataspace; +use clawhdf5_format::message_type::MessageType; +use clawhdf5_format::object_header::ObjectHeader; +use clawhdf5_format::selection::Selection; +use clawhdf5_format::superblock::Superblock; + +/// Elements per chunk along the chunked axis: 4 GiB + 8 bytes of `f64`. +const N: u64 = (1 << 29) + 1; + +fn fixture() -> PathBuf { + Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/fixtures/huge_chunks_filtered.h5") +} + +fn heavy() -> bool { + if std::env::var("CLAWHDF5_HUGE_CHUNKS").is_ok_and(|v| v == "1") { + return true; + } + eprintln!("SKIP: set CLAWHDF5_HUGE_CHUNKS=1 to decode 4 GiB chunks"); + false +} + +fn python() -> String { + std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string()) +} + +fn python_available() -> bool { + Command::new(python()) + .args(["-c", "import h5py"]) + .output() + .is_ok_and(|o| o.status.success()) +} + +fn interop_required() -> bool { + std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1") +} + +/// The raw layout message version, the parsed layout, the dataspace and the +/// chunks of `name` in the in-memory file `data`. +fn layout_of( + data: &[u8], + name: &str, +) -> ( + u8, + DataLayout, + Dataspace, + Vec, +) { + let sb = Superblock::parse(data, 0).unwrap(); + let addr = clawhdf5_format::group_v2::resolve_path_any(data, &sb, name).unwrap(); + let hdr = ObjectHeader::parse(data, addr as usize, sb.offset_size, sb.length_size).unwrap(); + let msg = |t| { + hdr.messages + .iter() + .find(|m| m.msg_type == t) + .unwrap_or_else(|| panic!("{name}: no {t:?} message")) + }; + let lm = msg(MessageType::DataLayout); + let layout = DataLayout::parse(&lm.data, sb.offset_size, sb.length_size).unwrap(); + let space = Dataspace::parse(&msg(MessageType::Dataspace).data, sb.length_size).unwrap(); + let (chunks, _) = + list_chunks(data, &layout, &space, 8, sb.offset_size, sb.length_size).unwrap(); + (lm.data[0], layout, space, chunks) +} + +/// The fixture's datasets: name, chunk index type, shape, and the scaled +/// origins of the chunks libhdf5 wrote. +const FILTERED: &[(&str, u8, &[u64], &[&[u64]])] = &[ + ("single", 1, &[N], &[&[0]]), + ("farray", 3, &[N + 10], &[&[0], &[N]]), + ("earray", 4, &[N + 10], &[&[0], &[N]]), + ( + "btree2", + 5, + &[2, N + 10], + &[&[0, 0], &[0, N], &[1, 0], &[1, N]], + ), +]; + +/// Every index of the fixture: layout version 5, its chunks listed at the +/// right origins with their stored (deflated) sizes. The index elements +/// store a size in 8 bytes (libhdf5's `H5F_SIZEOF_SIZE` under layout +/// version 5), where version 4 would use 6 for a chunk this size. +#[test] +fn filtered_huge_chunk_indexes_list() { + let data = std::fs::read(fixture()).unwrap(); + for &(name, index, shape, origins) in FILTERED { + let (version, layout, space, mut chunks) = layout_of(&data, name); + assert_eq!(version, 5, "{name}: layout message version"); + let DataLayout::Chunked { + chunk_dimensions, + chunk_index_type, + .. + } = &layout + else { + panic!("{name}: not chunked: {layout:?}"); + }; + assert_eq!(*chunk_index_type, Some(index), "{name}"); + assert_eq!(chunk_dimensions.last(), Some(&8), "{name}"); + assert_eq!(space.dimensions, shape, "{name}"); + chunks.sort_by(|a, b| a.offsets.cmp(&b.offsets)); + let got: Vec<&[u64]> = chunks.iter().map(|c| &c.offsets[..shape.len()]).collect(); + assert_eq!(got, origins, "{name}"); + for c in &chunks { + assert!( + (10_000..40_000).contains(&c.chunk_size), + "{name}: stored size {} of chunk {:?}", + c.chunk_size, + c.offsets + ); + assert_eq!(c.filter_mask, 0); + } + } +} + +/// `f64` values of `sel` in dataset `name` of `file`. +fn sel(file: &File, name: &str, start: &[u64], count: &[u64]) -> Vec { + let ds = file.dataset(name).unwrap(); + let s = Selection::Hyperslab { + start: start.to_vec(), + stride: vec![1; start.len()], + count: count.to_vec(), + block: vec![1; start.len()], + }; + ds.read_f64_selection(&s).unwrap() +} + +fn f(range: std::ops::Range) -> Vec { + range.map(f64::from).collect() +} + +/// Reads that touch a few elements of each 4 GiB chunk: written values, the +/// fill value (-1) next to them, and the edge of the dataset. +#[test] +fn filtered_huge_chunks_read() { + if !heavy() { + return; + } + let file = File::open(fixture()).unwrap(); + let mut first = f(0..10); + first.extend([-1.0; 2]); + for name in ["single", "farray", "earray"] { + assert_eq!(sel(&file, name, &[0], &[12]), first, "{name}"); + } + assert_eq!(sel(&file, "single", &[N - 2], &[2]), [-1.0; 2]); + let mut edge = vec![-1.0; 2]; + edge.extend(f(100..110)); + for name in ["farray", "earray"] { + assert_eq!(sel(&file, name, &[N - 2], &[12]), edge, "{name}"); + } + assert_eq!(sel(&file, "btree2", &[0, 0], &[1, 10]), f(0..10)); + assert_eq!(sel(&file, "btree2", &[1, N], &[1, 10]), f(300..310)); + assert_eq!( + sel(&file, "btree2", &[0, N - 1], &[2, 2]), + [-1.0, 100.0, -1.0, 300.0] + ); +} + +/// A directory under `tests/scratch/` (on disk: the sparse files must not +/// land on a tmpfs `/tmp`), removed when dropped. +fn scratch() -> tempfile::TempDir { + let root = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/scratch"); + std::fs::create_dir_all(&root).unwrap(); + tempfile::tempdir_in(root).unwrap() +} + +/// Have h5py (libhdf5 2.x) write `fixtures/gen_huge_chunks.py`'s file for +/// `mode` into `dir`; `None` when there is no h5py (a failure under +/// `CLAWHDF5_REQUIRE_INTEROP=1`). +fn generate(dir: &Path, mode: &str) -> Option { + if !python_available() { + assert!( + !interop_required(), + "CLAWHDF5_REQUIRE_INTEROP=1 but python3 with h5py is not available" + ); + eprintln!("SKIP: python3 with h5py not available"); + return None; + } + let script = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/fixtures/gen_huge_chunks.py"); + let path = dir.join(format!("{mode}.h5")); + let out = Command::new(python()) + .arg(&script) + .arg(mode) + .arg(&path) + .output() + .unwrap(); + assert!( + out.status.success(), + "gen_huge_chunks.py {mode} failed:\n{}", + String::from_utf8_lossy(&out.stderr) + ); + Some(path) +} + +/// Positioned reads of a file (no mmap), counting the bytes read. +struct Counting { + file: std::fs::File, + len: u64, + read: std::sync::atomic::AtomicU64, +} + +impl clawhdf5_format::storage::Storage for Counting { + fn read_at( + &self, + offset: u64, + len: usize, + ) -> Result, clawhdf5_format::error::FormatError> { + use std::os::unix::fs::FileExt; + let len = len.min(usize::try_from(self.len.saturating_sub(offset)).unwrap_or(usize::MAX)); + let mut buf = vec![0; len]; + self.file.read_exact_at(&mut buf, offset).unwrap(); + self.read + .fetch_add(len as u64, std::sync::atomic::Ordering::Relaxed); + Ok(buf.into()) + } + + fn len(&self) -> u64 { + self.len + } +} + +fn check_unfiltered(file: &File) { + let mut first = f(0..10); + first.extend([0.0; 2]); + for name in ["single", "implicit", "farray", "earray"] { + assert_eq!(sel(file, name, &[0], &[12]), first, "{name}"); + } + assert_eq!(sel(file, "single", &[N - 2], &[2]), [0.0; 2]); + let mut edge = vec![0.0; 2]; + edge.extend(f(100..110)); + for name in ["implicit", "farray", "earray"] { + assert_eq!(sel(file, name, &[N - 2], &[12]), edge, "{name}"); + } + let mut last = vec![0.0; 2]; + last.extend(f(500..510)); + assert_eq!(sel(file, "implicit", &[2 * N - 12], &[12]), last); + assert_eq!(sel(file, "btree2", &[0, 0], &[1, 10]), f(0..10)); + assert_eq!(sel(file, "btree2", &[1, N], &[1, 10]), f(300..310)); + assert_eq!( + sel(file, "btree2", &[0, N - 1], &[2, 2]), + [0.0, 100.0, 0.0, 300.0] + ); +} + +/// Unfiltered chunks of 4 GiB + 8 bytes in every index libhdf5 gives them +/// (Single Chunk, Implicit, Fixed Array, Extensible Array, v2 B-tree), in a +/// sparse file h5py writes: read through a memory map, and through +/// positioned reads, where a selection reads only the rows it needs. +#[test] +fn unfiltered_huge_chunks_read() { + if !heavy() { + return; + } + let dir = scratch(); + let Some(path) = generate(dir.path(), "unfiltered") else { + return; + }; + check_unfiltered(&File::open(&path).unwrap()); + + let file = std::fs::File::open(&path).unwrap(); + let len = file.metadata().unwrap().len(); + assert!(len > 40 << 30, "{len}"); + let storage = std::sync::Arc::new(Counting { + file, + len, + read: 0.into(), + }); + let positioned = File::open_storage(storage.clone()).unwrap(); + check_unfiltered(&positioned); + // Every read above together: metadata and a few rows, not 4 GiB chunks. + let read = storage.read.load(std::sync::atomic::Ordering::Relaxed); + assert!(read < 1 << 20, "{read} bytes read"); +} diff --git a/crates/clawhdf5/tests/parallel_integration.rs b/crates/clawhdf5/tests/parallel_integration.rs index bdcf4f0..964d3c8 100644 --- a/crates/clawhdf5/tests/parallel_integration.rs +++ b/crates/clawhdf5/tests/parallel_integration.rs @@ -283,7 +283,7 @@ mod parallel_tests { for (i, chunk) in compressed_chunks.iter().enumerate() { file_data[offset..offset + chunk.len()].copy_from_slice(chunk); chunk_infos.push(ChunkInfo { - chunk_size: chunk.len() as u32, + chunk_size: chunk.len() as u64, filter_mask: 0, offsets: vec![(i * chunk_elems) as u64, 0], address: offset as u64, From 5a20cf04e84cf511cf99a95f55e1fac6800ba0a7 Mon Sep 17 00:00:00 2001 From: osobh Date: Mon, 28 Sep 2026 23:18:36 -0500 Subject: [PATCH 2/7] Write chunks of 4 GiB or more; FileEditor refuses to rewrite them The writer gives a chunk of more than u32::MAX bytes layout message version 5 and, filtered, index elements whose stored size takes the file's size of lengths, as libhdf5 2.x does (the Fixed and Extensible Array structures match libhdf5's byte for byte). Chunk dimensions of 2^32 or more and filters that cannot take such a chunk (LZF, bitshuffle, bzip2, Blosc, pcodec) are refused instead of truncated. Chunks are extracted row by row and one at a time; deflate no longer cuts input at 4 GiB - 1 bytes, nor holds the worst-case bound of a large chunk; an LZ4 chunk of 4 GiB or more is read as the registered framing. FileEditor refuses writing values into, or pruning/allocating, chunks of 4 GiB or more before anything is written; growing the extent and attributes still work. Co-Authored-By: Claude Opus 5.5 (1M context) --- crates/clawhdf5-filters/src/fast_deflate.rs | 35 +- crates/clawhdf5-format/src/chunked_write.rs | 580 +++++++++++++++---- crates/clawhdf5-format/src/ea_writer.rs | 5 +- crates/clawhdf5-format/src/filters.rs | 86 ++- crates/clawhdf5/src/edit/earray.rs | 16 +- crates/clawhdf5/src/edit/farray.rs | 2 +- crates/clawhdf5/src/edit/mod.rs | 19 +- crates/clawhdf5/tests/huge_chunks_interop.rs | 220 +++++++ 8 files changed, 788 insertions(+), 175 deletions(-) diff --git a/crates/clawhdf5-filters/src/fast_deflate.rs b/crates/clawhdf5-filters/src/fast_deflate.rs index 6ce268f..2b4e5f0 100644 --- a/crates/clawhdf5-filters/src/fast_deflate.rs +++ b/crates/clawhdf5-filters/src/fast_deflate.rs @@ -327,24 +327,53 @@ fn inflate_bounded(data: &[u8], size_hint: usize, limit: usize) -> Result Result, String> { use flate2::{Compress, Compression, FlushCompress, Status}; // zlib's compressBound, plus the zlib header and trailer. let bound = data.len() + (data.len() >> 12) + (data.len() >> 14) + (data.len() >> 25) + 13 + 6; + // flate2's Rust backends (zlib-rs, miniz_oxide) zero the whole spare + // capacity on each call, so a large input's worst-case bound would be + // memory held for nothing (4 GiB for a 4 GiB chunk that deflates to a + // few MiB): past 64 MiB the output starts at 1/16 of the bound and + // doubles as needed. + let first = if bound <= DEFLATE_EXACT_BOUND { + bound + } else { + bound / 16 + }; let mut out = Vec::new(); - out.try_reserve_exact(bound) + out.try_reserve_exact(first) .map_err(|e| format!("deflate: cannot allocate output: {e}"))?; let mut deflater = Compress::new(Compression::new(level), true); loop { let (in_before, out_before) = (deflater.total_in(), deflater.total_out()); + let rest = &data[in_before as usize..]; + // zlib takes at most u32::MAX input bytes per call, and `Finish` + // ends the stream after the bytes it took: input of 4 GiB or more + // was cut at 4 GiB - 1. Finish only once the rest fits one call. + let flush = if rest.len() > u32::MAX as usize { + FlushCompress::None + } else { + FlushCompress::Finish + }; let status = deflater - .compress_vec(&data[in_before as usize..], &mut out, FlushCompress::Finish) + .compress_vec(rest, &mut out, flush) .map_err(|e| format!("deflate: {e}"))?; match status { - Status::StreamEnd => return Ok(out), + Status::StreamEnd => { + if out.capacity() - out.len() > DEFLATE_EXACT_BOUND { + out.shrink_to_fit(); + } + return Ok(out); + } + // Out of room (the bound makes it unreachable below + // `DEFLATE_EXACT_BOUND`): grow rather than fail. Status::Ok | Status::BufError if out.len() == out.capacity() => out .try_reserve(out.capacity().max(4096)) .map_err(|e| format!("deflate: cannot allocate output: {e}"))?, diff --git a/crates/clawhdf5-format/src/chunked_write.rs b/crates/clawhdf5-format/src/chunked_write.rs index 746a492..14c3e22 100644 --- a/crates/clawhdf5-format/src/chunked_write.rs +++ b/crates/clawhdf5-format/src/chunked_write.rs @@ -402,86 +402,83 @@ pub fn split_into_chunks( chunk_dims: &[u64], element_size: usize, ) -> Vec<(Vec, Vec)> { - let rank = shape.len(); - if rank == 0 { + if shape.is_empty() { return vec![(vec![], raw_data.to_vec())]; } + (0..chunk_count(shape, chunk_dims)) + .map(|i| extract_chunk(raw_data, shape, chunk_dims, element_size, i)) + .collect() +} - // Compute number of chunks per dimension - let mut num_chunks_per_dim = Vec::with_capacity(rank); - for d in 0..rank { - num_chunks_per_dim.push(shape[d].div_ceil(chunk_dims[d])); +/// Number of chunks of the current extent `shape`. +fn chunk_count(shape: &[u64], chunk_dims: &[u64]) -> u64 { + shape + .iter() + .zip(chunk_dims) + .map(|(&s, &c)| s.div_ceil(c)) + .product() +} + +/// The `linear_idx`-th chunk (row-major over the chunks of the current +/// extent) of the row-major dataset `raw_data`: its offset in dataset space +/// and its bytes, a whole chunk with the part past the dataset's edge zero. +/// Copied one row (a run along the last dimension) at a time. +fn extract_chunk( + raw_data: &[u8], + shape: &[u64], + chunk_dims: &[u64], + element_size: usize, + linear_idx: u64, +) -> (Vec, Vec) { + let rank = shape.len(); + let mut offsets = vec![0u64; rank]; + let mut remaining = linear_idx; + for d in (0..rank).rev() { + let n = shape[d].div_ceil(chunk_dims[d]); + offsets[d] = (remaining % n) * chunk_dims[d]; + remaining /= n; } - let total_chunks: u64 = num_chunks_per_dim.iter().product(); + let chunk_elements: usize = chunk_dims.iter().map(|&d| saturating_usize(d)).product(); + let mut chunk = vec![0u8; chunk_elements.saturating_mul(element_size)]; - // Dataset strides (row-major) + // Elements of the chunk inside the dataset, per dimension. + let valid: Vec = (0..rank) + .map(|d| saturating_usize(shape[d].saturating_sub(offsets[d]).min(chunk_dims[d]))) + .collect(); + if valid.contains(&0) { + return (offsets, chunk); + } let mut ds_strides = vec![1usize; rank]; - for i in (0..rank.saturating_sub(1)).rev() { - ds_strides[i] = ds_strides[i + 1] * saturating_usize(shape[i + 1]); - } - - // Chunk strides let mut chunk_strides = vec![1usize; rank]; - for i in (0..rank.saturating_sub(1)).rev() { - chunk_strides[i] = chunk_strides[i + 1] * saturating_usize(chunk_dims[i + 1]); + for d in (0..rank - 1).rev() { + ds_strides[d] = ds_strides[d + 1] * saturating_usize(shape[d + 1]); + chunk_strides[d] = chunk_strides[d + 1] * saturating_usize(chunk_dims[d + 1]); } - - let chunk_total_elements: usize = chunk_dims.iter().map(|&d| saturating_usize(d)).product(); - - let mut result = Vec::with_capacity(saturating_usize(total_chunks)); - - for linear_idx in 0..total_chunks { - // Convert linear index to chunk grid coordinates - let mut chunk_grid_coords = vec![0u64; rank]; - let mut remaining = linear_idx; - for d in (0..rank).rev() { - chunk_grid_coords[d] = remaining % num_chunks_per_dim[d]; - remaining /= num_chunks_per_dim[d]; + let row = valid[rank - 1] * element_size; + let mut idx = vec![0usize; rank]; + loop { + let src: usize = (0..rank) + .map(|d| (saturating_usize(offsets[d]) + idx[d]) * ds_strides[d]) + .sum::() + * element_size; + let dst: usize = (0..rank).map(|d| idx[d] * chunk_strides[d]).sum::() * element_size; + // Whole elements only, as far as `raw_data` reaches. + let n = row.min(raw_data.len().saturating_sub(src)) / element_size * element_size; + chunk[dst..dst + n].copy_from_slice(&raw_data[src..src + n]); + // Next row: advance every dimension but the last. + let mut d = rank - 1; + loop { + if d == 0 { + return (offsets, chunk); + } + d -= 1; + idx[d] += 1; + if idx[d] < valid[d] { + break; + } + idx[d] = 0; } - - // Chunk offset in dataset space - let offsets: Vec = (0..rank) - .map(|d| chunk_grid_coords[d] * chunk_dims[d]) - .collect(); - - // Extract chunk data - let mut chunk_bytes = vec![0u8; chunk_total_elements * element_size]; - - for flat_idx in 0..chunk_total_elements { - let mut remaining_idx = flat_idx; - let mut ds_flat = 0usize; - let mut out_of_bounds = false; - - for d in 0..rank { - let coord_in_chunk = remaining_idx / chunk_strides[d]; - remaining_idx %= chunk_strides[d]; - - let global_coord = saturating_usize(offsets[d]) + coord_in_chunk; - if global_coord >= saturating_usize(shape[d]) { - out_of_bounds = true; - break; - } - ds_flat += global_coord * ds_strides[d]; - } - - if out_of_bounds { - // Zero-filled (already initialized) - continue; - } - - let src_start = ds_flat * element_size; - let dst_start = flat_idx * element_size; - - if src_start + element_size <= raw_data.len() { - chunk_bytes[dst_start..dst_start + element_size] - .copy_from_slice(&raw_data[src_start..src_start + element_size]); - } - } - - result.push((offsets, chunk_bytes)); } - - result } /// Parallel compression threshold: use rayon when chunk count exceeds this. @@ -492,46 +489,68 @@ pub fn split_into_chunks( #[cfg(feature = "parallel")] const PARALLEL_COMPRESS_THRESHOLD: usize = 2; -/// Compress all chunks, using parallel compression when beneficial, and -/// return each chunk's stored bytes with its filter mask. +/// Largest chunk compressed in parallel: every thread holds a chunk and its +/// compressed copy at once, so chunks larger than this (up to 4 GiB and +/// more) are compressed one after another. +#[cfg(feature = "parallel")] +const PARALLEL_COMPRESS_MAX_CHUNK_BYTES: u64 = 64 << 20; + +/// Extract and compress every chunk of the dataset, returning each chunk's +/// raw size, stored bytes and filter mask, in chunk order. /// /// Chunks run through the pipeline as libhdf5 runs them /// ([`compress_chunk_masked`]): an optional filter that fails — LZF or Blosc /// output no smaller than its input — is skipped and its mask bit set. /// -/// With the `parallel` feature and more than [`PARALLEL_COMPRESS_THRESHOLD`] -/// filtered chunks, compression runs across rayon threads; otherwise it is -/// sequential. Output order matches input order, so per-chunk bytes are -/// identical to the sequential path. +/// Each chunk is extracted just before it is compressed and dropped after, +/// so at most one raw chunk per thread is held. With the `parallel` feature, +/// more than [`PARALLEL_COMPRESS_THRESHOLD`] filtered chunks, and chunks of +/// at most [`PARALLEL_COMPRESS_MAX_CHUNK_BYTES`], compression runs across +/// rayon threads; otherwise it is sequential. Output order matches chunk +/// order, so per-chunk bytes are identical to the sequential path. fn compress_all_chunks( - chunks: &[(Vec, Vec)], + raw_data: &[u8], + shape: &[u64], + chunk_dims: &[u64], + element_size: usize, + chunk_bytes: u64, pipeline: &Option, - element_size: u32, -) -> Result, u32)>, FormatError> { +) -> Result, u32)>, FormatError> { + let one = |i: u64| -> Result<(u64, Vec, u32), FormatError> { + let (_, raw) = if shape.is_empty() { + (Vec::new(), raw_data.to_vec()) + } else { + extract_chunk(raw_data, shape, chunk_dims, element_size, i) + }; + let raw_size = raw.len() as u64; + match pipeline { + Some(pl) => { + let (stored, mask) = compress_chunk_masked(&raw, pl, element_size as u32)?; + Ok((raw_size, stored, mask)) + } + None => Ok((raw_size, raw, 0)), + } + }; + let n = if shape.is_empty() { + 1 + } else { + chunk_count(shape, chunk_dims) + }; #[cfg(feature = "parallel")] { - if let Some(pl) = pipeline - && chunks.len() > PARALLEL_COMPRESS_THRESHOLD + if pipeline.is_some() + && n > PARALLEL_COMPRESS_THRESHOLD as u64 + && chunk_bytes <= PARALLEL_COMPRESS_MAX_CHUNK_BYTES { use rayon::prelude::*; - return chunks - .par_iter() - .map(|(_offsets, chunk_bytes)| compress_chunk_masked(chunk_bytes, pl, element_size)) - .collect(); + return (0..n).into_par_iter().map(one).collect(); } } + #[cfg(not(feature = "parallel"))] + let _ = chunk_bytes; // Sequential fallback - chunks - .iter() - .map(|(_offsets, chunk_bytes)| { - if let Some(pl) = pipeline { - compress_chunk_masked(chunk_bytes, pl, element_size) - } else { - Ok((chunk_bytes.clone(), 0)) - } - }) - .collect() + (0..n).map(one).collect() } /// Build the complete chunked dataset blob (chunk data + index) and return @@ -552,10 +571,12 @@ pub fn serialize_v4_single_chunk_pub( filter_mask, offset_size, element_size, + 4, ) } -/// Serialize a v4 single chunk layout message. +/// Serialize a v4 (or, for a chunk of 4 GiB or more, v5) single chunk +/// layout message. fn serialize_v4_single_chunk( chunk_dims: &[u32], chunk_address: u64, @@ -563,9 +584,10 @@ fn serialize_v4_single_chunk( filter_mask: Option, offset_size: u8, element_size: u32, + version: u8, ) -> Vec { let mut buf = Vec::new(); - buf.push(4); // version + buf.push(version); buf.push(2); // class = chunked // flags: bit 0 = unknown meaning in some files, bit 1 = filters for single chunk @@ -605,8 +627,9 @@ fn serialize_v4_fixed_array( offset_size: u8, element_size: u32, max_bits: u8, + version: u8, ) -> Vec { - let mut buf = layout_v4_chunked_prefix(chunk_dims, element_size); + let mut buf = layout_v4_chunked_prefix(chunk_dims, element_size, version); // chunk index type = 3 (Fixed Array) buf.push(3); @@ -645,9 +668,9 @@ pub(crate) fn push_v4_chunk_dims(buf: &mut Vec, chunk_dims: &[u32], element_ } } -fn layout_v4_chunked_prefix(chunk_dims: &[u32], element_size: u32) -> Vec { +fn layout_v4_chunked_prefix(chunk_dims: &[u32], element_size: u32, version: u8) -> Vec { let mut buf = Vec::new(); - buf.push(4); // version + buf.push(version); buf.push(2); // class = chunked let flags: u8 = 0x00; @@ -673,21 +696,53 @@ pub(crate) fn push_addr(buf: &mut Vec, addr: u64, offset_size: u8) { /// Width of the chunk-size field of a filtered chunk index element. Must /// match the library's `H5D_FARRAY_FILT_COMPUTE_CHUNK_SIZE_LEN` (the EA and -/// B-tree v2 indexes use the same formula): -/// `1 + ((log2(unfiltered chunk bytes) + 8) / 8)`, capped at 8. -pub(crate) fn filtered_chunk_size_len(slots: &[Option]) -> usize { +/// B-tree v2 indexes use the same formula): see [`chunk_size_len`]. Chunks +/// of more than `u32::MAX` bytes are written with layout version 5 +/// ([`layout_version_for`]). +pub(crate) fn filtered_chunk_size_len(slots: &[Option], length_size: u8) -> usize { let max_raw = slots .iter() .flatten() .map(|c| c.raw_size) .max() .unwrap_or(1); - let log2_val = if max_raw <= 1 { + chunk_size_len(max_raw, layout_version_for(max_raw), length_size) +} + +/// Largest chunk, in bytes, a layout message of version 4 or lower may +/// describe: libhdf5 writes a larger one with version 5 +/// (`H5D__chunk_construct`: "chunk size > 4GB requires H5F_LIBVER_V200"), +/// which libhdf5 before 2.0 cannot read. +pub const MAX_V4_CHUNK_BYTES: u64 = u32::MAX as u64; + +/// The layout message version clawhdf5 writes for chunks of `chunk_bytes` +/// bytes: 4, or 5 for a chunk larger than [`MAX_V4_CHUNK_BYTES`] (what +/// libhdf5 2.x writes for it; the chunk index is chosen as for version 4). +pub fn layout_version_for(chunk_bytes: u64) -> u8 { + if chunk_bytes > MAX_V4_CHUNK_BYTES { + 5 + } else { + 4 + } +} + +/// Width libhdf5 gives the stored-size field of a filtered chunk index +/// element (Fixed Array, Extensible Array, v2 B-tree) for chunks of +/// `chunk_bytes` bytes under layout message `layout_version` +/// (`H5D_FARRAY_FILT_COMPUTE_CHUNK_SIZE_LEN` and its EA and B-tree twins): +/// up to version 4, one byte more than the chunk's size needs, +/// `1 + ((log2(chunk_bytes) + 8) / 8)` capped at 8; from version 5 (HDF5 +/// 2.0), the file's size of lengths (`length_size`), whatever the chunk. +pub fn chunk_size_len(chunk_bytes: u64, layout_version: u8, length_size: u8) -> usize { + if layout_version >= 5 { + return usize::from(length_size); + } + let log2 = if chunk_bytes <= 1 { 0 } else { - 63 - max_raw.leading_zeros() + 63 - chunk_bytes.leading_zeros() }; - (1 + ((log2_val + 8) / 8) as usize).min(8) + (1 + ((log2 + 8) / 8) as usize).min(8) } /// Append one chunk index element: the chunk's address, plus its stored size @@ -734,7 +789,7 @@ pub fn build_fixed_array_at( let os = offset_size as usize; let num_elements = slots.len(); - let chunk_size_bytes = has_filters.then(|| filtered_chunk_size_len(slots)); + let chunk_size_bytes = has_filters.then(|| filtered_chunk_size_len(slots, length_size)); let elem_size = os + chunk_size_bytes.map_or(0, |n| n + 4); let client_id: u8 = if has_filters { 1 } else { 0 }; @@ -829,23 +884,23 @@ pub fn precompress_chunks( element_size: usize, options: &ChunkOptions, ) -> Result { - let chunk_bytes = chunk_dims - .iter() - .try_fold(element_size as u64, |acc, &d| acc.checked_mul(d)) - .and_then(|b| u32::try_from(b).ok()) - .unwrap_or(0); - let pipeline = options.build_pipeline_for_chunk(element_size as u32, chunk_bytes); + let (_, chunk_bytes) = checked_chunk_dims(chunk_dims, element_size)?; + if chunk_bytes > MAX_V4_CHUNK_BYTES { + check_huge_chunk_filters(options, chunk_bytes)?; + } + let pipeline = options + .build_pipeline_for_chunk(element_size as u32, u32::try_from(chunk_bytes).unwrap_or(0)); let has_filters = pipeline.is_some(); let pipeline_message = pipeline.as_ref().map(|pl| pl.serialize()); - let raw_chunks = split_into_chunks(raw_data, shape, chunk_dims, element_size); - let compressed = compress_all_chunks(&raw_chunks, &pipeline, element_size as u32)?; - - let chunks = raw_chunks - .into_iter() - .zip(compressed) - .map(|((_offsets, raw_bytes), (c, mask))| (raw_bytes.len() as u64, c, mask)) - .collect(); + let chunks = compress_all_chunks( + raw_data, + shape, + chunk_dims, + element_size, + chunk_bytes, + &pipeline, + )?; Ok(PrecompressedChunks { chunks, @@ -857,6 +912,67 @@ pub fn precompress_chunks( }) } +/// The chunk dimensions as the layout message stores them (each below +/// 2^32), and one chunk's size in bytes. A chunk dimension of 2^32 or more +/// (which HDF5 2.0 can store, in wider fields) is refused, as it is when +/// read: clawhdf5 holds chunk dimensions as `u32`. So is a chunk whose size +/// overflows 64 bits, or that this platform cannot hold in memory (a chunk +/// of 4 GiB or more on a 32-bit target). +fn checked_chunk_dims( + chunk_dims: &[u64], + element_size: usize, +) -> Result<(Vec, u64), FormatError> { + let dims = chunk_dims + .iter() + .map(|&d| { + u32::try_from(d).map_err(|_| { + FormatError::InvalidChunkDimensions(format!( + "chunk dimension {d} is 2^32 or more, which clawhdf5 does not support" + )) + }) + }) + .collect::, _>>()?; + let bytes = chunk_dims + .iter() + .try_fold(element_size as u64, |acc, &d| acc.checked_mul(d)) + .filter(|&b| usize::try_from(b).is_ok()) + .ok_or_else(|| { + FormatError::Overflow(format!( + "a chunk of {chunk_dims:?} x {element_size} bytes exceeds this platform's address space" + )) + })?; + Ok((dims, bytes)) +} + +/// The filters clawhdf5 can apply to a chunk of more than `u32::MAX` bytes: +/// shuffle, deflate, Zstandard, LZ4 (whose HDF5 framing records the size in +/// 64 bits and splits the chunk into blocks) and Fletcher32. The others +/// record the chunk size or their block lengths in 32 bits, or cannot take +/// a buffer that large (h5py's LZF, bitshuffle, bzip2, Blosc), and pcodec is +/// clawhdf5's own; they are refused rather than written into a chunk +/// libhdf5 could not decode. +fn check_huge_chunk_filters(options: &ChunkOptions, chunk_bytes: u64) -> Result<(), FormatError> { + let refused = if let Some(plugin) = &options.plugin { + Some(match plugin { + PluginFilter::Lzf => "LZF", + PluginFilter::Bitshuffle { .. } => "bitshuffle", + PluginFilter::Bzip2 { .. } => "bzip2", + PluginFilter::Blosc { .. } => "Blosc", + }) + } else if options.pcodec { + Some("pcodec") + } else { + None + }; + match refused { + Some(name) => Err(FormatError::FilterError(format!( + "{name} cannot compress a chunk of {chunk_bytes} bytes (4 GiB or more); \ + use smaller chunks, or deflate, Zstandard or LZ4" + ))), + None => Ok(()), + } +} + /// Lay out precompressed chunks at `base_address` and build index structures. /// /// This is the address-dependent half of chunk writing. Call it in Pass 1 @@ -910,7 +1026,8 @@ pub fn build_chunked_data_from_precompressed_libver( }); } - let chunk_dims_u32: Vec = pre.chunk_dims.iter().map(|&d| d as u32).collect(); + let (chunk_dims_u32, chunk_bytes) = checked_chunk_dims(&pre.chunk_dims, element_size)?; + let version = layout_version_for(chunk_bytes); let aligned_idx = align_to_cache_line(data_buf.len()); if aligned_idx > data_buf.len() { @@ -934,6 +1051,7 @@ pub fn build_chunked_data_from_precompressed_libver( ea_address, offset_size, element_size as u32, + version, ) } ChunkIndexPlan::SingleChunk => { @@ -951,6 +1069,7 @@ pub fn build_chunked_data_from_precompressed_libver( filter_mask, offset_size, element_size as u32, + version, ) } ChunkIndexPlan::FixedArray(grid, nslots) => { @@ -976,6 +1095,7 @@ pub fn build_chunked_data_from_precompressed_libver( offset_size, element_size as u32, FA_PAGE_BITS, + version, ) } ChunkIndexPlan::BTreeV2 => { @@ -1000,6 +1120,7 @@ pub fn build_chunked_data_from_precompressed_libver( offset_size, element_size as u32, node_size, + version, ) } }; @@ -1237,7 +1358,7 @@ fn build_btree_v2_chunk_index_at( let chunk_size_bytes = has_filters.then(|| { let slots: Vec> = records.iter().map(|(_, c)| Some((*c).clone())).collect(); - filtered_chunk_size_len(&slots) + filtered_chunk_size_len(&slots, length_size) }); let record_size = os + chunk_size_bytes.map_or(0, |n| n + 4) + 8 * rank; let record_size_u16 = u16::try_from(record_size) @@ -1287,8 +1408,9 @@ fn serialize_v4_btree_v2( offset_size: u8, element_size: u32, node_size: u32, + version: u8, ) -> Vec { - let mut buf = layout_v4_chunked_prefix(chunk_dims, element_size); + let mut buf = layout_v4_chunked_prefix(chunk_dims, element_size, version); buf.push(5); // chunk index type = 5 (version-2 B-tree) buf.extend_from_slice(&node_size.to_le_bytes()); buf.push(BT2_SPLIT_PERCENT); @@ -1868,7 +1990,7 @@ mod tests { #[test] fn serialize_v4_single_chunk_no_filters_roundtrip() { - let msg = serialize_v4_single_chunk(&[20], 0x1000, None, None, 8, 8); + let msg = serialize_v4_single_chunk(&[20], 0x1000, None, None, 8, 8, 4); let layout = DataLayout::parse(&msg, 8, 8).unwrap(); match layout { DataLayout::Chunked { @@ -1893,7 +2015,7 @@ mod tests { #[test] fn serialize_v4_single_chunk_with_filters_roundtrip() { - let msg = serialize_v4_single_chunk(&[100], 0x2000, Some(500), Some(0), 8, 8); + let msg = serialize_v4_single_chunk(&[100], 0x2000, Some(500), Some(0), 8, 8, 4); let layout = DataLayout::parse(&msg, 8, 8).unwrap(); match layout { DataLayout::Chunked { @@ -1912,7 +2034,7 @@ mod tests { #[test] fn serialize_v4_fixed_array_roundtrip() { - let msg = serialize_v4_fixed_array(&[20], 0x3000, 8, 8, 4); + let msg = serialize_v4_fixed_array(&[20], 0x3000, 8, 8, 4, 4); let layout = DataLayout::parse(&msg, 8, 8).unwrap(); match layout { DataLayout::Chunked { @@ -1956,11 +2078,213 @@ mod tests { assert_eq!(&fa[28..32], b"FADB"); } + // ---- Chunks of 4 GiB or more (layout message version 5) ---- + + /// 2^29 + 1 `f64`: 4 GiB + 8 bytes, the smallest `f64` chunk past + /// `u32::MAX`. + const HUGE_DIM: u64 = (1 << 29) + 1; + const HUGE_BYTES: u64 = HUGE_DIM * 8; + + fn huge_chunk(address: u64, compressed_size: u64) -> WrittenChunk { + WrittenChunk { + address, + compressed_size, + raw_size: HUGE_BYTES, + filter_mask: 0, + } + } + + #[test] + fn chunks_past_u32_max_take_layout_version_5() { + assert_eq!(layout_version_for(0), 4); + assert_eq!(layout_version_for(u64::from(u32::MAX)), 4); + assert_eq!(layout_version_for(u64::from(u32::MAX) + 1), 5); + assert_eq!(layout_version_for(HUGE_BYTES), 5); + } + + /// `H5D_FARRAY_FILT_COMPUTE_CHUNK_SIZE_LEN` (and its EA and v2 B-tree + /// twins) in libhdf5 2.2.0: one byte more than the chunk size needs up to + /// layout version 4, the size of lengths from version 5. + #[test] + fn chunk_size_len_follows_libhdf5() { + assert_eq!(chunk_size_len(1, 4, 8), 2); + assert_eq!(chunk_size_len(160, 4, 8), 2); + assert_eq!(chunk_size_len(255, 4, 8), 2); + assert_eq!(chunk_size_len(256, 4, 8), 3); + assert_eq!(chunk_size_len(u64::from(u32::MAX), 4, 8), 5); + assert_eq!(chunk_size_len(1 << 32, 4, 8), 6); + assert_eq!(chunk_size_len(u64::MAX, 4, 8), 8); + assert_eq!(chunk_size_len(HUGE_BYTES, 5, 8), 8); + assert_eq!(chunk_size_len(160, 5, 8), 8); + assert_eq!(chunk_size_len(HUGE_BYTES, 5, 4), 4); + let slots = [Some(huge_chunk(0x1000, 20_000)), None]; + assert_eq!(filtered_chunk_size_len(&slots, 8), 8); + let small = [Some(WrittenChunk { + raw_size: 160, + ..huge_chunk(0x1000, 100) + })]; + assert_eq!(filtered_chunk_size_len(&small, 8), 2); + } + + /// A 4 GiB + 8 byte chunk: every layout message is version 5 with the + /// dimensions libhdf5 writes (4 bytes each: 0x20000001 and 8), and a + /// filtered index element stores the chunk's size in 8 bytes. + #[test] + fn huge_chunk_layout_messages_and_index_elements() { + let dims = [HUGE_DIM as u32]; + let parsed = |msg: &[u8]| { + assert_eq!(msg[0], 5, "layout message version"); + // Class chunked, then (after the flags) 2 dimensions of 4 bytes. + assert_eq!(&msg[1..2], &[2]); + assert_eq!(&msg[3..5], &[2, 4]); + assert_eq!(&msg[5..13], &[1, 0, 0, 0x20, 8, 0, 0, 0]); + match DataLayout::parse(msg, 8, 8).unwrap() { + DataLayout::Chunked { + chunk_dimensions, + chunk_index_type, + .. + } => { + assert_eq!(chunk_dimensions, vec![HUGE_DIM as u32, 8]); + chunk_index_type.unwrap() + } + other => panic!("{other:?}"), + } + }; + let single = serialize_v4_single_chunk(&dims, 0x800, Some(20_000), Some(0), 8, 8, 5); + assert_eq!(parsed(&single), 1); + // Filtered size (8 bytes), filter mask, address. + assert_eq!(single.len(), 13 + 1 + 8 + 4 + 8); + assert_eq!( + parsed(&serialize_v4_fixed_array(&dims, 0x800, 8, 8, 10, 5)), + 3 + ); + let ea = ea_writer::serialize_v4_extensible_array(&dims, 0x800, 8, 8, 5); + assert_eq!(parsed(&ea), 4); + assert_eq!( + parsed(&serialize_v4_btree_v2(&dims, 0x800, 8, 8, 2048, 5)), + 5 + ); + + let slots = [Some(huge_chunk(0x1000, 20_000)), None]; + // FAHD: element size (address 8 + size 8 + mask 4) at byte 6. + let fa = build_fixed_array_at(&slots, 8, 8, true, 0x2000); + assert_eq!(fa[6], 20); + // FADB element 0: address, then the stored size in 8 bytes. + let fadb = 28; + let prefix = 4 + 1 + 1 + 8; + assert_eq!( + &fa[fadb + prefix..fadb + prefix + 8], + &0x1000u64.to_le_bytes() + ); + assert_eq!( + &fa[fadb + prefix + 8..fadb + prefix + 16], + &20_000u64.to_le_bytes() + ); + // AEHD: element size at byte 6 as well. + let ea = ea_writer::build_extensible_array_at(&slots, 8, 8, true, 0x2000); + assert_eq!(&ea[..4], b"EAHD"); + assert_eq!(ea[6], 20); + // BTHD: record size (element + 8-byte scaled offset) at bytes 10-11. + let chunk = huge_chunk(0x1000, 20_000); + let (bt, _) = + build_btree_v2_chunk_index_at(1, &[(vec![0], &chunk)], 8, 8, true, 0x2000).unwrap(); + assert_eq!(&bt[..4], b"BTHD"); + assert_eq!(u16::from_le_bytes([bt[10], bt[11]]), 28); + } + + #[test] + fn huge_chunks_refused_where_unsupported() { + // A chunk dimension of 2^32 or more. + assert!(matches!( + checked_chunk_dims(&[1 << 32], 1), + Err(FormatError::InvalidChunkDimensions(m)) if m.contains("2^32") + )); + // A chunk size that overflows u64. + assert!(matches!( + checked_chunk_dims(&[u32::MAX.into(), u32::MAX.into(), 2], 8), + Err(FormatError::Overflow(_)) + )); + assert_eq!( + checked_chunk_dims(&[HUGE_DIM], 8).unwrap(), + (vec![HUGE_DIM as u32], HUGE_BYTES) + ); + // Filters that cannot take a chunk that large. + for plugin in [ + PluginFilter::Lzf, + PluginFilter::Bzip2 { level: 9 }, + PluginFilter::Blosc { + codec: BloscCodec::Lz4, + level: 5, + shuffle: BloscShuffle::Byte, + }, + PluginFilter::Bitshuffle { + block_size: 0, + compression: BitshuffleCompression::Lz4, + }, + ] { + let options = ChunkOptions { + plugin: Some(plugin), + ..Default::default() + }; + assert!(matches!( + check_huge_chunk_filters(&options, HUGE_BYTES), + Err(FormatError::FilterError(m)) if m.contains("4 GiB") + )); + } + for options in [ + ChunkOptions { + deflate_level: Some(6), + shuffle: true, + fletcher32: true, + ..Default::default() + }, + ChunkOptions { + zstd_level: Some(3), + ..Default::default() + }, + ChunkOptions { + lz4: true, + ..Default::default() + }, + ] { + check_huge_chunk_filters(&options, HUGE_BYTES).unwrap(); + } + } + + /// Chunks are extracted row by row: the result is the element-by-element + /// split, edge padding included. + #[test] + fn extract_chunk_matches_elementwise_split() { + let shape = [5u64, 7, 3]; + let chunks = [2u64, 3, 2]; + let data: Vec = (0..5 * 7 * 3 * 2).map(|i| i as u8).collect(); + let n = chunk_count(&shape, &chunks); + assert_eq!(n, 3 * 3 * 2); + for i in 0..n { + let (offsets, chunk) = extract_chunk(&data, &shape, &chunks, 2, i); + assert_eq!(chunk.len(), 2 * 3 * 2 * 2); + for (e, pair) in chunk.chunks_exact(2).enumerate() { + let c = [e / 6, (e / 2) % 3, e % 2]; + let g: Vec = (0..3).map(|d| offsets[d] + c[d] as u64).collect(); + let expect = if (0..3).all(|d| g[d] < shape[d]) { + let flat = ((g[0] * 7 + g[1]) * 3 + g[2]) as usize; + [data[2 * flat], data[2 * flat + 1]] + } else { + [0, 0] + }; + assert_eq!(pair, expect, "chunk {i} element {e}"); + } + } + // Data shorter than the shape: the missing elements stay zero. + let (_, chunk) = extract_chunk(&data[..5], &[4], &[4], 2, 0); + assert_eq!(chunk, [0, 1, 2, 3, 0, 0, 0, 0]); + } + // ---- Extensible Array tests ---- #[test] fn serialize_v4_extensible_array_roundtrip() { - let msg = ea_writer::serialize_v4_extensible_array(&[10], 0x4000, 8, 8); + let msg = ea_writer::serialize_v4_extensible_array(&[10], 0x4000, 8, 8, 4); let layout = DataLayout::parse(&msg, 8, 8).unwrap(); match layout { DataLayout::Chunked { diff --git a/crates/clawhdf5-format/src/ea_writer.rs b/crates/clawhdf5-format/src/ea_writer.rs index de7859c..5951c50 100644 --- a/crates/clawhdf5-format/src/ea_writer.rs +++ b/crates/clawhdf5-format/src/ea_writer.rs @@ -18,9 +18,10 @@ pub(crate) fn serialize_v4_extensible_array( ea_address: u64, offset_size: u8, element_size: u32, + version: u8, ) -> Vec { let mut buf = Vec::new(); - buf.push(4); // version + buf.push(version); buf.push(2); // class = chunked buf.push(0x00); // flags @@ -87,7 +88,7 @@ pub fn build_extensible_array_at( ea_base_address: u64, ) -> Vec { let os = offset_size as usize; - let chunk_size_bytes = has_filters.then(|| filtered_chunk_size_len(slots)); + let chunk_size_bytes = has_filters.then(|| filtered_chunk_size_len(slots, length_size)); let elem_size = os + chunk_size_bytes.map_or(0, |n| n + 4); let client_id: u8 = if has_filters { 1 } else { 0 }; let arr_off_size = (MAX_NELMTS_BITS as usize).div_ceil(8); diff --git a/crates/clawhdf5-format/src/filters.rs b/crates/clawhdf5-format/src/filters.rs index 48553c1..ded4c0a 100644 --- a/crates/clawhdf5-format/src/filters.rs +++ b/crates/clawhdf5-format/src/filters.rs @@ -343,10 +343,10 @@ pub fn decompress_chunk_exact_with<'s>( // The bound is the output's size when only size-preserving // filters (shuffle, Fletcher32) remain to be undone; before // any other filter (a second deflate, N-Bit, ...) it is only - // a ceiling, and the output starts smaller and grows: the - // inflater writes every byte it reserves, so reserving a - // 4 GiB chunk's bound for a stage a few MiB long would hold - // twice the chunk's memory. + // a ceiling, and the output starts smaller and grows. + // Reserving the bound of a 4 GiB chunk for a stage a few MiB + // long doubled the peak memory of reading it (8.5 GiB, now + // 4.0, for `huge_chunks_filtered.h5`'s double deflate). let exact = pipeline.filters[..i].iter().enumerate().all(|(j, f)| { filter_skipped(filter_mask, j) || matches!(f.filter_id, FILTER_SHUFFLE | FILTER_FLETCHER32) @@ -442,7 +442,10 @@ pub fn compress_chunk_masked( "more than 32 filters in a pipeline".into(), )); } - let mut result = data.to_vec(); + // The input is not copied: the first filter reads it where it is (a + // chunk may be 4 GiB or more), and each filter's output replaces the + // previous one. + let mut owned: Option> = None; let mut mask = 0u32; for (i, filter) in pipeline.filters.iter().enumerate() { let ctx = FilterContext { @@ -450,7 +453,8 @@ pub fn compress_chunk_masked( element_size: element_size as usize, max_output: 0, }; - let out = match filter_registry::encode(&result, &ctx) { + let result: &[u8] = owned.as_deref().unwrap_or(data); + let out = match filter_registry::encode(result, &ctx) { Ok(out) if FAIL_UNLESS_SMALLER.contains(&filter.filter_id) && out.len() >= result.len() => { @@ -462,13 +466,13 @@ pub fn compress_chunk_masked( r => r, }; match out { - Ok(out) => result = out, + Ok(out) => owned = Some(out), Err(e @ FormatError::UnsupportedFilter(_)) => return Err(e), Err(_) if filter.flags & FILTER_FLAG_OPTIONAL != 0 => mask |= 1 << i, Err(e) => return Err(e), } } - Ok((result, mask)) + Ok((owned.unwrap_or_else(|| data.to_vec()), mask)) } /// The filters compiled into this build, sorted by ID (see @@ -1352,6 +1356,9 @@ fn deflate_compress(data: &[u8], level: u32) -> Result, FormatError> { deflate_bounded(data, level).map_err(FormatError::CompressionError) } +/// Largest compression output reserved at its worst-case size up front. +const DEFLATE_EXACT_BOUND: usize = 64 << 20; + /// Deflate `data` into a zlib stream in one pass, into a buffer sized for the /// worst case up front (the same reasoning as [`inflate_bounded`]). #[cfg(feature = "deflate")] @@ -1360,24 +1367,44 @@ pub(crate) fn deflate_bounded(data: &[u8], level: u32) -> Result, String // zlib's compressBound, plus the zlib header and trailer. let bound = data.len() + (data.len() >> 12) + (data.len() >> 14) + (data.len() >> 25) + 13 + 6; + // flate2's Rust backends (zlib-rs, miniz_oxide) zero the whole spare + // capacity on each call, so a large input's worst-case bound would be + // memory held for nothing (4 GiB for a 4 GiB chunk that deflates to a + // few MiB): past 64 MiB the output starts at 1/16 of the bound and + // doubles as needed. + let first = if bound <= DEFLATE_EXACT_BOUND { + bound + } else { + bound / 16 + }; let mut out = Vec::new(); - out.try_reserve_exact(bound) + out.try_reserve_exact(first) .map_err(|e| format!("deflate: cannot allocate output: {e}"))?; let mut deflater = Compress::new(Compression::new(level), true); loop { let (in_before, out_before) = (deflater.total_in(), deflater.total_out()); + let rest = &data[saturating_usize(in_before)..]; + // zlib takes at most u32::MAX input bytes per call, and `Finish` + // ends the stream after the bytes it took: a chunk of 4 GiB or more + // was cut at 4 GiB - 1. Finish only once the rest fits one call. + let flush = if rest.len() > u32::MAX as usize { + FlushCompress::None + } else { + FlushCompress::Finish + }; let status = deflater - .compress_vec( - &data[saturating_usize(in_before)..], - &mut out, - FlushCompress::Finish, - ) + .compress_vec(rest, &mut out, flush) .map_err(|e| format!("deflate: {e}"))?; match status { - Status::StreamEnd => return Ok(out), - // The bound should make running out of room unreachable; grow - // rather than fail if it happens. + Status::StreamEnd => { + if out.capacity() - out.len() > DEFLATE_EXACT_BOUND { + out.shrink_to_fit(); + } + return Ok(out); + } + // Out of room (the bound makes it unreachable below + // `DEFLATE_EXACT_BOUND`): grow rather than fail. Status::Ok | Status::BufError if out.len() == out.capacity() => out .try_reserve(out.capacity().max(4096)) .map_err(|e| format!("deflate: cannot allocate output: {e}"))?, @@ -1408,10 +1435,12 @@ const LZ4_DEFAULT_BLOCK_SIZE: usize = 1 << 30; /// * The legacy clawhdf5 framing (up to 2.7.0): a 4-byte little-endian size /// followed by one raw LZ4 block. libhdf5 cannot read it. /// -/// They are told apart unambiguously: an HDF5 chunk is smaller than 4 GiB, so -/// the registered format's big-endian `u64` size always starts with four zero -/// bytes and the whole chunk is at least 12 bytes; a legacy chunk starts with -/// four zero bytes only when it is empty, and is then 5 bytes long. +/// They are told apart unambiguously: for a chunk under 4 GiB the registered +/// format's big-endian `u64` size starts with four zero bytes and the whole +/// chunk is at least 12 bytes; a legacy chunk starts with four zero bytes +/// only when it is empty, and is then 5 bytes long. A chunk of 4 GiB or more +/// (HDF5 2.0) is always the registered format: clawhdf5 never wrote legacy +/// chunks that large. /// /// Every size read from the payload is bounded against `expected_bytes` (the /// pipeline's declared chunk size) before it sizes an allocation, so a crafted @@ -1429,14 +1458,18 @@ fn lz4_decompress(data: &[u8], expected_bytes: usize) -> Result, FormatE "lz4: declared size exceeds chunk size".into(), )); } - if size > MAX_DECOMPRESS_SIZE { + // Without a chunk size, a ceiling; with one, the chunk size is the + // bound (a chunk may be 4 GiB or more, like a deflated one). + if expected_bytes == 0 && size > MAX_DECOMPRESS_SIZE { return Err(FormatError::DecompressionError( "lz4: declared size exceeds limit".into(), )); } Ok(()) }; - if data.len() >= 12 && data[..4] == [0, 0, 0, 0] { + // A chunk of 4 GiB or more cannot be legacy (its size field is 32 + // bits), and its registered-format size does not start with zeros. + if data.len() >= 12 && (data[..4] == [0, 0, 0, 0] || expected_bytes > u32::MAX as usize) { return lz4_decompress_hdf5(data, check_size); } // Legacy clawhdf5 framing: 4-byte LE size + one LZ4 block. @@ -1454,9 +1487,10 @@ fn lz4_decompress_hdf5( ) -> Result, FormatError> { let err = |m: &str| FormatError::DecompressionError(format!("lz4: {m}")); let be32 = |b: &[u8]| u32::from_be_bytes([b[0], b[1], b[2], b[3]]) as usize; - // The first four bytes are zero (checked by the caller), so the size is - // the low 32 bits of the big-endian u64. - let orig_size = be32(&data[4..8]); + let orig_size = u64::from_be_bytes([ + data[0], data[1], data[2], data[3], data[4], data[5], data[6], data[7], + ]); + let orig_size = usize::try_from(orig_size).map_err(|_| err("chunk too large"))?; check_size(orig_size)?; let block_size = be32(&data[8..12]).min(orig_size); if block_size == 0 && orig_size != 0 { diff --git a/crates/clawhdf5/src/edit/earray.rs b/crates/clawhdf5/src/edit/earray.rs index bae196f..6c777dd 100644 --- a/crates/clawhdf5/src/edit/earray.rs +++ b/crates/clawhdf5/src/edit/earray.rs @@ -49,17 +49,9 @@ pub(crate) fn encode_elem( /// element for chunks of `chunk_bytes` bytes (`H5D__earray_idx_create`, /// `H5D__farray_idx_create`): one byte more than the nominal size needs — /// except under layout message version 5 (HDF5 2.0's own format), which -/// always uses 8 bytes. -pub(crate) fn chunk_size_len(chunk_bytes: u64, layout_version: u8) -> usize { - if layout_version >= 5 { - return 8; - } - let log2 = if chunk_bytes <= 1 { - 0 - } else { - 63 - chunk_bytes.leading_zeros() - }; - (1 + ((log2 + 8) / 8) as usize).min(8) +/// uses the file's size of lengths (`length_size`). +pub(crate) fn chunk_size_len(chunk_bytes: u64, layout_version: u8, length_size: u8) -> usize { + clawhdf5_format::chunked_write::chunk_size_len(chunk_bytes, layout_version, length_size) } /// Creation parameters, in the layout message's order. @@ -226,7 +218,7 @@ impl Ea { ) -> Result { let os = img.os; let elem_size = if filtered { - os as usize + chunk_size_len(chunk_bytes, layout_version) + 4 + os as usize + chunk_size_len(chunk_bytes, layout_version, img.ls) + 4 } else { os as usize }; diff --git a/crates/clawhdf5/src/edit/farray.rs b/crates/clawhdf5/src/edit/farray.rs index e7a1696..8e737fc 100644 --- a/crates/clawhdf5/src/edit/farray.rs +++ b/crates/clawhdf5/src/edit/farray.rs @@ -83,7 +83,7 @@ impl Fa { let os = img.os; let osz = os as usize; let elem_size = if filtered { - osz + chunk_size_len(chunk_bytes, layout_version) + 4 + osz + chunk_size_len(chunk_bytes, layout_version, img.ls) + 4 } else { osz }; diff --git a/crates/clawhdf5/src/edit/mod.rs b/crates/clawhdf5/src/edit/mod.rs index 28e1fa6..4d4f635 100644 --- a/crates/clawhdf5/src/edit/mod.rs +++ b/crates/clawhdf5/src/edit/mod.rs @@ -412,12 +412,24 @@ impl<'t> ChunkedEdit<'t> { .iter() .map(|&c| u64::from(c)) .collect(); + // Chunks of 4 GiB or more (HDF5 2.0 writes them with layout message + // version 5) are read, but not rewritten: the editor holds a chunk + // it rewrites in memory, compressed and not. Writing values, and a + // resize that prunes or allocates chunks, are refused here, before + // anything is written; growing the extent (late allocation) and + // setting attributes do not come here. let chunk_bytes = cd .iter() .try_fold(t.es as u64, |a, &c| a.checked_mul(c)) + .filter(|&b| b <= u64::from(u32::MAX)) .and_then(|b| usize::try_from(b).ok()) - .filter(|&b| b <= u32::MAX as usize) - .ok_or_else(|| Error::Unsupported("chunk larger than 4 GiB".into()))?; + .ok_or_else(|| { + Error::Unsupported( + "rewriting chunks of 4 GiB or more (writing values, or a resize that \ + prunes or allocates chunks)" + .into(), + ) + })?; let hdr = Header::load(img, t.addr)?; let layout_msg = hdr.find(MSG_LAYOUT).ok_or(Error::MissingMessage( clawhdf5_format::message_type::MessageType::DataLayout, @@ -600,7 +612,8 @@ impl<'t> ChunkedEdit<'t> { let d = self.hdr.data(img, self.layout_msg)?; let node_size = u32::from_le_bytes([d[at], d[at + 1], d[at + 2], d[at + 3]]); let size_len = if filtered { - earray::chunk_size_len(self.chunk_bytes as u64, self.lpos.version) + 4 + earray::chunk_size_len(self.chunk_bytes as u64, self.lpos.version, img.ls) + + 4 } else { 0 }; diff --git a/crates/clawhdf5/tests/huge_chunks_interop.rs b/crates/clawhdf5/tests/huge_chunks_interop.rs index bf2ff19..0fdf6f4 100644 --- a/crates/clawhdf5/tests/huge_chunks_interop.rs +++ b/crates/clawhdf5/tests/huge_chunks_interop.rs @@ -294,3 +294,223 @@ fn unfiltered_huge_chunks_read() { let read = storage.read.load(std::sync::atomic::Ordering::Relaxed); assert!(read < 1 << 20, "{read} bytes read"); } + +/// `FileEditor` does not rewrite chunks of 4 GiB or more: writing values, +/// or a resize that prunes or allocates chunks, is refused before anything +/// is written. Growing the extent and setting attributes still work. +#[test] +fn editor_refuses_rewriting_huge_chunks() { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("huge.h5"); + std::fs::copy(fixture(), &path).unwrap(); + let before = std::fs::read(&path).unwrap(); + let mut ed = clawhdf5::FileEditor::open(&path).unwrap(); + let one = Selection::Hyperslab { + start: vec![0], + stride: vec![1], + count: vec![1], + block: vec![1], + }; + for name in ["single", "farray", "earray"] { + let err = ed.write_values(name, &one, &[5.0f64]).unwrap_err(); + assert!( + matches!(&err, clawhdf5::Error::Unsupported(m) if m.contains("4 GiB")), + "{name}: {err:?}" + ); + } + // Shrinking prunes and fills chunks: refused. + for (name, shape) in [("earray", vec![10]), ("btree2", vec![1, N + 10])] { + let err = ed.resize(name, &shape).unwrap_err(); + assert!( + matches!(&err, clawhdf5::Error::Unsupported(m) if m.contains("4 GiB")), + "{name}: {err:?}" + ); + } + assert_eq!( + std::fs::read(&path).unwrap(), + before, + "a refused edit wrote" + ); + + // Growing without early allocation touches no chunk: the dataspace + // changes, as libhdf5's `H5Dset_extent` changes it. + ed.resize("earray", &[N + 20]).unwrap(); + ed.resize("btree2", &[3, N + 10]).unwrap(); + ed.set_attr("earray", "note", &clawhdf5::AttrValue::F64(1.5)) + .unwrap(); + let file = ed.reader().unwrap(); + assert_eq!(file.dataset("earray").unwrap().shape().unwrap(), [N + 20]); + assert_eq!( + file.dataset("btree2").unwrap().shape().unwrap(), + [3, N + 10] + ); + assert!(matches!( + file.dataset("earray").unwrap().attr("note").unwrap(), + Some(clawhdf5::AttrValue::F64(v)) if v == 1.5 + )); +} + +/// The Fixed and Extensible Array structures clawhdf5 builds for 4 GiB +/// chunks are libhdf5's byte for byte: built at the fixture's addresses +/// from the fixture's chunks, they match what libhdf5 2.0.0 wrote (the +/// header's own address fields aside, which point where each writer put +/// the next block). +#[test] +fn huge_chunk_array_indexes_match_libhdf5() { + use clawhdf5_format::chunked_write::WrittenChunk; + let data = std::fs::read(fixture()).unwrap(); + let u64_at = |at: usize| u64::from_le_bytes(data[at..at + 8].try_into().unwrap()); + for name in ["farray", "earray"] { + let (_, layout, _, mut chunks) = layout_of(&data, name); + let DataLayout::Chunked { + btree_address: Some(hdr), + .. + } = layout + else { + panic!("{name}: {layout:?}"); + }; + let hdr = hdr as usize; + chunks.sort_by_key(|c| c.offsets[0]); + let slots: Vec> = chunks + .iter() + .map(|c| { + Some(WrittenChunk { + address: c.address, + compressed_size: c.chunk_size, + raw_size: N * 8, + filter_mask: c.filter_mask, + }) + }) + .collect(); + if name == "farray" { + let ours = clawhdf5_format::chunked_write::build_fixed_array_at( + &slots, 8, 8, true, hdr as u64, + ); + // FAHD up to (not including) the data block address. + assert_eq!(&ours[..16], &data[hdr..hdr + 16], "FAHD"); + let dblk = u64_at(hdr + 16) as usize; + let fadb = &ours[28..]; + assert_eq!(fadb, &data[dblk..dblk + fadb.len()], "FADB"); + } else { + let ours = clawhdf5_format::ea_writer::build_extensible_array_at( + &slots, 8, 8, true, hdr as u64, + ); + // EAHD up to (not including) the index block address. + assert_eq!(&ours[..60], &data[hdr..hdr + 60], "EAHD"); + let iblk = u64_at(hdr + 60) as usize; + let eaib = &ours[72..]; + assert_eq!(eaib, &data[iblk..iblk + eaib.len()], "EAIB"); + } + } +} + +/// h5dump from libhdf5 2.x, when `CLAWHDF5_H5DUMP2` names one (Debian's +/// h5dump is 1.14, which cannot read layout message version 5). +fn h5dump2() -> Option { + std::env::var_os("CLAWHDF5_H5DUMP2").map(PathBuf::from) +} + +/// clawhdf5 writes chunks of 4 GiB + 8 bytes (deflated) in every index it +/// uses — Single Chunk, Fixed Array, Extensible Array, v2 B-tree — with +/// layout message version 5 and 8-byte stored sizes, as libhdf5 2.x does; +/// h5py (libhdf5 2.x) and h5dump 2.x read them. Each dataset holds ten +/// values in one 4 GiB chunk, so writing it holds one 4 GiB chunk. +#[test] +fn writer_huge_chunks_round_trip() { + if !heavy() { + return; + } + const UNLIM: u64 = u64::MAX; + let dir = scratch(); + let path = dir.path().join("ours.h5"); + let values: Vec = (0..10).map(f64::from).collect(); + let mut b = clawhdf5::FileBuilder::new(); + for (name, shape, max, chunk) in [ + ("single", vec![10], vec![N], vec![N]), + ("farray", vec![10], vec![N + 10], vec![N]), + ("earray", vec![10], vec![UNLIM], vec![N]), + ("btree2", vec![1, 10], vec![UNLIM, UNLIM], vec![1, N]), + ] { + b.create_dataset(name) + .with_f64_data(&values) + .with_shape(&shape) + .with_maxshape(&max) + .with_chunks(&chunk) + .with_deflate(6); + } + b.write(&path).unwrap(); + + let data = std::fs::read(&path).unwrap(); + assert!(data.len() < 64 << 20, "{} bytes", data.len()); + let mut stored = Vec::new(); + for (name, index) in [("single", 1), ("farray", 3), ("earray", 4), ("btree2", 5)] { + let (version, layout, _, chunks) = layout_of(&data, name); + assert_eq!(version, 5, "{name}"); + assert!( + matches!(layout, DataLayout::Chunked { chunk_index_type: Some(t), .. } if t == index), + "{name}: {layout:?}" + ); + assert_eq!(chunks.len(), 1, "{name}"); + stored.push((name, chunks[0].chunk_size)); + } + + let file = File::open(&path).unwrap(); + let mut expect = values.clone(); + expect.extend([0.0; 2]); + for name in ["single", "farray", "earray"] { + let got = file.dataset(name).unwrap().read_f64().unwrap(); + assert_eq!(got, values, "{name}"); + assert_eq!(sel(&file, name, &[0], &[10]), values, "{name}"); + } + assert_eq!(sel(&file, "btree2", &[0, 3], &[1, 7]), f(3..10)); + + if python_available() { + let out = Command::new(python()) + .arg("-c") + .arg( + r#" +import sys, h5py, numpy as np +with h5py.File(sys.argv[1], "r") as f: + for name in ("single", "farray", "earray", "btree2"): + d = f[name] + assert d.chunks[-1] == 2**29 + 1, (name, d.chunks) + got = d[0] if d.ndim == 2 else d[:] + assert list(got) == list(np.arange(10.0)), (name, got) +print("ok") +"#, + ) + .arg(&path) + .output() + .unwrap(); + assert!( + out.status.success(), + "h5py:\n{}", + String::from_utf8_lossy(&out.stderr) + ); + } else { + assert!(!interop_required(), "h5py not available"); + } + + // h5dump 2.2.0 reads the layouts and walks every index: the storage + // size it reports is the chunk's. (It cannot print the values: its + // deflate filter fails on any chunk over 4 GiB, libhdf5's own included, + // with "memory allocation failed for deflate uncompression".) + if let Some(h5dump) = h5dump2() { + for (name, size) in stored { + let out = Command::new(&h5dump) + .args(["-H", "-p", "-d", name]) + .arg(&path) + .output() + .unwrap(); + let text = format!( + "{}{}", + String::from_utf8_lossy(&out.stdout), + String::from_utf8_lossy(&out.stderr) + ); + assert!(out.status.success(), "h5dump {name}:\n{text}"); + assert!(text.contains("536870913 )"), "{name}:\n{text}"); + assert!(text.contains(&format!("SIZE {size} ")), "{name}:\n{text}"); + assert!(!text.contains("rror"), "{name}:\n{text}"); + } + } +} From 7a2e61ebd192a3dbec123752bb9d977ede4897ec Mon Sep 17 00:00:00 2001 From: osobh Date: Mon, 28 Sep 2026 23:37:34 -0500 Subject: [PATCH 3/7] Huge chunks: selection tests with fill values, LZ4 over 256 MiB, wasm32 check A selection of a chunked dataset with a fill value is compared with the full read in every index, through a map and through positioned reads. An LZ4 chunk larger than 256 MiB is bounded by the chunk size, not refused. The wasm package test reads the 4 GiB-chunk fixture and gets a clean error in every index. Co-Authored-By: Claude Opus 5.5 (1M context) --- crates/clawhdf5-format/src/chunked_read.rs | 4 +- crates/clawhdf5-format/src/chunked_write.rs | 4 +- .../clawhdf5-format/src/extensible_array.rs | 2 +- crates/clawhdf5-format/src/filters.rs | 33 +++++++ crates/clawhdf5-format/src/parallel_read.rs | 2 +- crates/clawhdf5-format/src/partial_read.rs | 5 +- .../clawhdf5-format/tests/raw_fetch_bounds.rs | 2 +- crates/clawhdf5-tools/src/check.rs | 4 +- crates/clawhdf5-tools/src/info.rs | 2 +- crates/clawhdf5/src/edit/mod.rs | 27 +++--- crates/clawhdf5/tests/huge_chunks_interop.rs | 4 +- .../tests/partial_read_equivalence.rs | 87 +++++++++++++++++++ examples/wasm-viewer/test/test.mjs | 14 +++ 13 files changed, 165 insertions(+), 25 deletions(-) diff --git a/crates/clawhdf5-format/src/chunked_read.rs b/crates/clawhdf5-format/src/chunked_read.rs index fa3f016..5f0528a 100644 --- a/crates/clawhdf5-format/src/chunked_read.rs +++ b/crates/clawhdf5-format/src/chunked_read.rs @@ -1491,7 +1491,7 @@ pub fn list_chunks_for_read_in( let chunk_bytes = checked_chunk_byte_len(&chunk_dims, elem_size)?; if let Some(c) = chunks .iter() - .find(|c| c.address != u64::MAX && c.chunk_size as usize != chunk_bytes) + .find(|c| c.address != u64::MAX && c.chunk_size != chunk_bytes as u64) { return Err(FormatError::ChunkedReadError(format!( "incorrect chunk size returned from index for unfiltered chunk at {:?}: \ @@ -2470,7 +2470,7 @@ mod tests { // Entries: key[i], child[i] pairs, then final key for chunk in chunks { // Key: chunk_size(4) + filter_mask(4) + ndims offsets - buf.extend_from_slice(&chunk.chunk_size.to_le_bytes()); + buf.extend_from_slice(&(chunk.chunk_size as u32).to_le_bytes()); buf.extend_from_slice(&chunk.filter_mask.to_le_bytes()); for d in 0..ndims { let off = if d < chunk.offsets.len() { diff --git a/crates/clawhdf5-format/src/chunked_write.rs b/crates/clawhdf5-format/src/chunked_write.rs index 14c3e22..89b0f7f 100644 --- a/crates/clawhdf5-format/src/chunked_write.rs +++ b/crates/clawhdf5-format/src/chunked_write.rs @@ -2263,7 +2263,7 @@ mod tests { for i in 0..n { let (offsets, chunk) = extract_chunk(&data, &shape, &chunks, 2, i); assert_eq!(chunk.len(), 2 * 3 * 2 * 2); - for (e, pair) in chunk.chunks_exact(2).enumerate() { + for (e, pair) in chunk.as_chunks::<2>().0.iter().enumerate() { let c = [e / 6, (e / 2) % 3, e % 2]; let g: Vec = (0..3).map(|d| offsets[d] + c[d] as u64).collect(); let expect = if (0..3).all(|d| g[d] < shape[d]) { @@ -2272,7 +2272,7 @@ mod tests { } else { [0, 0] }; - assert_eq!(pair, expect, "chunk {i} element {e}"); + assert_eq!(*pair, expect, "chunk {i} element {e}"); } } // Data shorter than the shape: the missing elements stay zero. diff --git a/crates/clawhdf5-format/src/extensible_array.rs b/crates/clawhdf5-format/src/extensible_array.rs index a17367b..889f2cf 100644 --- a/crates/clawhdf5-format/src/extensible_array.rs +++ b/crates/clawhdf5-format/src/extensible_array.rs @@ -946,7 +946,7 @@ mod tests { assert_eq!(chunks.len(), 2); assert_eq!(chunks[0].address, base_addr); assert_eq!(chunks[0].offsets, vec![0]); - assert_eq!(chunks[0].chunk_size, chunk_byte_size as u64); + assert_eq!(chunks[0].chunk_size, chunk_byte_size); assert_eq!(chunks[1].address, base_addr + chunk_byte_size); assert_eq!(chunks[1].offsets, vec![20]); } diff --git a/crates/clawhdf5-format/src/filters.rs b/crates/clawhdf5-format/src/filters.rs index ded4c0a..d902041 100644 --- a/crates/clawhdf5-format/src/filters.rs +++ b/crates/clawhdf5-format/src/filters.rs @@ -2764,6 +2764,39 @@ mod tests { assert_eq!(decompressed, data); } + /// A chunk's size bounds an LZ4 chunk, not the 256 MiB ceiling for an + /// unknown size: a 300 MiB chunk was refused ("declared size exceeds + /// limit"). + #[test] + #[cfg(feature = "lz4")] + fn lz4_chunks_over_256_mib_decode() { + let mut data = vec![0u8; 300 << 20]; + data[12345] = 7; + let compressed = lz4_compress(&data, &[]).unwrap(); + let decompressed = lz4_decompress(&compressed, data.len()).unwrap(); + assert!(decompressed == data); + // Without a chunk size the ceiling still applies. + assert!(lz4_decompress(&compressed, 0).is_err()); + } + + /// A chunk of 4 GiB or more is always in the registered framing, whose + /// big-endian size then does not start with four zero bytes: its size + /// is read whole (here larger than the chunk, so refused before any + /// allocation), not taken for a legacy 4-byte size. + #[test] + #[cfg(all(feature = "lz4", target_pointer_width = "64"))] + fn lz4_chunks_of_4_gib_use_the_registered_framing() { + let chunk = (1usize << 32) + 8; + let mut data = ((chunk + 8) as u64).to_be_bytes().to_vec(); + data.extend_from_slice(&(1u32 << 30).to_be_bytes()); + data.extend_from_slice(&[0; 8]); + let err = lz4_decompress(&data, chunk).unwrap_err(); + assert!( + matches!(&err, FormatError::DecompressionError(m) if m.contains("exceeds chunk size")), + "{err:?}" + ); + } + #[test] #[cfg(feature = "lz4")] fn pipeline_lz4_only() { diff --git a/crates/clawhdf5-format/src/parallel_read.rs b/crates/clawhdf5-format/src/parallel_read.rs index 0594e42..38288b4 100644 --- a/crates/clawhdf5-format/src/parallel_read.rs +++ b/crates/clawhdf5-format/src/parallel_read.rs @@ -255,7 +255,7 @@ pub fn decompress_chunks_lane_partitioned_in( for &local in &indices { let index = batch.start + local; let chunk_info = &chunks[index]; - let size = chunk_info.chunk_size as usize; + let size = crate::addr::saturating_usize(chunk_info.chunk_size); let raw_chunk = raw_bytes.get(index, &reqs[index])?; let decompressed = decompress_chunk_exact( diff --git a/crates/clawhdf5-format/src/partial_read.rs b/crates/clawhdf5-format/src/partial_read.rs index 269c45b..f9896de 100644 --- a/crates/clawhdf5-format/src/partial_read.rs +++ b/crates/clawhdf5-format/src/partial_read.rs @@ -368,7 +368,10 @@ pub fn read_selection_filled_in( fill: Option<&[u8]>, ) -> Result>, FormatError> { let dims = &dataspace.dimensions; - if dims.is_empty() || elem_size == 0 { + // A fill value that is not one element's bytes is the full path's to + // interpret. + let odd_fill = fill.is_some_and(|f| !f.is_empty() && f.len() != elem_size); + if dims.is_empty() || elem_size == 0 || odd_fill { return Ok(None); } let total = dataspace.checked_num_elements()?; diff --git a/crates/clawhdf5-format/tests/raw_fetch_bounds.rs b/crates/clawhdf5-format/tests/raw_fetch_bounds.rs index a65d611..c8b2471 100644 --- a/crates/clawhdf5-format/tests/raw_fetch_bounds.rs +++ b/crates/clawhdf5-format/tests/raw_fetch_bounds.rs @@ -151,7 +151,7 @@ fn crafted() -> (Vec, Chunked, Vec) { // v1 B-tree key (size, filter mask, offsets + 0) then the child // address. let mut pat = Vec::new(); - pat.extend_from_slice(&c.chunk_size.to_le_bytes()); + pat.extend_from_slice(&(c.chunk_size as u32).to_le_bytes()); pat.extend_from_slice(&c.filter_mask.to_le_bytes()); // The key holds one offset per dimension plus the element offset // (0); `offsets` may or may not list that last one. diff --git a/crates/clawhdf5-tools/src/check.rs b/crates/clawhdf5-tools/src/check.rs index 4a36554..1ba8239 100644 --- a/crates/clawhdf5-tools/src/check.rs +++ b/crates/clawhdf5-tools/src/check.rs @@ -916,7 +916,7 @@ impl Checker<'_> { bad.push("has size 0".into()); } else if !filtered && let Some(cb) = chunk_bytes - && u64::from(c.chunk_size) != cb + && c.chunk_size != cb { bad.push(format!( "is {} bytes; an unfiltered chunk is {cb}", @@ -933,7 +933,7 @@ impl Checker<'_> { ); } } - self.extent(c.address, u64::from(c.chunk_size), path); + self.extent(c.address, c.chunk_size, path); } if reported > 50 { self.problem( diff --git a/crates/clawhdf5-tools/src/info.rs b/crates/clawhdf5-tools/src/info.rs index d169f73..b0239f1 100644 --- a/crates/clawhdf5-tools/src/info.rs +++ b/crates/clawhdf5-tools/src/info.rs @@ -224,7 +224,7 @@ pub fn allocated_bytes(h5: &H5, info: &DsInfo) -> Result { let ds = info.ds.as_ref().map_err(Clone::clone)?; chunks(h5, layout, ds, dt)? .iter() - .map(|c| u64::from(c.chunk_size)) + .map(|c| c.chunk_size) .sum() } DataLayout::Virtual { .. } => 0, diff --git a/crates/clawhdf5/src/edit/mod.rs b/crates/clawhdf5/src/edit/mod.rs index 4d4f635..169c6d3 100644 --- a/crates/clawhdf5/src/edit/mod.rs +++ b/crates/clawhdf5/src/edit/mod.rs @@ -684,7 +684,7 @@ impl<'t> ChunkedEdit<'t> { } _ => return Err(Error::Unsupported("chunk index missing".into())), } - img.free(info.address, u64::from(info.chunk_size)); + img.free(info.address, info.chunk_size); Ok(()) } @@ -1638,7 +1638,7 @@ fn store_chunk( let len = bytes.len() as u64; let placed = match existing { Some(info) if t.pipeline.is_none() => { - if u64::from(info.chunk_size) != len { + if info.chunk_size != len { return Err(Error::Unsupported( "unfiltered chunk stored at an unexpected size".into(), )); @@ -1650,11 +1650,10 @@ fn store_chunk( // thing in the file (the chunk an append keeps rewriting usually is) // and can grow there. Some(info) - if len <= u64::from(info.chunk_size) - || img.grow_tail(info.address, u64::from(info.chunk_size), len)? => + if len <= info.chunk_size || img.grow_tail(info.address, info.chunk_size, len)? => { img.write(info.address, &bytes)?; - (len != u64::from(info.chunk_size) || info.filter_mask != mask).then_some(Elem { + (len != info.chunk_size || info.filter_mask != mask).then_some(Elem { addr: info.address, size: len, mask, @@ -1664,7 +1663,7 @@ fn store_chunk( let a = img.alloc(len)?; img.write(a, &bytes)?; if let Some(info) = existing { - img.free(info.address, u64::from(info.chunk_size)); + img.free(info.address, info.chunk_size); } Some(Elem { addr: a, @@ -1682,14 +1681,16 @@ fn store_chunk( fn img_read<'a>(f: &'a File, info: &ChunkInfo) -> Result<&'a [u8], Error> { let start = usize::try_from(info.address) .map_err(|_| Error::Unsupported("chunk address out of range".into()))?; - f.as_bytes() - .get(start..start + info.chunk_size as usize) - .ok_or_else(|| { - Error::Format(clawhdf5_format::error::FormatError::UnexpectedEof { - expected: start + info.chunk_size as usize, - available: f.as_bytes().len(), - }) + let end = usize::try_from(info.chunk_size) + .ok() + .and_then(|n| start.checked_add(n)) + .unwrap_or(usize::MAX); + f.as_bytes().get(start..end).ok_or_else(|| { + Error::Format(clawhdf5_format::error::FormatError::UnexpectedEof { + expected: end, + available: f.as_bytes().len(), }) + }) } fn decode_chunk( diff --git a/crates/clawhdf5/tests/huge_chunks_interop.rs b/crates/clawhdf5/tests/huge_chunks_interop.rs index 0fdf6f4..52885e7 100644 --- a/crates/clawhdf5/tests/huge_chunks_interop.rs +++ b/crates/clawhdf5/tests/huge_chunks_interop.rs @@ -88,7 +88,9 @@ fn layout_of( /// The fixture's datasets: name, chunk index type, shape, and the scaled /// origins of the chunks libhdf5 wrote. -const FILTERED: &[(&str, u8, &[u64], &[&[u64]])] = &[ +type Case = (&'static str, u8, &'static [u64], &'static [&'static [u64]]); + +const FILTERED: &[Case] = &[ ("single", 1, &[N], &[&[0]]), ("farray", 3, &[N + 10], &[&[0], &[N]]), ("earray", 4, &[N + 10], &[&[0], &[N]]), diff --git a/crates/clawhdf5/tests/partial_read_equivalence.rs b/crates/clawhdf5/tests/partial_read_equivalence.rs index a84b220..86fa568 100644 --- a/crates/clawhdf5/tests/partial_read_equivalence.rs +++ b/crates/clawhdf5/tests/partial_read_equivalence.rs @@ -195,3 +195,90 @@ fn out_of_bounds_selections_are_errors() { [99] ); } + +/// A chunked dataset with a fill value and chunks never written: a +/// selection is read over a box of fill values from the chunks it touches +/// (it used to be picked out of a full read), and through positioned reads +/// an unfiltered chunk is read row by row. Both equal the full read, in +/// every chunk index h5py writes. +#[test] +fn fill_value_selections_match_full_reads() { + let python = std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".into()); + let has_h5py = std::process::Command::new(&python) + .args(["-c", "import h5py"]) + .output() + .is_ok_and(|o| o.status.success()); + if !has_h5py { + assert!( + std::env::var("CLAWHDF5_REQUIRE_INTEROP").as_deref() != Ok("1"), + "CLAWHDF5_REQUIRE_INTEROP=1 but python3 with h5py is not available" + ); + eprintln!("SKIP: python3 with h5py not available"); + return; + } + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("fill.h5"); + let script = r#" +import sys, h5py, numpy as np +with h5py.File(sys.argv[1], "w", libver="latest") as f: + for name, maxshape, gz in [ + ("farray", None, False), ("farray_gz", None, True), + ("earray", (None, 23), False), ("earray_gz", (None, 23), True), + ("btree2", (None, None), False), ("btree2_gz", (None, None), True), + ]: + d = f.create_dataset(name, shape=(37, 23), maxshape=maxshape, chunks=(5, 4), + dtype=" pkg.openUrl("http://huge.invalid/x.h5", { fetch: mockFetch(farBytes, { total }) }), total > 2 ** 53 ? /2\^53 - 1/ : /4 GiB/, `length ${total}`); } + + // Nor can it hold a chunk of 4 GiB or more (HDF5 2.0, layout message + // version 5): the file opens and lists, and reading such a chunk is an + // error naming the size, in every chunk index. + const hugeChunks = join(import.meta.dirname, "..", "..", "..", + "crates/clawhdf5/tests/fixtures/huge_chunks_filtered.h5"); + const hc = pkg.open(new Uint8Array(readFileSync(hugeChunks))); + eq(hc.list("/").map((e) => e.name), ["btree2", "earray", "farray", "single"], "huge chunks: list"); + for (const name of ["single", "farray", "earray", "btree2"]) { + const [start, count] = name === "btree2" ? [[0, 0], [1, 4]] : [[0], [4]]; + await fails(() => hc.readHyperslab(`/${name}`, start, count), /exceeds the addressable size/, + `huge chunk: ${name}`); + } + hc.free(); } // A body of `total` bytes in 64 KiB pieces, made as they are read; `pulled()` From 740d1124a8e84be74050d0c3066fd717af0954e0 Mon Sep 17 00:00:00 2001 From: osobh Date: Mon, 28 Sep 2026 23:43:47 -0500 Subject: [PATCH 4/7] =?UTF-8?q?docs:=20chunks=20of=204=20GiB=20or=20more?= =?UTF-8?q?=20=E2=80=94=20CHANGELOG,=20known-issues,=20README?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three fixed entries (reading, writing, LZ4 over 256 MiB) and the open limits entry; the selection-read entry loses its fill-value case; the FileEditor limits gain the refused rewrites; the README tables name what is supported and what is not. Co-Authored-By: Claude Opus 5.5 (1M context) --- CHANGELOG.md | 74 +++++++++ README.md | 4 +- .../tests/fixtures/gen_huge_chunks.py | 5 +- docs/known-issues.md | 143 +++++++++++++++++- 4 files changed, 217 insertions(+), 9 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 95b763c..88e255b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -111,6 +111,80 @@ the pre-1.8 format (version-0 superblock, symbol-table groups) cannot be written. `docs/known-issues.md`. +### Chunks of 4 GiB or more (HDF5 2.0's layout message version 5) (2026-09-28) +- **What libhdf5 does** (read in libhdf5 2.2.0's `H5Dchunk.c`, `H5Dfarray.c`, + `H5Dearray.c`, `H5Dbtree2.c`, `H5Dbtree.c`, `H5Olayout.c`): a chunk of + more than 0xFFFFFFFF bytes requires layout message version 5, so a high + bound of `H5F_LIBVER_V200` ("chunk size > 4GB requires H5F_LIBVER_V200"), + whatever the low bound, and version 5 always uses the newer indexes. + Under version 5 a filtered Fixed Array, Extensible Array or v2 B-tree + element stores the chunk's size in "size of lengths" bytes (8); under + version 4 in one byte more than the chunk needs. A filtered Single Chunk + stores it in "size of lengths" bytes under both. Unfiltered elements + store no size, and the Implicit index none at all. A v1 B-tree key keeps + a 32-bit size: libhdf5 never writes a larger chunk there and refuses to + open one ("chunk size must be < 4GB with v1 b-tree index"). +- **Reading** such chunks works in every index: `ChunkInfo::chunk_size` and + `ChunkMapping::file_size` are now `u64` (**public API change**). The + Single Chunk, Implicit, Fixed and Extensible Array readers truncated an + unfiltered chunk's size (the read failed), and the v2 B-tree reader + refused any chunk over 4 GiB. Checked against a fixture libhdf5 2.0.0 + wrote (`tests/fixtures/huge_chunks_filtered.h5`, 188 KB: a 4 GiB + 8 + byte chunk in each filtered index, deflated twice) and a 44 GiB sparse + file h5py writes at test time (every unfiltered index). +- **Selections read what they touch:** a selection of a chunked dataset with + a non-default fill value is read over a box of fill values from the + chunks it touches (new `partial_read::read_selection_filled_in`); it + used to decode the whole dataset. An unfiltered chunk of a file that is + not in memory (`File::open_storage`) is read row by row instead of whole. + A filtered chunk is still decoded whole. +- **Writing:** a chunk of more than `u32::MAX` bytes gets layout message + version 5 and libhdf5's index element widths; the Fixed and Extensible + Array structures match libhdf5 2.0.0's byte for byte. New public helpers + `chunked_write::{layout_version_for, chunk_size_len, MAX_V4_CHUNK_BYTES}`. + Refused before anything is written: chunk dimensions of 2^32 or more + (they were cut to 32 bits) and, for such chunks, LZF, bitshuffle, bzip2, + Blosc and pcodec. Chunks are extracted row by row and one at a time + (compressed in parallel only up to 64 MiB), and a filter pipeline no + longer copies its input first. +- **Deflate** compressed only the first 4 GiB - 1 bytes of a larger chunk + (zlib takes at most that much per call, and `Finish` ended the stream + there; since the one-pass deflate of 2026-09-23, in `clawhdf5-format` and + `clawhdf5-filters`), and it reserved the compressor's worst-case bound, + which flate2's Rust backends zero on every call: 4 GiB kept per 4 GiB + chunk. Inputs over 64 MiB now start at 1/16 of the bound and grow, and a + result more than 64 MiB too large is shrunk. On the read side, an + intermediate deflate stage (one followed by a filter other than shuffle + or Fletcher-32) no longer reserves the chunk's whole bound: reading the + double-deflated fixture peaked at 8.5 GiB, now 4.0 GiB. +- **LZ4:** a chunk decoding to more than 256 MiB was refused ("lz4: + declared size exceeds limit") though the chunk size bounded it; the + ceiling now applies only to a decode of unknown size. A chunk of 4 GiB + or more is read as the registered HDF5 framing, whose 64-bit size no + longer starts with four zero bytes. +- **`FileEditor`** refuses, before writing anything, to write values into + or to prune, fill or allocate chunks of 4 GiB or more; growing the extent + (late allocation) and setting attributes work. It sizes a version-5 + filtered index element from the file's size of lengths (it assumed 8). +- **32-bit targets:** reading or writing such a chunk is + `FormatError::Overflow`; `scripts/check-32bit-casts.sh` passes, and the + wasm package test reads the fixture and gets that error in every index. +- Tests: `crates/clawhdf5/tests/huge_chunks_interop.rs` (index listing, + byte-for-byte array indexes and the editor always; with + `CLAWHDF5_HUGE_CHUNKS=1`, reads of every index filtered and unfiltered, + and a write of every index read back by clawhdf5, h5py 3.16 and — layout + and storage size — h5dump 2.2.0 via `CLAWHDF5_H5DUMP2`), + `partial_read_equivalence::fill_value_selections_match_full_reads`, unit + tests of the encodings (`chunked_write::tests`) and of LZ4. The opt-in + run (`cargo test --release -p clawhdf5 --test huge_chunks_interop -- + --test-threads=1`) peaked at 8.06 GiB resident, h5py included (tank, + 2026-09-28, commit `9143737`). libhdf5 2.2.0's h5dump cannot print these + chunks' values (its deflate filter fails on any chunk over 4 GiB, + libhdf5's own included); h5py 3.16 reads them. `docs/known-issues.md`: three fixed + entries and the open "Chunks of 4 GiB or more: limits" (chunk dimensions + of 2^32 or more, the refused filters, memory, the untested unfiltered + write). + ### A dropped `FileEditor` releases its lock at once (2026-09-28) - `FileEditor`'s `flock` could outlive the editor for a moment when another thread forked to spawn a process: the child shared the locked diff --git a/README.md b/README.md index a87b5a2..d7f61da 100644 --- a/README.md +++ b/README.md @@ -116,9 +116,9 @@ Limits and open issues, with dates, are in | **File format** | Superblock v0–v3, user blocks, v1/v2 object headers; writing the HDF5 1.10 format (default) or, with `libver_bounds(LibVer::V18, LibVer::V18)`, files HDF5 1.8 reads (checked with HDF5 1.8.23) | Metadata cache images | Writing the pre-1.8 format (version-0 superblock, symbol-table groups) | | **Groups and links** | Symbol-table, compact and dense groups (tested to 100 000 links), creation order, soft and hard links; writing external links | | Following external links (explicit error); user-defined links are skipped | | **Datatypes** | Integers and IEEE floats of every width and byte order (incl. `f16`), enums, compounds (every version, incl. HDF5 2.0's v5), arrays, fixed-length strings, opaque, complex: h5py's `{r, i}` compound (`with_complex_f64_data`) and HDF5 2.0's native class 11 (`with_native_complex_f64_data`, opt-in: only libhdf5 2.0+ reads it; reads surface it as `{r, i}`) | Variable-length strings and sequences, object references; HDF5 2.x's small floats (bfloat16, FP8 E4M3/E5M2, FP6 E2M3/E3M2, FP4 E2M1: every bit pattern decoded as libhdf5 2.2.0 decodes it) and other non-IEEE floats up to 64 bits | Writing variable-length data; writing non-IEEE floats; decoding region and attribute references; x87 long double and binary128 | -| **Layouts and chunk indexes** | Compact, contiguous and chunked; chunk indexes single chunk, Fixed Array, Extensible Array and v2 B-tree (the writer picks one as libhdf5 does), and v1 B-tree (every chunked dataset under the 1.8 bound, split as libhdf5 splits it); fill values; resizable datasets; virtual datasets (read limits in known-issues) | The implicit chunk index (the editor also changes it) | External raw data files (explicit error) | +| **Layouts and chunk indexes** | Compact, contiguous and chunked; chunk indexes single chunk, Fixed Array, Extensible Array and v2 B-tree (the writer picks one as libhdf5 does), and v1 B-tree (every chunked dataset under the 1.8 bound, split as libhdf5 splits it); chunks of 4 GiB or more (HDF5 2.0's layout message version 5); fill values; resizable datasets; virtual datasets (read limits in known-issues) | The implicit chunk index (the editor also changes it) | External raw data files (explicit error); chunk dimensions of 2^32 or more | | **Filters** | deflate (pure-Rust zlib-rs), shuffle, Fletcher-32, LZ4 (opt-in), Zstd (C, opt-in); plugins LZF, bitshuffle, bzip2, Blosc 1 | N-Bit, scale-offset, SZIP (C, opt-in); plugins Blosc2 and ZFP | Other filter IDs, unless you register a codec (`filter_registry::register_filter`) | -| **Editing in place** | `FileEditor`: overwrite values, grow and shrink chunked datasets (every index), set attributes (compact and dense), in files from h5py or clawhdf5 | | Creating or deleting objects in an existing file; deleting attributes; new chunks in implicit indexes; VL data; filters this build cannot encode (refused before any write) | +| **Editing in place** | `FileEditor`: overwrite values, grow and shrink chunked datasets (every index), set attributes (compact and dense), in files from h5py or clawhdf5 | | Creating or deleting objects in an existing file; deleting attributes; new chunks in implicit indexes; VL data; rewriting chunks of 4 GiB or more; filters this build cannot encode (refused before any write) | | **Access** | Local files (mmap or buffered), bytes in memory, any `Storage` backend, HTTP(S) and S3/GCS/Azure via `clawhdf5-remote`, SWMR reading (`File::open_swmr`, `Dataset::refresh`) | Remote files and the browser are read-only | SWMR writing; remote SWMR; MPI collective I/O (`clawhdf5-io`'s `mpi-io` reads on one rank and broadcasts) | | **Bindings** | Python (read, `'w'` for numeric and complex arrays, `'r+'` editing, URLs), NetCDF-4 (CF scale/offset/fill) | WebAssembly (`open(bytes)`, `openUrl`); no Zstd/SZIP/pcodec, no compound, reference, opaque, bitfield, time or VL-sequence datasets | Node.js (the package does not work; see known-issues) | diff --git a/crates/clawhdf5/tests/fixtures/gen_huge_chunks.py b/crates/clawhdf5/tests/fixtures/gen_huge_chunks.py index 16c18c4..f901fd5 100644 --- a/crates/clawhdf5/tests/fixtures/gen_huge_chunks.py +++ b/crates/clawhdf5/tests/fixtures/gen_huge_chunks.py @@ -32,10 +32,11 @@ chunk still inflates to 4 GiB + 8 bytes. writes only the elements written (an unfiltered chunk larger than the chunk cache is written in place) and the file is sparse: tens of GiB long but a few blocks on disk. Unwritten elements read as whatever the file holds there: -zeros. Do not put it on tmpfs, which is memory. +zeros. Do not put it on tmpfs, which is memory. Its Single Chunk is +allocated early (see `make`). libhdf5 holds a whole filtered chunk in memory while it writes it, so the -`filtered` run needs about 4.5 GiB of memory (and about 3 minutes). +`filtered` run needs about 4 GiB of memory and 100 s (tank). Generated 2026-09-28 with h5py 3.16.0 (libhdf5 2.0.0). """ diff --git a/docs/known-issues.md b/docs/known-issues.md index 9d932d0..89e0f1c 100644 --- a/docs/known-issues.md +++ b/docs/known-issues.md @@ -15,6 +15,10 @@ Checked against `main` at `9b5803f` on 2026-09-28. | Issue | Kind | Since | |---|---|---| + +| [NetCDF-4: variables' dimensions are guessed from sizes](#netcdf-4-variables-dimensions-are-guessed-from-sizes) | **wrong metadata** (`Variable::dimensions`, pure dimension scales listed as variables; values and dimension sizes are right) | 2026-09-28 | +| [Chunks of 4 GiB or more: limits](#chunks-of-4-gib-or-more-limits) | refused (chunk dimensions of 2^32 or more, five filters, editor rewrites), memory (a decoded chunk is held whole) | 2026-09-28 | +| [NetCDF-4: an unlimited dimension reports size 0](#netcdf-4-an-unlimited-dimension-reports-size-0) | **wrong metadata** (dimension size; variable shapes and values are right) | 2026-09-28 | | [Small floats decode as libhdf5 does, not as the OCP MX specification](#small-floats-decode-as-libhdf5-does-not-as-the-ocp-mx-specification) | deliberate: libhdf5's values (FP4/FP6/FP8 E4M3 all-ones exponent is inf/NaN) | 2026-09-28 | | [In-place modification (`FileEditor`) limits](#in-place-modification-fileeditor-limits) | refused edits (`Error::Unsupported`), space reuse per editor, no journal | 2026-09-26 | | [Python in-place editing limits](#python-in-place-editing-clawhdf5filepath-r-limits) | refused writes (`NotImplementedError`), deliberate conversion differences | 2026-09-27 | @@ -57,6 +61,10 @@ writing anything: message (they have no dense storage); - partial edge chunks stored unfiltered (`H5Pset_chunk_opts`), external raw data files, virtual datasets; +- values of a dataset whose chunks are 4 GiB or more (HDF5 2.0), and a + resize of one that prunes, fills or allocates chunks (added 2026-09-28; + growing its extent under late allocation, and its attributes, work): + see [Chunks of 4 GiB or more](#chunks-of-4-gib-or-more-limits); - files with a metadata cache image, paged or persistent free-space management, a driver info block, or version-3 consistency flags set. @@ -141,18 +149,79 @@ before anything is written. On top of them: read (and so the Python `ds[...]`) of contiguous data copies just the selected runs; of chunked data it materialises only the selection's bounding box when that box covers at most half the dataset -(`partial_read`). It decodes the whole dataset and extracts the selection -instead when: +(`partial_read`), with the dataset's fill value where no chunk was +written. It decodes the whole dataset and extracts the selection instead +when: - the bounding box covers more than half the dataset — which a strided selection across a chunked dataset (`ds[::100]`) always does, although it may touch few chunks; -- the dataset is compact or virtual, or has no storage; -- it is chunked with a non-default fill value (the box path does not fill - unallocated chunks, so the fill-aware full read is used). +- the dataset is compact or virtual, or has no storage. + +A chunked dataset with a non-default fill value was read whole for any +selection until 2026-09-28 (`partial_read::read_selection_filled_in` fills +the box now; `partial_read_equivalence::fill_value_selections_match_full_reads`). Values are correct in every case; this is cost only. Selections other than `Selection::All` also bypass the file's chunk cache. +## Chunks of 4 GiB or more: limits + +**Status:** open (documented 2026-09-28). HDF5 2.0 writes a chunk of more +than 0xFFFFFFFF bytes with layout message version 5 (`H5D__chunk_construct` +in libhdf5 2.2.0: "chunk size > 4GB requires H5F_LIBVER_V200"), which +also makes a filtered chunk index element store the chunk's size in "size +of lengths" bytes (8) instead of one byte more than the chunk needs. +clawhdf5 reads and writes such chunks in every index libhdf5 gives them +(Single Chunk, Implicit, Fixed Array, Extensible Array, v2 B-tree; libhdf5 +never puts one in a v1 B-tree and refuses to open one, as clawhdf5 does). +The [fix](#chunks-of-4-gib-or-more-could-not-be-read) is tested by +`crates/clawhdf5/tests/huge_chunks_interop.rs`; its decoding tests are +opt-in (`CLAWHDF5_HUGE_CHUNKS=1`, one at a time: `--test-threads=1`). +What remains: +- **Chunk dimensions of 2^32 or more** (which libhdf5 2.x also allows under + `H5F_LIBVER_V200`) are refused, when read (`InvalidChunkDimensions`) + and when written: `DataLayout::Chunked::chunk_dimensions` is `Vec`. + A 4 GiB chunk of 1-byte elements needs one. +- **Filters the writer refuses for such a chunk** (`FilterError`, before + anything is written): LZF, bitshuffle, bzip2 and Blosc (their HDF5 + filters record sizes or block lengths in 32 bits, or cannot take such a + buffer) and pcodec. Deflate, shuffle, Fletcher-32, LZ4 and Zstd are + written; only shuffle + deflate is tested end to end (LZ4's framing and + the 256 MiB limit it had are unit-tested; Zstd is not tested at this + size). +- **Unfiltered chunks this size are written but not tested end to end:** + `FileBuilder` assembles the whole file in memory, several copies of the + chunk (well over the 12 GiB the tests may use). Their layout messages + are unit-tested (`chunked_write::tests::huge_chunk_layout_messages_and_index_elements`). +- **`FileEditor`** refuses to rewrite such chunks (see + [its limits](#in-place-modification-fileeditor-limits)). +- **Memory:** decoding a filtered chunk holds all of it (4 GiB and more), + twice when it is shuffled (the inflated and the unshuffled copy, as + libhdf5's filters do too); writing one holds the chunk and its shuffled + copy. A selection decodes only the chunks it touches; one of an + unfiltered chunk reads only the rows it selects, through a memory map or + positioned reads (`File::open_storage`): every selection of + `unfiltered_huge_chunks_read` together read under 1 MiB. Peak resident + memory of `cargo test --release -p clawhdf5 --test huge_chunks_interop + -- --test-threads=1` with `CLAWHDF5_HUGE_CHUNKS=1` (all six tests, h5py + included) was 8.06 GiB (tank, 2026-09-28, commit `9143737`); selection + reads of the double-deflated fixture alone, 4.0 GiB. +- **32-bit targets (wasm32):** such a chunk cannot be held in memory: + reading or writing one is `FormatError::Overflow` ("exceeds the + addressable size" / "address space"); the file still opens and lists + (`examples/wasm-viewer/test/test.mjs`, "huge chunk"). +- **libhdf5 2.2.0's h5dump cannot print these values:** built here from tag + `2.2.0`, it fails to inflate any deflated chunk over 4 GiB, those + libhdf5 2.0.0 writes included ("memory allocation failed for deflate + uncompression", `H5Zdeflate.c`), while h5py 3.16 (libhdf5 2.0.0) reads + them. The tests check h5dump 2.2.0 on layouts and storage sizes only + (`CLAWHDF5_H5DUMP2`). +- libhdf5 2.0.0 itself (h5py 3.16) drops a write to an unallocated, + unfiltered Single Chunk this large when the fill time is "never" (the + dataset stays unallocated); `fixtures/gen_huge_chunks.py` allocates it + early instead. Not a clawhdf5 issue; recorded because the generator + depends on it. + ## HDF5 features still unsupported **Status:** open. What remains of the gaps the @@ -564,6 +633,70 @@ under load from other builds) a freshly opened file with 8192 chunks per dataset reads small selections 1.2x to 2.3x slower through a version-1 B-tree (see `BENCHMARKS.md`, "HDF5 1.8 format"). +## Chunks of 4 GiB or more could not be read + +**Status:** fixed 2026-09-28 (branch `feat/huge-chunks`); affected every +release (v2.1.0 to v2.7.0). Errors, and cost; no wrong values were +returned. Nothing for users to do but upgrade. + +HDF5 2.0 writes chunks of more than 4 GiB - 1 bytes (layout message +version 5). `ChunkInfo::chunk_size` was a `u32`, so the Single Chunk, +Implicit, Fixed Array and Extensible Array readers truncated an +unfiltered chunk's size and the read failed ("incorrect chunk size +returned from index for unfiltered chunk"); a v2 B-tree index refused any +such chunk ("chunk larger than 4 GiB"), filtered or not. Filtered chunks +in the other indexes read, because their stored sizes were small. A +selection of a chunked dataset with a non-default fill value decoded the +whole dataset (8 GiB of output for the fixture's 2-D dataset), and a +selection of an unfiltered chunk fetched the whole chunk from a file that +is not in memory. `ChunkInfo::chunk_size` and `ChunkMapping::file_size` are now +`u64`; the selection path fills its box with the fill value and reads an +unfiltered chunk's rows only. Tests: `huge_chunks_interop` +(`filtered_huge_chunk_indexes_list` always; with `CLAWHDF5_HUGE_CHUNKS=1` +`filtered_huge_chunks_read`, over a fixture libhdf5 2.0.0 wrote, and +`unfiltered_huge_chunks_read`, over a 44 GiB sparse file h5py writes at +test time). Limits that remain: +[Chunks of 4 GiB or more](#chunks-of-4-gib-or-more-limits). + +## Chunks of 4 GiB or more were written unreadable + +**Status:** fixed 2026-09-28 (branch `feat/huge-chunks`). The deflate +truncation was before any release (the one-pass deflate dates from +2026-09-23); the rest affected every release (v2.1.0 to v2.7.0). Files +clawhdf5 wrote with a chunk of 4 GiB or more should be written again. + +The writer gave such a chunk layout message version 4 (which libhdf5 +before 2.0 cannot read, and which libhdf5 2.x never writes for it; whether +2.x reads what clawhdf5 wrote was not checked), cut a chunk dimension of +2^32 or more to 32 bits, and, since 2026-09-23, deflated only the first +4 GiB - 1 bytes of the chunk (zlib takes at most that much per call and +`Finish` ended the stream there): the chunk failed to decode ("decoded to +4294967295 bytes, expected 4294967304"). It now writes layout version 5 +with libhdf5's index element widths (the Fixed and Extensible Array +structures match libhdf5 2.0.0's byte for byte: +`huge_chunk_array_indexes_match_libhdf5`), refuses chunk dimensions of +2^32 or more and the filters that cannot take such a chunk, deflates the +whole chunk, and no longer keeps the compressor's worst-case bound (4 GiB +of zeroed memory for a chunk that deflates to 4 MiB): writing the four +datasets of `writer_huge_chunks_round_trip` went past the tests' 12 GiB +cap, and compressing one shuffled 4 GiB chunk now peaks at 4.3 GiB +(tank, 2026-09-28). h5py 3.16 reads what it writes. + +## LZ4 chunks larger than 256 MiB were refused + +**Status:** fixed 2026-09-28 (branch `feat/huge-chunks`); affected every +release (v2.1.0 to v2.7.0). An error, never wrong data. Nothing for users +to do but upgrade. + +The LZ4 decoder refused a chunk that decodes to more than 256 MiB +("lz4: declared size exceeds limit") even when the dataset's chunk size +bounded it; that ceiling is meant for a decode whose size is unknown, and +now applies only then (deflate never had it). A chunk of 4 GiB or more is +always read as the registered HDF5 framing, whose 64-bit size then no +longer starts with four zero bytes. Tests: +`filters::tests::lz4_chunks_over_256_mib_decode`, +`lz4_chunks_of_4_gib_use_the_registered_framing`. + ## A dropped `FileEditor` could keep its file locked for a moment **Status:** fixed 2026-09-28 (#23), before any release From b55768ff76b740eb52d312bbe9890711a80728db Mon Sep 17 00:00:00 2001 From: osobh Date: Mon, 28 Sep 2026 23:46:04 -0500 Subject: [PATCH 5/7] chunked writer: chunks of 4 GiB or more never use the v1 B-tree; need high bound V200 Stacking feat/libver-v18 under feat/huge-chunks sent every chunked dataset of a file with a 1.8 low bound to a version-1 B-tree, whose key holds a 32-bit chunk size. As in libhdf5, such a chunk now takes layout version 5 whatever the low bound, and a high bound below 2.0 is FormatError::LibverBound. Co-Authored-By: Claude Opus 5.5 (1M context) --- crates/clawhdf5-format/src/chunked_write.rs | 57 ++++++++++++++++++++- crates/clawhdf5-format/src/file_writer.rs | 6 ++- 2 files changed, 60 insertions(+), 3 deletions(-) diff --git a/crates/clawhdf5-format/src/chunked_write.rs b/crates/clawhdf5-format/src/chunked_write.rs index 89b0f7f..d750fd8 100644 --- a/crates/clawhdf5-format/src/chunked_write.rs +++ b/crates/clawhdf5-format/src/chunked_write.rs @@ -984,7 +984,13 @@ pub fn build_chunked_data_from_precompressed( base_address: u64, maxshape: Option<&[u64]>, ) -> Result { - build_chunked_data_from_precompressed_libver(pre, base_address, maxshape, LibVer::Latest) + build_chunked_data_from_precompressed_libver( + pre, + base_address, + maxshape, + LibVer::Latest, + LibVer::Latest, + ) } /// [`build_chunked_data_from_precompressed`] for a file whose low library @@ -992,13 +998,28 @@ pub fn build_chunked_data_from_precompressed( /// every chunked dataset gets a version-3 layout message and a version-1 /// B-tree chunk index, whatever its maximum shape, as libhdf5 writes it; /// otherwise the version-4 layout and the index libhdf5 picks for it. +/// +/// A chunk of 4 GiB or more (over [`MAX_V4_CHUNK_BYTES`]) takes layout +/// message version 5 whatever `low` is, as in libhdf5 (a version-1 B-tree +/// key holds a 32-bit size), and needs a `high` bound of at least +/// [`LibVer::V200`] ([`FormatError::LibverBound`] otherwise). pub fn build_chunked_data_from_precompressed_libver( pre: &PrecompressedChunks, base_address: u64, maxshape: Option<&[u64]>, low: LibVer, + high: LibVer, ) -> Result { - if low < LibVer::V110 { + let (_, chunk_bytes) = checked_chunk_dims(&pre.chunk_dims, pre.element_size)?; + if chunk_bytes > MAX_V4_CHUNK_BYTES { + if high < LibVer::V200 { + return Err(FormatError::LibverBound { + what: format!("a chunk of {chunk_bytes} bytes (4 GiB or more)"), + needs: LibVer::V200, + high, + }); + } + } else if low < LibVer::V110 { return build_btree_v1_chunked_data(pre, base_address, maxshape); } let index = ChunkIndexPlan::new(&pre.shape, maxshape, &pre.chunk_dims)?; @@ -2102,6 +2123,38 @@ mod tests { assert_eq!(layout_version_for(HUGE_BYTES), 5); } + /// A chunk of 4 GiB or more never goes into a version-1 B-tree (its key + /// holds a 32-bit size): under a 1.8 low bound it still takes layout + /// version 5, and a high bound below 2.0 refuses it, as in libhdf5. + #[test] + fn huge_chunks_ignore_the_v18_low_bound_and_need_v200() { + // One filtered chunk; the stored bytes stand in for its compression. + let pre = PrecompressedChunks { + chunks: vec![(HUGE_BYTES, vec![0u8; 16], 0)], + has_filters: true, + element_size: 8, + shape: vec![HUGE_DIM], + chunk_dims: vec![HUGE_DIM], + pipeline_message: None, + }; + let r = build_chunked_data_from_precompressed_libver( + &pre, + 4096, + None, + LibVer::V18, + LibVer::Latest, + ) + .unwrap(); + assert_eq!(r.layout_message[0], 5, "layout message version"); + for high in [LibVer::V18, LibVer::V114] { + match build_chunked_data_from_precompressed_libver(&pre, 4096, None, LibVer::V18, high) + { + Err(FormatError::LibverBound { needs, .. }) => assert_eq!(needs, LibVer::V200), + other => panic!("high bound {high}: {:?}", other.map(|r| r.layout_message)), + } + } + } + /// `H5D_FARRAY_FILT_COMPUTE_CHUNK_SIZE_LEN` (and its EA and v2 B-tree /// twins) in libhdf5 2.2.0: one byte more than the chunk size needs up to /// layout version 4, the size of lengths from version 5. diff --git a/crates/clawhdf5-format/src/file_writer.rs b/crates/clawhdf5-format/src/file_writer.rs index 6dcfe47..4eb59ee 100644 --- a/crates/clawhdf5-format/src/file_writer.rs +++ b/crates/clawhdf5-format/src/file_writer.rs @@ -1517,7 +1517,9 @@ impl FileWriter { /// (virtual datasets, a paged file, the 1.12 reference types, native /// complex numbers). With a low bound of 1.8 and a later high bound /// such objects are written in the newer format, as libhdf5 writes them; - /// the rest of the file stays readable by 1.8. + /// the rest of the file stays readable by 1.8. A chunk of 4 GiB or more + /// always takes HDF5 2.0's version-5 layout (never a version-1 B-tree) + /// and so needs a high bound of at least [`LibVer::V200`]. /// /// A low bound above the high bound makes [`Self::finish`] fail. pub fn libver_bounds(&mut self, low: LibVer, high: LibVer) -> &mut Self { @@ -1830,6 +1832,7 @@ impl FileWriter { dummy_cursor, d.maxshape.as_deref(), low, + high, )?; dummy_cursor += result.data_bytes.len() as u64; let oh = build_chunked_dataset_oh( @@ -1991,6 +1994,7 @@ impl FileWriter { base_address, d.maxshape.as_deref(), low, + high, )?; cursor2 += result.data_bytes.len(); let oh = build_chunked_dataset_oh( From 23e13b3d9c966236546ecaf7d2fa9dfb60d35a16 Mon Sep 17 00:00:00 2001 From: osobh Date: Mon, 28 Sep 2026 23:47:01 -0500 Subject: [PATCH 6/7] BENCHMARKS: idle re-run of the HDF5 1.8 format A/B (supersedes the loaded run) Co-Authored-By: Claude Opus 5.5 (1M context) --- BENCHMARKS.md | 50 +++++++++++++++++++++++++++++++++++++++++++++++++- 1 file changed, 49 insertions(+), 1 deletion(-) diff --git a/BENCHMARKS.md b/BENCHMARKS.md index 4024ecb..c81101a 100644 --- a/BENCHMARKS.md +++ b/BENCHMARKS.md @@ -536,7 +536,10 @@ The rows and columns of the uncompressed layouts are within 20% (chunked column 0.45 -> 0.49 ms, contiguous column 2.55 -> 2.61 ms). This run does not explain the slower windows. -### HDF5 1.8 format: version-1 B-tree chunk indexes (2026-09-28, tank) +### HDF5 1.8 format: version-1 B-tree chunk indexes (2026-09-28, tank, loaded) + +**Superseded** by the idle re-run below; kept as the record the default +was first decided on. Measured 2026-09-28 on tank (AMD Ryzen 7 7800X3D), branch `feat/libver-v18` at `3c61635`, to decide whether the writer's default should become the @@ -598,6 +601,51 @@ about 150 per dataset here) against a Fixed Array's few blocks. Writing costs th same; files grow by about 36 bytes per chunk. The default therefore stays the 1.10 format; the 1.8 format is opt-in. +### HDF5 1.8 format: version-1 B-tree chunk indexes, idle re-run (2026-09-28, tank) + +Measured 2026-09-28 on tank (AMD Ryzen 7 7800X3D), idle: 1-minute load +average 1.66 to 1.84 at the start of each run. Stacked branch +`feat/huge-chunks` at `b55768f` (contains `feat/libver-v18`). Default and +`--v18` runs alternated, five of each per chunk size; median (min-max), ms. + +> **Run:** `cargo run --release -p clawhdf5-bench --bin read_harness -- --chunk N [--v18]` +> with N = 256 (128 chunks per dataset) and N = 32 (8192 chunks per dataset) + +| chunks | layout | read | 1.10 | 1.8 | 1.8 / 1.10 | +|---|---|---|---:|---:|---:| +| 256 | chunked + deflate | full (first) | 5.9 (5.5-6.2) | 6.2 (6.0-6.7) | 1.05x | +| 256 | chunked + deflate | full (repeat) | 4.4 (4.1-4.9) | 4.3 (4.2-4.4) | 0.98x | +| 256 | chunked + deflate | 64 x 64 window (1 chunk) | 0.15 (0.15-0.17) | 0.16 (0.15-0.16) | 1.07x | +| 256 | chunked + deflate | 512 x 512 window (4-9 chunks) | 1.23 (1.23-1.36) | 1.22 (1.22-1.25) | 0.99x | +| 256 | chunked + deflate | one row | 0.95 (0.94-1.04) | 0.94 (0.93-0.98) | 0.99x | +| 256 | chunked + deflate | one column | 1.92 (1.91-2.12) | 1.90 (1.90-1.91) | 0.99x | +| 256 | chunked | full (first) | 11.1 (10.7-11.6) | 10.9 (10.6-11.9) | 0.98x | +| 256 | chunked | full (repeat) | 5.1 (4.9-5.2) | 5.2 (4.7-5.6) | 1.02x | +| 256 | chunked | 64 x 64 window (1 chunk) | 0.03 (0.03-0.03) | 0.04 (0.04-0.04) | 1.33x | +| 256 | chunked | 512 x 512 window (4-9 chunks) | 0.36 (0.36-0.38) | 0.37 (0.36-0.38) | 1.03x | +| 256 | chunked | one row | 0.05 (0.04-0.05) | 0.05 (0.05-0.05) | 1.00x | +| 256 | chunked | one column | 0.45 (0.43-0.45) | 0.44 (0.44-0.45) | 0.98x | +| 32 | chunked + deflate | full (first) | 11.7 (11.2-12.4) | 12.5 (11.8-13.2) | 1.07x | +| 32 | chunked + deflate | full (repeat) | 9.5 (9.3-9.8) | 9.4 (9.3-11.2) | 0.99x | +| 32 | chunked + deflate | 64 x 64 window (1 chunk) | 0.46 (0.45-0.49) | 0.89 (0.89-0.90) | 1.93x | +| 32 | chunked + deflate | 512 x 512 window (4-9 chunks) | 1.95 (1.94-2.04) | 2.40 (2.39-2.51) | 1.23x | +| 32 | chunked + deflate | one row | 0.69 (0.67-0.72) | 1.15 (1.13-1.18) | 1.67x | +| 32 | chunked + deflate | one column | 1.26 (1.24-1.28) | 1.72 (1.69-1.73) | 1.37x | +| 32 | chunked | full (first) | 12.5 (11.9-12.7) | 12.6 (12.5-13.4) | 1.01x | +| 32 | chunked | full (repeat) | 5.9 (5.8-6.0) | 6.2 (6.0-6.4) | 1.05x | +| 32 | chunked | 64 x 64 window (1 chunk) | 0.36 (0.35-0.36) | 0.84 (0.83-0.86) | 2.33x | +| 32 | chunked | 512 x 512 window (4-9 chunks) | 0.70 (0.69-0.74) | 1.17 (1.16-1.17) | 1.67x | +| 32 | chunked | one row | 0.38 (0.38-0.41) | 0.85 (0.84-0.86) | 2.24x | +| 32 | chunked | one column | 1.03 (1.02-1.21) | 1.22 (1.20-1.25) | 1.18x | + +Writes: 377 vs 374 ms (128 chunks), 399 vs 409 ms (8192 chunks); file +bytes as in the loaded run. The loaded run's 0.25x and 0.52x windows at 128 +chunks were the load: idle, every 128-chunk read is within 7% except the +0.03 ms single-chunk window (one timer tick). The 8192-chunk result stands: +a selection on a freshly opened file costs 0.2 to 0.5 ms more through the +version-1 B-tree (1.2x to 2.3x), full reads are within 7%. The default stays +the 1.10 format. + ## Local file speed after range reads ### `ObjectHeader::parse` back at 8f59b2e's speed (2026-09-27, tank) From 4b075656e79dbbf0235d2eba4273b67528faa1d1 Mon Sep 17 00:00:00 2001 From: osobh Date: Tue, 29 Sep 2026 00:00:28 -0500 Subject: [PATCH 7/7] =?UTF-8?q?docs:=20known-issues=20=E2=80=94=20cite=20P?= =?UTF-8?q?Rs=20#27,=20#28,=20#29?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-Authored-By: Claude Opus 5.5 (1M context) --- docs/known-issues.md | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/docs/known-issues.md b/docs/known-issues.md index 89e0f1c..09fd916 100644 --- a/docs/known-issues.md +++ b/docs/known-issues.md @@ -542,7 +542,7 @@ given. ## NetCDF-4: variables' dimensions are guessed from sizes -**Status:** fixed 2026-09-28 (branch `fix/netcdf-dimension-list`). Affected +**Status:** fixed 2026-09-28 (#27). Affected every release (v2.1.0 to v2.7.0: size matching dates from the crate's first version). Wrong metadata only: stored values were always read right. Users who read `_Netcdf4Coordinates` themselves can use @@ -594,7 +594,7 @@ comparison over the conformance corpus's netCDF-readable files. ## HDF5 1.8 could not read the files we wrote -**Status:** fixed 2026-09-28 (branch `feat/libver-v18`), as an opt-in. +**Status:** fixed 2026-09-28 (#28), as an opt-in. Affected every release (v2.1.0 to v2.7.0): the writer only ever wrote the HDF5 1.10 format. Users who need HDF5 1.8 to read their files call `FileBuilder::libver_bounds(LibVer::V18, LibVer::V18)` (format crate: @@ -635,7 +635,7 @@ B-tree (see `BENCHMARKS.md`, "HDF5 1.8 format"). ## Chunks of 4 GiB or more could not be read -**Status:** fixed 2026-09-28 (branch `feat/huge-chunks`); affected every +**Status:** fixed 2026-09-28 (#29); affected every release (v2.1.0 to v2.7.0). Errors, and cost; no wrong values were returned. Nothing for users to do but upgrade. @@ -660,7 +660,7 @@ test time). Limits that remain: ## Chunks of 4 GiB or more were written unreadable -**Status:** fixed 2026-09-28 (branch `feat/huge-chunks`). The deflate +**Status:** fixed 2026-09-28 (#29). The deflate truncation was before any release (the one-pass deflate dates from 2026-09-23); the rest affected every release (v2.1.0 to v2.7.0). Files clawhdf5 wrote with a chunk of 4 GiB or more should be written again. @@ -684,7 +684,7 @@ cap, and compressing one shuffled 4 GiB chunk now peaks at 4.3 GiB ## LZ4 chunks larger than 256 MiB were refused -**Status:** fixed 2026-09-28 (branch `feat/huge-chunks`); affected every +**Status:** fixed 2026-09-28 (#29); affected every release (v2.1.0 to v2.7.0). An error, never wrong data. Nothing for users to do but upgrade.