From ac7871fff5684f9934e5c75f7d0528c4dca4a768 Mon Sep 17 00:00:00 2001 From: osobh Date: Mon, 28 Sep 2026 22:53:12 -0500 Subject: [PATCH] Read chunks of 4 GiB or more in every chunk index ChunkInfo::chunk_size (and ChunkMapping::file_size) are u64: sizes past u32 were truncated for Single Chunk, Implicit, Fixed and Extensible Array indexes, and a v2 B-tree index refused them. A selection of a chunked dataset with a non-default fill value is read over a box of fill values instead of a full read, an unfiltered chunk of a file that is not in memory is read row by row, and an intermediate deflate stage no longer reserves the chunk's whole bound. Co-Authored-By: Claude Opus 5.5 (1M context) --- .gitignore | 3 + crates/clawhdf5-format/src/chunk_cache.rs | 2 +- crates/clawhdf5-format/src/chunk_index.rs | 4 +- crates/clawhdf5-format/src/chunked_read.rs | 45 +-- crates/clawhdf5-format/src/chunked_write.rs | 2 +- .../clawhdf5-format/src/extensible_array.rs | 6 +- crates/clawhdf5-format/src/filters.rs | 17 +- crates/clawhdf5-format/src/fixed_array.rs | 8 +- crates/clawhdf5-format/src/parallel_read.rs | 2 +- crates/clawhdf5-format/src/partial_read.rs | 114 +++++++ .../clawhdf5-format/tests/raw_fetch_bounds.rs | 4 +- crates/clawhdf5/src/reader.rs | 33 +- .../tests/fixtures/gen_huge_chunks.py | 109 +++++++ .../tests/fixtures/huge_chunks_filtered.h5 | Bin 0 -> 188488 bytes crates/clawhdf5/tests/huge_chunks_interop.rs | 296 ++++++++++++++++++ crates/clawhdf5/tests/parallel_integration.rs | 2 +- 16 files changed, 604 insertions(+), 43 deletions(-) create mode 100644 crates/clawhdf5/tests/fixtures/gen_huge_chunks.py create mode 100644 crates/clawhdf5/tests/fixtures/huge_chunks_filtered.h5 create mode 100644 crates/clawhdf5/tests/huge_chunks_interop.rs diff --git a/.gitignore b/.gitignore index cf90c2c..d0a22c0 100644 --- a/.gitignore +++ b/.gitignore @@ -7,3 +7,6 @@ weights/ .venv __pycache__/ .pytest_cache/ + +# Scratch files the heavy tests generate (huge_chunks_interop) +crates/*/tests/scratch/ diff --git a/crates/clawhdf5-format/src/chunk_cache.rs b/crates/clawhdf5-format/src/chunk_cache.rs index 6e3fee8..cca133d 100644 --- a/crates/clawhdf5-format/src/chunk_cache.rs +++ b/crates/clawhdf5-format/src/chunk_cache.rs @@ -899,7 +899,7 @@ mod tests { fn make_chunk(offsets: Vec, address: u64, size: u32) -> ChunkInfo { ChunkInfo { - chunk_size: size, + chunk_size: u64::from(size), filter_mask: 0, offsets, address, diff --git a/crates/clawhdf5-format/src/chunk_index.rs b/crates/clawhdf5-format/src/chunk_index.rs index 706bad0..1e083a0 100644 --- a/crates/clawhdf5-format/src/chunk_index.rs +++ b/crates/clawhdf5-format/src/chunk_index.rs @@ -109,7 +109,7 @@ pub struct ChunkMapping { /// File byte address of the compressed chunk. pub file_offset: u64, /// Size of the compressed chunk in the file. - pub file_size: u32, + pub file_size: u64, /// Filter mask (0 = all filters applied). pub filter_mask: u32, /// Pre-computed row-copy operations for assembling this chunk into output. @@ -338,7 +338,7 @@ mod tests { fn make_chunk(offsets: Vec, address: u64, size: u32) -> ChunkInfo { ChunkInfo { - chunk_size: size, + chunk_size: u64::from(size), filter_mask: 0, offsets, address, diff --git a/crates/clawhdf5-format/src/chunked_read.rs b/crates/clawhdf5-format/src/chunked_read.rs index 4d97adb..fa3f016 100644 --- a/crates/clawhdf5-format/src/chunked_read.rs +++ b/crates/clawhdf5-format/src/chunked_read.rs @@ -314,7 +314,7 @@ pub(crate) fn chunk_req( chunk_bytes: usize, wanted: bool, ) -> ExtentReq { - let len = c.chunk_size as usize; + let len = crate::addr::saturating_usize(c.chunk_size); ExtentReq { addr: c.address, len, @@ -521,7 +521,7 @@ pub fn decompress_all_chunks_with_stats_in( #[derive(Debug, Clone)] pub struct ChunkInfo { /// Size of chunk data in the file (after compression). - pub chunk_size: u32, + pub chunk_size: u64, /// Bitmask of filters that were NOT applied (0 = all applied). pub filter_mask: u32, /// N-dimensional offset of this chunk in dataset space. @@ -610,7 +610,7 @@ fn stored_element_size(dt: &Datatype, offset_size: u8) -> u64 { /// (`H5D__chunk_set_sizes`: "stored datatype size in chunk layout does not /// match datatype description"). Reading it anyway laid the chunks out with /// the wrong element size. -pub(crate) fn check_chunk_element_size( +pub fn check_chunk_element_size( layout: &DataLayout, datatype: &Datatype, offset_size: u8, @@ -1028,7 +1028,7 @@ fn parse_chunk_node( if node_level == 0 { chunks.push(stored.len()); stored.push(ChunkInfo { - chunk_size, + chunk_size: u64::from(chunk_size), filter_mask, offsets: keys[k..].to_vec(), address, @@ -1131,7 +1131,7 @@ pub fn generate_implicit_chunks_in_grid( } chunks.push(ChunkInfo { - chunk_size: chunk_byte_size as u32, + chunk_size: chunk_byte_size, filter_mask: 0, offsets, address: base_address.saturating_add(grid_idx.saturating_mul(chunk_byte_size)), @@ -1190,9 +1190,15 @@ fn read_btree_v2_chunks( } _ => return Err(bad("tree is not a chunk index")), }; - let unfiltered_bytes = checked_chunk_byte_len(chunk_dims, elem_size)?; - let unfiltered_bytes = - u32::try_from(unfiltered_bytes).map_err(|_| bad("chunk larger than 4 GiB"))?; + // A u64 whatever the platform: an unfiltered chunk's size is only + // recorded here; a chunk this platform cannot address fails when read. + let unfiltered_bytes = u64::try_from( + chunk_dims + .iter() + .try_fold(elem_size as u128, |acc, &c| acc.checked_mul(c as u128)) + .ok_or_else(|| bad("chunk size overflows"))?, + ) + .map_err(|_| bad("chunk size overflows"))?; let records = collect_btree_v2_records_in(file_data, &header, offset_size, length_size)?; let mut chunks = Vec::with_capacity(records.len()); @@ -1213,10 +1219,7 @@ fn read_btree_v2_chunks( pos += size_len; let mask = u32::from_le_bytes([data[pos], data[pos + 1], data[pos + 2], data[pos + 3]]); pos += 4; - ( - u32::try_from(size).map_err(|_| bad("stored chunk larger than 4 GiB"))?, - mask, - ) + (size, mask) }; let mut offsets = Vec::with_capacity(rank); for &dim in chunk_dims { @@ -1334,9 +1337,9 @@ pub fn list_chunks_in( // Single chunk — one chunk covering the entire dataset let chunk_byte_size = checked_chunk_byte_len(&chunk_dims, elem_size)?; let (csize, fmask) = if let Some(fs) = single_filtered_size { - (fs as u32, single_filter_mask.unwrap_or(0)) + (fs, single_filter_mask.unwrap_or(0)) } else { - (chunk_byte_size as u32, 0) + (chunk_byte_size as u64, 0) }; vec![ChunkInfo { chunk_size: csize, @@ -2124,7 +2127,7 @@ pub fn read_chunked_data_indexed_in( .iter() .zip(&hits) .map(|(m, hit)| { - let len = m.file_size as usize; + let len = crate::addr::saturating_usize(m.file_size); ExtentReq { addr: m.file_offset, len, @@ -2757,7 +2760,7 @@ mod tests { } chunk_infos.push(ChunkInfo { - chunk_size: chunk_bytes as u32, + chunk_size: chunk_bytes as u64, filter_mask: 0, offsets: vec![start as u64, 0], address: data_offset as u64, @@ -2869,7 +2872,7 @@ mod tests { .collect(); let stored = crate::filters::compress_chunk(&chunk, &pipeline, 4).unwrap(); chunks.push(ChunkInfo { - chunk_size: stored.len() as u32, + chunk_size: stored.len() as u64, filter_mask: 0, offsets: vec![r0 as u64, c0 as u64, 0], address: file.len() as u64, @@ -2943,7 +2946,7 @@ mod tests { let short = crate::filters::compress_chunk(&[1u8; 64], &pipeline, 4).unwrap(); for bad in [5usize, 11, 40] { chunks[bad].address = file.len() as u64; - chunks[bad].chunk_size = short.len() as u32; + chunks[bad].chunk_size = short.len() as u64; file.extend_from_slice(&short); } for _ in 0..20 { @@ -3133,7 +3136,7 @@ mod tests { file_data[data_offset..data_offset + compressed.len()].copy_from_slice(&compressed); chunk_infos.push(ChunkInfo { - chunk_size: compressed.len() as u32, + chunk_size: compressed.len() as u64, filter_mask: 0, offsets: vec![start as u64, 0], address: data_offset as u64, @@ -3215,7 +3218,7 @@ mod tests { file_data[data_offset..data_offset + chunk_size].copy_from_slice(&chunk_bytes); chunk_infos.push(ChunkInfo { - chunk_size: chunk_size as u32, + chunk_size: chunk_size as u64, filter_mask: 0, offsets: vec![row_start as u64, col_start as u64, 0], address: data_offset as u64, @@ -3340,7 +3343,7 @@ mod tests { assert_eq!(c.address, 0x1000 + i as u64 * chunk_byte_size as u64); assert_eq!(c.offsets, vec![i as u64 * 20]); assert_eq!(c.filter_mask, 0); - assert_eq!(c.chunk_size, chunk_byte_size as u32); + assert_eq!(c.chunk_size, chunk_byte_size as u64); } } diff --git a/crates/clawhdf5-format/src/chunked_write.rs b/crates/clawhdf5-format/src/chunked_write.rs index 10f1099..746a492 100644 --- a/crates/clawhdf5-format/src/chunked_write.rs +++ b/crates/clawhdf5-format/src/chunked_write.rs @@ -2125,7 +2125,7 @@ mod tests { for info in &infos { // Skipped chunks are stored at the chunk's size (shuffled). assert_eq!( - info.chunk_size == (c * 8) as u32, + info.chunk_size == (c * 8) as u64, info.filter_mask != 0, "{info:?}" ); diff --git a/crates/clawhdf5-format/src/extensible_array.rs b/crates/clawhdf5-format/src/extensible_array.rs index 42cf595..a17367b 100644 --- a/crates/clawhdf5-format/src/extensible_array.rs +++ b/crates/clawhdf5-format/src/extensible_array.rs @@ -224,7 +224,7 @@ fn read_element( }; Ok(( Some(ChunkInfo { - chunk_size: chunk_byte_size as u32, + chunk_size: chunk_byte_size, filter_mask: 0, offsets, address, @@ -259,7 +259,7 @@ fn read_element( }; Ok(( Some(ChunkInfo { - chunk_size: chunk_size as u32, + chunk_size, filter_mask, offsets, address, @@ -946,7 +946,7 @@ mod tests { assert_eq!(chunks.len(), 2); assert_eq!(chunks[0].address, base_addr); assert_eq!(chunks[0].offsets, vec![0]); - assert_eq!(chunks[0].chunk_size, chunk_byte_size as u32); + assert_eq!(chunks[0].chunk_size, chunk_byte_size as u64); assert_eq!(chunks[1].address, base_addr + chunk_byte_size); assert_eq!(chunks[1].offsets, vec![20]); } diff --git a/crates/clawhdf5-format/src/filters.rs b/crates/clawhdf5-format/src/filters.rs index 28070d4..48553c1 100644 --- a/crates/clawhdf5-format/src/filters.rs +++ b/crates/clawhdf5-format/src/filters.rs @@ -340,10 +340,23 @@ pub fn decompress_chunk_exact_with<'s>( } else { MAX_DECOMPRESS_SIZE }; - let size_hint = if ctx.max_output != 0 { + // The bound is the output's size when only size-preserving + // filters (shuffle, Fletcher32) remain to be undone; before + // any other filter (a second deflate, N-Bit, ...) it is only + // a ceiling, and the output starts smaller and grows: the + // inflater writes every byte it reserves, so reserving a + // 4 GiB chunk's bound for a stage a few MiB long would hold + // twice the chunk's memory. + let exact = pipeline.filters[..i].iter().enumerate().all(|(j, f)| { + filter_skipped(filter_mask, j) + || matches!(f.filter_id, FILTER_SHUFFLE | FILTER_FLETCHER32) + }); + let size_hint = if ctx.max_output == 0 { + input.len().saturating_mul(4).min(1 << 20) + } else if exact { ctx.max_output } else { - input.len().saturating_mul(4).min(1 << 20) + input.len().saturating_mul(4).min(ctx.max_output) }; let inflater = scratch .inflater diff --git a/crates/clawhdf5-format/src/fixed_array.rs b/crates/clawhdf5-format/src/fixed_array.rs index de80b7a..dbe00ae 100644 --- a/crates/clawhdf5-format/src/fixed_array.rs +++ b/crates/clawhdf5-format/src/fixed_array.rs @@ -383,7 +383,7 @@ fn parse_fa_element( offset_size: u8, element_size: u8, chunk_byte_size: u64, -) -> Result, FormatError> { +) -> Result, FormatError> { let os = offset_size as usize; if client_id == 0 { // Non-filtered: element is just the chunk address. @@ -393,7 +393,7 @@ fn parse_fa_element( return Ok(None); } let address = read_offset(file_data, abs, offset_size)?; - Ok(Some((address, chunk_byte_size as u32, 0))) + Ok(Some((address, chunk_byte_size, 0))) } else { // Filtered: address(offset_size) + chunk_size(variable) + filter_mask(4) let es = element_size as usize; @@ -418,7 +418,7 @@ fn parse_fa_element( file_data[fm_off + 2], file_data[fm_off + 3], ]); - Ok(Some((address, chunk_size as u32, filter_mask))) + Ok(Some((address, chunk_size, filter_mask))) } } @@ -688,7 +688,7 @@ mod tests { assert_eq!(c.address, base_addr + i as u64 * chunk_byte_size as u64); assert_eq!(c.offsets, vec![i as u64 * 20]); assert_eq!(c.filter_mask, 0); - assert_eq!(c.chunk_size, chunk_byte_size as u32); + assert_eq!(c.chunk_size, chunk_byte_size as u64); } } diff --git a/crates/clawhdf5-format/src/parallel_read.rs b/crates/clawhdf5-format/src/parallel_read.rs index 606173c..0594e42 100644 --- a/crates/clawhdf5-format/src/parallel_read.rs +++ b/crates/clawhdf5-format/src/parallel_read.rs @@ -426,7 +426,7 @@ mod tests { for i in 0..8u64 { let len = if short && i == 5 { 16 } else { 32 }; infos.push(ChunkInfo { - chunk_size: len as u32, + chunk_size: len as u64, filter_mask: 0, offsets: vec![i * 8], address: file.len() as u64, diff --git a/crates/clawhdf5-format/src/partial_read.rs b/crates/clawhdf5-format/src/partial_read.rs index 3361485..269c45b 100644 --- a/crates/clawhdf5-format/src/partial_read.rs +++ b/crates/clawhdf5-format/src/partial_read.rs @@ -243,6 +243,58 @@ fn copy_overlap( } } +/// Copy the part of unfiltered chunk `chunk` (`chunk_bytes` long, shape +/// `chunk_shape`) that overlaps the box into `out`, reading only the runs of +/// the overlap from the file. The whole chunk must still lie inside the +/// file, as it must when it is fetched whole. +#[allow(clippy::too_many_arguments)] +fn read_unfiltered_overlap( + file_data: &S, + chunk: &crate::chunked_read::ChunkInfo, + chunk_bytes: usize, + chunk_shape: &[u64], + elem_size: usize, + out: &mut [u8], + box_start: &[u64], + box_extent: &[u64], +) -> Result<(), FormatError> { + let rank = chunk_shape.len(); + let origin = &chunk.offsets[..rank]; + let (lo, extent): (Vec, Vec) = (0..rank) + .map(|d| { + let lo = origin[d].max(box_start[d]); + let hi = origin[d] + .saturating_add(chunk_shape[d]) + .min(box_start[d] + box_extent[d]); + (lo, hi.saturating_sub(lo)) + }) + .unzip(); + let file_len = crate::storage::len_usize(file_data); + let base = crate::addr::to_usize(chunk.address)?; + if base > file_len || chunk_bytes > file_len - base { + return Err(FormatError::UnexpectedEof { + expected: base.saturating_add(chunk_bytes), + available: file_len, + }); + } + let overlap = Selection::Hyperslab { + start: lo.iter().zip(origin).map(|(l, o)| l - o).collect(), + stride: vec![1; rank], + count: extent.clone(), + block: vec![1; rank], + }; + let rows = crate::gather::gather_storage( + file_data, + chunk.address, + chunk_bytes, + chunk_shape, + elem_size, + &overlap, + )?; + copy_overlap(&rows, &lo, &extent, out, box_start, box_extent, elem_size); + Ok(()) +} + /// Read `selection` without materialising the whole dataset, when that is /// possible and worthwhile. `Ok(None)` means "use the full-read path": an /// `All`/`None`/invalid selection, a layout this doesn't handle (compact, @@ -281,6 +333,39 @@ pub fn read_selection_in( offset_size: u8, length_size: u8, selection: &Selection, +) -> Result>, FormatError> { + read_selection_filled_in( + file_data, + layout, + dataspace, + elem_size, + pipeline, + offset_size, + length_size, + selection, + None, + ) +} + +/// [`read_selection_in`] for a dataset whose fill value is `fill` (one +/// element's bytes; `None` or all zeros is the default fill): the elements +/// of a chunked dataset's selection that lie in chunks never written read as +/// `fill`, as they do in a full read. Only the chunks the selection's +/// bounding box overlaps are read, so a selection of a few elements of a +/// dataset whose chunks are 4 GiB or more costs one decoded chunk (a +/// filtered chunk has to be decoded whole) or, unfiltered, only the bytes +/// it selects. +#[allow(clippy::too_many_arguments)] +pub fn read_selection_filled_in( + file_data: &S, + layout: &DataLayout, + dataspace: &Dataspace, + elem_size: usize, + pipeline: Option<&FilterPipeline>, + offset_size: u8, + length_size: u8, + selection: &Selection, + fill: Option<&[u8]>, ) -> Result>, FormatError> { let dims = &dataspace.dimensions; if dims.is_empty() || elem_size == 0 { @@ -341,8 +426,18 @@ pub fn read_selection_in( return Ok(None); } let mut boxed = alloc_output(checked_byte_len(box_elements, elem_size)?)?; + if let Some(fill) = fill.filter(|f| <[u8]>::len(f) == elem_size && f.iter().any(|&b| b != 0)) { + for element in boxed.chunks_exact_mut(elem_size) { + element.copy_from_slice(fill); + } + } match layout { + // No chunk was ever written: every element is the fill value. + DataLayout::Chunked { + btree_address: None, + .. + } if fill.is_some() => {} DataLayout::Chunked { btree_address: Some(_), .. @@ -373,6 +468,25 @@ pub fn read_selection_in( }) }) .collect(); + // An unfiltered chunk of a file that is not in memory: fetch + // only the rows the box needs, not the whole chunk (which may be + // 4 GiB or more). + let (direct, wanted): (Vec<_>, Vec<_>) = wanted.into_iter().partition(|c| { + file_data.as_contiguous().is_none() + && pipeline.is_none_or(|pl| all_filters_skipped(pl, c.filter_mask)) + }); + for chunk in direct { + read_unfiltered_overlap( + file_data, + chunk, + chunk_bytes, + &chunk_shape, + elem_size, + &mut boxed, + &box_start, + &box_extent, + )?; + } // Their stored bytes, batch by batch when the file is not in // memory; each batch's chunks are decoded into this thread's // reusable buffers before the next batch is fetched. diff --git a/crates/clawhdf5-format/tests/raw_fetch_bounds.rs b/crates/clawhdf5-format/tests/raw_fetch_bounds.rs index 8217323..a65d611 100644 --- a/crates/clawhdf5-format/tests/raw_fetch_bounds.rs +++ b/crates/clawhdf5-format/tests/raw_fetch_bounds.rs @@ -173,7 +173,7 @@ fn crafted() -> (Vec, Chunked, Vec) { assert!( chunks .iter() - .all(|c| c.chunk_size == HUGE && c.address == blob) + .all(|c| c.chunk_size == u64::from(HUGE) && c.address == blob) ); (bytes, ds, chunks) } @@ -313,7 +313,7 @@ fn large_reads_are_fetched_in_batches() { let data: Vec = (0..2 * CHUNK).map(|i| (i % 251) as u8).collect(); let chunks: Vec = (0..40u64) .map(|i| ChunkInfo { - chunk_size: CHUNK as u32, + chunk_size: CHUNK as u64, filter_mask: 0, offsets: vec![i * CHUNK as u64], address: (i % 2) * CHUNK as u64, diff --git a/crates/clawhdf5/src/reader.rs b/crates/clawhdf5/src/reader.rs index 4456d66..4426d39 100644 --- a/crates/clawhdf5/src/reader.rs +++ b/crates/clawhdf5/src/reader.rs @@ -1528,11 +1528,11 @@ impl<'f> Dataset<'f> { let ds = self.dataspace()?; let dl = self.data_layout()?; let pipeline = self.filter_pipeline()?; - // The selection reader knows nothing about fill values. When they - // matter — no storage at all, or a non-zero fill on a chunked (possibly - // sparse) dataset — select from a fill-aware full read instead. (The - // selection reader currently decodes the full dataset too, so this - // costs nothing extra.) + // Fill values matter when there is no storage at all, or a non-zero + // fill on a chunked (possibly sparse) dataset: a chunked dataset's + // selection is then read over a box of fill values; anything else + // (and a selection whose box is most of the dataset) is selected + // from a fill-aware full read. let fill = clawhdf5_format::fill_value::dataset_fill_value_from_storage( &self.file.data, &self.header.messages, @@ -1545,6 +1545,29 @@ impl<'f> Dataset<'f> { && !clawhdf5_format::fill_value::is_default(fill.as_deref())); if fill_matters { clawhdf5_format::partial_read::validate(selection, &ds.dimensions)?; + // A chunked dataset: read the chunks the selection touches over + // a box of fill values, not the whole dataset (which may be many + // GiB when its chunks are large). + if matches!(dl, DataLayout::Chunked { .. }) { + clawhdf5_format::chunked_read::check_chunk_element_size( + &dl, + &dt, + self.file.offset_size(), + )?; + if let Some(selected) = clawhdf5_format::partial_read::read_selection_filled_in( + &self.file.data, + &dl, + &ds, + dt.type_size() as usize, + pipeline.as_ref(), + self.file.offset_size(), + self.file.length_size(), + selection, + Some(fill.as_deref().unwrap_or(&[])), + )? { + return Ok(selected); + } + } let full = self.read_raw()?; return Ok(data_read::extract_selection_from_buffer( &full, diff --git a/crates/clawhdf5/tests/fixtures/gen_huge_chunks.py b/crates/clawhdf5/tests/fixtures/gen_huge_chunks.py new file mode 100644 index 0000000..16c18c4 --- /dev/null +++ b/crates/clawhdf5/tests/fixtures/gen_huge_chunks.py @@ -0,0 +1,109 @@ +"""Write HDF5 files whose chunks are 4 GiB or more, one dataset per chunk index. + + python gen_huge_chunks.py filtered OUT.h5 # huge_chunks_filtered.h5 + python gen_huge_chunks.py unfiltered OUT.h5 # generated at test time + +libhdf5 2.x writes a chunk of more than 0xFFFFFFFF bytes with layout message +version 5 (`H5D__chunk_construct`: "chunk size > 4GB requires +H5F_LIBVER_V200"), so never with a version-1 B-tree; a filtered chunk index +element of a version-5 layout stores the chunk's size in "size of lengths" +bytes (8). Every dataset is ` N: + ds[N : N + 10] = np.arange(100.0, 110.0) + if index == "implicit": + ds[2 * N - 10 : 2 * N] = np.arange(500.0, 510.0) + dsid.close() + + +def main(): + mode, out = sys.argv[1], sys.argv[2] + filtered = {"filtered": True, "unfiltered": False}[mode] + fapl = h5py.h5p.create(h5py.h5p.FILE_ACCESS) + fapl.set_libver_bounds(h5py.h5f.LIBVER_EARLIEST, h5py.h5f.LIBVER_V200) + fid = h5py.h5f.create(out.encode(), h5py.h5f.ACC_TRUNC, fapl=fapl) + indexes = ["single", "farray", "earray", "btree2"] + if not filtered: + indexes.insert(1, "implicit") + for index in indexes: + make(fid, index, index, filtered) + fid.close() + + +if __name__ == "__main__": + main() diff --git a/crates/clawhdf5/tests/fixtures/huge_chunks_filtered.h5 b/crates/clawhdf5/tests/fixtures/huge_chunks_filtered.h5 new file mode 100644 index 0000000000000000000000000000000000000000..60a8931a889d10a8a6d6839ef56fa2e8c79e6e13 GIT binary patch literal 188488 zcmeHw2~<;8yEcd-Dg|0w0cG;mS<7e<5Q&Jh0uEJ+We`CSL}q19s0i4CUO3Kc{YL{w%%L<|@i!yJJAb#EN>wVyWE;+--f3)!j{*zZp37PTFh(9PPp2?Du z;|I`h4e}EBRSLaNzVo>T4SkaQSuL@AhQxA0O7tIzDQNza?@yANBsnpcdkV?tjy(Ry zW8^1hNEnEYOH$7h*&ay=#{*7o_Pz%srjaM1pYeA7`{Oba<2^NH(xgd~$wrc2Ao;bW z=tJYbCcYB={|j*o$hN%J%XR%@i^$V=T3D`?L{>tB{15t37#e^5N5AoN-z@!)&KN(A zt&abc|M)!ISPgN1{>Nwk?|t__n^!Z?P?4?2umAhz)%eIxnIthWucW4*z{C9S=MkO> ze?KPcO*gKUkocz-kZ<@mA7)we##$LUp@IAOdjE~q2{iIP z#uxiadM^|W$2)L9!Rdm)`&OqzE(=n2XJ#_k2e5iku;euwYjwwLu*nv6$r-pd;C=<`Os ztTvT~n)}meuf&IZM#uaUVkBv7_TYXkaYH_Q^wZf|8d}*i;9V#l!Tr^?7L{_Rb@nMW zTdZds6MgGy--4+tesz>x6zP;!@K$_1Jf@;J#R~CD6_{i=&s`URNC(|kgIgv z;gR!jU=B+>f_GiCQk<=8W3NR_nCr1?M+*i$}0FGod;o%u!Y;A}g+nsUd2dLUQ+G^99OM zqNf#qMf8?@H(hD#?uC|3;dp-S{+4krS~A%t=uF~Aai9G9ZA?_hRQBO>ilexHZE2cI zJCuH#*3H>~4_PqPyI<+FEHrMk{WL}N$T1l{V(JL*hzrDcm7 z%Gxc(bF*ToWkFoDWZTQzZ)))MO0L=JD@)xOic9HQBc5Ltspn~1OF}E|68Aj9{uMtLU8}4t0$cCm90HK#QPc_N~w(XLk*d!=B_n}8*+!PSd6TTtew`-O7RHl zceeF^(omL)V0Sw9iyxA;oFLD*O(mh;2j6}?gs<1SLn|>_vW8!_)9;daex>Ysp}xRj zwR4Sa+jV>oPA6OUkDB^}nFBLNK83%7nFBKiW)8eL@a9N!2;LlcbKuQE&TL1IRA>`&DHwU>na?Oyag4`VB=AZ-tB?u@%pmYM1AfN;Rl~$;Ie3|;fAeg$&OX(WYe^Vdus^GMb7|(Yk4c;}eO+!`6$zE9BjFA-`A(mPs>_Z_Ud?-x=T?%0vlh+mgf=ci_&!8iMRE+0{ zXH=kidkE3nlldA$OZ)9sfo zSMO^(GZEV2tFFW|6q={r$#7$7sMWc`2NZZAkX0Q;CaI3wvqP4SV^`KXS9u^XPYc35 z*bo>pq1?vIKo}Yp!`S#6j8tf}QB$Aq@Tm39kdFbzeeeQ&j-EG>A;Qz+6fw5IBhJ-O zRqCz1^)n1gS$uS58ftDh-{}Ab%v^t9C5PHO>qB}c1%qlH>VJp83)1%A`?JLOy(+8z zIS4OfAfL6TrVyhN){k%y?Bn&2y1eBU{H1;gr5nnWda}6!6${+p;O3;{>+(emHuwEH zjPPl=y4RvK^oUY|H@@1>O%_X4rFOFmtm-f(<>ukDVZO5G!bc7TQAneqRrv-824$)= zt2qmUd1eXLBS=a`ap#+XbUaD63k?ynVn9$@T)hK-X;WM8c3o-Xu0UEO1tDXua8Bt6 z>#-X)3aAo~K1?~7axmq<%8^UPYG1H&u&~0y3Ja^qflLHn4tzNXA0T|d@eYI!5V1nU z3P}>uJAfn!kTH=7a4rS_R}PSe&-R-FH{^1X2PV_}!4=zW;}?v#%2(|4!Y1~SLi zoKEAi(zbja={qy`_Fq-795_#0^`Dl8 zZW{jdtv;p!Io(}Vwp(9XwuoC*p00tOauRQp&f#{t$r{LIarV$K(4|>U&F`L0i>@F$ z=o~Z-`G(MQFz4Wb08{eMFv3sTE9<7R8G>JgtsZH#QDTf+n9QdY5DSPAzMZh2##_75 z3@Gw|A`d9?fFchl@JIRwIEAUysX z)_LHL0`4f_juKf&;En=bcaX0D`3jJ)Ah!zxcNB0(0e2K|M*(*fat?tz3M#EoX@yEF zR9Z!mGE`bYbCJ?q1d2SM$ODQzWEX*z5m*_4l@W+Q@%{)P0tN0_;GPA#V^PQy+75Kb zKz9rb+rY3*nuEJQ>JFstKMrJVb=EdX zcXD`S6ul%8!;YRcS$sW5YSyKY9~!WPbk10^e1=YoUT}Wn;VGi4h1Sk1FRQaPl9nxQ zDC06U#1pTZlPK_&HTWzoimQ&5!MZnCPVG?gywCUdPvVX@r79;~v9CkvWy6Bh#{2^l2I=U(mE%L8)7Z2h^$lUQ7*9U zdO&f|w!L@LQk9BecfKF66m!*94Dnvjjii$z-aZ+K#S7=ql@z3`oLwJcY+Qk5nj33o zjQGm7ACBAQei>_H*BfgdP(QxHe)lJQ3(C=zsdY-f$z6GN317@RtNLk4=(>Zfw?}b9 zG#Q*?H`!rXJ1xH-Ohuifp6v5c`b)O;zU!GvM}Sm8{L3=gkM!{ zY9wvlX`-ga!K2f~4G%Vw?tf6WWCf-W*E_71a8JG{QG4kiz6IG5&KozR$GyF6oA_cD zISnq34v+pV)8o)e+>nGqp1r`mtLO5vBRLulV>D zq+5Be`qH&Jc8!;l9D>FIXu$N zOU@ID=VsLiy`avrG_?BV!)zh$5gxtY_f(AJp4`~T(i}WG{J^>BCh`y8duEPh%eN+$ zUi%c*v*7v9A@~-;dIP?lgT%tWGsGI`JG+j~MguL=*-&qRX^-2wgU_9mWU{$!?qb#Q z-xpX}R1Bh2gHp4FEg_zp?ZGtGrxb@rE(dL_3EU$Av_K;ygPz({THI=lhdLzx6uJq~ z(~7(4`is^bgQos7ot_lnTX;O2cmAX#SNoO<)d4+W4+rKnh$YqUZaN#6k?$+36pNn-W~&NBYf<%-dF2f%rf7d^AqkS7hZOJh(a@%IWTibe@P$A9GE#UbKuQ^H%Gd2 z;LU+I2i_b6t=L(HpcR5v2wEX%h1?t&D?)A#a&wTIgWMcWFG2|dN)S+jfD!~Q$}~d> z0xGRgX@yEF%B21dl~$;<{tE#mkBHJ{)hqK7Yc4VU=!s%|n`m*JrdZ7{PRC z?2bWKEg4bce!o7NrG!p%9-iIUjqyjBXRCQzbrBG1*fH~SF=9rqyC$Pd$#a(4k`;$A zWFl_IkcP69D?irB`4UD}+_iqX&|!5@&EB@_xHFwjw(cL*$oq^Hxe9+Wy}Kr%L#by@ zvt|r#Z&HCV3n5`F{U7Gn&zK}?+})ePZz9lX$_3Ix6vQX3t=~8_LV8UY`(uc=O*qk3mKtmXj74NGs=pd-@;-Lcx@SDF;&ytenVHt@Z^g2Ma4Ktgx_(%)yBY z@a4dlgYW^u2V@R3j39h~h!rAMNRmL3L_*>O%TVyAsu2VYOCq}P$$E@s3HIRP3hI{Xx7z?vwr;Wu_FWF%krl4f+ zICf{Fxu1zq-LsNVQ-8xiTQQMK*~rH?K6OglWNSG0F?Wz*dU#}~n{0M%9;?|>4Dxy3 znAFvYC`6hb^7Hk?Se?^)?rLhk%NG^e-y_7FGSE2aIhb?sK!7PBZT0`i2p1O_&=rI= z45H~1!Ap;$356*>M5Uszk4X~_#uAd{85?h{Gu^mWLh?(E$qJLFix!9f;MX`@Q9@!I zD=8uIVBE4x9Nv4scb(~`H4>8U;v?>Y@ymQ|IdQpPwvIE4E&9VTLi{svmW0HAbPT`u z|I2IYI&Oxfgp7o!=f=Aa{k=%kU6PZ=58zRmfl9FC8-MK^DG4Rf=Ren=p(~L;t0k7t zkXTNP!(}C=NGOQjpCmO&a{PVroSwqT;^TijM!qvc!a#Jsr1*=++b1dEc)-cc-uHmS zH1Z_$^N)_pj5kwa$|MOX5waECh6ed1Q^@8yF47nleO%%TacjxDB=mA!$B#-#toaVj z*cs%T^Uxv3MU7%~@MLOUb_6|pj?acy=mF>f=mF?~zwZJ6`cF-FU5gM;E|>B9<7GSM zx13fmQ}K=n`(fI~O|B8|w=K$<6dw3s>yKuOu3ik&*mb`4m2K!A1(iw8(;kF3-`co#_Ztq<~XIu|nzrF-wh`x`P!ymXYMg!Xej zgj3XNJjilX*_#?mI%@Wm|(p+EJ;c4<&ib1BLZqv};Qz+hflfb8#P6xdMkDfw% zG@PNVI{Z?*5)X<6P+q%@F~YFat4Xp!MGA7|FyQI{hj}eDsg>YN|@ni;6p+ z*ka1nH{Ijgg1BvT%yfD^Kk0%hxo0UaIa-K!w*y{~PlY6l@Bu#uhXX}p#;&fU9 zX#pQK@dldATO-AevhAPhQyag-TUl;0AW>cpNUTZ>A$(;`0D^%Inc_L}ZQC&eaeyHEHsWeskjGT-AiWzT?_n51&qM6J z%pT96yNY7&SD%60@jXj{IV?=*?4p(8JeAx~<+1R}1m$X2BHG6kZ`!wQYAZ__cuty&G}M zE;Mem{e&s5tp%Fn8>(!({Ems*vD&S^yra&P-0js+);_^3-Le2}vTA#I`%MkLUdc6E zePyXTLvblxYsB-*BK16NYe{Iuo&0C!_#U3mE`8FEwxT_sK`{`gt0$cCm90HK#GBYq zRT(SVP^IRsMQy0MLsu+D)&((3Pwe*UZ0rAo_Lw2sYsY?ySvtrwZc|C9_rbRxsjXPCvXAEoIjW^#u;Aooj6WvK7tLAIuz>Idb;@4rUI_9GE%q=Ki!D7`!>~ z=D?dH*T>Ly@a7EiPwxKY z@Rrlxx3|1qvV4(Yd;2ukgKpj~Ucb$i`Qq79-$`Dc@e9i4EIRq_do#KH8uVU5y|pzU zrsoG1dkS(zFZ1+oo~=gG(FN<5YWUc%&^tD!Uq(1 zA&^xaMItMX+p|N&y2Wdqt2_{xrv>32YzRz{tK7y!^tIbq3}fSOFjAq>MooRb!=u(a zLp}x=_rVMBIeOkih6qoOQ^fEPk2qIDRV1Kn{S1Rr79U-ihMF7BcRGLpGuIzjAtm2^ zvp%GEQZT6Iq5gLWydZ7=y+2DV4P0f_KL_Du4CJ%+)D&WuVEqUO!9HFOsmoh#!C&f^ zP`aT^sVAE&P_e)b4sK3LzAj(HU~}KE!w8>-t9vaO%_X4rFOFmtm<$T zk%ZSqPX+TKsufz+l7V*Sur3e zEw0{yzqF~Xce}2%aaSNMl7f&iS2(923SzrqqX0^@8m1gfIhb-_<vCHK)_Kth6nkNBYjpJ^o%U{!63d(N5ZC+RKyd z3^}^=5Vhu|!?#)NWDI_BdGm8PVz6b+i)x~XfxLL0T7*WPT=lCsj2jX1t>&vsowfIW zhO4s{gb!bkp#sZ}l-B$?5K@vfcX9vPImg@^lUKl#~0; z(m7m2n`*REU^<=zzwb||Uk8{Ed9AzJvolxDJaO{ir)8(FKy5}D2 zDfH#==|r@A?9}eUmYjl!A;RPbA&0nM%U37ZOhHNsW0*tq+6nm#f!1uIvXD;`R`xOk zcEU(vM4jknjCv4*tg!^5c{GJF#uV=AvlFt2)KRUZKT+i=7Jnnm3Y<&G6NSzCR&5K$ zdB#q>$PBPXU|2@&ajUy{!E)9%Nk_cwM$tQrdQnN0F{Lp~$u{mSO@)_fV zUxzWz*1YnvI$I+VrxtgD@Qafu@Rc?AEG>$QIh>8UH&{;XQ1ZOb_xDf26lFQ-ihUhQ zFB=x5Hm=2pOVqk&CNdsHw`(zxo1PMLPX@yCu0LajYp8ARDZiYQjCxs+)+y$WtF;zZ zMAj+wC>PjvJ)joYw)bvYswAh@`+*6)Jh{rC#|-JkF+C`VVO)+zlacjeh7d@=8=>Zc{4>khKs z9>opOWN?bzM1)^{KVUk+GH>q(h>Q+V2lkCpgkSs-Cz`61g&FtV=my*){HkhGBWdeS z6E!ss9-S_3c(9Rl|AVq6D{y?IcUUdqo_tZF_R>Lo3$i7gH*N@QyuEFk_+l104K9uj zkNz#wy}opCM0!FlMm;Aw8X7hrGCD*Z7$Bf{4F$m&%oxd(OGmr}SHwNS z5wv$881hiHZcyxva)|jYD1vWUINyL1Dzv{#fenCz{IrPn@%^(=V)GezVU)*JBc z9E61a&JY8yANF|CvrtD89YnLIBM@fv1*C8iUMTei?Ss6#{@m_EU4 z$n+RQG=9r7HSaKRkL0PZQbqKHZ@OC9UU%?N7uY^caaiq9_i{2UW@q}dDAlX zQ+$>Cf^A=vAeidH zh&r%lg4eKe1s_?B*?SdhCcNR;3kZPT;yTpZV}NahkDb>0YMqN&=DTx#!Xxd%%Z?9G zXa+L}W)2mX^uf%5nFBKi-W+&yWTNRZ54<_>=D?eSpmjX$3_&Xdtq`noL+p+2vwb}Cgqg$gpPDpW;@7LX z$xP{uA_yh*N}uDbQ8_Jghs`a9SkUtEgl6O+!&Q}}EMv6g`Z;2Nh!Kyz3K63@voCpP zP|AmkkH{k#$p^&2qq8tlL9IMISXVmJFYPZP5=dOIC#1F-n~G6-$geSyh3w9E6M~lI zJdvPfN<%6|f%lO$t{HjAb3&GJuSKOEseITPSG8V@!Klcyt3=|JR{l2{Z)3p0*3C_b zDep!n?R+vNE?DY4ml-YDX>;f=;(|!OFIat$!aWq&*=<;V zDDXzJ#$_32|3UfiY&8!_Fv=DWJ7#`9j<>k1*Ik3S_&jH+Em=Y7_r>iPLQHwC{8%Sv ztl!7EYyEVg!|I@#y?@c~LrS2m$W<6@k=|XC(4o{br&%)ww>PQ4m?e^Be3)NPQQ!%` z8CO1(o)8yIYwI^g+?&PPCd!8~DCIeS0CDjpg-1SnK`DV=eZ2}X<<$$X)LBtDm1T;3 zNC~w0{_Z~$HzJHy@sPbGMy_!A0|$(x$0D+#CX^4gDn%j^7u?e+)JDeSoq)l=a8W0fZ0#=aD4C9^ElEUkG^yF>y89wrrEpTo)ndo@VM9 zA$KH^@#AMFy~8{+4QjniPwf-V5KO6R>ArsO3%>kxMxuY>XND(K-*JZN!syCXYI(#l z&gkrB(Y%?*>FB(zQX0BW>1>}E3gjKLvNP2{`uQ5}y?bIT%#NKl7E`@shh><8lD*^D zosH&xCPsD7kTRpcVW6#;$fa!L;~SqkrERh`ocowN$S^%T5^>^X*XFUBEyW<8_l-$i zorprD=^;N~PmI+$t>><$_Pcyhq5VBV%qatngPwyq2M+|864F-xkBo3}kpW#nSi>Ni zJ`ue1IGRwH;zLv_3j3He;b1HwS)Q@+mWU_MwFvR#aT&iqUbbU?%V`BO74L|!AEs^G zcudPUFU0G*@o^>P?_XB?LpYL#p?Cn+%%h7c0@}3#)y`%l{PYTjcBzKWVJEs z#~8$LEI|>W89bWwieDMStY??3Qq#u~stSo- zHxtZDLb-zXH*;T#QLELrKFG)ET+BF>?oH_cXOwv9C`$?L=X?mKsJUul%QqWIyA|M?C>;D{{Az7l&;Vm5pW4J5OM}u@12g3Ozff4uVTWn{*hiyRVjK=apx0T zOu6=^dmI+bZKGqRv+Mau7gR~ia$a(@5Tjdz<=u6W47S{hy&-9sz0dmNfjFI(Kw7{@ zO$?yTyfsqnDBJ$2KDCij?Y`}0y_JU0t$8o;_CMaJm(`}yP;-A8Z2}fN<}ZQ;vj_KM zSn!9BemYxAMU-n3m}OgwN>WoC@xiU1pj?}}0!d>pigZdVpa9jH1NRVAIG<^JL=gk3 z_r4jk*Om4?+W8kiwfgZad)s!*Ko}s%zKyur7UVJ2F*mSb2t5z6_c41sgYGJdx!-&S za>p^tfjKNp=;)%A;yjfERC_GEGC{c(mWXi7@n#vj?eJG9Cj{0x%4RoPUU)bIqmk2G zh1$weM!ADaR$xhSy+bOB)d;Skt<{gA*m-p8-G~iuW^-jM?(!rPTC)hVoSk2baV|Nd zHnNC=PA@qB4}j`~>I_owJR&Quim4%TH5HP(ADbiHZqd_*j30hb)-t-H2OmA(B-8gefkq1)Af4 zYTGVKjlI6Sqt28>dpDG|PcTciEI=UZwwJfx)ZputT(i|zmbx<(m(sNcQ!u8Ur)@0> zt+pt>?v1gKVX*P;N`cj$`6$hyed zX-%NLJKOp{A&fbqy>#rSm?eWek+i*Y zjqP8c=%)T)=0+);eD>>dwCP}g-uK!DSR zQPe)FRMr!1RS|*i#3rvG!&NKMJ|D#Ht1O@hbk!Oi(T*;a^Y_8bk;C;ncyr**fj5_> z7=(Gk_Y{h^ize_XdhS8*z@w+o#=@SXx}y|NIG8yIS|MnKpcSp68b&D)wEk`85WG3a z%|UKX6c|;BfN7AMgWQ~kXg(+;LeL5&2q-~72?Ccs-J1Od4@@kTGm%|UJsa&wTI zyNSyvA-hzi)4OyJYzy!}j)Rt_R(`UA%sqEAz#( zrM{EAJmVLX%~^Ev-S=j4`!(pjgnDahKupgMEcO)SieBdF-#lB5q$f+RCE;({pH+?2 zGk0Zs?51k)u8D0(1JTJ|Sl^716&)d#S|sd4AtHPTiz+6sEBpnO;CY@wM+B)DlM>IU zK=t+zqPHg#D-HVi1q8=OOUC9lSF$Nk3g<yQZnKYVX(RH*I}}{hO2umN|8okf;VpGBIW=fX!0 z1yM+&p;h??sBmVgG^;sFtnKMLORye6QYwl&-wdSVNwQsNh>#Tng3{vZ9r#O|+IqL^ zN*i|t(jqAc8FPhm3gUya8#W4{L@Qy+DQ#mhjE%p+BiLx8ras@{QR|%{9|H_E;sy8| zJ#QjIgr~p?6c$#9SRqLQ5i3r)Kwb`#B#@Vb@PYIV z5bO|A!BBmGR4`N@kTF>826;IsctMR73SLlSMdq*yst-`s2PDbgmi39wrn{TX2<1}y zOgX1%YbPtt|5^E7xcjlN%Tx3|On-Mu$M@;G(`*BoV{1;Qaan0wK9BUBnS1=bT>O_t z$D^IJ&9s*%*%@+l=^<*(ONVc>*vS});PU3@aKvE8nithX5d(SgJhcdoJh|#uaX5(~ zGmpW_j{}A)yq%?NE=IQKHt6%ZvJIsOe#8v-kY3QcmPv7c`K`3sgyQ|7}>r2ZP zajVMHHPBN|(!r;5xQI5-KrV~3hlYVJ&2nmf_jFox1=&I8pmE4Igr0*r2M+|8l7EH~ ze$rl9HVG%~zlSgS^)KV+%>SEn@VO7_?B?S0;T3uSdH{L= zdH{L=df=bk10sstEyai;cZB(NuJY`S3fA$9%1(87m;V^)8EABF&84GDROw|JjUG?q z*U9My>&bZcYO~wxALoP_ILbaSJE6Kc;n)}Lr^`Lfbk9B3Q|Qa#(}`&L*s0xxEja}d zLxjl>LJo1imak5*nSzuO#xRHIwG;9g0)!7<}I5oLU z4N<=%!cR`3z*pAbv$QBK=5RLZ-e5VkL&@_#-`_t8Q? zsCCbfnsU+YT1@1sr^MWof%fyPKVyb>4YawZ{Blw<>SaM%r>E*S*O&aTwvSv zfC{kN-n(h3lAN0F2POzVc`xWj(n%3-pA5v}5Ps-N3feZ3T_0j>T!C?mjWshyd}Z4Y z$L(^zj5q$&8*3g=Kfc0#_a}S{%F&gnbxOa$@z&*mRsx~!}w(c}hQ{&*# z>Eeb58%g&+C|j}u=STGpt0ml%FG|#2I*4yUwuJM>4FOcQw`~(&%p#}3#nIu>zh!zH zT8SHyP{^~_mky3dPsqi)o)R4m4I2;{Eh3u>5Kt6Rg5V5hjAY8CBVK|l;vV4$+B?uT zsyj#313T)sNkRylnGxU-9uRNVoD_^`&nY-|!aG{;x%8 zljZkx_I+NTb)VYiQO+9Co+DzyYuYVNrI_DKO6o*pw1_%5Pe?K0(F^JJ$>l&GkIPj;x*3tN=zplw``$vP=|;*IDLZGnCUTy zXndDvYTjYs9?4T*rHbf@-*mOIz3$+lF0g%?;;`By@wWF&Jkr%4ycY8{@}_0%rx*_R z1>3$TLEE%;$>CWTpPRg1UwYXZ{(}eNskGl`WTiT?H_pR>ISt~?J9jsojmyaQ6;TJ* zOz;|4uHYl9F?+9K%|tdV_5#`s)Z#kS+hc%jgpZxp`)ZwwS?0TQe!?T|!pn{iQD_D; z2WAd|OZ36aftdp{2i_cbbEE@$nFrn+cyr**LC{KC1BGM=S|MnKpcR5v$jyE zHwU>n$j#yOB9tJY1W{a54kZYjTZa+^lpvte3YAu|(k@h5q0$PK)_)D4wM1`iposw8EZqb=9Z5d%bwc=T0>7}c45$vcBm zK3;r89tjUVAPyd#g^>zs<>A4)(wTl~e^EY0;=(;4wbj^EjM77Xjgc&5cfOmDKrH8p zBoI>?QYogskF0Uc=tG_pvdnueD)mU^Vp*Sp}@}Wy2B$6|6GF!&FH>hk!Ezqpew~k zI+`^u%RKuJ%ExD`c}Rj;ws_bv^Yd}M#bv$j8blfBIZJKH3QE5>ZpRR!Ep_F`Iyqzg zUd~nlZS&Nd?9%ku3AW{CbK4SNP4i z^0D-UxNurqzcErdF4i_tK8!&r&-nvLf;lNX^4SYYBKPX+Rfs9KUU;R>io&TZQ|v<$ zxy|=?|5+jzVYG^e>@6{Jh07l}U?e>jkrg$ee5_R|Qa--mo=&kY!<36c)WI<2V9JTg zE{R~}_K$* zYOL3B!5YE`DC@JKtp6uHW(XhtFNY7q9^ElEUkG^yF>y89wrrEpTo)ndo@VM9A$KH^ z@#AMFy~8{+4QjniPwf-V5KO6R>ArsO3%>kxMxuY>XND(K-*JZN!syCXYI(#l&gkrB z(#w{Z$LZ+2tx_7gPU&o)*s^-Ztn5rRkbb^~d+(kY3$tUVjm11=Y3;RS0|zn zX?n=d*AruPPV2d=sr@crRA_&X5Oc~vjDTnjSSlkWI(s@{noPo=No=$c%J z9GXt&GW?0o{R4$UEBhYHD59sZhA1l>D<$5ue1%O6xA#QbLC>C8VKmsCiwZsLFzC3en1y{_;=_Me??{uzWk>Q9EAAshFTDJAZ+s zDqECyA#V&*mHkSy=_1*{vfXXRD3s+&x&dMX<2CU+1v@aN(p;d8sLH|)_a#q+@JF_6 zAbHfj|56R3U1b-Z6!EBK@1Us43b=L$93Jgyzdu3pmd0vZgm~17I6lr3RAuIsBvsj$ zwwAU+3Kh!K93t7ktXF8!G0HSgm8OE&z(%?RgY=u24eUDo9HJ_FtUpp=j$1-@9m%7% zYnBLw-gbM(pOjQ*^KvpMsnN(S-PIzhvhbR?1d6JRmr^bw zJ*;^Vg{_T|N2xqXdRP!tm~V$WAm^s~hiaXDznK^K?8Gg}aPL?|@~A}>UBNr7*jKlU zc+~d%jQjGLpCA>H2u{yGN|C%>AIMcfBfVibxXX=VWK8WK85yICxMk%`4biMtNFJ$f z6OkTrKRlt>z_MB+5$T};E1%V@g*%|}wLA|I#MLff{Q>Qp)8YE+bBlHu?}yN^sx8{ z?;S;Y*w9;yNDq^`n?1S{a0j@5j6{r#N^Wu8jo;yx-2TBxL>GCv2S1w51}@)`bde=z zF5|xJvmutukt$e|ZbNZdx(GBkpppLEgt>Bpk85y00zBwf6A^TxK zFGYITIm||+hZTO^SwmZJ2do$t@`S z79)2UXHH7mU8=c5aam^46Rx0M8O?kr< z>EY2aPegji$sKCoU&9@cIBJg=8KoQEuUzp)jKkOCo{b^tBEigonUjz>0cOq$%p90G z@aDjqLoSns5qNXp&4D)uL8~Z!s)C>uf>sDxA!vo%9I}V|;~+N&xjD$qL2eFbjiCeq zB?u@%KnVg<{6Gl;Dy>jyg-R=#Jh=;%R;aZ8>i{K0Z2rpin+#+h8Ek!bG3wN0&2^hU zJACKIFnhC;7gkR#y*?{JTJ_k)M|sBtPklkD5jMN-HADny_DalXx)^WKzRdOxBBk3V*SXh$0tlq_$D5ag zrV6g|vvK9W`sOT>M~yY`b^=*Z`PNG(L_s*iF0vA19BgTtl?Z>cFpsI)F)r9yWf+oO zXkHrH?dj)pS5d6wx09tAiEtV-c&0vB@b7xo`Q37-k@8^4!Yg)|x-Tk$i%99>&2rz? zQc$C%s}95=`8-;Oi6(a7eQ%M}ef@srWVg@;a;B-bng+QSi$~EFIdBcKr-fUUth=e#TAI4J1iey)kx=&!^F#&@baZ^>Sp2{xlIWl}31Cj<>`;1kk z=tcaxUvPV?Bl!)ey>fml6oatZyC=?N0&4F zb1LRJtTxO%bn+y=T-K|d`pQyw2IIO~F-FTGWiLtH$Ef*X0%>IOL+O)#(_yV(RTHcd z8X(9vMnDZ_5B=y>+}?h+P+wK5pt$M96I*eQz?6e22U8BLoY?S#m4k&97FJkTNp}=M zQoxr3Uk<_t2p>pmu$BYi14OJ4u|krBvf_u^=dTLGiW;s~Nvr-Y*UM|= z3-nA1OT&ijdaUh=S8kQh#h4c9R#(G0QfKXbk7BG!%CKk6(-MT}39_L$Wp1WBZ4$8u z(S5$gboP48q^_$DtG|c{Pq?ok8ZL$8^yXcfs#2RT+2mSNh@Ka;Ze-G)40Z}#u48z~ zhp~N#R7Gt?GBf`t3}6|0TZTArhF3u2uxAWC2XhV{2rwm+MAr50|G)?UNJTl5GBr0=gyzs)Y+@UPSnB>AGn_WqIJe@;D_QNdhf~)(^u3r0e|HgT8a@VZF zWq(#aU3IB-?qbJyIZcNru?K~ z(l!*N7?Q2owWr6buvm~ji&afrIJC!y7$&$v{?<#pZoTdthVAge`&$cz9z+LYB%SDF znhTq*4$z#4K6N2aV~oq-5yEuFuq9DR2<3@)4B-aOpE&XaEJh+h%w-BT{B-#Hq`~n{ zQb=T9wcc79dPK>t4O1ss=PmC>oMvCm_Pa(2I6reT+*}%Jbx!`{N{qcs>vmxypyhgY z=hy(n{uNVX<1P{Q066)mu5bv*uOYtCW&-~ z=7=XQln}A!4o$Qc8bzmj$C+Zo>ll4+pe$9N`_|G4vwv}T zbD8Eeq>}tDGe!JSXM@FxsGC$v_84N?Fh#8wd|%n`!%J*jD0S(3RxyZ{Z`OyfQ40(V zKoDQ@CntNoaYG|4u$;_2SXruG(A@`cc@LlYx=PJl_!SBvSjD zyHJlhZ{$?vb@k&}_FnZ^m|HI?J_>b{Tx)6)_7|cOK7<%a-(J&tCzApm*Dc#H4m=K_ zfX4%?zEMG%;fwN;3x$~EaO5LPZGpq11O9n81(^9RfW}pu@38s^(c5T69eZeEXi*Y6 zcxNE)SA1+F_^ER>lKuP5pc7-Dl`Wny+3G*R!X>qq+1%PfNMX%TVWI;v7{?Ss$Yf%fTrm50E{yAC#

(;I zB(^P5m2$E-xzR1|q}t4alyDNV9iGLjp-2wx1P#RqpDi1^;=Uc8!rjhISCHI#7wy7I z3bSolv#?a88cx@xNDi6uBW|On{&QxfP?F}wn{p5so2y;$3=edUze_WsZd|gwYcmDe zKJ&smUIf{GO+mKzvZ{+rbrH#7!w8p2xyHgydtGVkh| z;iYinS{-E%cJHPAN^zf8Mh5E(kd#sDxr8E7~AU8+42NIVdHwU>n$jw1+PGk*M`$7o< zN)S+jfD#0hAS5Id5~0!xl~$;r#}%hlbEN3SrDd0JS-v$gmy|)AmX(k~ z0gId&JUXdY(_O=z5X{rL)AmZ|qI$%c$6~?h6BYe!`;q$KE%x;{wis{_K>OH+2<(D~ z0*K``VglRh!(YzmQ0mDxKg)N(Es3+piC-n-QS_znG~5!sj`%HgNEP(C+y#6Kc_+^g zpdtKBePJZN(-`%xJ4h^YVWaJZE{s4)61?D(YMcdj=~o>gS;H%qDBp{fYqA3 literal 0 HcmV?d00001 diff --git a/crates/clawhdf5/tests/huge_chunks_interop.rs b/crates/clawhdf5/tests/huge_chunks_interop.rs new file mode 100644 index 0000000..bf2ff19 --- /dev/null +++ b/crates/clawhdf5/tests/huge_chunks_interop.rs @@ -0,0 +1,296 @@ +//! Chunks of 4 GiB or more, which HDF5 2.0 writes (layout message version 5, +//! `H5F_LIBVER_V200`), in every chunk index libhdf5 uses for them. +//! +//! `fixtures/huge_chunks_filtered.h5` (written by libhdf5 2.0.0 through h5py +//! 3.16, `fixtures/gen_huge_chunks.py filtered`) holds one dataset per +//! filtered index — Single Chunk, Fixed Array, Extensible Array, v2 B-tree — +//! whose chunks are 2^29 + 1 `f64` (4 GiB + 8 bytes), stored deflated twice +//! so each takes about 20 KiB. Listing their chunks is cheap and always runs; +//! decoding one inflates 4 GiB, so those tests are opt-in: +//! `CLAWHDF5_HUGE_CHUNKS=1` (run them one at a time, `--test-threads=1`: +//! each needs about 4.5 GiB of memory). The same variable enables the +//! unfiltered tests, which have h5py write a sparse file (about 44 GiB long, +//! a few blocks on disk) under `tests/scratch/`, and the writer tests. +//! +//! libhdf5 never writes such a chunk with a version-1 B-tree (a chunk of more +//! than 0xFFFFFFFF bytes forces layout version 5 and with it the newer +//! indexes) and refuses to open one; `chunked_read`'s unit tests cover that. + +use std::path::{Path, PathBuf}; +use std::process::Command; + +use clawhdf5::File; +use clawhdf5_format::chunked_read::list_chunks; +use clawhdf5_format::data_layout::DataLayout; +use clawhdf5_format::dataspace::Dataspace; +use clawhdf5_format::message_type::MessageType; +use clawhdf5_format::object_header::ObjectHeader; +use clawhdf5_format::selection::Selection; +use clawhdf5_format::superblock::Superblock; + +/// Elements per chunk along the chunked axis: 4 GiB + 8 bytes of `f64`. +const N: u64 = (1 << 29) + 1; + +fn fixture() -> PathBuf { + Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/fixtures/huge_chunks_filtered.h5") +} + +fn heavy() -> bool { + if std::env::var("CLAWHDF5_HUGE_CHUNKS").is_ok_and(|v| v == "1") { + return true; + } + eprintln!("SKIP: set CLAWHDF5_HUGE_CHUNKS=1 to decode 4 GiB chunks"); + false +} + +fn python() -> String { + std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string()) +} + +fn python_available() -> bool { + Command::new(python()) + .args(["-c", "import h5py"]) + .output() + .is_ok_and(|o| o.status.success()) +} + +fn interop_required() -> bool { + std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1") +} + +/// The raw layout message version, the parsed layout, the dataspace and the +/// chunks of `name` in the in-memory file `data`. +fn layout_of( + data: &[u8], + name: &str, +) -> ( + u8, + DataLayout, + Dataspace, + Vec, +) { + let sb = Superblock::parse(data, 0).unwrap(); + let addr = clawhdf5_format::group_v2::resolve_path_any(data, &sb, name).unwrap(); + let hdr = ObjectHeader::parse(data, addr as usize, sb.offset_size, sb.length_size).unwrap(); + let msg = |t| { + hdr.messages + .iter() + .find(|m| m.msg_type == t) + .unwrap_or_else(|| panic!("{name}: no {t:?} message")) + }; + let lm = msg(MessageType::DataLayout); + let layout = DataLayout::parse(&lm.data, sb.offset_size, sb.length_size).unwrap(); + let space = Dataspace::parse(&msg(MessageType::Dataspace).data, sb.length_size).unwrap(); + let (chunks, _) = + list_chunks(data, &layout, &space, 8, sb.offset_size, sb.length_size).unwrap(); + (lm.data[0], layout, space, chunks) +} + +/// The fixture's datasets: name, chunk index type, shape, and the scaled +/// origins of the chunks libhdf5 wrote. +const FILTERED: &[(&str, u8, &[u64], &[&[u64]])] = &[ + ("single", 1, &[N], &[&[0]]), + ("farray", 3, &[N + 10], &[&[0], &[N]]), + ("earray", 4, &[N + 10], &[&[0], &[N]]), + ( + "btree2", + 5, + &[2, N + 10], + &[&[0, 0], &[0, N], &[1, 0], &[1, N]], + ), +]; + +/// Every index of the fixture: layout version 5, its chunks listed at the +/// right origins with their stored (deflated) sizes. The index elements +/// store a size in 8 bytes (libhdf5's `H5F_SIZEOF_SIZE` under layout +/// version 5), where version 4 would use 6 for a chunk this size. +#[test] +fn filtered_huge_chunk_indexes_list() { + let data = std::fs::read(fixture()).unwrap(); + for &(name, index, shape, origins) in FILTERED { + let (version, layout, space, mut chunks) = layout_of(&data, name); + assert_eq!(version, 5, "{name}: layout message version"); + let DataLayout::Chunked { + chunk_dimensions, + chunk_index_type, + .. + } = &layout + else { + panic!("{name}: not chunked: {layout:?}"); + }; + assert_eq!(*chunk_index_type, Some(index), "{name}"); + assert_eq!(chunk_dimensions.last(), Some(&8), "{name}"); + assert_eq!(space.dimensions, shape, "{name}"); + chunks.sort_by(|a, b| a.offsets.cmp(&b.offsets)); + let got: Vec<&[u64]> = chunks.iter().map(|c| &c.offsets[..shape.len()]).collect(); + assert_eq!(got, origins, "{name}"); + for c in &chunks { + assert!( + (10_000..40_000).contains(&c.chunk_size), + "{name}: stored size {} of chunk {:?}", + c.chunk_size, + c.offsets + ); + assert_eq!(c.filter_mask, 0); + } + } +} + +/// `f64` values of `sel` in dataset `name` of `file`. +fn sel(file: &File, name: &str, start: &[u64], count: &[u64]) -> Vec { + let ds = file.dataset(name).unwrap(); + let s = Selection::Hyperslab { + start: start.to_vec(), + stride: vec![1; start.len()], + count: count.to_vec(), + block: vec![1; start.len()], + }; + ds.read_f64_selection(&s).unwrap() +} + +fn f(range: std::ops::Range) -> Vec { + range.map(f64::from).collect() +} + +/// Reads that touch a few elements of each 4 GiB chunk: written values, the +/// fill value (-1) next to them, and the edge of the dataset. +#[test] +fn filtered_huge_chunks_read() { + if !heavy() { + return; + } + let file = File::open(fixture()).unwrap(); + let mut first = f(0..10); + first.extend([-1.0; 2]); + for name in ["single", "farray", "earray"] { + assert_eq!(sel(&file, name, &[0], &[12]), first, "{name}"); + } + assert_eq!(sel(&file, "single", &[N - 2], &[2]), [-1.0; 2]); + let mut edge = vec![-1.0; 2]; + edge.extend(f(100..110)); + for name in ["farray", "earray"] { + assert_eq!(sel(&file, name, &[N - 2], &[12]), edge, "{name}"); + } + assert_eq!(sel(&file, "btree2", &[0, 0], &[1, 10]), f(0..10)); + assert_eq!(sel(&file, "btree2", &[1, N], &[1, 10]), f(300..310)); + assert_eq!( + sel(&file, "btree2", &[0, N - 1], &[2, 2]), + [-1.0, 100.0, -1.0, 300.0] + ); +} + +/// A directory under `tests/scratch/` (on disk: the sparse files must not +/// land on a tmpfs `/tmp`), removed when dropped. +fn scratch() -> tempfile::TempDir { + let root = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/scratch"); + std::fs::create_dir_all(&root).unwrap(); + tempfile::tempdir_in(root).unwrap() +} + +/// Have h5py (libhdf5 2.x) write `fixtures/gen_huge_chunks.py`'s file for +/// `mode` into `dir`; `None` when there is no h5py (a failure under +/// `CLAWHDF5_REQUIRE_INTEROP=1`). +fn generate(dir: &Path, mode: &str) -> Option { + if !python_available() { + assert!( + !interop_required(), + "CLAWHDF5_REQUIRE_INTEROP=1 but python3 with h5py is not available" + ); + eprintln!("SKIP: python3 with h5py not available"); + return None; + } + let script = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/fixtures/gen_huge_chunks.py"); + let path = dir.join(format!("{mode}.h5")); + let out = Command::new(python()) + .arg(&script) + .arg(mode) + .arg(&path) + .output() + .unwrap(); + assert!( + out.status.success(), + "gen_huge_chunks.py {mode} failed:\n{}", + String::from_utf8_lossy(&out.stderr) + ); + Some(path) +} + +/// Positioned reads of a file (no mmap), counting the bytes read. +struct Counting { + file: std::fs::File, + len: u64, + read: std::sync::atomic::AtomicU64, +} + +impl clawhdf5_format::storage::Storage for Counting { + fn read_at( + &self, + offset: u64, + len: usize, + ) -> Result, clawhdf5_format::error::FormatError> { + use std::os::unix::fs::FileExt; + let len = len.min(usize::try_from(self.len.saturating_sub(offset)).unwrap_or(usize::MAX)); + let mut buf = vec![0; len]; + self.file.read_exact_at(&mut buf, offset).unwrap(); + self.read + .fetch_add(len as u64, std::sync::atomic::Ordering::Relaxed); + Ok(buf.into()) + } + + fn len(&self) -> u64 { + self.len + } +} + +fn check_unfiltered(file: &File) { + let mut first = f(0..10); + first.extend([0.0; 2]); + for name in ["single", "implicit", "farray", "earray"] { + assert_eq!(sel(file, name, &[0], &[12]), first, "{name}"); + } + assert_eq!(sel(file, "single", &[N - 2], &[2]), [0.0; 2]); + let mut edge = vec![0.0; 2]; + edge.extend(f(100..110)); + for name in ["implicit", "farray", "earray"] { + assert_eq!(sel(file, name, &[N - 2], &[12]), edge, "{name}"); + } + let mut last = vec![0.0; 2]; + last.extend(f(500..510)); + assert_eq!(sel(file, "implicit", &[2 * N - 12], &[12]), last); + assert_eq!(sel(file, "btree2", &[0, 0], &[1, 10]), f(0..10)); + assert_eq!(sel(file, "btree2", &[1, N], &[1, 10]), f(300..310)); + assert_eq!( + sel(file, "btree2", &[0, N - 1], &[2, 2]), + [0.0, 100.0, 0.0, 300.0] + ); +} + +/// Unfiltered chunks of 4 GiB + 8 bytes in every index libhdf5 gives them +/// (Single Chunk, Implicit, Fixed Array, Extensible Array, v2 B-tree), in a +/// sparse file h5py writes: read through a memory map, and through +/// positioned reads, where a selection reads only the rows it needs. +#[test] +fn unfiltered_huge_chunks_read() { + if !heavy() { + return; + } + let dir = scratch(); + let Some(path) = generate(dir.path(), "unfiltered") else { + return; + }; + check_unfiltered(&File::open(&path).unwrap()); + + let file = std::fs::File::open(&path).unwrap(); + let len = file.metadata().unwrap().len(); + assert!(len > 40 << 30, "{len}"); + let storage = std::sync::Arc::new(Counting { + file, + len, + read: 0.into(), + }); + let positioned = File::open_storage(storage.clone()).unwrap(); + check_unfiltered(&positioned); + // Every read above together: metadata and a few rows, not 4 GiB chunks. + let read = storage.read.load(std::sync::atomic::Ordering::Relaxed); + assert!(read < 1 << 20, "{read} bytes read"); +} diff --git a/crates/clawhdf5/tests/parallel_integration.rs b/crates/clawhdf5/tests/parallel_integration.rs index bdcf4f0..964d3c8 100644 --- a/crates/clawhdf5/tests/parallel_integration.rs +++ b/crates/clawhdf5/tests/parallel_integration.rs @@ -283,7 +283,7 @@ mod parallel_tests { for (i, chunk) in compressed_chunks.iter().enumerate() { file_data[offset..offset + chunk.len()].copy_from_slice(chunk); chunk_infos.push(ChunkInfo { - chunk_size: chunk.len() as u32, + chunk_size: chunk.len() as u64, filter_mask: 0, offsets: vec![(i * chunk_elems) as u64, 0], address: offset as u64,