From 7a2e61ebd192a3dbec123752bb9d977ede4897ec Mon Sep 17 00:00:00 2001 From: osobh Date: Mon, 28 Sep 2026 23:37:34 -0500 Subject: [PATCH] Huge chunks: selection tests with fill values, LZ4 over 256 MiB, wasm32 check A selection of a chunked dataset with a fill value is compared with the full read in every index, through a map and through positioned reads. An LZ4 chunk larger than 256 MiB is bounded by the chunk size, not refused. The wasm package test reads the 4 GiB-chunk fixture and gets a clean error in every index. Co-Authored-By: Claude Opus 5.5 (1M context) --- crates/clawhdf5-format/src/chunked_read.rs | 4 +- crates/clawhdf5-format/src/chunked_write.rs | 4 +- .../clawhdf5-format/src/extensible_array.rs | 2 +- crates/clawhdf5-format/src/filters.rs | 33 +++++++ crates/clawhdf5-format/src/parallel_read.rs | 2 +- crates/clawhdf5-format/src/partial_read.rs | 5 +- .../clawhdf5-format/tests/raw_fetch_bounds.rs | 2 +- crates/clawhdf5-tools/src/check.rs | 4 +- crates/clawhdf5-tools/src/info.rs | 2 +- crates/clawhdf5/src/edit/mod.rs | 27 +++--- crates/clawhdf5/tests/huge_chunks_interop.rs | 4 +- .../tests/partial_read_equivalence.rs | 87 +++++++++++++++++++ examples/wasm-viewer/test/test.mjs | 14 +++ 13 files changed, 165 insertions(+), 25 deletions(-) diff --git a/crates/clawhdf5-format/src/chunked_read.rs b/crates/clawhdf5-format/src/chunked_read.rs index fa3f016..5f0528a 100644 --- a/crates/clawhdf5-format/src/chunked_read.rs +++ b/crates/clawhdf5-format/src/chunked_read.rs @@ -1491,7 +1491,7 @@ pub fn list_chunks_for_read_in( let chunk_bytes = checked_chunk_byte_len(&chunk_dims, elem_size)?; if let Some(c) = chunks .iter() - .find(|c| c.address != u64::MAX && c.chunk_size as usize != chunk_bytes) + .find(|c| c.address != u64::MAX && c.chunk_size != chunk_bytes as u64) { return Err(FormatError::ChunkedReadError(format!( "incorrect chunk size returned from index for unfiltered chunk at {:?}: \ @@ -2470,7 +2470,7 @@ mod tests { // Entries: key[i], child[i] pairs, then final key for chunk in chunks { // Key: chunk_size(4) + filter_mask(4) + ndims offsets - buf.extend_from_slice(&chunk.chunk_size.to_le_bytes()); + buf.extend_from_slice(&(chunk.chunk_size as u32).to_le_bytes()); buf.extend_from_slice(&chunk.filter_mask.to_le_bytes()); for d in 0..ndims { let off = if d < chunk.offsets.len() { diff --git a/crates/clawhdf5-format/src/chunked_write.rs b/crates/clawhdf5-format/src/chunked_write.rs index 14c3e22..89b0f7f 100644 --- a/crates/clawhdf5-format/src/chunked_write.rs +++ b/crates/clawhdf5-format/src/chunked_write.rs @@ -2263,7 +2263,7 @@ mod tests { for i in 0..n { let (offsets, chunk) = extract_chunk(&data, &shape, &chunks, 2, i); assert_eq!(chunk.len(), 2 * 3 * 2 * 2); - for (e, pair) in chunk.chunks_exact(2).enumerate() { + for (e, pair) in chunk.as_chunks::<2>().0.iter().enumerate() { let c = [e / 6, (e / 2) % 3, e % 2]; let g: Vec = (0..3).map(|d| offsets[d] + c[d] as u64).collect(); let expect = if (0..3).all(|d| g[d] < shape[d]) { @@ -2272,7 +2272,7 @@ mod tests { } else { [0, 0] }; - assert_eq!(pair, expect, "chunk {i} element {e}"); + assert_eq!(*pair, expect, "chunk {i} element {e}"); } } // Data shorter than the shape: the missing elements stay zero. diff --git a/crates/clawhdf5-format/src/extensible_array.rs b/crates/clawhdf5-format/src/extensible_array.rs index a17367b..889f2cf 100644 --- a/crates/clawhdf5-format/src/extensible_array.rs +++ b/crates/clawhdf5-format/src/extensible_array.rs @@ -946,7 +946,7 @@ mod tests { assert_eq!(chunks.len(), 2); assert_eq!(chunks[0].address, base_addr); assert_eq!(chunks[0].offsets, vec![0]); - assert_eq!(chunks[0].chunk_size, chunk_byte_size as u64); + assert_eq!(chunks[0].chunk_size, chunk_byte_size); assert_eq!(chunks[1].address, base_addr + chunk_byte_size); assert_eq!(chunks[1].offsets, vec![20]); } diff --git a/crates/clawhdf5-format/src/filters.rs b/crates/clawhdf5-format/src/filters.rs index ded4c0a..d902041 100644 --- a/crates/clawhdf5-format/src/filters.rs +++ b/crates/clawhdf5-format/src/filters.rs @@ -2764,6 +2764,39 @@ mod tests { assert_eq!(decompressed, data); } + /// A chunk's size bounds an LZ4 chunk, not the 256 MiB ceiling for an + /// unknown size: a 300 MiB chunk was refused ("declared size exceeds + /// limit"). + #[test] + #[cfg(feature = "lz4")] + fn lz4_chunks_over_256_mib_decode() { + let mut data = vec![0u8; 300 << 20]; + data[12345] = 7; + let compressed = lz4_compress(&data, &[]).unwrap(); + let decompressed = lz4_decompress(&compressed, data.len()).unwrap(); + assert!(decompressed == data); + // Without a chunk size the ceiling still applies. + assert!(lz4_decompress(&compressed, 0).is_err()); + } + + /// A chunk of 4 GiB or more is always in the registered framing, whose + /// big-endian size then does not start with four zero bytes: its size + /// is read whole (here larger than the chunk, so refused before any + /// allocation), not taken for a legacy 4-byte size. + #[test] + #[cfg(all(feature = "lz4", target_pointer_width = "64"))] + fn lz4_chunks_of_4_gib_use_the_registered_framing() { + let chunk = (1usize << 32) + 8; + let mut data = ((chunk + 8) as u64).to_be_bytes().to_vec(); + data.extend_from_slice(&(1u32 << 30).to_be_bytes()); + data.extend_from_slice(&[0; 8]); + let err = lz4_decompress(&data, chunk).unwrap_err(); + assert!( + matches!(&err, FormatError::DecompressionError(m) if m.contains("exceeds chunk size")), + "{err:?}" + ); + } + #[test] #[cfg(feature = "lz4")] fn pipeline_lz4_only() { diff --git a/crates/clawhdf5-format/src/parallel_read.rs b/crates/clawhdf5-format/src/parallel_read.rs index 0594e42..38288b4 100644 --- a/crates/clawhdf5-format/src/parallel_read.rs +++ b/crates/clawhdf5-format/src/parallel_read.rs @@ -255,7 +255,7 @@ pub fn decompress_chunks_lane_partitioned_in( for &local in &indices { let index = batch.start + local; let chunk_info = &chunks[index]; - let size = chunk_info.chunk_size as usize; + let size = crate::addr::saturating_usize(chunk_info.chunk_size); let raw_chunk = raw_bytes.get(index, &reqs[index])?; let decompressed = decompress_chunk_exact( diff --git a/crates/clawhdf5-format/src/partial_read.rs b/crates/clawhdf5-format/src/partial_read.rs index 269c45b..f9896de 100644 --- a/crates/clawhdf5-format/src/partial_read.rs +++ b/crates/clawhdf5-format/src/partial_read.rs @@ -368,7 +368,10 @@ pub fn read_selection_filled_in( fill: Option<&[u8]>, ) -> Result>, FormatError> { let dims = &dataspace.dimensions; - if dims.is_empty() || elem_size == 0 { + // A fill value that is not one element's bytes is the full path's to + // interpret. + let odd_fill = fill.is_some_and(|f| !f.is_empty() && f.len() != elem_size); + if dims.is_empty() || elem_size == 0 || odd_fill { return Ok(None); } let total = dataspace.checked_num_elements()?; diff --git a/crates/clawhdf5-format/tests/raw_fetch_bounds.rs b/crates/clawhdf5-format/tests/raw_fetch_bounds.rs index a65d611..c8b2471 100644 --- a/crates/clawhdf5-format/tests/raw_fetch_bounds.rs +++ b/crates/clawhdf5-format/tests/raw_fetch_bounds.rs @@ -151,7 +151,7 @@ fn crafted() -> (Vec, Chunked, Vec) { // v1 B-tree key (size, filter mask, offsets + 0) then the child // address. let mut pat = Vec::new(); - pat.extend_from_slice(&c.chunk_size.to_le_bytes()); + pat.extend_from_slice(&(c.chunk_size as u32).to_le_bytes()); pat.extend_from_slice(&c.filter_mask.to_le_bytes()); // The key holds one offset per dimension plus the element offset // (0); `offsets` may or may not list that last one. diff --git a/crates/clawhdf5-tools/src/check.rs b/crates/clawhdf5-tools/src/check.rs index 4a36554..1ba8239 100644 --- a/crates/clawhdf5-tools/src/check.rs +++ b/crates/clawhdf5-tools/src/check.rs @@ -916,7 +916,7 @@ impl Checker<'_> { bad.push("has size 0".into()); } else if !filtered && let Some(cb) = chunk_bytes - && u64::from(c.chunk_size) != cb + && c.chunk_size != cb { bad.push(format!( "is {} bytes; an unfiltered chunk is {cb}", @@ -933,7 +933,7 @@ impl Checker<'_> { ); } } - self.extent(c.address, u64::from(c.chunk_size), path); + self.extent(c.address, c.chunk_size, path); } if reported > 50 { self.problem( diff --git a/crates/clawhdf5-tools/src/info.rs b/crates/clawhdf5-tools/src/info.rs index d169f73..b0239f1 100644 --- a/crates/clawhdf5-tools/src/info.rs +++ b/crates/clawhdf5-tools/src/info.rs @@ -224,7 +224,7 @@ pub fn allocated_bytes(h5: &H5, info: &DsInfo) -> Result { let ds = info.ds.as_ref().map_err(Clone::clone)?; chunks(h5, layout, ds, dt)? .iter() - .map(|c| u64::from(c.chunk_size)) + .map(|c| c.chunk_size) .sum() } DataLayout::Virtual { .. } => 0, diff --git a/crates/clawhdf5/src/edit/mod.rs b/crates/clawhdf5/src/edit/mod.rs index 4d4f635..169c6d3 100644 --- a/crates/clawhdf5/src/edit/mod.rs +++ b/crates/clawhdf5/src/edit/mod.rs @@ -684,7 +684,7 @@ impl<'t> ChunkedEdit<'t> { } _ => return Err(Error::Unsupported("chunk index missing".into())), } - img.free(info.address, u64::from(info.chunk_size)); + img.free(info.address, info.chunk_size); Ok(()) } @@ -1638,7 +1638,7 @@ fn store_chunk( let len = bytes.len() as u64; let placed = match existing { Some(info) if t.pipeline.is_none() => { - if u64::from(info.chunk_size) != len { + if info.chunk_size != len { return Err(Error::Unsupported( "unfiltered chunk stored at an unexpected size".into(), )); @@ -1650,11 +1650,10 @@ fn store_chunk( // thing in the file (the chunk an append keeps rewriting usually is) // and can grow there. Some(info) - if len <= u64::from(info.chunk_size) - || img.grow_tail(info.address, u64::from(info.chunk_size), len)? => + if len <= info.chunk_size || img.grow_tail(info.address, info.chunk_size, len)? => { img.write(info.address, &bytes)?; - (len != u64::from(info.chunk_size) || info.filter_mask != mask).then_some(Elem { + (len != info.chunk_size || info.filter_mask != mask).then_some(Elem { addr: info.address, size: len, mask, @@ -1664,7 +1663,7 @@ fn store_chunk( let a = img.alloc(len)?; img.write(a, &bytes)?; if let Some(info) = existing { - img.free(info.address, u64::from(info.chunk_size)); + img.free(info.address, info.chunk_size); } Some(Elem { addr: a, @@ -1682,14 +1681,16 @@ fn store_chunk( fn img_read<'a>(f: &'a File, info: &ChunkInfo) -> Result<&'a [u8], Error> { let start = usize::try_from(info.address) .map_err(|_| Error::Unsupported("chunk address out of range".into()))?; - f.as_bytes() - .get(start..start + info.chunk_size as usize) - .ok_or_else(|| { - Error::Format(clawhdf5_format::error::FormatError::UnexpectedEof { - expected: start + info.chunk_size as usize, - available: f.as_bytes().len(), - }) + let end = usize::try_from(info.chunk_size) + .ok() + .and_then(|n| start.checked_add(n)) + .unwrap_or(usize::MAX); + f.as_bytes().get(start..end).ok_or_else(|| { + Error::Format(clawhdf5_format::error::FormatError::UnexpectedEof { + expected: end, + available: f.as_bytes().len(), }) + }) } fn decode_chunk( diff --git a/crates/clawhdf5/tests/huge_chunks_interop.rs b/crates/clawhdf5/tests/huge_chunks_interop.rs index 0fdf6f4..52885e7 100644 --- a/crates/clawhdf5/tests/huge_chunks_interop.rs +++ b/crates/clawhdf5/tests/huge_chunks_interop.rs @@ -88,7 +88,9 @@ fn layout_of( /// The fixture's datasets: name, chunk index type, shape, and the scaled /// origins of the chunks libhdf5 wrote. -const FILTERED: &[(&str, u8, &[u64], &[&[u64]])] = &[ +type Case = (&'static str, u8, &'static [u64], &'static [&'static [u64]]); + +const FILTERED: &[Case] = &[ ("single", 1, &[N], &[&[0]]), ("farray", 3, &[N + 10], &[&[0], &[N]]), ("earray", 4, &[N + 10], &[&[0], &[N]]), diff --git a/crates/clawhdf5/tests/partial_read_equivalence.rs b/crates/clawhdf5/tests/partial_read_equivalence.rs index a84b220..86fa568 100644 --- a/crates/clawhdf5/tests/partial_read_equivalence.rs +++ b/crates/clawhdf5/tests/partial_read_equivalence.rs @@ -195,3 +195,90 @@ fn out_of_bounds_selections_are_errors() { [99] ); } + +/// A chunked dataset with a fill value and chunks never written: a +/// selection is read over a box of fill values from the chunks it touches +/// (it used to be picked out of a full read), and through positioned reads +/// an unfiltered chunk is read row by row. Both equal the full read, in +/// every chunk index h5py writes. +#[test] +fn fill_value_selections_match_full_reads() { + let python = std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".into()); + let has_h5py = std::process::Command::new(&python) + .args(["-c", "import h5py"]) + .output() + .is_ok_and(|o| o.status.success()); + if !has_h5py { + assert!( + std::env::var("CLAWHDF5_REQUIRE_INTEROP").as_deref() != Ok("1"), + "CLAWHDF5_REQUIRE_INTEROP=1 but python3 with h5py is not available" + ); + eprintln!("SKIP: python3 with h5py not available"); + return; + } + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("fill.h5"); + let script = r#" +import sys, h5py, numpy as np +with h5py.File(sys.argv[1], "w", libver="latest") as f: + for name, maxshape, gz in [ + ("farray", None, False), ("farray_gz", None, True), + ("earray", (None, 23), False), ("earray_gz", (None, 23), True), + ("btree2", (None, None), False), ("btree2_gz", (None, None), True), + ]: + d = f.create_dataset(name, shape=(37, 23), maxshape=maxshape, chunks=(5, 4), + dtype=" pkg.openUrl("http://huge.invalid/x.h5", { fetch: mockFetch(farBytes, { total }) }), total > 2 ** 53 ? /2\^53 - 1/ : /4 GiB/, `length ${total}`); } + + // Nor can it hold a chunk of 4 GiB or more (HDF5 2.0, layout message + // version 5): the file opens and lists, and reading such a chunk is an + // error naming the size, in every chunk index. + const hugeChunks = join(import.meta.dirname, "..", "..", "..", + "crates/clawhdf5/tests/fixtures/huge_chunks_filtered.h5"); + const hc = pkg.open(new Uint8Array(readFileSync(hugeChunks))); + eq(hc.list("/").map((e) => e.name), ["btree2", "earray", "farray", "single"], "huge chunks: list"); + for (const name of ["single", "farray", "earray", "btree2"]) { + const [start, count] = name === "btree2" ? [[0, 0], [1, 4]] : [[0], [4]]; + await fails(() => hc.readHyperslab(`/${name}`, start, count), /exceeds the addressable size/, + `huge chunk: ${name}`); + } + hc.free(); } // A body of `total` bytes in 64 KiB pieces, made as they are read; `pulled()`