Huge chunks: selection tests with fill values, LZ4 over 256 MiB, wasm32 check
A selection of a chunked dataset with a fill value is compared with the full read in every index, through a map and through positioned reads. An LZ4 chunk larger than 256 MiB is bounded by the chunk size, not refused. The wasm package test reads the 4 GiB-chunk fixture and gets a clean error in every index. Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
@@ -1491,7 +1491,7 @@ pub fn list_chunks_for_read_in<S: Storage + ?Sized>(
|
||||
let chunk_bytes = checked_chunk_byte_len(&chunk_dims, elem_size)?;
|
||||
if let Some(c) = chunks
|
||||
.iter()
|
||||
.find(|c| c.address != u64::MAX && c.chunk_size as usize != chunk_bytes)
|
||||
.find(|c| c.address != u64::MAX && c.chunk_size != chunk_bytes as u64)
|
||||
{
|
||||
return Err(FormatError::ChunkedReadError(format!(
|
||||
"incorrect chunk size returned from index for unfiltered chunk at {:?}: \
|
||||
@@ -2470,7 +2470,7 @@ mod tests {
|
||||
// Entries: key[i], child[i] pairs, then final key
|
||||
for chunk in chunks {
|
||||
// Key: chunk_size(4) + filter_mask(4) + ndims offsets
|
||||
buf.extend_from_slice(&chunk.chunk_size.to_le_bytes());
|
||||
buf.extend_from_slice(&(chunk.chunk_size as u32).to_le_bytes());
|
||||
buf.extend_from_slice(&chunk.filter_mask.to_le_bytes());
|
||||
for d in 0..ndims {
|
||||
let off = if d < chunk.offsets.len() {
|
||||
|
||||
@@ -2263,7 +2263,7 @@ mod tests {
|
||||
for i in 0..n {
|
||||
let (offsets, chunk) = extract_chunk(&data, &shape, &chunks, 2, i);
|
||||
assert_eq!(chunk.len(), 2 * 3 * 2 * 2);
|
||||
for (e, pair) in chunk.chunks_exact(2).enumerate() {
|
||||
for (e, pair) in chunk.as_chunks::<2>().0.iter().enumerate() {
|
||||
let c = [e / 6, (e / 2) % 3, e % 2];
|
||||
let g: Vec<u64> = (0..3).map(|d| offsets[d] + c[d] as u64).collect();
|
||||
let expect = if (0..3).all(|d| g[d] < shape[d]) {
|
||||
@@ -2272,7 +2272,7 @@ mod tests {
|
||||
} else {
|
||||
[0, 0]
|
||||
};
|
||||
assert_eq!(pair, expect, "chunk {i} element {e}");
|
||||
assert_eq!(*pair, expect, "chunk {i} element {e}");
|
||||
}
|
||||
}
|
||||
// Data shorter than the shape: the missing elements stay zero.
|
||||
|
||||
@@ -946,7 +946,7 @@ mod tests {
|
||||
assert_eq!(chunks.len(), 2);
|
||||
assert_eq!(chunks[0].address, base_addr);
|
||||
assert_eq!(chunks[0].offsets, vec![0]);
|
||||
assert_eq!(chunks[0].chunk_size, chunk_byte_size as u64);
|
||||
assert_eq!(chunks[0].chunk_size, chunk_byte_size);
|
||||
assert_eq!(chunks[1].address, base_addr + chunk_byte_size);
|
||||
assert_eq!(chunks[1].offsets, vec![20]);
|
||||
}
|
||||
|
||||
@@ -2764,6 +2764,39 @@ mod tests {
|
||||
assert_eq!(decompressed, data);
|
||||
}
|
||||
|
||||
/// A chunk's size bounds an LZ4 chunk, not the 256 MiB ceiling for an
|
||||
/// unknown size: a 300 MiB chunk was refused ("declared size exceeds
|
||||
/// limit").
|
||||
#[test]
|
||||
#[cfg(feature = "lz4")]
|
||||
fn lz4_chunks_over_256_mib_decode() {
|
||||
let mut data = vec![0u8; 300 << 20];
|
||||
data[12345] = 7;
|
||||
let compressed = lz4_compress(&data, &[]).unwrap();
|
||||
let decompressed = lz4_decompress(&compressed, data.len()).unwrap();
|
||||
assert!(decompressed == data);
|
||||
// Without a chunk size the ceiling still applies.
|
||||
assert!(lz4_decompress(&compressed, 0).is_err());
|
||||
}
|
||||
|
||||
/// A chunk of 4 GiB or more is always in the registered framing, whose
|
||||
/// big-endian size then does not start with four zero bytes: its size
|
||||
/// is read whole (here larger than the chunk, so refused before any
|
||||
/// allocation), not taken for a legacy 4-byte size.
|
||||
#[test]
|
||||
#[cfg(all(feature = "lz4", target_pointer_width = "64"))]
|
||||
fn lz4_chunks_of_4_gib_use_the_registered_framing() {
|
||||
let chunk = (1usize << 32) + 8;
|
||||
let mut data = ((chunk + 8) as u64).to_be_bytes().to_vec();
|
||||
data.extend_from_slice(&(1u32 << 30).to_be_bytes());
|
||||
data.extend_from_slice(&[0; 8]);
|
||||
let err = lz4_decompress(&data, chunk).unwrap_err();
|
||||
assert!(
|
||||
matches!(&err, FormatError::DecompressionError(m) if m.contains("exceeds chunk size")),
|
||||
"{err:?}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
#[cfg(feature = "lz4")]
|
||||
fn pipeline_lz4_only() {
|
||||
|
||||
@@ -255,7 +255,7 @@ pub fn decompress_chunks_lane_partitioned_in<S: Storage + ?Sized>(
|
||||
for &local in &indices {
|
||||
let index = batch.start + local;
|
||||
let chunk_info = &chunks[index];
|
||||
let size = chunk_info.chunk_size as usize;
|
||||
let size = crate::addr::saturating_usize(chunk_info.chunk_size);
|
||||
let raw_chunk = raw_bytes.get(index, &reqs[index])?;
|
||||
|
||||
let decompressed = decompress_chunk_exact(
|
||||
|
||||
@@ -368,7 +368,10 @@ pub fn read_selection_filled_in<S: Storage + ?Sized>(
|
||||
fill: Option<&[u8]>,
|
||||
) -> Result<Option<Vec<u8>>, FormatError> {
|
||||
let dims = &dataspace.dimensions;
|
||||
if dims.is_empty() || elem_size == 0 {
|
||||
// A fill value that is not one element's bytes is the full path's to
|
||||
// interpret.
|
||||
let odd_fill = fill.is_some_and(|f| !f.is_empty() && f.len() != elem_size);
|
||||
if dims.is_empty() || elem_size == 0 || odd_fill {
|
||||
return Ok(None);
|
||||
}
|
||||
let total = dataspace.checked_num_elements()?;
|
||||
|
||||
@@ -151,7 +151,7 @@ fn crafted() -> (Vec<u8>, Chunked, Vec<ChunkInfo>) {
|
||||
// v1 B-tree key (size, filter mask, offsets + 0) then the child
|
||||
// address.
|
||||
let mut pat = Vec::new();
|
||||
pat.extend_from_slice(&c.chunk_size.to_le_bytes());
|
||||
pat.extend_from_slice(&(c.chunk_size as u32).to_le_bytes());
|
||||
pat.extend_from_slice(&c.filter_mask.to_le_bytes());
|
||||
// The key holds one offset per dimension plus the element offset
|
||||
// (0); `offsets` may or may not list that last one.
|
||||
|
||||
@@ -916,7 +916,7 @@ impl Checker<'_> {
|
||||
bad.push("has size 0".into());
|
||||
} else if !filtered
|
||||
&& let Some(cb) = chunk_bytes
|
||||
&& u64::from(c.chunk_size) != cb
|
||||
&& c.chunk_size != cb
|
||||
{
|
||||
bad.push(format!(
|
||||
"is {} bytes; an unfiltered chunk is {cb}",
|
||||
@@ -933,7 +933,7 @@ impl Checker<'_> {
|
||||
);
|
||||
}
|
||||
}
|
||||
self.extent(c.address, u64::from(c.chunk_size), path);
|
||||
self.extent(c.address, c.chunk_size, path);
|
||||
}
|
||||
if reported > 50 {
|
||||
self.problem(
|
||||
|
||||
@@ -224,7 +224,7 @@ pub fn allocated_bytes(h5: &H5, info: &DsInfo) -> Result<u64> {
|
||||
let ds = info.ds.as_ref().map_err(Clone::clone)?;
|
||||
chunks(h5, layout, ds, dt)?
|
||||
.iter()
|
||||
.map(|c| u64::from(c.chunk_size))
|
||||
.map(|c| c.chunk_size)
|
||||
.sum()
|
||||
}
|
||||
DataLayout::Virtual { .. } => 0,
|
||||
|
||||
@@ -684,7 +684,7 @@ impl<'t> ChunkedEdit<'t> {
|
||||
}
|
||||
_ => return Err(Error::Unsupported("chunk index missing".into())),
|
||||
}
|
||||
img.free(info.address, u64::from(info.chunk_size));
|
||||
img.free(info.address, info.chunk_size);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
@@ -1638,7 +1638,7 @@ fn store_chunk(
|
||||
let len = bytes.len() as u64;
|
||||
let placed = match existing {
|
||||
Some(info) if t.pipeline.is_none() => {
|
||||
if u64::from(info.chunk_size) != len {
|
||||
if info.chunk_size != len {
|
||||
return Err(Error::Unsupported(
|
||||
"unfiltered chunk stored at an unexpected size".into(),
|
||||
));
|
||||
@@ -1650,11 +1650,10 @@ fn store_chunk(
|
||||
// thing in the file (the chunk an append keeps rewriting usually is)
|
||||
// and can grow there.
|
||||
Some(info)
|
||||
if len <= u64::from(info.chunk_size)
|
||||
|| img.grow_tail(info.address, u64::from(info.chunk_size), len)? =>
|
||||
if len <= info.chunk_size || img.grow_tail(info.address, info.chunk_size, len)? =>
|
||||
{
|
||||
img.write(info.address, &bytes)?;
|
||||
(len != u64::from(info.chunk_size) || info.filter_mask != mask).then_some(Elem {
|
||||
(len != info.chunk_size || info.filter_mask != mask).then_some(Elem {
|
||||
addr: info.address,
|
||||
size: len,
|
||||
mask,
|
||||
@@ -1664,7 +1663,7 @@ fn store_chunk(
|
||||
let a = img.alloc(len)?;
|
||||
img.write(a, &bytes)?;
|
||||
if let Some(info) = existing {
|
||||
img.free(info.address, u64::from(info.chunk_size));
|
||||
img.free(info.address, info.chunk_size);
|
||||
}
|
||||
Some(Elem {
|
||||
addr: a,
|
||||
@@ -1682,14 +1681,16 @@ fn store_chunk(
|
||||
fn img_read<'a>(f: &'a File, info: &ChunkInfo) -> Result<&'a [u8], Error> {
|
||||
let start = usize::try_from(info.address)
|
||||
.map_err(|_| Error::Unsupported("chunk address out of range".into()))?;
|
||||
f.as_bytes()
|
||||
.get(start..start + info.chunk_size as usize)
|
||||
.ok_or_else(|| {
|
||||
Error::Format(clawhdf5_format::error::FormatError::UnexpectedEof {
|
||||
expected: start + info.chunk_size as usize,
|
||||
available: f.as_bytes().len(),
|
||||
})
|
||||
let end = usize::try_from(info.chunk_size)
|
||||
.ok()
|
||||
.and_then(|n| start.checked_add(n))
|
||||
.unwrap_or(usize::MAX);
|
||||
f.as_bytes().get(start..end).ok_or_else(|| {
|
||||
Error::Format(clawhdf5_format::error::FormatError::UnexpectedEof {
|
||||
expected: end,
|
||||
available: f.as_bytes().len(),
|
||||
})
|
||||
})
|
||||
}
|
||||
|
||||
fn decode_chunk(
|
||||
|
||||
@@ -88,7 +88,9 @@ fn layout_of(
|
||||
|
||||
/// The fixture's datasets: name, chunk index type, shape, and the scaled
|
||||
/// origins of the chunks libhdf5 wrote.
|
||||
const FILTERED: &[(&str, u8, &[u64], &[&[u64]])] = &[
|
||||
type Case = (&'static str, u8, &'static [u64], &'static [&'static [u64]]);
|
||||
|
||||
const FILTERED: &[Case] = &[
|
||||
("single", 1, &[N], &[&[0]]),
|
||||
("farray", 3, &[N + 10], &[&[0], &[N]]),
|
||||
("earray", 4, &[N + 10], &[&[0], &[N]]),
|
||||
|
||||
@@ -195,3 +195,90 @@ fn out_of_bounds_selections_are_errors() {
|
||||
[99]
|
||||
);
|
||||
}
|
||||
|
||||
/// A chunked dataset with a fill value and chunks never written: a
|
||||
/// selection is read over a box of fill values from the chunks it touches
|
||||
/// (it used to be picked out of a full read), and through positioned reads
|
||||
/// an unfiltered chunk is read row by row. Both equal the full read, in
|
||||
/// every chunk index h5py writes.
|
||||
#[test]
|
||||
fn fill_value_selections_match_full_reads() {
|
||||
let python = std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".into());
|
||||
let has_h5py = std::process::Command::new(&python)
|
||||
.args(["-c", "import h5py"])
|
||||
.output()
|
||||
.is_ok_and(|o| o.status.success());
|
||||
if !has_h5py {
|
||||
assert!(
|
||||
std::env::var("CLAWHDF5_REQUIRE_INTEROP").as_deref() != Ok("1"),
|
||||
"CLAWHDF5_REQUIRE_INTEROP=1 but python3 with h5py is not available"
|
||||
);
|
||||
eprintln!("SKIP: python3 with h5py not available");
|
||||
return;
|
||||
}
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let path = dir.path().join("fill.h5");
|
||||
let script = r#"
|
||||
import sys, h5py, numpy as np
|
||||
with h5py.File(sys.argv[1], "w", libver="latest") as f:
|
||||
for name, maxshape, gz in [
|
||||
("farray", None, False), ("farray_gz", None, True),
|
||||
("earray", (None, 23), False), ("earray_gz", (None, 23), True),
|
||||
("btree2", (None, None), False), ("btree2_gz", (None, None), True),
|
||||
]:
|
||||
d = f.create_dataset(name, shape=(37, 23), maxshape=maxshape, chunks=(5, 4),
|
||||
dtype="<i4", fillvalue=-5,
|
||||
compression="gzip" if gz else None)
|
||||
d[3:17, 2:11] = np.arange(14 * 9, dtype="<i4").reshape(14, 9)
|
||||
d[30, 20] = 7
|
||||
"#;
|
||||
let out = std::process::Command::new(&python)
|
||||
.args(["-c", script])
|
||||
.arg(&path)
|
||||
.output()
|
||||
.unwrap();
|
||||
assert!(
|
||||
out.status.success(),
|
||||
"{}",
|
||||
String::from_utf8_lossy(&out.stderr)
|
||||
);
|
||||
|
||||
let dims = [37u64, 23];
|
||||
let mapped = File::open(&path).unwrap();
|
||||
let storage = clawhdf5::FileStorage::open(&path).unwrap();
|
||||
let positioned = File::open_storage(std::sync::Arc::new(storage)).unwrap();
|
||||
let mut rng = Rng(11);
|
||||
for name in [
|
||||
"farray",
|
||||
"farray_gz",
|
||||
"earray",
|
||||
"earray_gz",
|
||||
"btree2",
|
||||
"btree2_gz",
|
||||
] {
|
||||
let full = mapped.dataset(name).unwrap().read_i32().unwrap();
|
||||
assert_eq!(full[0], -5, "{name}");
|
||||
assert_eq!(full[3 * 23 + 2], 0, "{name}");
|
||||
for case in 0..60 {
|
||||
let selection = if case % 5 == 4 {
|
||||
let points = (0..1 + rng.below(12))
|
||||
.map(|_| dims.iter().map(|&d| rng.below(d)).collect())
|
||||
.collect();
|
||||
Selection::Points(points)
|
||||
} else {
|
||||
random_hyperslab(&mut rng, &dims)
|
||||
};
|
||||
let want = reference(&full, &dims, &selection);
|
||||
for (how, file) in [("mapped", &mapped), ("positioned", &positioned)] {
|
||||
assert_eq!(
|
||||
file.dataset(name)
|
||||
.unwrap()
|
||||
.read_i32_selection(&selection)
|
||||
.unwrap(),
|
||||
want,
|
||||
"{name} {how} case {case}: {selection:?}"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -500,6 +500,20 @@ async function limitTests() {
|
||||
await fails(() => pkg.openUrl("http://huge.invalid/x.h5", { fetch: mockFetch(farBytes, { total }) }),
|
||||
total > 2 ** 53 ? /2\^53 - 1/ : /4 GiB/, `length ${total}`);
|
||||
}
|
||||
|
||||
// Nor can it hold a chunk of 4 GiB or more (HDF5 2.0, layout message
|
||||
// version 5): the file opens and lists, and reading such a chunk is an
|
||||
// error naming the size, in every chunk index.
|
||||
const hugeChunks = join(import.meta.dirname, "..", "..", "..",
|
||||
"crates/clawhdf5/tests/fixtures/huge_chunks_filtered.h5");
|
||||
const hc = pkg.open(new Uint8Array(readFileSync(hugeChunks)));
|
||||
eq(hc.list("/").map((e) => e.name), ["btree2", "earray", "farray", "single"], "huge chunks: list");
|
||||
for (const name of ["single", "farray", "earray", "btree2"]) {
|
||||
const [start, count] = name === "btree2" ? [[0, 0], [1, 4]] : [[0], [4]];
|
||||
await fails(() => hc.readHyperslab(`/${name}`, start, count), /exceeds the addressable size/,
|
||||
`huge chunk: ${name}`);
|
||||
}
|
||||
hc.free();
|
||||
}
|
||||
|
||||
// A body of `total` bytes in 64 KiB pieces, made as they are read; `pulled()`
|
||||
|
||||
Reference in New Issue
Block a user