Chunks of 4 GiB or more: read in every index, write as libhdf5 2.x does #29

Merged
osobh merged 7 commits from feat/huge-chunks into main 2026-09-29 11:57:48 +00:00
13 changed files with 165 additions and 25 deletions
Showing only changes of commit 7a2e61ebd1 - Show all commits
+2 -2
View File
@@ -1491,7 +1491,7 @@ pub fn list_chunks_for_read_in<S: Storage + ?Sized>(
let chunk_bytes = checked_chunk_byte_len(&chunk_dims, elem_size)?;
if let Some(c) = chunks
.iter()
.find(|c| c.address != u64::MAX && c.chunk_size as usize != chunk_bytes)
.find(|c| c.address != u64::MAX && c.chunk_size != chunk_bytes as u64)
{
return Err(FormatError::ChunkedReadError(format!(
"incorrect chunk size returned from index for unfiltered chunk at {:?}: \
@@ -2470,7 +2470,7 @@ mod tests {
// Entries: key[i], child[i] pairs, then final key
for chunk in chunks {
// Key: chunk_size(4) + filter_mask(4) + ndims offsets
buf.extend_from_slice(&chunk.chunk_size.to_le_bytes());
buf.extend_from_slice(&(chunk.chunk_size as u32).to_le_bytes());
buf.extend_from_slice(&chunk.filter_mask.to_le_bytes());
for d in 0..ndims {
let off = if d < chunk.offsets.len() {
+2 -2
View File
@@ -2263,7 +2263,7 @@ mod tests {
for i in 0..n {
let (offsets, chunk) = extract_chunk(&data, &shape, &chunks, 2, i);
assert_eq!(chunk.len(), 2 * 3 * 2 * 2);
for (e, pair) in chunk.chunks_exact(2).enumerate() {
for (e, pair) in chunk.as_chunks::<2>().0.iter().enumerate() {
let c = [e / 6, (e / 2) % 3, e % 2];
let g: Vec<u64> = (0..3).map(|d| offsets[d] + c[d] as u64).collect();
let expect = if (0..3).all(|d| g[d] < shape[d]) {
@@ -2272,7 +2272,7 @@ mod tests {
} else {
[0, 0]
};
assert_eq!(pair, expect, "chunk {i} element {e}");
assert_eq!(*pair, expect, "chunk {i} element {e}");
}
}
// Data shorter than the shape: the missing elements stay zero.
@@ -946,7 +946,7 @@ mod tests {
assert_eq!(chunks.len(), 2);
assert_eq!(chunks[0].address, base_addr);
assert_eq!(chunks[0].offsets, vec![0]);
assert_eq!(chunks[0].chunk_size, chunk_byte_size as u64);
assert_eq!(chunks[0].chunk_size, chunk_byte_size);
assert_eq!(chunks[1].address, base_addr + chunk_byte_size);
assert_eq!(chunks[1].offsets, vec![20]);
}
+33
View File
@@ -2764,6 +2764,39 @@ mod tests {
assert_eq!(decompressed, data);
}
/// A chunk's size bounds an LZ4 chunk, not the 256 MiB ceiling for an
/// unknown size: a 300 MiB chunk was refused ("declared size exceeds
/// limit").
#[test]
#[cfg(feature = "lz4")]
fn lz4_chunks_over_256_mib_decode() {
let mut data = vec![0u8; 300 << 20];
data[12345] = 7;
let compressed = lz4_compress(&data, &[]).unwrap();
let decompressed = lz4_decompress(&compressed, data.len()).unwrap();
assert!(decompressed == data);
// Without a chunk size the ceiling still applies.
assert!(lz4_decompress(&compressed, 0).is_err());
}
/// A chunk of 4 GiB or more is always in the registered framing, whose
/// big-endian size then does not start with four zero bytes: its size
/// is read whole (here larger than the chunk, so refused before any
/// allocation), not taken for a legacy 4-byte size.
#[test]
#[cfg(all(feature = "lz4", target_pointer_width = "64"))]
fn lz4_chunks_of_4_gib_use_the_registered_framing() {
let chunk = (1usize << 32) + 8;
let mut data = ((chunk + 8) as u64).to_be_bytes().to_vec();
data.extend_from_slice(&(1u32 << 30).to_be_bytes());
data.extend_from_slice(&[0; 8]);
let err = lz4_decompress(&data, chunk).unwrap_err();
assert!(
matches!(&err, FormatError::DecompressionError(m) if m.contains("exceeds chunk size")),
"{err:?}"
);
}
#[test]
#[cfg(feature = "lz4")]
fn pipeline_lz4_only() {
+1 -1
View File
@@ -255,7 +255,7 @@ pub fn decompress_chunks_lane_partitioned_in<S: Storage + ?Sized>(
for &local in &indices {
let index = batch.start + local;
let chunk_info = &chunks[index];
let size = chunk_info.chunk_size as usize;
let size = crate::addr::saturating_usize(chunk_info.chunk_size);
let raw_chunk = raw_bytes.get(index, &reqs[index])?;
let decompressed = decompress_chunk_exact(
+4 -1
View File
@@ -368,7 +368,10 @@ pub fn read_selection_filled_in<S: Storage + ?Sized>(
fill: Option<&[u8]>,
) -> Result<Option<Vec<u8>>, FormatError> {
let dims = &dataspace.dimensions;
if dims.is_empty() || elem_size == 0 {
// A fill value that is not one element's bytes is the full path's to
// interpret.
let odd_fill = fill.is_some_and(|f| !f.is_empty() && f.len() != elem_size);
if dims.is_empty() || elem_size == 0 || odd_fill {
return Ok(None);
}
let total = dataspace.checked_num_elements()?;
@@ -151,7 +151,7 @@ fn crafted() -> (Vec<u8>, Chunked, Vec<ChunkInfo>) {
// v1 B-tree key (size, filter mask, offsets + 0) then the child
// address.
let mut pat = Vec::new();
pat.extend_from_slice(&c.chunk_size.to_le_bytes());
pat.extend_from_slice(&(c.chunk_size as u32).to_le_bytes());
pat.extend_from_slice(&c.filter_mask.to_le_bytes());
// The key holds one offset per dimension plus the element offset
// (0); `offsets` may or may not list that last one.
+2 -2
View File
@@ -916,7 +916,7 @@ impl Checker<'_> {
bad.push("has size 0".into());
} else if !filtered
&& let Some(cb) = chunk_bytes
&& u64::from(c.chunk_size) != cb
&& c.chunk_size != cb
{
bad.push(format!(
"is {} bytes; an unfiltered chunk is {cb}",
@@ -933,7 +933,7 @@ impl Checker<'_> {
);
}
}
self.extent(c.address, u64::from(c.chunk_size), path);
self.extent(c.address, c.chunk_size, path);
}
if reported > 50 {
self.problem(
+1 -1
View File
@@ -224,7 +224,7 @@ pub fn allocated_bytes(h5: &H5, info: &DsInfo) -> Result<u64> {
let ds = info.ds.as_ref().map_err(Clone::clone)?;
chunks(h5, layout, ds, dt)?
.iter()
.map(|c| u64::from(c.chunk_size))
.map(|c| c.chunk_size)
.sum()
}
DataLayout::Virtual { .. } => 0,
+14 -13
View File
@@ -684,7 +684,7 @@ impl<'t> ChunkedEdit<'t> {
}
_ => return Err(Error::Unsupported("chunk index missing".into())),
}
img.free(info.address, u64::from(info.chunk_size));
img.free(info.address, info.chunk_size);
Ok(())
}
@@ -1638,7 +1638,7 @@ fn store_chunk(
let len = bytes.len() as u64;
let placed = match existing {
Some(info) if t.pipeline.is_none() => {
if u64::from(info.chunk_size) != len {
if info.chunk_size != len {
return Err(Error::Unsupported(
"unfiltered chunk stored at an unexpected size".into(),
));
@@ -1650,11 +1650,10 @@ fn store_chunk(
// thing in the file (the chunk an append keeps rewriting usually is)
// and can grow there.
Some(info)
if len <= u64::from(info.chunk_size)
|| img.grow_tail(info.address, u64::from(info.chunk_size), len)? =>
if len <= info.chunk_size || img.grow_tail(info.address, info.chunk_size, len)? =>
{
img.write(info.address, &bytes)?;
(len != u64::from(info.chunk_size) || info.filter_mask != mask).then_some(Elem {
(len != info.chunk_size || info.filter_mask != mask).then_some(Elem {
addr: info.address,
size: len,
mask,
@@ -1664,7 +1663,7 @@ fn store_chunk(
let a = img.alloc(len)?;
img.write(a, &bytes)?;
if let Some(info) = existing {
img.free(info.address, u64::from(info.chunk_size));
img.free(info.address, info.chunk_size);
}
Some(Elem {
addr: a,
@@ -1682,14 +1681,16 @@ fn store_chunk(
fn img_read<'a>(f: &'a File, info: &ChunkInfo) -> Result<&'a [u8], Error> {
let start = usize::try_from(info.address)
.map_err(|_| Error::Unsupported("chunk address out of range".into()))?;
f.as_bytes()
.get(start..start + info.chunk_size as usize)
.ok_or_else(|| {
Error::Format(clawhdf5_format::error::FormatError::UnexpectedEof {
expected: start + info.chunk_size as usize,
available: f.as_bytes().len(),
})
let end = usize::try_from(info.chunk_size)
.ok()
.and_then(|n| start.checked_add(n))
.unwrap_or(usize::MAX);
f.as_bytes().get(start..end).ok_or_else(|| {
Error::Format(clawhdf5_format::error::FormatError::UnexpectedEof {
expected: end,
available: f.as_bytes().len(),
})
})
}
fn decode_chunk(
+3 -1
View File
@@ -88,7 +88,9 @@ fn layout_of(
/// The fixture's datasets: name, chunk index type, shape, and the scaled
/// origins of the chunks libhdf5 wrote.
const FILTERED: &[(&str, u8, &[u64], &[&[u64]])] = &[
type Case = (&'static str, u8, &'static [u64], &'static [&'static [u64]]);
const FILTERED: &[Case] = &[
("single", 1, &[N], &[&[0]]),
("farray", 3, &[N + 10], &[&[0], &[N]]),
("earray", 4, &[N + 10], &[&[0], &[N]]),
@@ -195,3 +195,90 @@ fn out_of_bounds_selections_are_errors() {
[99]
);
}
/// A chunked dataset with a fill value and chunks never written: a
/// selection is read over a box of fill values from the chunks it touches
/// (it used to be picked out of a full read), and through positioned reads
/// an unfiltered chunk is read row by row. Both equal the full read, in
/// every chunk index h5py writes.
#[test]
fn fill_value_selections_match_full_reads() {
let python = std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".into());
let has_h5py = std::process::Command::new(&python)
.args(["-c", "import h5py"])
.output()
.is_ok_and(|o| o.status.success());
if !has_h5py {
assert!(
std::env::var("CLAWHDF5_REQUIRE_INTEROP").as_deref() != Ok("1"),
"CLAWHDF5_REQUIRE_INTEROP=1 but python3 with h5py is not available"
);
eprintln!("SKIP: python3 with h5py not available");
return;
}
let dir = tempfile::tempdir().unwrap();
let path = dir.path().join("fill.h5");
let script = r#"
import sys, h5py, numpy as np
with h5py.File(sys.argv[1], "w", libver="latest") as f:
for name, maxshape, gz in [
("farray", None, False), ("farray_gz", None, True),
("earray", (None, 23), False), ("earray_gz", (None, 23), True),
("btree2", (None, None), False), ("btree2_gz", (None, None), True),
]:
d = f.create_dataset(name, shape=(37, 23), maxshape=maxshape, chunks=(5, 4),
dtype="<i4", fillvalue=-5,
compression="gzip" if gz else None)
d[3:17, 2:11] = np.arange(14 * 9, dtype="<i4").reshape(14, 9)
d[30, 20] = 7
"#;
let out = std::process::Command::new(&python)
.args(["-c", script])
.arg(&path)
.output()
.unwrap();
assert!(
out.status.success(),
"{}",
String::from_utf8_lossy(&out.stderr)
);
let dims = [37u64, 23];
let mapped = File::open(&path).unwrap();
let storage = clawhdf5::FileStorage::open(&path).unwrap();
let positioned = File::open_storage(std::sync::Arc::new(storage)).unwrap();
let mut rng = Rng(11);
for name in [
"farray",
"farray_gz",
"earray",
"earray_gz",
"btree2",
"btree2_gz",
] {
let full = mapped.dataset(name).unwrap().read_i32().unwrap();
assert_eq!(full[0], -5, "{name}");
assert_eq!(full[3 * 23 + 2], 0, "{name}");
for case in 0..60 {
let selection = if case % 5 == 4 {
let points = (0..1 + rng.below(12))
.map(|_| dims.iter().map(|&d| rng.below(d)).collect())
.collect();
Selection::Points(points)
} else {
random_hyperslab(&mut rng, &dims)
};
let want = reference(&full, &dims, &selection);
for (how, file) in [("mapped", &mapped), ("positioned", &positioned)] {
assert_eq!(
file.dataset(name)
.unwrap()
.read_i32_selection(&selection)
.unwrap(),
want,
"{name} {how} case {case}: {selection:?}"
);
}
}
}
}
+14
View File
@@ -500,6 +500,20 @@ async function limitTests() {
await fails(() => pkg.openUrl("http://huge.invalid/x.h5", { fetch: mockFetch(farBytes, { total }) }),
total > 2 ** 53 ? /2\^53 - 1/ : /4 GiB/, `length ${total}`);
}
// Nor can it hold a chunk of 4 GiB or more (HDF5 2.0, layout message
// version 5): the file opens and lists, and reading such a chunk is an
// error naming the size, in every chunk index.
const hugeChunks = join(import.meta.dirname, "..", "..", "..",
"crates/clawhdf5/tests/fixtures/huge_chunks_filtered.h5");
const hc = pkg.open(new Uint8Array(readFileSync(hugeChunks)));
eq(hc.list("/").map((e) => e.name), ["btree2", "earray", "farray", "single"], "huge chunks: list");
for (const name of ["single", "farray", "earray", "btree2"]) {
const [start, count] = name === "btree2" ? [[0, 0], [1, 4]] : [[0], [4]];
await fails(() => hc.readHyperslab(`/${name}`, start, count), /exceeds the addressable size/,
`huge chunk: ${name}`);
}
hc.free();
}
// A body of `total` bytes in 64 KiB pieces, made as they are read; `pulled()`