Chunks of 4 GiB or more: read in every index, write as libhdf5 2.x does #29
@@ -1491,7 +1491,7 @@ pub fn list_chunks_for_read_in<S: Storage + ?Sized>(
|
|||||||
let chunk_bytes = checked_chunk_byte_len(&chunk_dims, elem_size)?;
|
let chunk_bytes = checked_chunk_byte_len(&chunk_dims, elem_size)?;
|
||||||
if let Some(c) = chunks
|
if let Some(c) = chunks
|
||||||
.iter()
|
.iter()
|
||||||
.find(|c| c.address != u64::MAX && c.chunk_size as usize != chunk_bytes)
|
.find(|c| c.address != u64::MAX && c.chunk_size != chunk_bytes as u64)
|
||||||
{
|
{
|
||||||
return Err(FormatError::ChunkedReadError(format!(
|
return Err(FormatError::ChunkedReadError(format!(
|
||||||
"incorrect chunk size returned from index for unfiltered chunk at {:?}: \
|
"incorrect chunk size returned from index for unfiltered chunk at {:?}: \
|
||||||
@@ -2470,7 +2470,7 @@ mod tests {
|
|||||||
// Entries: key[i], child[i] pairs, then final key
|
// Entries: key[i], child[i] pairs, then final key
|
||||||
for chunk in chunks {
|
for chunk in chunks {
|
||||||
// Key: chunk_size(4) + filter_mask(4) + ndims offsets
|
// Key: chunk_size(4) + filter_mask(4) + ndims offsets
|
||||||
buf.extend_from_slice(&chunk.chunk_size.to_le_bytes());
|
buf.extend_from_slice(&(chunk.chunk_size as u32).to_le_bytes());
|
||||||
buf.extend_from_slice(&chunk.filter_mask.to_le_bytes());
|
buf.extend_from_slice(&chunk.filter_mask.to_le_bytes());
|
||||||
for d in 0..ndims {
|
for d in 0..ndims {
|
||||||
let off = if d < chunk.offsets.len() {
|
let off = if d < chunk.offsets.len() {
|
||||||
|
|||||||
@@ -2263,7 +2263,7 @@ mod tests {
|
|||||||
for i in 0..n {
|
for i in 0..n {
|
||||||
let (offsets, chunk) = extract_chunk(&data, &shape, &chunks, 2, i);
|
let (offsets, chunk) = extract_chunk(&data, &shape, &chunks, 2, i);
|
||||||
assert_eq!(chunk.len(), 2 * 3 * 2 * 2);
|
assert_eq!(chunk.len(), 2 * 3 * 2 * 2);
|
||||||
for (e, pair) in chunk.chunks_exact(2).enumerate() {
|
for (e, pair) in chunk.as_chunks::<2>().0.iter().enumerate() {
|
||||||
let c = [e / 6, (e / 2) % 3, e % 2];
|
let c = [e / 6, (e / 2) % 3, e % 2];
|
||||||
let g: Vec<u64> = (0..3).map(|d| offsets[d] + c[d] as u64).collect();
|
let g: Vec<u64> = (0..3).map(|d| offsets[d] + c[d] as u64).collect();
|
||||||
let expect = if (0..3).all(|d| g[d] < shape[d]) {
|
let expect = if (0..3).all(|d| g[d] < shape[d]) {
|
||||||
@@ -2272,7 +2272,7 @@ mod tests {
|
|||||||
} else {
|
} else {
|
||||||
[0, 0]
|
[0, 0]
|
||||||
};
|
};
|
||||||
assert_eq!(pair, expect, "chunk {i} element {e}");
|
assert_eq!(*pair, expect, "chunk {i} element {e}");
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
// Data shorter than the shape: the missing elements stay zero.
|
// Data shorter than the shape: the missing elements stay zero.
|
||||||
|
|||||||
@@ -946,7 +946,7 @@ mod tests {
|
|||||||
assert_eq!(chunks.len(), 2);
|
assert_eq!(chunks.len(), 2);
|
||||||
assert_eq!(chunks[0].address, base_addr);
|
assert_eq!(chunks[0].address, base_addr);
|
||||||
assert_eq!(chunks[0].offsets, vec![0]);
|
assert_eq!(chunks[0].offsets, vec![0]);
|
||||||
assert_eq!(chunks[0].chunk_size, chunk_byte_size as u64);
|
assert_eq!(chunks[0].chunk_size, chunk_byte_size);
|
||||||
assert_eq!(chunks[1].address, base_addr + chunk_byte_size);
|
assert_eq!(chunks[1].address, base_addr + chunk_byte_size);
|
||||||
assert_eq!(chunks[1].offsets, vec![20]);
|
assert_eq!(chunks[1].offsets, vec![20]);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -2764,6 +2764,39 @@ mod tests {
|
|||||||
assert_eq!(decompressed, data);
|
assert_eq!(decompressed, data);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// A chunk's size bounds an LZ4 chunk, not the 256 MiB ceiling for an
|
||||||
|
/// unknown size: a 300 MiB chunk was refused ("declared size exceeds
|
||||||
|
/// limit").
|
||||||
|
#[test]
|
||||||
|
#[cfg(feature = "lz4")]
|
||||||
|
fn lz4_chunks_over_256_mib_decode() {
|
||||||
|
let mut data = vec![0u8; 300 << 20];
|
||||||
|
data[12345] = 7;
|
||||||
|
let compressed = lz4_compress(&data, &[]).unwrap();
|
||||||
|
let decompressed = lz4_decompress(&compressed, data.len()).unwrap();
|
||||||
|
assert!(decompressed == data);
|
||||||
|
// Without a chunk size the ceiling still applies.
|
||||||
|
assert!(lz4_decompress(&compressed, 0).is_err());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A chunk of 4 GiB or more is always in the registered framing, whose
|
||||||
|
/// big-endian size then does not start with four zero bytes: its size
|
||||||
|
/// is read whole (here larger than the chunk, so refused before any
|
||||||
|
/// allocation), not taken for a legacy 4-byte size.
|
||||||
|
#[test]
|
||||||
|
#[cfg(all(feature = "lz4", target_pointer_width = "64"))]
|
||||||
|
fn lz4_chunks_of_4_gib_use_the_registered_framing() {
|
||||||
|
let chunk = (1usize << 32) + 8;
|
||||||
|
let mut data = ((chunk + 8) as u64).to_be_bytes().to_vec();
|
||||||
|
data.extend_from_slice(&(1u32 << 30).to_be_bytes());
|
||||||
|
data.extend_from_slice(&[0; 8]);
|
||||||
|
let err = lz4_decompress(&data, chunk).unwrap_err();
|
||||||
|
assert!(
|
||||||
|
matches!(&err, FormatError::DecompressionError(m) if m.contains("exceeds chunk size")),
|
||||||
|
"{err:?}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
#[cfg(feature = "lz4")]
|
#[cfg(feature = "lz4")]
|
||||||
fn pipeline_lz4_only() {
|
fn pipeline_lz4_only() {
|
||||||
|
|||||||
@@ -255,7 +255,7 @@ pub fn decompress_chunks_lane_partitioned_in<S: Storage + ?Sized>(
|
|||||||
for &local in &indices {
|
for &local in &indices {
|
||||||
let index = batch.start + local;
|
let index = batch.start + local;
|
||||||
let chunk_info = &chunks[index];
|
let chunk_info = &chunks[index];
|
||||||
let size = chunk_info.chunk_size as usize;
|
let size = crate::addr::saturating_usize(chunk_info.chunk_size);
|
||||||
let raw_chunk = raw_bytes.get(index, &reqs[index])?;
|
let raw_chunk = raw_bytes.get(index, &reqs[index])?;
|
||||||
|
|
||||||
let decompressed = decompress_chunk_exact(
|
let decompressed = decompress_chunk_exact(
|
||||||
|
|||||||
@@ -368,7 +368,10 @@ pub fn read_selection_filled_in<S: Storage + ?Sized>(
|
|||||||
fill: Option<&[u8]>,
|
fill: Option<&[u8]>,
|
||||||
) -> Result<Option<Vec<u8>>, FormatError> {
|
) -> Result<Option<Vec<u8>>, FormatError> {
|
||||||
let dims = &dataspace.dimensions;
|
let dims = &dataspace.dimensions;
|
||||||
if dims.is_empty() || elem_size == 0 {
|
// A fill value that is not one element's bytes is the full path's to
|
||||||
|
// interpret.
|
||||||
|
let odd_fill = fill.is_some_and(|f| !f.is_empty() && f.len() != elem_size);
|
||||||
|
if dims.is_empty() || elem_size == 0 || odd_fill {
|
||||||
return Ok(None);
|
return Ok(None);
|
||||||
}
|
}
|
||||||
let total = dataspace.checked_num_elements()?;
|
let total = dataspace.checked_num_elements()?;
|
||||||
|
|||||||
@@ -151,7 +151,7 @@ fn crafted() -> (Vec<u8>, Chunked, Vec<ChunkInfo>) {
|
|||||||
// v1 B-tree key (size, filter mask, offsets + 0) then the child
|
// v1 B-tree key (size, filter mask, offsets + 0) then the child
|
||||||
// address.
|
// address.
|
||||||
let mut pat = Vec::new();
|
let mut pat = Vec::new();
|
||||||
pat.extend_from_slice(&c.chunk_size.to_le_bytes());
|
pat.extend_from_slice(&(c.chunk_size as u32).to_le_bytes());
|
||||||
pat.extend_from_slice(&c.filter_mask.to_le_bytes());
|
pat.extend_from_slice(&c.filter_mask.to_le_bytes());
|
||||||
// The key holds one offset per dimension plus the element offset
|
// The key holds one offset per dimension plus the element offset
|
||||||
// (0); `offsets` may or may not list that last one.
|
// (0); `offsets` may or may not list that last one.
|
||||||
|
|||||||
@@ -916,7 +916,7 @@ impl Checker<'_> {
|
|||||||
bad.push("has size 0".into());
|
bad.push("has size 0".into());
|
||||||
} else if !filtered
|
} else if !filtered
|
||||||
&& let Some(cb) = chunk_bytes
|
&& let Some(cb) = chunk_bytes
|
||||||
&& u64::from(c.chunk_size) != cb
|
&& c.chunk_size != cb
|
||||||
{
|
{
|
||||||
bad.push(format!(
|
bad.push(format!(
|
||||||
"is {} bytes; an unfiltered chunk is {cb}",
|
"is {} bytes; an unfiltered chunk is {cb}",
|
||||||
@@ -933,7 +933,7 @@ impl Checker<'_> {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
self.extent(c.address, u64::from(c.chunk_size), path);
|
self.extent(c.address, c.chunk_size, path);
|
||||||
}
|
}
|
||||||
if reported > 50 {
|
if reported > 50 {
|
||||||
self.problem(
|
self.problem(
|
||||||
|
|||||||
@@ -224,7 +224,7 @@ pub fn allocated_bytes(h5: &H5, info: &DsInfo) -> Result<u64> {
|
|||||||
let ds = info.ds.as_ref().map_err(Clone::clone)?;
|
let ds = info.ds.as_ref().map_err(Clone::clone)?;
|
||||||
chunks(h5, layout, ds, dt)?
|
chunks(h5, layout, ds, dt)?
|
||||||
.iter()
|
.iter()
|
||||||
.map(|c| u64::from(c.chunk_size))
|
.map(|c| c.chunk_size)
|
||||||
.sum()
|
.sum()
|
||||||
}
|
}
|
||||||
DataLayout::Virtual { .. } => 0,
|
DataLayout::Virtual { .. } => 0,
|
||||||
|
|||||||
@@ -684,7 +684,7 @@ impl<'t> ChunkedEdit<'t> {
|
|||||||
}
|
}
|
||||||
_ => return Err(Error::Unsupported("chunk index missing".into())),
|
_ => return Err(Error::Unsupported("chunk index missing".into())),
|
||||||
}
|
}
|
||||||
img.free(info.address, u64::from(info.chunk_size));
|
img.free(info.address, info.chunk_size);
|
||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1638,7 +1638,7 @@ fn store_chunk(
|
|||||||
let len = bytes.len() as u64;
|
let len = bytes.len() as u64;
|
||||||
let placed = match existing {
|
let placed = match existing {
|
||||||
Some(info) if t.pipeline.is_none() => {
|
Some(info) if t.pipeline.is_none() => {
|
||||||
if u64::from(info.chunk_size) != len {
|
if info.chunk_size != len {
|
||||||
return Err(Error::Unsupported(
|
return Err(Error::Unsupported(
|
||||||
"unfiltered chunk stored at an unexpected size".into(),
|
"unfiltered chunk stored at an unexpected size".into(),
|
||||||
));
|
));
|
||||||
@@ -1650,11 +1650,10 @@ fn store_chunk(
|
|||||||
// thing in the file (the chunk an append keeps rewriting usually is)
|
// thing in the file (the chunk an append keeps rewriting usually is)
|
||||||
// and can grow there.
|
// and can grow there.
|
||||||
Some(info)
|
Some(info)
|
||||||
if len <= u64::from(info.chunk_size)
|
if len <= info.chunk_size || img.grow_tail(info.address, info.chunk_size, len)? =>
|
||||||
|| img.grow_tail(info.address, u64::from(info.chunk_size), len)? =>
|
|
||||||
{
|
{
|
||||||
img.write(info.address, &bytes)?;
|
img.write(info.address, &bytes)?;
|
||||||
(len != u64::from(info.chunk_size) || info.filter_mask != mask).then_some(Elem {
|
(len != info.chunk_size || info.filter_mask != mask).then_some(Elem {
|
||||||
addr: info.address,
|
addr: info.address,
|
||||||
size: len,
|
size: len,
|
||||||
mask,
|
mask,
|
||||||
@@ -1664,7 +1663,7 @@ fn store_chunk(
|
|||||||
let a = img.alloc(len)?;
|
let a = img.alloc(len)?;
|
||||||
img.write(a, &bytes)?;
|
img.write(a, &bytes)?;
|
||||||
if let Some(info) = existing {
|
if let Some(info) = existing {
|
||||||
img.free(info.address, u64::from(info.chunk_size));
|
img.free(info.address, info.chunk_size);
|
||||||
}
|
}
|
||||||
Some(Elem {
|
Some(Elem {
|
||||||
addr: a,
|
addr: a,
|
||||||
@@ -1682,14 +1681,16 @@ fn store_chunk(
|
|||||||
fn img_read<'a>(f: &'a File, info: &ChunkInfo) -> Result<&'a [u8], Error> {
|
fn img_read<'a>(f: &'a File, info: &ChunkInfo) -> Result<&'a [u8], Error> {
|
||||||
let start = usize::try_from(info.address)
|
let start = usize::try_from(info.address)
|
||||||
.map_err(|_| Error::Unsupported("chunk address out of range".into()))?;
|
.map_err(|_| Error::Unsupported("chunk address out of range".into()))?;
|
||||||
f.as_bytes()
|
let end = usize::try_from(info.chunk_size)
|
||||||
.get(start..start + info.chunk_size as usize)
|
.ok()
|
||||||
.ok_or_else(|| {
|
.and_then(|n| start.checked_add(n))
|
||||||
Error::Format(clawhdf5_format::error::FormatError::UnexpectedEof {
|
.unwrap_or(usize::MAX);
|
||||||
expected: start + info.chunk_size as usize,
|
f.as_bytes().get(start..end).ok_or_else(|| {
|
||||||
available: f.as_bytes().len(),
|
Error::Format(clawhdf5_format::error::FormatError::UnexpectedEof {
|
||||||
})
|
expected: end,
|
||||||
|
available: f.as_bytes().len(),
|
||||||
})
|
})
|
||||||
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
fn decode_chunk(
|
fn decode_chunk(
|
||||||
|
|||||||
@@ -88,7 +88,9 @@ fn layout_of(
|
|||||||
|
|
||||||
/// The fixture's datasets: name, chunk index type, shape, and the scaled
|
/// The fixture's datasets: name, chunk index type, shape, and the scaled
|
||||||
/// origins of the chunks libhdf5 wrote.
|
/// origins of the chunks libhdf5 wrote.
|
||||||
const FILTERED: &[(&str, u8, &[u64], &[&[u64]])] = &[
|
type Case = (&'static str, u8, &'static [u64], &'static [&'static [u64]]);
|
||||||
|
|
||||||
|
const FILTERED: &[Case] = &[
|
||||||
("single", 1, &[N], &[&[0]]),
|
("single", 1, &[N], &[&[0]]),
|
||||||
("farray", 3, &[N + 10], &[&[0], &[N]]),
|
("farray", 3, &[N + 10], &[&[0], &[N]]),
|
||||||
("earray", 4, &[N + 10], &[&[0], &[N]]),
|
("earray", 4, &[N + 10], &[&[0], &[N]]),
|
||||||
|
|||||||
@@ -195,3 +195,90 @@ fn out_of_bounds_selections_are_errors() {
|
|||||||
[99]
|
[99]
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// A chunked dataset with a fill value and chunks never written: a
|
||||||
|
/// selection is read over a box of fill values from the chunks it touches
|
||||||
|
/// (it used to be picked out of a full read), and through positioned reads
|
||||||
|
/// an unfiltered chunk is read row by row. Both equal the full read, in
|
||||||
|
/// every chunk index h5py writes.
|
||||||
|
#[test]
|
||||||
|
fn fill_value_selections_match_full_reads() {
|
||||||
|
let python = std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".into());
|
||||||
|
let has_h5py = std::process::Command::new(&python)
|
||||||
|
.args(["-c", "import h5py"])
|
||||||
|
.output()
|
||||||
|
.is_ok_and(|o| o.status.success());
|
||||||
|
if !has_h5py {
|
||||||
|
assert!(
|
||||||
|
std::env::var("CLAWHDF5_REQUIRE_INTEROP").as_deref() != Ok("1"),
|
||||||
|
"CLAWHDF5_REQUIRE_INTEROP=1 but python3 with h5py is not available"
|
||||||
|
);
|
||||||
|
eprintln!("SKIP: python3 with h5py not available");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
let dir = tempfile::tempdir().unwrap();
|
||||||
|
let path = dir.path().join("fill.h5");
|
||||||
|
let script = r#"
|
||||||
|
import sys, h5py, numpy as np
|
||||||
|
with h5py.File(sys.argv[1], "w", libver="latest") as f:
|
||||||
|
for name, maxshape, gz in [
|
||||||
|
("farray", None, False), ("farray_gz", None, True),
|
||||||
|
("earray", (None, 23), False), ("earray_gz", (None, 23), True),
|
||||||
|
("btree2", (None, None), False), ("btree2_gz", (None, None), True),
|
||||||
|
]:
|
||||||
|
d = f.create_dataset(name, shape=(37, 23), maxshape=maxshape, chunks=(5, 4),
|
||||||
|
dtype="<i4", fillvalue=-5,
|
||||||
|
compression="gzip" if gz else None)
|
||||||
|
d[3:17, 2:11] = np.arange(14 * 9, dtype="<i4").reshape(14, 9)
|
||||||
|
d[30, 20] = 7
|
||||||
|
"#;
|
||||||
|
let out = std::process::Command::new(&python)
|
||||||
|
.args(["-c", script])
|
||||||
|
.arg(&path)
|
||||||
|
.output()
|
||||||
|
.unwrap();
|
||||||
|
assert!(
|
||||||
|
out.status.success(),
|
||||||
|
"{}",
|
||||||
|
String::from_utf8_lossy(&out.stderr)
|
||||||
|
);
|
||||||
|
|
||||||
|
let dims = [37u64, 23];
|
||||||
|
let mapped = File::open(&path).unwrap();
|
||||||
|
let storage = clawhdf5::FileStorage::open(&path).unwrap();
|
||||||
|
let positioned = File::open_storage(std::sync::Arc::new(storage)).unwrap();
|
||||||
|
let mut rng = Rng(11);
|
||||||
|
for name in [
|
||||||
|
"farray",
|
||||||
|
"farray_gz",
|
||||||
|
"earray",
|
||||||
|
"earray_gz",
|
||||||
|
"btree2",
|
||||||
|
"btree2_gz",
|
||||||
|
] {
|
||||||
|
let full = mapped.dataset(name).unwrap().read_i32().unwrap();
|
||||||
|
assert_eq!(full[0], -5, "{name}");
|
||||||
|
assert_eq!(full[3 * 23 + 2], 0, "{name}");
|
||||||
|
for case in 0..60 {
|
||||||
|
let selection = if case % 5 == 4 {
|
||||||
|
let points = (0..1 + rng.below(12))
|
||||||
|
.map(|_| dims.iter().map(|&d| rng.below(d)).collect())
|
||||||
|
.collect();
|
||||||
|
Selection::Points(points)
|
||||||
|
} else {
|
||||||
|
random_hyperslab(&mut rng, &dims)
|
||||||
|
};
|
||||||
|
let want = reference(&full, &dims, &selection);
|
||||||
|
for (how, file) in [("mapped", &mapped), ("positioned", &positioned)] {
|
||||||
|
assert_eq!(
|
||||||
|
file.dataset(name)
|
||||||
|
.unwrap()
|
||||||
|
.read_i32_selection(&selection)
|
||||||
|
.unwrap(),
|
||||||
|
want,
|
||||||
|
"{name} {how} case {case}: {selection:?}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -500,6 +500,20 @@ async function limitTests() {
|
|||||||
await fails(() => pkg.openUrl("http://huge.invalid/x.h5", { fetch: mockFetch(farBytes, { total }) }),
|
await fails(() => pkg.openUrl("http://huge.invalid/x.h5", { fetch: mockFetch(farBytes, { total }) }),
|
||||||
total > 2 ** 53 ? /2\^53 - 1/ : /4 GiB/, `length ${total}`);
|
total > 2 ** 53 ? /2\^53 - 1/ : /4 GiB/, `length ${total}`);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Nor can it hold a chunk of 4 GiB or more (HDF5 2.0, layout message
|
||||||
|
// version 5): the file opens and lists, and reading such a chunk is an
|
||||||
|
// error naming the size, in every chunk index.
|
||||||
|
const hugeChunks = join(import.meta.dirname, "..", "..", "..",
|
||||||
|
"crates/clawhdf5/tests/fixtures/huge_chunks_filtered.h5");
|
||||||
|
const hc = pkg.open(new Uint8Array(readFileSync(hugeChunks)));
|
||||||
|
eq(hc.list("/").map((e) => e.name), ["btree2", "earray", "farray", "single"], "huge chunks: list");
|
||||||
|
for (const name of ["single", "farray", "earray", "btree2"]) {
|
||||||
|
const [start, count] = name === "btree2" ? [[0, 0], [1, 4]] : [[0], [4]];
|
||||||
|
await fails(() => hc.readHyperslab(`/${name}`, start, count), /exceeds the addressable size/,
|
||||||
|
`huge chunk: ${name}`);
|
||||||
|
}
|
||||||
|
hc.free();
|
||||||
}
|
}
|
||||||
|
|
||||||
// A body of `total` bytes in 64 KiB pieces, made as they are read; `pulled()`
|
// A body of `total` bytes in 64 KiB pieces, made as they are read; `pulled()`
|
||||||
|
|||||||
Reference in New Issue
Block a user