Files
clawhdf5/crates/clawhdf5/tests/huge_chunks_interop.rs
T
osobhandClaude Opus 5.5 7a2e61ebd1 Huge chunks: selection tests with fill values, LZ4 over 256 MiB, wasm32 check
A selection of a chunked dataset with a fill value is compared with the
full read in every index, through a map and through positioned reads. An
LZ4 chunk larger than 256 MiB is bounded by the chunk size, not refused.
The wasm package test reads the 4 GiB-chunk fixture and gets a clean
error in every index.

Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
2026-09-28 23:44:48 -05:00

519 lines
19 KiB
Rust

//! Chunks of 4 GiB or more, which HDF5 2.0 writes (layout message version 5,
//! `H5F_LIBVER_V200`), in every chunk index libhdf5 uses for them.
//!
//! `fixtures/huge_chunks_filtered.h5` (written by libhdf5 2.0.0 through h5py
//! 3.16, `fixtures/gen_huge_chunks.py filtered`) holds one dataset per
//! filtered index — Single Chunk, Fixed Array, Extensible Array, v2 B-tree —
//! whose chunks are 2^29 + 1 `f64` (4 GiB + 8 bytes), stored deflated twice
//! so each takes about 20 KiB. Listing their chunks is cheap and always runs;
//! decoding one inflates 4 GiB, so those tests are opt-in:
//! `CLAWHDF5_HUGE_CHUNKS=1` (run them one at a time, `--test-threads=1`:
//! each needs about 4.5 GiB of memory). The same variable enables the
//! unfiltered tests, which have h5py write a sparse file (about 44 GiB long,
//! a few blocks on disk) under `tests/scratch/`, and the writer tests.
//!
//! libhdf5 never writes such a chunk with a version-1 B-tree (a chunk of more
//! than 0xFFFFFFFF bytes forces layout version 5 and with it the newer
//! indexes) and refuses to open one; `chunked_read`'s unit tests cover that.
use std::path::{Path, PathBuf};
use std::process::Command;
use clawhdf5::File;
use clawhdf5_format::chunked_read::list_chunks;
use clawhdf5_format::data_layout::DataLayout;
use clawhdf5_format::dataspace::Dataspace;
use clawhdf5_format::message_type::MessageType;
use clawhdf5_format::object_header::ObjectHeader;
use clawhdf5_format::selection::Selection;
use clawhdf5_format::superblock::Superblock;
/// Elements per chunk along the chunked axis: 4 GiB + 8 bytes of `f64`.
const N: u64 = (1 << 29) + 1;
fn fixture() -> PathBuf {
Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/fixtures/huge_chunks_filtered.h5")
}
fn heavy() -> bool {
if std::env::var("CLAWHDF5_HUGE_CHUNKS").is_ok_and(|v| v == "1") {
return true;
}
eprintln!("SKIP: set CLAWHDF5_HUGE_CHUNKS=1 to decode 4 GiB chunks");
false
}
fn python() -> String {
std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string())
}
fn python_available() -> bool {
Command::new(python())
.args(["-c", "import h5py"])
.output()
.is_ok_and(|o| o.status.success())
}
fn interop_required() -> bool {
std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1")
}
/// The raw layout message version, the parsed layout, the dataspace and the
/// chunks of `name` in the in-memory file `data`.
fn layout_of(
data: &[u8],
name: &str,
) -> (
u8,
DataLayout,
Dataspace,
Vec<clawhdf5_format::chunked_read::ChunkInfo>,
) {
let sb = Superblock::parse(data, 0).unwrap();
let addr = clawhdf5_format::group_v2::resolve_path_any(data, &sb, name).unwrap();
let hdr = ObjectHeader::parse(data, addr as usize, sb.offset_size, sb.length_size).unwrap();
let msg = |t| {
hdr.messages
.iter()
.find(|m| m.msg_type == t)
.unwrap_or_else(|| panic!("{name}: no {t:?} message"))
};
let lm = msg(MessageType::DataLayout);
let layout = DataLayout::parse(&lm.data, sb.offset_size, sb.length_size).unwrap();
let space = Dataspace::parse(&msg(MessageType::Dataspace).data, sb.length_size).unwrap();
let (chunks, _) =
list_chunks(data, &layout, &space, 8, sb.offset_size, sb.length_size).unwrap();
(lm.data[0], layout, space, chunks)
}
/// The fixture's datasets: name, chunk index type, shape, and the scaled
/// origins of the chunks libhdf5 wrote.
type Case = (&'static str, u8, &'static [u64], &'static [&'static [u64]]);
const FILTERED: &[Case] = &[
("single", 1, &[N], &[&[0]]),
("farray", 3, &[N + 10], &[&[0], &[N]]),
("earray", 4, &[N + 10], &[&[0], &[N]]),
(
"btree2",
5,
&[2, N + 10],
&[&[0, 0], &[0, N], &[1, 0], &[1, N]],
),
];
/// Every index of the fixture: layout version 5, its chunks listed at the
/// right origins with their stored (deflated) sizes. The index elements
/// store a size in 8 bytes (libhdf5's `H5F_SIZEOF_SIZE` under layout
/// version 5), where version 4 would use 6 for a chunk this size.
#[test]
fn filtered_huge_chunk_indexes_list() {
let data = std::fs::read(fixture()).unwrap();
for &(name, index, shape, origins) in FILTERED {
let (version, layout, space, mut chunks) = layout_of(&data, name);
assert_eq!(version, 5, "{name}: layout message version");
let DataLayout::Chunked {
chunk_dimensions,
chunk_index_type,
..
} = &layout
else {
panic!("{name}: not chunked: {layout:?}");
};
assert_eq!(*chunk_index_type, Some(index), "{name}");
assert_eq!(chunk_dimensions.last(), Some(&8), "{name}");
assert_eq!(space.dimensions, shape, "{name}");
chunks.sort_by(|a, b| a.offsets.cmp(&b.offsets));
let got: Vec<&[u64]> = chunks.iter().map(|c| &c.offsets[..shape.len()]).collect();
assert_eq!(got, origins, "{name}");
for c in &chunks {
assert!(
(10_000..40_000).contains(&c.chunk_size),
"{name}: stored size {} of chunk {:?}",
c.chunk_size,
c.offsets
);
assert_eq!(c.filter_mask, 0);
}
}
}
/// `f64` values of `sel` in dataset `name` of `file`.
fn sel(file: &File, name: &str, start: &[u64], count: &[u64]) -> Vec<f64> {
let ds = file.dataset(name).unwrap();
let s = Selection::Hyperslab {
start: start.to_vec(),
stride: vec![1; start.len()],
count: count.to_vec(),
block: vec![1; start.len()],
};
ds.read_f64_selection(&s).unwrap()
}
fn f(range: std::ops::Range<i32>) -> Vec<f64> {
range.map(f64::from).collect()
}
/// Reads that touch a few elements of each 4 GiB chunk: written values, the
/// fill value (-1) next to them, and the edge of the dataset.
#[test]
fn filtered_huge_chunks_read() {
if !heavy() {
return;
}
let file = File::open(fixture()).unwrap();
let mut first = f(0..10);
first.extend([-1.0; 2]);
for name in ["single", "farray", "earray"] {
assert_eq!(sel(&file, name, &[0], &[12]), first, "{name}");
}
assert_eq!(sel(&file, "single", &[N - 2], &[2]), [-1.0; 2]);
let mut edge = vec![-1.0; 2];
edge.extend(f(100..110));
for name in ["farray", "earray"] {
assert_eq!(sel(&file, name, &[N - 2], &[12]), edge, "{name}");
}
assert_eq!(sel(&file, "btree2", &[0, 0], &[1, 10]), f(0..10));
assert_eq!(sel(&file, "btree2", &[1, N], &[1, 10]), f(300..310));
assert_eq!(
sel(&file, "btree2", &[0, N - 1], &[2, 2]),
[-1.0, 100.0, -1.0, 300.0]
);
}
/// A directory under `tests/scratch/` (on disk: the sparse files must not
/// land on a tmpfs `/tmp`), removed when dropped.
fn scratch() -> tempfile::TempDir {
let root = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/scratch");
std::fs::create_dir_all(&root).unwrap();
tempfile::tempdir_in(root).unwrap()
}
/// Have h5py (libhdf5 2.x) write `fixtures/gen_huge_chunks.py`'s file for
/// `mode` into `dir`; `None` when there is no h5py (a failure under
/// `CLAWHDF5_REQUIRE_INTEROP=1`).
fn generate(dir: &Path, mode: &str) -> Option<PathBuf> {
if !python_available() {
assert!(
!interop_required(),
"CLAWHDF5_REQUIRE_INTEROP=1 but python3 with h5py is not available"
);
eprintln!("SKIP: python3 with h5py not available");
return None;
}
let script = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/fixtures/gen_huge_chunks.py");
let path = dir.join(format!("{mode}.h5"));
let out = Command::new(python())
.arg(&script)
.arg(mode)
.arg(&path)
.output()
.unwrap();
assert!(
out.status.success(),
"gen_huge_chunks.py {mode} failed:\n{}",
String::from_utf8_lossy(&out.stderr)
);
Some(path)
}
/// Positioned reads of a file (no mmap), counting the bytes read.
struct Counting {
file: std::fs::File,
len: u64,
read: std::sync::atomic::AtomicU64,
}
impl clawhdf5_format::storage::Storage for Counting {
fn read_at(
&self,
offset: u64,
len: usize,
) -> Result<std::borrow::Cow<'_, [u8]>, clawhdf5_format::error::FormatError> {
use std::os::unix::fs::FileExt;
let len = len.min(usize::try_from(self.len.saturating_sub(offset)).unwrap_or(usize::MAX));
let mut buf = vec![0; len];
self.file.read_exact_at(&mut buf, offset).unwrap();
self.read
.fetch_add(len as u64, std::sync::atomic::Ordering::Relaxed);
Ok(buf.into())
}
fn len(&self) -> u64 {
self.len
}
}
fn check_unfiltered(file: &File) {
let mut first = f(0..10);
first.extend([0.0; 2]);
for name in ["single", "implicit", "farray", "earray"] {
assert_eq!(sel(file, name, &[0], &[12]), first, "{name}");
}
assert_eq!(sel(file, "single", &[N - 2], &[2]), [0.0; 2]);
let mut edge = vec![0.0; 2];
edge.extend(f(100..110));
for name in ["implicit", "farray", "earray"] {
assert_eq!(sel(file, name, &[N - 2], &[12]), edge, "{name}");
}
let mut last = vec![0.0; 2];
last.extend(f(500..510));
assert_eq!(sel(file, "implicit", &[2 * N - 12], &[12]), last);
assert_eq!(sel(file, "btree2", &[0, 0], &[1, 10]), f(0..10));
assert_eq!(sel(file, "btree2", &[1, N], &[1, 10]), f(300..310));
assert_eq!(
sel(file, "btree2", &[0, N - 1], &[2, 2]),
[0.0, 100.0, 0.0, 300.0]
);
}
/// Unfiltered chunks of 4 GiB + 8 bytes in every index libhdf5 gives them
/// (Single Chunk, Implicit, Fixed Array, Extensible Array, v2 B-tree), in a
/// sparse file h5py writes: read through a memory map, and through
/// positioned reads, where a selection reads only the rows it needs.
#[test]
fn unfiltered_huge_chunks_read() {
if !heavy() {
return;
}
let dir = scratch();
let Some(path) = generate(dir.path(), "unfiltered") else {
return;
};
check_unfiltered(&File::open(&path).unwrap());
let file = std::fs::File::open(&path).unwrap();
let len = file.metadata().unwrap().len();
assert!(len > 40 << 30, "{len}");
let storage = std::sync::Arc::new(Counting {
file,
len,
read: 0.into(),
});
let positioned = File::open_storage(storage.clone()).unwrap();
check_unfiltered(&positioned);
// Every read above together: metadata and a few rows, not 4 GiB chunks.
let read = storage.read.load(std::sync::atomic::Ordering::Relaxed);
assert!(read < 1 << 20, "{read} bytes read");
}
/// `FileEditor` does not rewrite chunks of 4 GiB or more: writing values,
/// or a resize that prunes or allocates chunks, is refused before anything
/// is written. Growing the extent and setting attributes still work.
#[test]
fn editor_refuses_rewriting_huge_chunks() {
let dir = tempfile::tempdir().unwrap();
let path = dir.path().join("huge.h5");
std::fs::copy(fixture(), &path).unwrap();
let before = std::fs::read(&path).unwrap();
let mut ed = clawhdf5::FileEditor::open(&path).unwrap();
let one = Selection::Hyperslab {
start: vec![0],
stride: vec![1],
count: vec![1],
block: vec![1],
};
for name in ["single", "farray", "earray"] {
let err = ed.write_values(name, &one, &[5.0f64]).unwrap_err();
assert!(
matches!(&err, clawhdf5::Error::Unsupported(m) if m.contains("4 GiB")),
"{name}: {err:?}"
);
}
// Shrinking prunes and fills chunks: refused.
for (name, shape) in [("earray", vec![10]), ("btree2", vec![1, N + 10])] {
let err = ed.resize(name, &shape).unwrap_err();
assert!(
matches!(&err, clawhdf5::Error::Unsupported(m) if m.contains("4 GiB")),
"{name}: {err:?}"
);
}
assert_eq!(
std::fs::read(&path).unwrap(),
before,
"a refused edit wrote"
);
// Growing without early allocation touches no chunk: the dataspace
// changes, as libhdf5's `H5Dset_extent` changes it.
ed.resize("earray", &[N + 20]).unwrap();
ed.resize("btree2", &[3, N + 10]).unwrap();
ed.set_attr("earray", "note", &clawhdf5::AttrValue::F64(1.5))
.unwrap();
let file = ed.reader().unwrap();
assert_eq!(file.dataset("earray").unwrap().shape().unwrap(), [N + 20]);
assert_eq!(
file.dataset("btree2").unwrap().shape().unwrap(),
[3, N + 10]
);
assert!(matches!(
file.dataset("earray").unwrap().attr("note").unwrap(),
Some(clawhdf5::AttrValue::F64(v)) if v == 1.5
));
}
/// The Fixed and Extensible Array structures clawhdf5 builds for 4 GiB
/// chunks are libhdf5's byte for byte: built at the fixture's addresses
/// from the fixture's chunks, they match what libhdf5 2.0.0 wrote (the
/// header's own address fields aside, which point where each writer put
/// the next block).
#[test]
fn huge_chunk_array_indexes_match_libhdf5() {
use clawhdf5_format::chunked_write::WrittenChunk;
let data = std::fs::read(fixture()).unwrap();
let u64_at = |at: usize| u64::from_le_bytes(data[at..at + 8].try_into().unwrap());
for name in ["farray", "earray"] {
let (_, layout, _, mut chunks) = layout_of(&data, name);
let DataLayout::Chunked {
btree_address: Some(hdr),
..
} = layout
else {
panic!("{name}: {layout:?}");
};
let hdr = hdr as usize;
chunks.sort_by_key(|c| c.offsets[0]);
let slots: Vec<Option<WrittenChunk>> = chunks
.iter()
.map(|c| {
Some(WrittenChunk {
address: c.address,
compressed_size: c.chunk_size,
raw_size: N * 8,
filter_mask: c.filter_mask,
})
})
.collect();
if name == "farray" {
let ours = clawhdf5_format::chunked_write::build_fixed_array_at(
&slots, 8, 8, true, hdr as u64,
);
// FAHD up to (not including) the data block address.
assert_eq!(&ours[..16], &data[hdr..hdr + 16], "FAHD");
let dblk = u64_at(hdr + 16) as usize;
let fadb = &ours[28..];
assert_eq!(fadb, &data[dblk..dblk + fadb.len()], "FADB");
} else {
let ours = clawhdf5_format::ea_writer::build_extensible_array_at(
&slots, 8, 8, true, hdr as u64,
);
// EAHD up to (not including) the index block address.
assert_eq!(&ours[..60], &data[hdr..hdr + 60], "EAHD");
let iblk = u64_at(hdr + 60) as usize;
let eaib = &ours[72..];
assert_eq!(eaib, &data[iblk..iblk + eaib.len()], "EAIB");
}
}
}
/// h5dump from libhdf5 2.x, when `CLAWHDF5_H5DUMP2` names one (Debian's
/// h5dump is 1.14, which cannot read layout message version 5).
fn h5dump2() -> Option<PathBuf> {
std::env::var_os("CLAWHDF5_H5DUMP2").map(PathBuf::from)
}
/// clawhdf5 writes chunks of 4 GiB + 8 bytes (deflated) in every index it
/// uses — Single Chunk, Fixed Array, Extensible Array, v2 B-tree — with
/// layout message version 5 and 8-byte stored sizes, as libhdf5 2.x does;
/// h5py (libhdf5 2.x) and h5dump 2.x read them. Each dataset holds ten
/// values in one 4 GiB chunk, so writing it holds one 4 GiB chunk.
#[test]
fn writer_huge_chunks_round_trip() {
if !heavy() {
return;
}
const UNLIM: u64 = u64::MAX;
let dir = scratch();
let path = dir.path().join("ours.h5");
let values: Vec<f64> = (0..10).map(f64::from).collect();
let mut b = clawhdf5::FileBuilder::new();
for (name, shape, max, chunk) in [
("single", vec![10], vec![N], vec![N]),
("farray", vec![10], vec![N + 10], vec![N]),
("earray", vec![10], vec![UNLIM], vec![N]),
("btree2", vec![1, 10], vec![UNLIM, UNLIM], vec![1, N]),
] {
b.create_dataset(name)
.with_f64_data(&values)
.with_shape(&shape)
.with_maxshape(&max)
.with_chunks(&chunk)
.with_deflate(6);
}
b.write(&path).unwrap();
let data = std::fs::read(&path).unwrap();
assert!(data.len() < 64 << 20, "{} bytes", data.len());
let mut stored = Vec::new();
for (name, index) in [("single", 1), ("farray", 3), ("earray", 4), ("btree2", 5)] {
let (version, layout, _, chunks) = layout_of(&data, name);
assert_eq!(version, 5, "{name}");
assert!(
matches!(layout, DataLayout::Chunked { chunk_index_type: Some(t), .. } if t == index),
"{name}: {layout:?}"
);
assert_eq!(chunks.len(), 1, "{name}");
stored.push((name, chunks[0].chunk_size));
}
let file = File::open(&path).unwrap();
let mut expect = values.clone();
expect.extend([0.0; 2]);
for name in ["single", "farray", "earray"] {
let got = file.dataset(name).unwrap().read_f64().unwrap();
assert_eq!(got, values, "{name}");
assert_eq!(sel(&file, name, &[0], &[10]), values, "{name}");
}
assert_eq!(sel(&file, "btree2", &[0, 3], &[1, 7]), f(3..10));
if python_available() {
let out = Command::new(python())
.arg("-c")
.arg(
r#"
import sys, h5py, numpy as np
with h5py.File(sys.argv[1], "r") as f:
for name in ("single", "farray", "earray", "btree2"):
d = f[name]
assert d.chunks[-1] == 2**29 + 1, (name, d.chunks)
got = d[0] if d.ndim == 2 else d[:]
assert list(got) == list(np.arange(10.0)), (name, got)
print("ok")
"#,
)
.arg(&path)
.output()
.unwrap();
assert!(
out.status.success(),
"h5py:\n{}",
String::from_utf8_lossy(&out.stderr)
);
} else {
assert!(!interop_required(), "h5py not available");
}
// h5dump 2.2.0 reads the layouts and walks every index: the storage
// size it reports is the chunk's. (It cannot print the values: its
// deflate filter fails on any chunk over 4 GiB, libhdf5's own included,
// with "memory allocation failed for deflate uncompression".)
if let Some(h5dump) = h5dump2() {
for (name, size) in stored {
let out = Command::new(&h5dump)
.args(["-H", "-p", "-d", name])
.arg(&path)
.output()
.unwrap();
let text = format!(
"{}{}",
String::from_utf8_lossy(&out.stdout),
String::from_utf8_lossy(&out.stderr)
);
assert!(out.status.success(), "h5dump {name}:\n{text}");
assert!(text.contains("536870913 )"), "{name}:\n{text}");
assert!(text.contains(&format!("SIZE {size} ")), "{name}:\n{text}");
assert!(!text.contains("rror"), "{name}:\n{text}");
}
}
}