Files
clawhdf5/crates/clawhdf5/tests/partial_read_equivalence.rs
T
osobhandClaude Opus 5.5 7a2e61ebd1 Huge chunks: selection tests with fill values, LZ4 over 256 MiB, wasm32 check
A selection of a chunked dataset with a fill value is compared with the
full read in every index, through a map and through positioned reads. An
LZ4 chunk larger than 256 MiB is bounded by the chunk size, not refused.
The wasm package test reads the 4 GiB-chunk fixture and gets a clean
error in every index.

Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
2026-09-28 23:44:48 -05:00

285 lines
9.5 KiB
Rust

//! Selection reads must return exactly what a full read followed by element
//! extraction returns — for every layout, rank and selection shape — while
//! touching only what the selection needs.
use clawhdf5::{File, FileBuilder};
use clawhdf5_format::selection::Selection;
struct Rng(u64);
impl Rng {
fn next(&mut self) -> u64 {
self.0 = self.0.wrapping_add(0x9E37_79B9_7F4A_7C15);
let mut z = self.0;
z = (z ^ (z >> 30)).wrapping_mul(0xBF58_476D_1CE4_E5B9);
z = (z ^ (z >> 27)).wrapping_mul(0x94D0_49BB_1331_11EB);
z ^ (z >> 31)
}
fn below(&mut self, n: u64) -> u64 {
self.next() % n.max(1)
}
}
/// Row-major reference extraction from a full read.
fn reference(full: &[i32], dims: &[u64], selection: &Selection) -> Vec<i32> {
let strides: Vec<u64> = (0..dims.len())
.map(|d| dims[d + 1..].iter().product())
.collect();
let at =
|coord: &[u64]| full[coord.iter().zip(&strides).map(|(c, s)| c * s).sum::<u64>() as usize];
match selection {
Selection::Points(points) => points.iter().map(|p| at(p)).collect(),
Selection::Hyperslab {
start,
stride,
count,
block,
} => {
// Selected indices per dimension, then their cartesian product.
let per_dim: Vec<Vec<u64>> = (0..dims.len())
.map(|d| {
(0..count[d])
.flat_map(|c| (0..block[d]).map(move |b| (c, b)))
.map(|(c, b)| start[d] + c * stride[d] + b)
.collect()
})
.collect();
let mut out = Vec::new();
let mut idx = vec![0usize; dims.len()];
loop {
let coord: Vec<u64> = idx
.iter()
.enumerate()
.map(|(d, &i)| per_dim[d][i])
.collect();
out.push(at(&coord));
let mut d = dims.len();
loop {
if d == 0 {
return out;
}
d -= 1;
idx[d] += 1;
if idx[d] < per_dim[d].len() {
break;
}
idx[d] = 0;
}
}
}
_ => unreachable!(),
}
}
fn random_hyperslab(rng: &mut Rng, dims: &[u64]) -> Selection {
let mut start = Vec::new();
let mut stride = Vec::new();
let mut count = Vec::new();
let mut block = Vec::new();
for &dim in dims {
let b = 1 + rng.below(3);
let st = b + rng.below(4); // stride >= block: no overlap
let s = rng.below(dim - b + 1);
let max_count = (dim - s - b) / st + 1;
let c = 1 + rng.below(max_count.min(6));
start.push(s);
stride.push(st);
count.push(c);
block.push(b);
}
Selection::Hyperslab {
start,
stride,
count,
block,
}
}
#[test]
fn selection_reads_match_full_reads_for_every_layout() {
let dir = tempfile::tempdir().unwrap();
let mut rng = Rng(7);
// (dims, chunk dims)
let shapes: [(&[u64], &[u64]); 3] = [
(&[97], &[10]),
(&[41, 53], &[8, 9]),
(&[11, 13, 17], &[4, 5, 6]),
];
for (dims, chunks) in shapes {
let n: u64 = dims.iter().product();
let data: Vec<i32> = (0..n as i32).map(|v| v * 3 - 7).collect();
let path = dir.path().join(format!("r{}.h5", dims.len()));
let mut builder = FileBuilder::new();
builder
.create_dataset("contiguous")
.with_i32_data(&data)
.with_shape(dims);
builder
.create_dataset("chunked")
.with_i32_data(&data)
.with_shape(dims)
.with_chunks(chunks);
builder
.create_dataset("deflated")
.with_i32_data(&data)
.with_shape(dims)
.with_chunks(chunks)
.with_deflate(3);
builder.write(&path).unwrap();
let file = File::open(&path).unwrap();
for name in ["contiguous", "chunked", "deflated"] {
let ds = file.dataset(name).unwrap();
let full = ds.read_i32().unwrap();
assert_eq!(full, data, "{name} full read");
for case in 0..60 {
let selection = if case % 5 == 4 {
let points = (0..1 + rng.below(12))
.map(|_| dims.iter().map(|&d| rng.below(d)).collect())
.collect();
Selection::Points(points)
} else {
random_hyperslab(&mut rng, dims)
};
assert_eq!(
ds.read_i32_selection(&selection).unwrap(),
reference(&full, dims, &selection),
"{name} rank {} case {case}: {selection:?}",
dims.len()
);
}
}
}
}
#[test]
fn out_of_bounds_selections_are_errors() {
let dir = tempfile::tempdir().unwrap();
let path = dir.path().join("oob.h5");
let mut builder = FileBuilder::new();
builder
.create_dataset("d")
.with_i32_data(&(0..100).collect::<Vec<i32>>())
.with_shape(&[10, 10])
.with_chunks(&[4, 4]);
builder.write(&path).unwrap();
let file = File::open(&path).unwrap();
let ds = file.dataset("d").unwrap();
let beyond = Selection::Hyperslab {
start: vec![8, 8],
stride: vec![1, 1],
count: vec![5, 5],
block: vec![1, 1],
};
use clawhdf5::Error;
use clawhdf5_format::error::FormatError;
let is_oob = |s: &Selection| {
matches!(
ds.read_i32_selection(s),
Err(Error::Format(FormatError::SelectionOutOfBounds(_)))
)
};
// Used to come back padded with zeros.
assert!(is_oob(&beyond));
// Row out of range.
assert!(is_oob(&Selection::Points(vec![vec![10, 0]])));
// Column out of range: used to wrap into the next row and return its value.
assert!(is_oob(&Selection::Points(vec![vec![0, 12]])));
// Wrong rank.
assert!(is_oob(&Selection::Points(vec![vec![3]])));
// In range is fine.
assert_eq!(
ds.read_i32_selection(&Selection::Points(vec![vec![9, 9]]))
.unwrap(),
[99]
);
}
/// A chunked dataset with a fill value and chunks never written: a
/// selection is read over a box of fill values from the chunks it touches
/// (it used to be picked out of a full read), and through positioned reads
/// an unfiltered chunk is read row by row. Both equal the full read, in
/// every chunk index h5py writes.
#[test]
fn fill_value_selections_match_full_reads() {
let python = std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".into());
let has_h5py = std::process::Command::new(&python)
.args(["-c", "import h5py"])
.output()
.is_ok_and(|o| o.status.success());
if !has_h5py {
assert!(
std::env::var("CLAWHDF5_REQUIRE_INTEROP").as_deref() != Ok("1"),
"CLAWHDF5_REQUIRE_INTEROP=1 but python3 with h5py is not available"
);
eprintln!("SKIP: python3 with h5py not available");
return;
}
let dir = tempfile::tempdir().unwrap();
let path = dir.path().join("fill.h5");
let script = r#"
import sys, h5py, numpy as np
with h5py.File(sys.argv[1], "w", libver="latest") as f:
for name, maxshape, gz in [
("farray", None, False), ("farray_gz", None, True),
("earray", (None, 23), False), ("earray_gz", (None, 23), True),
("btree2", (None, None), False), ("btree2_gz", (None, None), True),
]:
d = f.create_dataset(name, shape=(37, 23), maxshape=maxshape, chunks=(5, 4),
dtype="<i4", fillvalue=-5,
compression="gzip" if gz else None)
d[3:17, 2:11] = np.arange(14 * 9, dtype="<i4").reshape(14, 9)
d[30, 20] = 7
"#;
let out = std::process::Command::new(&python)
.args(["-c", script])
.arg(&path)
.output()
.unwrap();
assert!(
out.status.success(),
"{}",
String::from_utf8_lossy(&out.stderr)
);
let dims = [37u64, 23];
let mapped = File::open(&path).unwrap();
let storage = clawhdf5::FileStorage::open(&path).unwrap();
let positioned = File::open_storage(std::sync::Arc::new(storage)).unwrap();
let mut rng = Rng(11);
for name in [
"farray",
"farray_gz",
"earray",
"earray_gz",
"btree2",
"btree2_gz",
] {
let full = mapped.dataset(name).unwrap().read_i32().unwrap();
assert_eq!(full[0], -5, "{name}");
assert_eq!(full[3 * 23 + 2], 0, "{name}");
for case in 0..60 {
let selection = if case % 5 == 4 {
let points = (0..1 + rng.below(12))
.map(|_| dims.iter().map(|&d| rng.below(d)).collect())
.collect();
Selection::Points(points)
} else {
random_hyperslab(&mut rng, &dims)
};
let want = reference(&full, &dims, &selection);
for (how, file) in [("mapped", &mapped), ("positioned", &positioned)] {
assert_eq!(
file.dataset(name)
.unwrap()
.read_i32_selection(&selection)
.unwrap(),
want,
"{name} {how} case {case}: {selection:?}"
);
}
}
}
}