Files
clawhdf5/crates/clawhdf5/tests/h5py_chunked_read_tests.rs
T
osobhandClaude Opus 5.5 d074385944 fix(format): read partial edge chunks stored unfiltered
Layout message v4 flag bit 0 (H5D_CHUNK_DONT_FILTER_PARTIAL_CHUNKS, set
with H5Pset_chunk_opts) makes libhdf5 store every chunk that extends past
the dataset's extent without the filter pipeline, while its filter mask
still reads 0. The parser ignored the flag, so readers tried to inflate
raw bytes: libhdf5's own h5fc_edge_v3.h5 failed with "deflate: ...
unknown compression method".

DataLayout::Chunked gains dont_filter_partial_edge_chunks (always false
for v3), and list_chunks — the one place every read path gets its chunk
list from — marks such partial chunks as having skipped every filter, so
the full, cached, indexed, parallel and selection readers all copy them
as-is. chunked_write.rs gets `..` in one exhaustive test pattern for the
new field.

Regression: libhdf5_edge_chunk_fixture_reads (h5fc_edge_v3.h5 from the
HDF5 tools test files, committed as a 2.5 KB fixture), and
h5py_unfiltered_partial_edge_chunks_read (the flag set through h5py's
bundled libhdf5 via ctypes, as h5py has no binding for it: fixed array,
extensible array and B-tree v2 indexes, 1-D and 2-D, plus a hyperslab
of the last chunk), and v4_chunked_dont_filter_partial_edge_chunks_flag.

Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
2026-09-25 21:08:36 -05:00

346 lines
13 KiB
Rust

//! Chunked-read regressions against files written by h5py / libhdf5.
//!
//! Each test builds its input with h5py (or uses a small committed fixture
//! when h5py cannot produce the feature) and compares clawhdf5's read with the
//! known contents. Tests are skipped if python3 with h5py is not available,
//! unless `CLAWHDF5_REQUIRE_INTEROP=1`.
use std::process::Command;
use clawhdf5::File;
use clawhdf5_format::selection::Selection;
/// The Python interpreter to drive interop checks with (see
/// `h5py_interop_tests.rs`).
fn python() -> String {
std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string())
}
fn interop_required() -> bool {
std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1")
}
fn python_available() -> bool {
Command::new(python())
.args(["-c", "import h5py, numpy"])
.output()
.map(|o| o.status.success())
.unwrap_or(false)
}
macro_rules! skip_if_no_python {
() => {
if !python_available() {
assert!(
!interop_required(),
"CLAWHDF5_REQUIRE_INTEROP=1 but python3 with h5py is not available"
);
eprintln!("SKIP: python3 with h5py not available");
return;
}
};
}
fn run_python(script: &str) {
let output = Command::new(python())
.args(["-c", script])
.output()
.expect("failed to run python3");
if !output.status.success() {
let stderr = String::from_utf8_lossy(&output.stderr);
let stdout = String::from_utf8_lossy(&output.stdout);
panic!("Python script failed:\nSTDOUT: {stdout}\nSTDERR: {stderr}");
}
}
// ---------------------------------------------------------------------------
// Files with 4-byte addresses (superblock size-of-offsets = 4)
// ---------------------------------------------------------------------------
/// The chunk B-tree (v1, type 1) stores each chunk offset in its keys as a
/// fixed 8-byte value whatever the file's size-of-offsets. Reading them with
/// the offset width misparsed every key in a 4-byte-offset file: unfiltered
/// datasets came back as zeros and filtered ones failed to inflate.
#[test]
fn h5py_four_byte_offsets_chunked_reads() {
skip_if_no_python!();
let dir = tempfile::tempdir().unwrap();
let path = dir.path().join("sizes_4.h5");
let p = path.display().to_string();
run_python(&format!(
r#"
import h5py, numpy as np
for name, lengths in (("{p}", 4), ("{p}.l8", 8)):
fcpl = h5py.h5p.create(h5py.h5p.FILE_CREATE)
fcpl.set_sizes(4, lengths)
fid = h5py.h5f.create(name.encode(), h5py.h5f.ACC_TRUNC, fcpl=fcpl)
with h5py.File(fid) as f:
f.create_dataset("plain", data=np.arange(100.0), chunks=(10,))
f.create_dataset("gzip", data=np.arange(100.0), chunks=(10,), compression="gzip")
f.create_dataset("grid", data=np.arange(35 * 13, dtype="<i4").reshape(35, 13),
chunks=(8, 5))
f.create_dataset("grid_gzip", data=np.arange(35 * 13, dtype="<i4").reshape(35, 13),
chunks=(8, 5), compression="gzip", shuffle=True)
"#
));
let expect: Vec<f64> = (0..100).map(f64::from).collect();
let grid: Vec<i32> = (0..35 * 13).collect();
for name in [p.clone(), format!("{p}.l8")] {
let file = File::open(&name).unwrap();
for ds in ["plain", "gzip"] {
assert_eq!(
file.dataset(ds).unwrap().read_f64().unwrap(),
expect,
"{name}:{ds}"
);
}
for ds in ["grid", "grid_gzip"] {
assert_eq!(
file.dataset(ds).unwrap().read_i32().unwrap(),
grid,
"{name}:{ds}"
);
}
}
}
// ---------------------------------------------------------------------------
// Per-chunk filter masks
// ---------------------------------------------------------------------------
/// A chunk's filter mask has one bit per pipeline filter: bit i set means
/// filter i was not applied to that chunk. Any nonzero mask used to skip the
/// whole pipeline, so a chunk that skipped only gzip was handed back still
/// shuffled.
#[test]
fn h5py_partial_filter_mask_skips_only_masked_filters() {
skip_if_no_python!();
let dir = tempfile::tempdir().unwrap();
let path = dir.path().join("mask.h5");
let p = path.display().to_string();
run_python(&format!(
r#"
import h5py, numpy as np, zlib
def shuffle(b, es):
a = np.frombuffer(b, dtype=np.uint8).reshape(-1, es)
return a.T.copy().tobytes()
with h5py.File("{p}", "w") as f:
# shuffle (0) + gzip (1). Even chunks: both applied. Odd chunks: mask
# 0b10, gzip skipped, shuffle applied. Chunk 3: mask 0b11, raw.
ds = f.create_dataset("shuf_gzip", shape=(32,), chunks=(8,), dtype="<i4",
compression="gzip", shuffle=True)
for i in range(4):
raw = np.arange(1000 + i * 8, 1008 + i * 8, dtype="<i4").tobytes()
if i == 3:
ds.id.write_direct_chunk((i * 8,), raw, filter_mask=0b11)
elif i % 2:
ds.id.write_direct_chunk((i * 8,), shuffle(raw, 4), filter_mask=0b10)
else:
ds.id.write_direct_chunk((i * 8,), zlib.compress(shuffle(raw, 4)), filter_mask=0)
# shuffle (0) + gzip (1) on a 2-D dataset, chunk (0, 1) skips shuffle only.
ds = f.create_dataset("grid", shape=(8, 8), chunks=(4, 4), dtype="<f8",
compression="gzip", shuffle=True)
full = np.arange(64, dtype="<f8").reshape(8, 8)
for r in (0, 4):
for c in (0, 4):
raw = np.ascontiguousarray(full[r:r + 4, c:c + 4]).tobytes()
if (r, c) == (0, 4):
ds.id.write_direct_chunk((r, c), zlib.compress(raw), filter_mask=0b01)
else:
ds.id.write_direct_chunk((r, c), zlib.compress(shuffle(raw, 8)), filter_mask=0)
assert (f["grid"][...] == full).all()
assert (f["shuf_gzip"][...] == np.arange(1000, 1032)).all()
"#
));
let file = File::open(&path).unwrap();
let want: Vec<i32> = (1000..1032).collect();
assert_eq!(file.dataset("shuf_gzip").unwrap().read_i32().unwrap(), want);
let grid: Vec<f64> = (0..64).map(f64::from).collect();
assert_eq!(file.dataset("grid").unwrap().read_f64().unwrap(), grid);
// The selection path decodes chunks on its own.
let part = file
.dataset("shuf_gzip")
.unwrap()
.read_selection(&Selection::Hyperslab {
start: vec![6],
stride: vec![1],
count: vec![1],
block: vec![20],
})
.unwrap();
let want: Vec<u8> = (1006..1026i32).flat_map(i32::to_le_bytes).collect();
assert_eq!(part, want);
}
// ---------------------------------------------------------------------------
// Size-changing filters ahead of a codec
// ---------------------------------------------------------------------------
/// Fletcher32 placed before the compressor (NetCDF-4's ordering) makes the
/// codec's decoded output 4 bytes larger than the chunk. The decompression
/// cap was the chunk size for every stage, so these files failed with
/// "deflate: output exceeds size limit".
#[test]
fn h5py_fletcher32_before_deflate_reads() {
skip_if_no_python!();
let dir = tempfile::tempdir().unwrap();
let path = dir.path().join("fletcher_first.h5");
let p = path.display().to_string();
run_python(&format!(
r#"
import h5py, numpy as np
arr = np.sin(np.arange(5000) / 50.0)
grid = np.arange(37 * 21, dtype="<i4").reshape(37, 21)
ids = {{"fletcher32": 3, "shuffle": 2, "deflate": 1}}
with h5py.File("{p}", "w") as f:
for name, steps, data, chunk in (
("fl_shuf_gzip", ("fletcher32", "shuffle", "deflate"), arr, (500,)),
("fl_gzip", ("fletcher32", "deflate"), arr, (500,)),
("shuf_fl_gzip", ("shuffle", "fletcher32", "deflate"), arr, (500,)),
("grid_fl_shuf_gzip", ("fletcher32", "shuffle", "deflate"), grid, (8, 5)),
):
dcpl = h5py.h5p.create(h5py.h5p.DATASET_CREATE)
dcpl.set_chunk(chunk)
for s in steps:
if s == "deflate":
dcpl.set_deflate(4)
elif s == "shuffle":
dcpl.set_shuffle()
else:
dcpl.set_fletcher32()
tid = h5py.h5t.py_create(data.dtype)
space = h5py.h5s.create_simple(data.shape)
d = h5py.h5d.create(f.id, name.encode(), tid, space, dcpl=dcpl)
d.write(h5py.h5s.ALL, h5py.h5s.ALL, np.ascontiguousarray(data))
order = [d.get_create_plist().get_filter(i)[0] for i in range(len(steps))]
assert order == [ids[s] for s in steps], order
"#
));
let file = File::open(&path).unwrap();
let arr: Vec<f64> = (0..5000).map(|i| (f64::from(i) / 50.0).sin()).collect();
for name in ["fl_shuf_gzip", "fl_gzip", "shuf_fl_gzip"] {
let values = file.dataset(name).unwrap().read_f64().unwrap();
assert_eq!(values.len(), arr.len(), "{name}");
for (i, (v, w)) in values.iter().zip(&arr).enumerate() {
assert!((v - w).abs() < 1e-12, "{name}[{i}]: {v} vs {w}");
}
}
let grid: Vec<i32> = (0..37 * 21).collect();
assert_eq!(
file.dataset("grid_fl_shuf_gzip")
.unwrap()
.read_i32()
.unwrap(),
grid
);
}
// ---------------------------------------------------------------------------
// "Don't filter partial edge chunks"
// ---------------------------------------------------------------------------
/// libhdf5's own test file for `H5Pset_chunk_opts(H5D_CHUNK_DONT_FILTER_PARTIAL_CHUNKS)`:
/// a 12x6 f32 dataset in 5x5 gzip chunks whose edge chunks are stored raw
/// with a filter mask of 0. Reading it tried to inflate the raw edge chunks
/// ("deflate: ... unknown compression method").
#[test]
fn libhdf5_edge_chunk_fixture_reads() {
let path = concat!(
env!("CARGO_MANIFEST_DIR"),
"/tests/fixtures/h5fc_edge_v3.h5"
);
let file = File::open(path).unwrap();
let ds = file.dataset("DSET_EDGE").unwrap();
assert_eq!(ds.shape().unwrap(), vec![12, 6]);
assert_eq!(ds.read_f32().unwrap(), vec![100.0f32; 72]);
}
/// The same layout flag on datasets with varied contents, set through the
/// libhdf5 that h5py ships (h5py has no binding for `H5Pset_chunk_opts`).
#[test]
fn h5py_unfiltered_partial_edge_chunks_read() {
skip_if_no_python!();
let dir = tempfile::tempdir().unwrap();
let path = dir.path().join("edge.h5");
let p = path.display().to_string();
let script = format!(
r#"
import ctypes, glob, os, sys
import h5py, numpy as np
here = os.path.dirname(h5py.__file__)
libs = glob.glob(os.path.join(here, "..", "h5py.libs", "libhdf5-*.so*"))
libs += glob.glob(os.path.join(here, ".dylibs", "libhdf5*.dylib"))
if not libs:
print("NO_LIBHDF5")
sys.exit(0)
lib = ctypes.CDLL(libs[0])
lib.H5Pset_chunk_opts.argtypes = [ctypes.c_int64, ctypes.c_uint]
def make(f, name, data, chunk, maxshape, shuffle):
dcpl = h5py.h5p.create(h5py.h5p.DATASET_CREATE)
dcpl.set_chunk(chunk)
if shuffle:
dcpl.set_shuffle()
dcpl.set_deflate(6)
assert lib.H5Pset_chunk_opts(dcpl.id, 0x0002) >= 0
space = h5py.h5s.create_simple(data.shape, maxshape)
d = h5py.h5d.create(f.id, name.encode(), h5py.h5t.py_create(data.dtype), space, dcpl=dcpl)
d.write(h5py.h5s.ALL, h5py.h5s.ALL, np.ascontiguousarray(data))
with h5py.File("{p}", "w") as f:
line = np.sin(np.arange(1000) / 7.0)
grid = np.arange(37 * 53, dtype="<f8").reshape(37, 53) * 0.5
make(f, "fixed_1d", line, (64,), None, False)
make(f, "ea_1d", line, (64,), (h5py.h5s.UNLIMITED,), True)
make(f, "bt2_2d", grid, (8, 8), (h5py.h5s.UNLIMITED,) * 2, True)
make(f, "fixed_2d", grid, (8, 8), None, True)
print("OK")
"#
);
let out = Command::new(python())
.args(["-c", &script])
.output()
.expect("failed to run python3");
assert!(
out.status.success(),
"{}",
String::from_utf8_lossy(&out.stderr)
);
if String::from_utf8_lossy(&out.stdout).contains("NO_LIBHDF5") {
eprintln!("SKIP: h5py's bundled libhdf5 not found (needed for H5Pset_chunk_opts)");
return;
}
let file = File::open(&path).unwrap();
let line: Vec<f64> = (0..1000).map(|i| (f64::from(i) / 7.0).sin()).collect();
for name in ["fixed_1d", "ea_1d"] {
let values = file.dataset(name).unwrap().read_f64().unwrap();
assert_eq!(values.len(), line.len(), "{name}");
for (i, (v, w)) in values.iter().zip(&line).enumerate() {
assert!((v - w).abs() < 1e-12, "{name}[{i}]: {v} vs {w}");
}
}
let grid: Vec<f64> = (0..37 * 53).map(|i| f64::from(i) * 0.5).collect();
for name in ["bt2_2d", "fixed_2d"] {
assert_eq!(
file.dataset(name).unwrap().read_f64().unwrap(),
grid,
"{name}"
);
}
// A selection touching only the last (partial, unfiltered) chunk.
let tail = file
.dataset("fixed_1d")
.unwrap()
.read_selection(&Selection::Hyperslab {
start: vec![990],
stride: vec![1],
count: vec![1],
block: vec![10],
})
.unwrap();
let want: Vec<u8> = line[990..].iter().flat_map(|v| v.to_le_bytes()).collect();
assert_eq!(tail, want);
}