Merge branch 'fix/p0-chunked-read' into fix/phase0-correctness
This commit is contained in:
@@ -0,0 +1,87 @@
|
||||
//! A `File` is `Send + Sync` and keeps one chunk cache for all its datasets.
|
||||
//! Threads reading different chunked datasets through the same `File` must
|
||||
//! each get their own dataset's data.
|
||||
|
||||
use std::sync::Arc;
|
||||
|
||||
use clawhdf5::{File, FileBuilder};
|
||||
|
||||
const DATASETS: usize = 24;
|
||||
const THREADS: usize = 16;
|
||||
const ROUNDS: usize = 40;
|
||||
|
||||
/// Contents of dataset `k`: distinct from every other dataset's, element for
|
||||
/// element, so any chunk served from the wrong dataset shows.
|
||||
fn values(k: usize, n: usize) -> Vec<f64> {
|
||||
(0..n).map(|i| (k * 100_000 + i) as f64).collect()
|
||||
}
|
||||
|
||||
fn build() -> File {
|
||||
let mut b = FileBuilder::new();
|
||||
for k in 0..DATASETS {
|
||||
let ds = b.create_dataset(&format!("d{k:02}"));
|
||||
match k % 3 {
|
||||
// 1-D, compressed: chunk offsets 0, 8, 16, ... in every dataset.
|
||||
0 => {
|
||||
ds.with_f64_data(&values(k, 64)).with_shape(&[64]);
|
||||
ds.with_chunks(&[8]).with_deflate(1);
|
||||
}
|
||||
// 1-D, shuffle + compressed, a different length.
|
||||
1 => {
|
||||
ds.with_f64_data(&values(k, 40)).with_shape(&[40]);
|
||||
ds.with_chunks(&[8]).with_shuffle().with_deflate(1);
|
||||
}
|
||||
// 2-D, compressed: coordinates (0,0), (0,4), (4,0), ... overlap
|
||||
// the other datasets' in the first dimension.
|
||||
_ => {
|
||||
ds.with_f64_data(&values(k, 64)).with_shape(&[8, 8]);
|
||||
ds.with_chunks(&[4, 4]).with_deflate(1);
|
||||
}
|
||||
}
|
||||
}
|
||||
File::from_bytes(b.finish().unwrap()).unwrap()
|
||||
}
|
||||
|
||||
fn expected(k: usize) -> Vec<f64> {
|
||||
values(k, if k % 3 == 1 { 40 } else { 64 })
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn threads_reading_different_datasets_get_their_own_chunks() {
|
||||
let file = Arc::new(build());
|
||||
// Sequential sanity check first.
|
||||
for k in 0..DATASETS {
|
||||
let got = file.dataset(&format!("d{k:02}")).unwrap().read_f64();
|
||||
assert_eq!(got.unwrap(), expected(k), "sequential d{k:02}");
|
||||
}
|
||||
|
||||
let handles: Vec<_> = (0..THREADS)
|
||||
.map(|t| {
|
||||
let file = Arc::clone(&file);
|
||||
std::thread::spawn(move || {
|
||||
let mut wrong = Vec::new();
|
||||
for round in 0..ROUNDS {
|
||||
let k = (t * 7 + round * 5) % DATASETS;
|
||||
let name = format!("d{k:02}");
|
||||
match file.dataset(&name).unwrap().read_f64() {
|
||||
Ok(v) if v == expected(k) => {}
|
||||
Ok(v) => wrong.push(format!("{name}: wrong data, first {:?}", &v[..4])),
|
||||
Err(e) => wrong.push(format!("{name}: {e}")),
|
||||
}
|
||||
}
|
||||
wrong
|
||||
})
|
||||
})
|
||||
.collect();
|
||||
let failures: Vec<String> = handles
|
||||
.into_iter()
|
||||
.flat_map(|h| h.join().unwrap())
|
||||
.collect();
|
||||
assert!(
|
||||
failures.is_empty(),
|
||||
"{} of {} concurrent reads were wrong, e.g. {:?}",
|
||||
failures.len(),
|
||||
THREADS * ROUNDS,
|
||||
&failures[..failures.len().min(5)]
|
||||
);
|
||||
}
|
||||
BIN
Binary file not shown.
@@ -0,0 +1,345 @@
|
||||
//! Chunked-read regressions against files written by h5py / libhdf5.
|
||||
//!
|
||||
//! Each test builds its input with h5py (or uses a small committed fixture
|
||||
//! when h5py cannot produce the feature) and compares clawhdf5's read with the
|
||||
//! known contents. Tests are skipped if python3 with h5py is not available,
|
||||
//! unless `CLAWHDF5_REQUIRE_INTEROP=1`.
|
||||
|
||||
use std::process::Command;
|
||||
|
||||
use clawhdf5::File;
|
||||
use clawhdf5_format::selection::Selection;
|
||||
|
||||
/// The Python interpreter to drive interop checks with (see
|
||||
/// `h5py_interop_tests.rs`).
|
||||
fn python() -> String {
|
||||
std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string())
|
||||
}
|
||||
|
||||
fn interop_required() -> bool {
|
||||
std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1")
|
||||
}
|
||||
|
||||
fn python_available() -> bool {
|
||||
Command::new(python())
|
||||
.args(["-c", "import h5py, numpy"])
|
||||
.output()
|
||||
.map(|o| o.status.success())
|
||||
.unwrap_or(false)
|
||||
}
|
||||
|
||||
macro_rules! skip_if_no_python {
|
||||
() => {
|
||||
if !python_available() {
|
||||
assert!(
|
||||
!interop_required(),
|
||||
"CLAWHDF5_REQUIRE_INTEROP=1 but python3 with h5py is not available"
|
||||
);
|
||||
eprintln!("SKIP: python3 with h5py not available");
|
||||
return;
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
fn run_python(script: &str) {
|
||||
let output = Command::new(python())
|
||||
.args(["-c", script])
|
||||
.output()
|
||||
.expect("failed to run python3");
|
||||
if !output.status.success() {
|
||||
let stderr = String::from_utf8_lossy(&output.stderr);
|
||||
let stdout = String::from_utf8_lossy(&output.stdout);
|
||||
panic!("Python script failed:\nSTDOUT: {stdout}\nSTDERR: {stderr}");
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Files with 4-byte addresses (superblock size-of-offsets = 4)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// The chunk B-tree (v1, type 1) stores each chunk offset in its keys as a
|
||||
/// fixed 8-byte value whatever the file's size-of-offsets. Reading them with
|
||||
/// the offset width misparsed every key in a 4-byte-offset file: unfiltered
|
||||
/// datasets came back as zeros and filtered ones failed to inflate.
|
||||
#[test]
|
||||
fn h5py_four_byte_offsets_chunked_reads() {
|
||||
skip_if_no_python!();
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let path = dir.path().join("sizes_4.h5");
|
||||
let p = path.display().to_string();
|
||||
run_python(&format!(
|
||||
r#"
|
||||
import h5py, numpy as np
|
||||
for name, lengths in (("{p}", 4), ("{p}.l8", 8)):
|
||||
fcpl = h5py.h5p.create(h5py.h5p.FILE_CREATE)
|
||||
fcpl.set_sizes(4, lengths)
|
||||
fid = h5py.h5f.create(name.encode(), h5py.h5f.ACC_TRUNC, fcpl=fcpl)
|
||||
with h5py.File(fid) as f:
|
||||
f.create_dataset("plain", data=np.arange(100.0), chunks=(10,))
|
||||
f.create_dataset("gzip", data=np.arange(100.0), chunks=(10,), compression="gzip")
|
||||
f.create_dataset("grid", data=np.arange(35 * 13, dtype="<i4").reshape(35, 13),
|
||||
chunks=(8, 5))
|
||||
f.create_dataset("grid_gzip", data=np.arange(35 * 13, dtype="<i4").reshape(35, 13),
|
||||
chunks=(8, 5), compression="gzip", shuffle=True)
|
||||
"#
|
||||
));
|
||||
|
||||
let expect: Vec<f64> = (0..100).map(f64::from).collect();
|
||||
let grid: Vec<i32> = (0..35 * 13).collect();
|
||||
for name in [p.clone(), format!("{p}.l8")] {
|
||||
let file = File::open(&name).unwrap();
|
||||
for ds in ["plain", "gzip"] {
|
||||
assert_eq!(
|
||||
file.dataset(ds).unwrap().read_f64().unwrap(),
|
||||
expect,
|
||||
"{name}:{ds}"
|
||||
);
|
||||
}
|
||||
for ds in ["grid", "grid_gzip"] {
|
||||
assert_eq!(
|
||||
file.dataset(ds).unwrap().read_i32().unwrap(),
|
||||
grid,
|
||||
"{name}:{ds}"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Per-chunk filter masks
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// A chunk's filter mask has one bit per pipeline filter: bit i set means
|
||||
/// filter i was not applied to that chunk. Any nonzero mask used to skip the
|
||||
/// whole pipeline, so a chunk that skipped only gzip was handed back still
|
||||
/// shuffled.
|
||||
#[test]
|
||||
fn h5py_partial_filter_mask_skips_only_masked_filters() {
|
||||
skip_if_no_python!();
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let path = dir.path().join("mask.h5");
|
||||
let p = path.display().to_string();
|
||||
run_python(&format!(
|
||||
r#"
|
||||
import h5py, numpy as np, zlib
|
||||
def shuffle(b, es):
|
||||
a = np.frombuffer(b, dtype=np.uint8).reshape(-1, es)
|
||||
return a.T.copy().tobytes()
|
||||
with h5py.File("{p}", "w") as f:
|
||||
# shuffle (0) + gzip (1). Even chunks: both applied. Odd chunks: mask
|
||||
# 0b10, gzip skipped, shuffle applied. Chunk 3: mask 0b11, raw.
|
||||
ds = f.create_dataset("shuf_gzip", shape=(32,), chunks=(8,), dtype="<i4",
|
||||
compression="gzip", shuffle=True)
|
||||
for i in range(4):
|
||||
raw = np.arange(1000 + i * 8, 1008 + i * 8, dtype="<i4").tobytes()
|
||||
if i == 3:
|
||||
ds.id.write_direct_chunk((i * 8,), raw, filter_mask=0b11)
|
||||
elif i % 2:
|
||||
ds.id.write_direct_chunk((i * 8,), shuffle(raw, 4), filter_mask=0b10)
|
||||
else:
|
||||
ds.id.write_direct_chunk((i * 8,), zlib.compress(shuffle(raw, 4)), filter_mask=0)
|
||||
# shuffle (0) + gzip (1) on a 2-D dataset, chunk (0, 1) skips shuffle only.
|
||||
ds = f.create_dataset("grid", shape=(8, 8), chunks=(4, 4), dtype="<f8",
|
||||
compression="gzip", shuffle=True)
|
||||
full = np.arange(64, dtype="<f8").reshape(8, 8)
|
||||
for r in (0, 4):
|
||||
for c in (0, 4):
|
||||
raw = np.ascontiguousarray(full[r:r + 4, c:c + 4]).tobytes()
|
||||
if (r, c) == (0, 4):
|
||||
ds.id.write_direct_chunk((r, c), zlib.compress(raw), filter_mask=0b01)
|
||||
else:
|
||||
ds.id.write_direct_chunk((r, c), zlib.compress(shuffle(raw, 8)), filter_mask=0)
|
||||
assert (f["grid"][...] == full).all()
|
||||
assert (f["shuf_gzip"][...] == np.arange(1000, 1032)).all()
|
||||
"#
|
||||
));
|
||||
|
||||
let file = File::open(&path).unwrap();
|
||||
let want: Vec<i32> = (1000..1032).collect();
|
||||
assert_eq!(file.dataset("shuf_gzip").unwrap().read_i32().unwrap(), want);
|
||||
let grid: Vec<f64> = (0..64).map(f64::from).collect();
|
||||
assert_eq!(file.dataset("grid").unwrap().read_f64().unwrap(), grid);
|
||||
// The selection path decodes chunks on its own.
|
||||
let part = file
|
||||
.dataset("shuf_gzip")
|
||||
.unwrap()
|
||||
.read_selection(&Selection::Hyperslab {
|
||||
start: vec![6],
|
||||
stride: vec![1],
|
||||
count: vec![1],
|
||||
block: vec![20],
|
||||
})
|
||||
.unwrap();
|
||||
let want: Vec<u8> = (1006..1026i32).flat_map(i32::to_le_bytes).collect();
|
||||
assert_eq!(part, want);
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Size-changing filters ahead of a codec
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Fletcher32 placed before the compressor (NetCDF-4's ordering) makes the
|
||||
/// codec's decoded output 4 bytes larger than the chunk. The decompression
|
||||
/// cap was the chunk size for every stage, so these files failed with
|
||||
/// "deflate: output exceeds size limit".
|
||||
#[test]
|
||||
fn h5py_fletcher32_before_deflate_reads() {
|
||||
skip_if_no_python!();
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let path = dir.path().join("fletcher_first.h5");
|
||||
let p = path.display().to_string();
|
||||
run_python(&format!(
|
||||
r#"
|
||||
import h5py, numpy as np
|
||||
arr = np.sin(np.arange(5000) / 50.0)
|
||||
grid = np.arange(37 * 21, dtype="<i4").reshape(37, 21)
|
||||
ids = {{"fletcher32": 3, "shuffle": 2, "deflate": 1}}
|
||||
with h5py.File("{p}", "w") as f:
|
||||
for name, steps, data, chunk in (
|
||||
("fl_shuf_gzip", ("fletcher32", "shuffle", "deflate"), arr, (500,)),
|
||||
("fl_gzip", ("fletcher32", "deflate"), arr, (500,)),
|
||||
("shuf_fl_gzip", ("shuffle", "fletcher32", "deflate"), arr, (500,)),
|
||||
("grid_fl_shuf_gzip", ("fletcher32", "shuffle", "deflate"), grid, (8, 5)),
|
||||
):
|
||||
dcpl = h5py.h5p.create(h5py.h5p.DATASET_CREATE)
|
||||
dcpl.set_chunk(chunk)
|
||||
for s in steps:
|
||||
if s == "deflate":
|
||||
dcpl.set_deflate(4)
|
||||
elif s == "shuffle":
|
||||
dcpl.set_shuffle()
|
||||
else:
|
||||
dcpl.set_fletcher32()
|
||||
tid = h5py.h5t.py_create(data.dtype)
|
||||
space = h5py.h5s.create_simple(data.shape)
|
||||
d = h5py.h5d.create(f.id, name.encode(), tid, space, dcpl=dcpl)
|
||||
d.write(h5py.h5s.ALL, h5py.h5s.ALL, np.ascontiguousarray(data))
|
||||
order = [d.get_create_plist().get_filter(i)[0] for i in range(len(steps))]
|
||||
assert order == [ids[s] for s in steps], order
|
||||
"#
|
||||
));
|
||||
|
||||
let file = File::open(&path).unwrap();
|
||||
let arr: Vec<f64> = (0..5000).map(|i| (f64::from(i) / 50.0).sin()).collect();
|
||||
for name in ["fl_shuf_gzip", "fl_gzip", "shuf_fl_gzip"] {
|
||||
let values = file.dataset(name).unwrap().read_f64().unwrap();
|
||||
assert_eq!(values.len(), arr.len(), "{name}");
|
||||
for (i, (v, w)) in values.iter().zip(&arr).enumerate() {
|
||||
assert!((v - w).abs() < 1e-12, "{name}[{i}]: {v} vs {w}");
|
||||
}
|
||||
}
|
||||
let grid: Vec<i32> = (0..37 * 21).collect();
|
||||
assert_eq!(
|
||||
file.dataset("grid_fl_shuf_gzip")
|
||||
.unwrap()
|
||||
.read_i32()
|
||||
.unwrap(),
|
||||
grid
|
||||
);
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// "Don't filter partial edge chunks"
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// libhdf5's own test file for `H5Pset_chunk_opts(H5D_CHUNK_DONT_FILTER_PARTIAL_CHUNKS)`:
|
||||
/// a 12x6 f32 dataset in 5x5 gzip chunks whose edge chunks are stored raw
|
||||
/// with a filter mask of 0. Reading it tried to inflate the raw edge chunks
|
||||
/// ("deflate: ... unknown compression method").
|
||||
#[test]
|
||||
fn libhdf5_edge_chunk_fixture_reads() {
|
||||
let path = concat!(
|
||||
env!("CARGO_MANIFEST_DIR"),
|
||||
"/tests/fixtures/h5fc_edge_v3.h5"
|
||||
);
|
||||
let file = File::open(path).unwrap();
|
||||
let ds = file.dataset("DSET_EDGE").unwrap();
|
||||
assert_eq!(ds.shape().unwrap(), vec![12, 6]);
|
||||
assert_eq!(ds.read_f32().unwrap(), vec![100.0f32; 72]);
|
||||
}
|
||||
|
||||
/// The same layout flag on datasets with varied contents, set through the
|
||||
/// libhdf5 that h5py ships (h5py has no binding for `H5Pset_chunk_opts`).
|
||||
#[test]
|
||||
fn h5py_unfiltered_partial_edge_chunks_read() {
|
||||
skip_if_no_python!();
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let path = dir.path().join("edge.h5");
|
||||
let p = path.display().to_string();
|
||||
let script = format!(
|
||||
r#"
|
||||
import ctypes, glob, os, sys
|
||||
import h5py, numpy as np
|
||||
here = os.path.dirname(h5py.__file__)
|
||||
libs = glob.glob(os.path.join(here, "..", "h5py.libs", "libhdf5-*.so*"))
|
||||
libs += glob.glob(os.path.join(here, ".dylibs", "libhdf5*.dylib"))
|
||||
if not libs:
|
||||
print("NO_LIBHDF5")
|
||||
sys.exit(0)
|
||||
lib = ctypes.CDLL(libs[0])
|
||||
lib.H5Pset_chunk_opts.argtypes = [ctypes.c_int64, ctypes.c_uint]
|
||||
def make(f, name, data, chunk, maxshape, shuffle):
|
||||
dcpl = h5py.h5p.create(h5py.h5p.DATASET_CREATE)
|
||||
dcpl.set_chunk(chunk)
|
||||
if shuffle:
|
||||
dcpl.set_shuffle()
|
||||
dcpl.set_deflate(6)
|
||||
assert lib.H5Pset_chunk_opts(dcpl.id, 0x0002) >= 0
|
||||
space = h5py.h5s.create_simple(data.shape, maxshape)
|
||||
d = h5py.h5d.create(f.id, name.encode(), h5py.h5t.py_create(data.dtype), space, dcpl=dcpl)
|
||||
d.write(h5py.h5s.ALL, h5py.h5s.ALL, np.ascontiguousarray(data))
|
||||
with h5py.File("{p}", "w") as f:
|
||||
line = np.sin(np.arange(1000) / 7.0)
|
||||
grid = np.arange(37 * 53, dtype="<f8").reshape(37, 53) * 0.5
|
||||
make(f, "fixed_1d", line, (64,), None, False)
|
||||
make(f, "ea_1d", line, (64,), (h5py.h5s.UNLIMITED,), True)
|
||||
make(f, "bt2_2d", grid, (8, 8), (h5py.h5s.UNLIMITED,) * 2, True)
|
||||
make(f, "fixed_2d", grid, (8, 8), None, True)
|
||||
print("OK")
|
||||
"#
|
||||
);
|
||||
let out = Command::new(python())
|
||||
.args(["-c", &script])
|
||||
.output()
|
||||
.expect("failed to run python3");
|
||||
assert!(
|
||||
out.status.success(),
|
||||
"{}",
|
||||
String::from_utf8_lossy(&out.stderr)
|
||||
);
|
||||
if String::from_utf8_lossy(&out.stdout).contains("NO_LIBHDF5") {
|
||||
eprintln!("SKIP: h5py's bundled libhdf5 not found (needed for H5Pset_chunk_opts)");
|
||||
return;
|
||||
}
|
||||
|
||||
let file = File::open(&path).unwrap();
|
||||
let line: Vec<f64> = (0..1000).map(|i| (f64::from(i) / 7.0).sin()).collect();
|
||||
for name in ["fixed_1d", "ea_1d"] {
|
||||
let values = file.dataset(name).unwrap().read_f64().unwrap();
|
||||
assert_eq!(values.len(), line.len(), "{name}");
|
||||
for (i, (v, w)) in values.iter().zip(&line).enumerate() {
|
||||
assert!((v - w).abs() < 1e-12, "{name}[{i}]: {v} vs {w}");
|
||||
}
|
||||
}
|
||||
let grid: Vec<f64> = (0..37 * 53).map(|i| f64::from(i) * 0.5).collect();
|
||||
for name in ["bt2_2d", "fixed_2d"] {
|
||||
assert_eq!(
|
||||
file.dataset(name).unwrap().read_f64().unwrap(),
|
||||
grid,
|
||||
"{name}"
|
||||
);
|
||||
}
|
||||
// A selection touching only the last (partial, unfiltered) chunk.
|
||||
let tail = file
|
||||
.dataset("fixed_1d")
|
||||
.unwrap()
|
||||
.read_selection(&Selection::Hyperslab {
|
||||
start: vec![990],
|
||||
stride: vec![1],
|
||||
count: vec![1],
|
||||
block: vec![10],
|
||||
})
|
||||
.unwrap();
|
||||
let want: Vec<u8> = line[990..].iter().flat_map(|v| v.to_le_bytes()).collect();
|
||||
assert_eq!(tail, want);
|
||||
}
|
||||
Reference in New Issue
Block a user