format: read a contiguous selection's runs merged across small gaps
gather_storage merged only runs that touch, so a strided selection of a contiguous dataset over a Storage became one range (and one owned Vec) per element: a stride-2 read of 32M f32 through File::open_storage made 16,777,232 read_at calls, took 2.0 s and peaked at 2.09 GB. The selection is now walked twice. The first walk checks the runs and plans spans: runs in increasing order at most 4 KiB apart (GATHER_GAP_BYTES) are read as one span up to 8 MiB (GATHER_SPAN_BYTES; a longer run is split), so nothing is stored per run. The spans are fetched in RAW_BATCH_BYTES batches while the second walk copies each run out of its span. Same checks and errors as before. The same read is now 32 reads and 0.31 s (File::open: 0.08 s). contiguous_read_interop: every h5py-checked selection is also read through File::open_storage and must give libhdf5's bytes; a new test bounds the range reads of strided, blocked, column and point selections (stride 2: at most 1 data read; 563,200 before). Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
@@ -12,9 +12,11 @@
|
||||
|
||||
use std::path::Path;
|
||||
use std::process::Command;
|
||||
use std::sync::Arc;
|
||||
|
||||
use clawhdf5::File;
|
||||
use clawhdf5_format::selection::Selection;
|
||||
use clawhdf5_format::storage::CountingStorage;
|
||||
|
||||
fn python() -> String {
|
||||
std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string())
|
||||
@@ -364,9 +366,31 @@ with h5py.File("{path}", "r") as f:
|
||||
));
|
||||
|
||||
let file = File::open(&path).unwrap();
|
||||
// The same file over a storage that serves only range reads: the
|
||||
// selections read only their runs (see `strided_selections_over_storage_
|
||||
// read_few_ranges`) and must give the same bytes.
|
||||
let remote = File::open_storage(Arc::new(CountingStorage::new(
|
||||
std::fs::read(&path).unwrap(),
|
||||
)))
|
||||
.unwrap();
|
||||
for (k, (name, code, sel)) in cases.iter().enumerate() {
|
||||
let ds = file.dataset(name).unwrap();
|
||||
let want_bytes = std::fs::read(dir.path().join(format!("sel_{k}.bin"))).unwrap();
|
||||
let rds = remote.dataset(name).unwrap();
|
||||
assert!(
|
||||
rds.read_selection(sel).unwrap() == want_bytes,
|
||||
"{name} {sel:?}: bytes over a range storage differ from libhdf5's"
|
||||
);
|
||||
assert_eq!(
|
||||
rds.read_f64_selection(sel).unwrap(),
|
||||
ds.read_f64_selection(sel).unwrap(),
|
||||
"{name} {sel:?}: f64 over a range storage"
|
||||
);
|
||||
assert_eq!(
|
||||
rds.read_i32_selection(sel).unwrap(),
|
||||
ds.read_i32_selection(sel).unwrap(),
|
||||
"{name} {sel:?}: i32 over a range storage"
|
||||
);
|
||||
let got_bytes = ds.read_selection(sel).unwrap();
|
||||
assert!(
|
||||
got_bytes == want_bytes,
|
||||
@@ -401,3 +425,64 @@ with h5py.File("{path}", "r") as f:
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// Over a storage without the file in memory, a selection of a contiguous
|
||||
/// dataset reads its runs merged across small gaps: a strided selection is
|
||||
/// a few large reads, not one per element (563 200 for the stride-2 case
|
||||
/// before), and the values are libhdf5's.
|
||||
#[test]
|
||||
fn strided_selections_over_storage_read_few_ranges() {
|
||||
skip_if_no_python!();
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let path = dir.path().join("contig.h5");
|
||||
write_file(&path);
|
||||
let storage = Arc::new(CountingStorage::new(std::fs::read(&path).unwrap()));
|
||||
let remote = File::open_storage(storage.clone()).unwrap();
|
||||
let local = File::open(&path).unwrap();
|
||||
// f4le_big is 1100 x 1024 f32: 4 KiB rows, 4.4 MB in all.
|
||||
let name = "f4le_big";
|
||||
let every_100th: Vec<Vec<u64>> = (0..1100u64 * 1024)
|
||||
.step_by(100)
|
||||
.map(|i| vec![i / 1024, i % 1024])
|
||||
.collect();
|
||||
let backwards: Vec<Vec<u64>> = every_100th.iter().rev().take(50).cloned().collect();
|
||||
// (selection, most range reads it may take)
|
||||
let cases: Vec<(Selection, u64)> = vec![
|
||||
// Every other element of every row: one read of the whole dataset.
|
||||
(slab(&[(0, 1, 1100, 1), (0, 2, 512, 1)]), 1),
|
||||
// Blocks of 3 every 7 on every other row: rows are 4 KiB apart, so
|
||||
// one read per selected row at most.
|
||||
(slab(&[(0, 2, 550, 1), (1, 7, 146, 3)]), 550),
|
||||
// Every 100th element, in order: 400-byte gaps, one read.
|
||||
(Selection::Points(every_100th), 1),
|
||||
// Points going backwards are not merged.
|
||||
(Selection::Points(backwards), 50),
|
||||
// A column: 4 KiB apart, merged.
|
||||
(slab(&[(0, 1, 1100, 1), (5, 1, 1, 1)]), 1),
|
||||
];
|
||||
let ds = remote.dataset(name).unwrap();
|
||||
for (sel, most) in cases {
|
||||
storage.reset();
|
||||
let got = ds.read_f32_selection(&sel).unwrap();
|
||||
let reads = storage.reads();
|
||||
assert_eq!(
|
||||
got,
|
||||
local
|
||||
.dataset(name)
|
||||
.unwrap()
|
||||
.read_f32_selection(&sel)
|
||||
.unwrap(),
|
||||
"{sel:?}"
|
||||
);
|
||||
// A few reads of metadata besides the data.
|
||||
assert!(reads <= most + 8, "{sel:?}: {reads} range reads");
|
||||
storage.reset();
|
||||
let bytes = ds.read_selection(&sel).unwrap();
|
||||
assert!(
|
||||
storage.reads() <= most + 8,
|
||||
"{sel:?}: {} reads",
|
||||
storage.reads()
|
||||
);
|
||||
assert_eq!(bytes.len(), got.len() * 4);
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user