LazyStorage holds the blocks of a remote file fetched so far. An operation runs as passes over it: a read that misses records the missing blocks and fails, the pass's result is dropped whatever it is (a parser may have caught the error and carried on), and the caller fetches the reported ranges and re-runs the pass. No block is evicted while an operation is in flight, so every pass that does not finish asks for at least one new block and the operation ends. Blocks sit in an LRU with a byte budget, trimmed between operations, bulk (raw data) blocks first. Reader::open_storage opens a file through any Storage, and variable- length strings resolve through the file's storage instead of File::as_bytes, which panics for a file not in memory. tests/lazy.rs compares, file by file, what the viewer can show (kinds, listings, attributes, info, whole reads and hyperslabs) read lazily with the facade's range-storage path and the in-memory reader: files built here at 512 B to 1 MiB blocks, the h5py/netCDF4 fixture, and with CLAWHDF5_WASM_CORPUS the conformance corpus (656 files agree). A listing plus a small read of a 48 MB file fetches 3 ranges. Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
392 lines
14 KiB
Rust
392 lines
14 KiB
Rust
//! The restartable ("NeedBytes") reader against the in-memory one: every
|
|
//! file must list, describe and read the same through a [`LazyStorage`]
|
|
//! that starts empty and is fed only the ranges its passes ask for, as the
|
|
//! browser's `openUrl` feeds it from HTTP range requests.
|
|
//!
|
|
//! - Files written here with `FileBuilder`, at several block sizes (512 B
|
|
//! blocks make almost every structure read a miss).
|
|
//! - The h5py/netCDF4 fixture of `examples/wasm-viewer/test/make_fixture.py`
|
|
//! (skipped without h5py, unless `CLAWHDF5_REQUIRE_INTEROP=1`;
|
|
//! `CLAWHDF5_PYTHON` names the interpreter).
|
|
//! - `CLAWHDF5_WASM_CORPUS=dir[:dir...]`: every HDF5 file under those
|
|
//! directories up to 64 MiB (e.g. `conformance/.cache/corpus`).
|
|
//!
|
|
//! Also the request budget: listing and reading one small dataset of a large
|
|
//! file fetches a few blocks, not the file.
|
|
|
|
use std::ops::Range;
|
|
use std::path::{Path, PathBuf};
|
|
use std::process::Command;
|
|
use std::sync::Arc;
|
|
|
|
use clawhdf5::{AttrValue, FileBuilder};
|
|
use clawhdf5_format::storage::CountingStorage;
|
|
use clawhdf5_wasm::core::{Hyperslab, Kind, Reader};
|
|
use clawhdf5_wasm::lazy::{LazyConfig, LazyStorage};
|
|
|
|
/// The API the JavaScript side calls, one operation at a time.
|
|
trait Api {
|
|
fn call<T>(&self, op: impl Fn(&Reader) -> T) -> T;
|
|
}
|
|
|
|
struct Local(Reader);
|
|
|
|
impl Api for Local {
|
|
fn call<T>(&self, op: impl Fn(&Reader) -> T) -> T {
|
|
op(&self.0)
|
|
}
|
|
}
|
|
|
|
/// A lazily read file and the "server" it fetches from.
|
|
struct Lazy {
|
|
data: Arc<Vec<u8>>,
|
|
storage: Arc<LazyStorage>,
|
|
reader: Reader,
|
|
}
|
|
|
|
fn fetch(data: &[u8], r: Range<u64>) -> Result<Vec<u8>, String> {
|
|
Ok(data[r.start as usize..r.end as usize].to_vec())
|
|
}
|
|
|
|
impl Lazy {
|
|
/// Open as `openUrl` does: the first block comes with the probe that
|
|
/// learns the length, then the open is run until it has its bytes.
|
|
fn open(data: Vec<u8>, config: LazyConfig) -> Result<Lazy, String> {
|
|
let data = Arc::new(data);
|
|
let storage = Arc::new(LazyStorage::new(data.len() as u64, config));
|
|
let first = (storage.config().block_size as usize).min(data.len());
|
|
storage.supply(0, &data[..first])?;
|
|
let s = storage.clone();
|
|
let reader =
|
|
storage.run_blocking(|| Reader::open_storage(s.clone()), |r| fetch(&data, r))??;
|
|
Ok(Lazy {
|
|
data,
|
|
storage,
|
|
reader,
|
|
})
|
|
}
|
|
}
|
|
|
|
impl Api for Lazy {
|
|
fn call<T>(&self, op: impl Fn(&Reader) -> T) -> T {
|
|
self.storage
|
|
.run_blocking(|| op(&self.reader), |r| fetch(&self.data, r))
|
|
.expect("serving from memory cannot fail")
|
|
}
|
|
}
|
|
|
|
/// Everything the viewer can show of a file, as text: each object's kind,
|
|
/// listing, attributes (and attribute errors), dataset info, whole value
|
|
/// and a hyperslab — or the error each gives.
|
|
fn transcript(api: &impl Api) -> Vec<String> {
|
|
let mut out = Vec::new();
|
|
let mut todo = vec![("/".to_string(), 0usize)];
|
|
while let Some((path, depth)) = todo.pop() {
|
|
if out.len() > 4000 {
|
|
out.push("... (truncated)".into());
|
|
break;
|
|
}
|
|
let kind = api.call(|r| r.kind(&path));
|
|
out.push(format!("{path}: {kind:?}"));
|
|
out.push(format!("{path} attrs: {:?}", api.call(|r| r.attrs(&path))));
|
|
match kind {
|
|
Ok(Kind::Group) => {
|
|
let list = api.call(|r| r.list(&path));
|
|
out.push(format!("{path} list: {list:?}"));
|
|
if let Ok(children) = list
|
|
&& depth < 12
|
|
{
|
|
for c in children.into_iter().rev() {
|
|
let child = if path == "/" {
|
|
format!("/{}", c.name)
|
|
} else {
|
|
format!("{path}/{}", c.name)
|
|
};
|
|
todo.push((child, depth + 1));
|
|
}
|
|
}
|
|
}
|
|
Ok(Kind::Dataset) => {
|
|
let info = api.call(|r| r.info(&path));
|
|
out.push(format!("{path} info: {info:?}"));
|
|
let Ok(info) = info else { continue };
|
|
let n = info
|
|
.shape
|
|
.iter()
|
|
.chain(&info.element_shape)
|
|
.try_fold(1u64, |a, &d| a.checked_mul(d));
|
|
if n.is_none_or(|n| n > 4_000_000) {
|
|
out.push(format!("{path}: not read ({n:?} values)"));
|
|
continue;
|
|
}
|
|
out.push(format!(
|
|
"{path} read: {:?}",
|
|
api.call(|r| r.read(&path, None))
|
|
));
|
|
if !info.shape.is_empty() && info.shape.iter().all(|&d| d > 1) {
|
|
let slab = Hyperslab {
|
|
start: info.shape.iter().map(|_| 1).collect(),
|
|
count: info.shape.iter().map(|&d| d / 2).collect(),
|
|
stride: None,
|
|
block: None,
|
|
};
|
|
let part = api.call(|r| r.read(&path, Some(&slab)));
|
|
out.push(format!("{path} slab: {part:?}"));
|
|
}
|
|
}
|
|
Err(_) => {}
|
|
}
|
|
}
|
|
out
|
|
}
|
|
|
|
/// The lazy transcript of `data` at `block` bytes per block equals the
|
|
/// transcript of the same file through a range storage that has every byte
|
|
/// (`CountingStorage`: the facade's `Storage` path, the one the lazy reader
|
|
/// takes), and agrees with the in-memory one: the same values, and an error
|
|
/// wherever it has one (a malformed file can fail at a different check,
|
|
/// with a different message, when read by ranges). Returns what the lazy
|
|
/// reader fetched and its transcript.
|
|
fn check_equal(name: &str, data: &[u8], block: u64) -> (u64, u64, Vec<String>) {
|
|
let ctx = format!("{name} (blocks of {block} B)");
|
|
let ranged = Reader::open_storage(Arc::new(CountingStorage::new(data.to_vec())));
|
|
let local = Reader::open(data.to_vec());
|
|
let lazy = Lazy::open(data.to_vec(), config(block));
|
|
let (ranged, local, lazy) = match (ranged, local, lazy) {
|
|
(Ok(r), Ok(l), Ok(z)) => (r, l, z),
|
|
(Err(r), Err(_), Err(z)) => {
|
|
assert_eq!(z, r, "{ctx}: open error");
|
|
return (0, 0, Vec::new());
|
|
}
|
|
(r, l, z) => panic!(
|
|
"{ctx}: opens differently: ranged {:?}, in memory {:?}, lazily {:?}",
|
|
r.err(),
|
|
l.err(),
|
|
z.err()
|
|
),
|
|
};
|
|
let got = transcript(&lazy);
|
|
let want = transcript(&Local(ranged));
|
|
for (i, (w, g)) in want.iter().zip(&got).enumerate() {
|
|
assert_eq!(g, w, "{ctx}, line {i}");
|
|
}
|
|
assert_eq!(got.len(), want.len(), "{ctx}: transcript length");
|
|
let local = transcript(&Local(local));
|
|
for (i, (l, g)) in local.iter().zip(&got).enumerate() {
|
|
let both_errors = match (l.split_once("Err("), g.split_once("Err(")) {
|
|
(Some((a, _)), Some((b, _))) => a == b,
|
|
_ => false,
|
|
};
|
|
assert!(
|
|
l == g || both_errors,
|
|
"{ctx}, line {i}: in memory\n {l}\nlazily\n {g}"
|
|
);
|
|
}
|
|
assert_eq!(got.len(), local.len(), "{ctx}: transcript length");
|
|
let st = lazy.storage.stats();
|
|
(st.requests, st.bytes_fetched, got)
|
|
}
|
|
|
|
fn config(block: u64) -> LazyConfig {
|
|
LazyConfig {
|
|
block_size: block,
|
|
// A small budget, so eviction between operations is exercised.
|
|
capacity: 16 * block,
|
|
max_request: 8 * block,
|
|
}
|
|
}
|
|
|
|
fn builder_file() -> Vec<u8> {
|
|
let mut b = FileBuilder::new();
|
|
b.create_dataset("grid")
|
|
.with_f64_data(&(0..20_000).map(f64::from).collect::<Vec<_>>())
|
|
.with_shape(&[100, 200])
|
|
.with_chunks(&[10, 25])
|
|
.with_deflate(4);
|
|
b.create_dataset("contiguous")
|
|
.with_i32_data(&(0..50_000).collect::<Vec<_>>());
|
|
b.create_dataset("bytes").with_u8_data(&[1, 2, 250]);
|
|
let mut g = b.create_group("sensors");
|
|
for i in 0..40 {
|
|
g.create_dataset(&format!("t{i}"))
|
|
.with_f32_data(&[i as f32, 1.5, -2.25]);
|
|
}
|
|
g.set_attr("location", AttrValue::String("lab".into()));
|
|
b.add_group(g.finish());
|
|
b.set_attr("version", AttrValue::I64(3));
|
|
b.set_attr("scale", AttrValue::F64Array(vec![0.5, 2.0]));
|
|
b.finish().unwrap()
|
|
}
|
|
|
|
#[test]
|
|
fn builder_files_read_the_same_at_every_block_size() {
|
|
let data = builder_file();
|
|
for block in [512, 4096, 1 << 20] {
|
|
let (requests, _, lines) = check_equal("builder", &data, block);
|
|
assert!(requests > 0);
|
|
// The transcript covers every object, values included.
|
|
assert!(lines.iter().any(|l| l.starts_with("/grid read: Ok")));
|
|
assert!(lines.iter().any(|l| l.starts_with("/grid slab: Ok")));
|
|
assert!(lines.iter().any(|l| l.starts_with("/sensors/t39 read: Ok")));
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn garbage_fails_to_open_as_in_memory() {
|
|
check_equal("zeros", &[0u8; 5000], 512);
|
|
check_equal("empty", &[], 512);
|
|
let mut cut = builder_file();
|
|
cut.truncate(cut.len() / 3);
|
|
check_equal("truncated", &cut, 512);
|
|
}
|
|
|
|
/// Listing a large file and reading one small dataset fetches a few blocks,
|
|
/// not the file.
|
|
#[test]
|
|
fn a_small_read_of_a_large_file_fetches_a_few_blocks() {
|
|
let mut b = FileBuilder::new();
|
|
b.create_dataset("small").with_f64_data(&[1.0, 2.0, 3.0]);
|
|
// 48 MB of raw data, written after the small dataset's metadata.
|
|
b.create_dataset("big")
|
|
.with_f64_data(&(0..6_000_000).map(f64::from).collect::<Vec<_>>());
|
|
let mut g = b.create_group("group");
|
|
g.create_dataset("inner").with_i32_data(&[7, 8]);
|
|
b.add_group(g.finish());
|
|
let data = b.finish().unwrap();
|
|
let lazy = Lazy::open(data.clone(), LazyConfig::default()).unwrap();
|
|
let list = lazy.call(|r| r.list("/")).unwrap();
|
|
assert_eq!(list.len(), 3);
|
|
assert_eq!(
|
|
format!("{:?}", lazy.call(|r| r.read("/small", None)).unwrap().data),
|
|
"F64([1.0, 2.0, 3.0])"
|
|
);
|
|
assert_eq!(
|
|
format!(
|
|
"{:?}",
|
|
lazy.call(|r| r.read("/group/inner", None)).unwrap().data
|
|
),
|
|
"I32([7, 8])"
|
|
);
|
|
// A window of the big dataset reads only its block(s).
|
|
let slab = Hyperslab {
|
|
start: vec![3_000_000],
|
|
count: vec![4],
|
|
stride: None,
|
|
block: None,
|
|
};
|
|
assert_eq!(
|
|
format!(
|
|
"{:?}",
|
|
lazy.call(|r| r.read("/big", Some(&slab))).unwrap().data
|
|
),
|
|
"F64([3000000.0, 3000001.0, 3000002.0, 3000003.0])"
|
|
);
|
|
let st = lazy.storage.stats();
|
|
eprintln!("{} bytes: {st:?}", data.len());
|
|
assert!(st.requests <= 6, "{st:?}");
|
|
assert!(st.bytes_fetched <= 6 << 20, "{st:?}");
|
|
assert!(st.bytes_fetched * 8 < data.len() as u64, "{st:?}");
|
|
}
|
|
|
|
fn python() -> String {
|
|
std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string())
|
|
}
|
|
|
|
fn python_available() -> bool {
|
|
Command::new(python())
|
|
.args(["-c", "import h5py, netCDF4, numpy"])
|
|
.output()
|
|
.is_ok_and(|o| o.status.success())
|
|
}
|
|
|
|
#[test]
|
|
fn h5py_and_netcdf4_files_read_the_same_lazily() {
|
|
if !python_available() {
|
|
assert!(
|
|
!std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1"),
|
|
"CLAWHDF5_REQUIRE_INTEROP=1 but {} lacks h5py/netCDF4/numpy",
|
|
python()
|
|
);
|
|
eprintln!("skipping: {} lacks h5py/netCDF4/numpy", python());
|
|
return;
|
|
}
|
|
let dir = tempfile::tempdir().unwrap();
|
|
let generator = Path::new(env!("CARGO_MANIFEST_DIR"))
|
|
.join("../../examples/wasm-viewer/test/make_fixture.py");
|
|
let out = Command::new(python())
|
|
.arg(&generator)
|
|
.arg(dir.path())
|
|
.output()
|
|
.unwrap();
|
|
assert!(
|
|
out.status.success(),
|
|
"{}",
|
|
String::from_utf8_lossy(&out.stderr)
|
|
);
|
|
for name in ["fixture.h5", "fixture.nc"] {
|
|
let data = std::fs::read(dir.path().join(name)).unwrap();
|
|
for block in [512, 64 * 1024] {
|
|
let (_, _, lines) = check_equal(name, &data, block);
|
|
assert!(lines.iter().filter(|l| l.contains(" read: Ok")).count() >= 2);
|
|
}
|
|
}
|
|
}
|
|
|
|
fn hdf5_files(dir: &Path, out: &mut Vec<PathBuf>) {
|
|
let Ok(entries) = std::fs::read_dir(dir) else {
|
|
return;
|
|
};
|
|
for e in entries.flatten() {
|
|
let p = e.path();
|
|
if p.is_dir() {
|
|
hdf5_files(&p, out);
|
|
} else if std::fs::read(&p)
|
|
.ok()
|
|
.is_some_and(|b| b.len() <= 64 << 20 && is_hdf5(&b))
|
|
{
|
|
out.push(p);
|
|
}
|
|
}
|
|
}
|
|
|
|
/// The HDF5 signature at 0 or a power-of-two user-block offset.
|
|
fn is_hdf5(b: &[u8]) -> bool {
|
|
const SIG: &[u8] = b"\x89HDF\r\n\x1a\n";
|
|
let mut at = 0usize;
|
|
loop {
|
|
if b.get(at..at + 8) == Some(SIG) {
|
|
return true;
|
|
}
|
|
at = if at == 0 { 512 } else { at * 2 };
|
|
if at >= b.len() {
|
|
return false;
|
|
}
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn corpus_files_read_the_same_lazily() {
|
|
let Ok(dirs) = std::env::var("CLAWHDF5_WASM_CORPUS") else {
|
|
eprintln!("CLAWHDF5_WASM_CORPUS not set; skipping the corpus");
|
|
return;
|
|
};
|
|
let mut files = Vec::new();
|
|
for d in std::env::split_paths(&dirs) {
|
|
hdf5_files(&d, &mut files);
|
|
}
|
|
files.sort();
|
|
assert!(!files.is_empty(), "no HDF5 files under {dirs}");
|
|
let (mut requests, mut bytes, mut total) = (0u64, 0u64, 0u64);
|
|
for f in &files {
|
|
let data = std::fs::read(f).unwrap();
|
|
total += data.len() as u64;
|
|
let (r, b, _) = check_equal(&f.display().to_string(), &data, 64 * 1024);
|
|
requests += r;
|
|
bytes += b;
|
|
}
|
|
eprintln!(
|
|
"{} files ({total} bytes): {requests} requests, {bytes} bytes fetched",
|
|
files.len()
|
|
);
|
|
}
|