A read longer than isize::MAX (2 GiB on wasm32) aborted the module in LazyStorage::assemble (capacity_overflow), taking every open file on the page with it, and a hostile server only had to claim a large length and serve a heap collection of 2 GiB + 4 KiB to get there (after fetching 2 GiB). Reading a large u8 dataset whole aborted the same way when its values were widened to 64 bits. - LazyConfig::max_fetch (openUrl option maxFetch, default 512 MiB, at most 1 GiB): a read longer than it fails at once, before anything is fetched, and an operation whose passes would fetch more than it fails before fetching (Operation::charge). assemble reserves fallibly. - Reader::read refuses a read that would use more than 1 GiB while decoding (core::MAX_READ_BYTES: stored bytes + 64-bit values + result) with an error naming readHyperslab, before reading. - openUrl refuses a file of 4 GiB or more at open on wasm32: the format code turns offsets into usize, so nothing past 4 GiB can be read there (shown by a new test: data at 3 GiB reads, a 4 GiB file is refused). maxDownload is bounded to 1 GiB. Tests: make_fixture.py writes limits.h5 (a sparse 2^28 + 1024 byte u8 dataset), hostile_vl.h5 (the reviewer's collection) and far.h5 (data at 3 GiB); test.mjs (wasm32) and tests/lazy.rs (native) check each is an error or reads, and that the module survives. Before: RuntimeError: unreachable in Node; the native test read the huge dataset and fetched 2 GiB. Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
476 lines
16 KiB
Rust
476 lines
16 KiB
Rust
//! The restartable ("NeedBytes") reader against the in-memory one: every
|
|
//! file must list, describe and read the same through a [`LazyStorage`]
|
|
//! that starts empty and is fed only the ranges its passes ask for, as the
|
|
//! browser's `openUrl` feeds it from HTTP range requests.
|
|
//!
|
|
//! - Files written here with `FileBuilder`, at several block sizes (512 B
|
|
//! blocks make almost every structure read a miss).
|
|
//! - The h5py/netCDF4 fixture of `examples/wasm-viewer/test/make_fixture.py`
|
|
//! (skipped without h5py, unless `CLAWHDF5_REQUIRE_INTEROP=1`;
|
|
//! `CLAWHDF5_PYTHON` names the interpreter).
|
|
//! - `CLAWHDF5_WASM_CORPUS=dir[:dir...]`: every HDF5 file under those
|
|
//! directories up to 64 MiB (e.g. `conformance/.cache/corpus`).
|
|
//!
|
|
//! Also the request budget: listing and reading one small dataset of a large
|
|
//! file fetches a few blocks, not the file.
|
|
|
|
use std::ops::Range;
|
|
use std::path::{Path, PathBuf};
|
|
use std::process::Command;
|
|
use std::sync::Arc;
|
|
|
|
use clawhdf5::{AttrValue, FileBuilder};
|
|
use clawhdf5_format::storage::CountingStorage;
|
|
use clawhdf5_wasm::core::{Hyperslab, Kind, Reader};
|
|
use clawhdf5_wasm::lazy::{LazyConfig, LazyStorage};
|
|
|
|
/// The API the JavaScript side calls, one operation at a time.
|
|
trait Api {
|
|
fn call<T>(&self, op: impl Fn(&Reader) -> T) -> T;
|
|
}
|
|
|
|
struct Local(Reader);
|
|
|
|
impl Api for Local {
|
|
fn call<T>(&self, op: impl Fn(&Reader) -> T) -> T {
|
|
op(&self.0)
|
|
}
|
|
}
|
|
|
|
/// A lazily read file and the "server" it fetches from.
|
|
struct Lazy {
|
|
data: Arc<Vec<u8>>,
|
|
storage: Arc<LazyStorage>,
|
|
reader: Reader,
|
|
}
|
|
|
|
fn fetch(data: &[u8], r: Range<u64>) -> Result<Vec<u8>, String> {
|
|
Ok(data[r.start as usize..r.end as usize].to_vec())
|
|
}
|
|
|
|
impl Lazy {
|
|
/// Open as `openUrl` does: the first block comes with the probe that
|
|
/// learns the length, then the open is run until it has its bytes.
|
|
fn open(data: Vec<u8>, config: LazyConfig) -> Result<Lazy, String> {
|
|
let data = Arc::new(data);
|
|
let storage = Arc::new(LazyStorage::new(data.len() as u64, config));
|
|
let first = (storage.config().block_size as usize).min(data.len());
|
|
storage.supply(0, &data[..first])?;
|
|
let s = storage.clone();
|
|
let reader =
|
|
storage.run_blocking(|| Reader::open_storage(s.clone()), |r| fetch(&data, r))??;
|
|
Ok(Lazy {
|
|
data,
|
|
storage,
|
|
reader,
|
|
})
|
|
}
|
|
}
|
|
|
|
impl Api for Lazy {
|
|
fn call<T>(&self, op: impl Fn(&Reader) -> T) -> T {
|
|
self.storage
|
|
.run_blocking(|| op(&self.reader), |r| fetch(&self.data, r))
|
|
.expect("serving from memory cannot fail")
|
|
}
|
|
}
|
|
|
|
/// Everything the viewer can show of a file, as text: each object's kind,
|
|
/// listing, attributes (and attribute errors), dataset info, whole value
|
|
/// and a hyperslab — or the error each gives.
|
|
fn transcript(api: &impl Api) -> Vec<String> {
|
|
let mut out = Vec::new();
|
|
let mut todo = vec![("/".to_string(), 0usize)];
|
|
while let Some((path, depth)) = todo.pop() {
|
|
if out.len() > 4000 {
|
|
out.push("... (truncated)".into());
|
|
break;
|
|
}
|
|
let kind = api.call(|r| r.kind(&path));
|
|
out.push(format!("{path}: {kind:?}"));
|
|
out.push(format!("{path} attrs: {:?}", api.call(|r| r.attrs(&path))));
|
|
match kind {
|
|
Ok(Kind::Group) => {
|
|
let list = api.call(|r| r.list(&path));
|
|
out.push(format!("{path} list: {list:?}"));
|
|
if let Ok(children) = list
|
|
&& depth < 12
|
|
{
|
|
for c in children.into_iter().rev() {
|
|
let child = if path == "/" {
|
|
format!("/{}", c.name)
|
|
} else {
|
|
format!("{path}/{}", c.name)
|
|
};
|
|
todo.push((child, depth + 1));
|
|
}
|
|
}
|
|
}
|
|
Ok(Kind::Dataset) => {
|
|
let info = api.call(|r| r.info(&path));
|
|
out.push(format!("{path} info: {info:?}"));
|
|
let Ok(info) = info else { continue };
|
|
let n = info
|
|
.shape
|
|
.iter()
|
|
.chain(&info.element_shape)
|
|
.try_fold(1u64, |a, &d| a.checked_mul(d));
|
|
if n.is_none_or(|n| n > 4_000_000) {
|
|
out.push(format!("{path}: not read ({n:?} values)"));
|
|
continue;
|
|
}
|
|
out.push(format!(
|
|
"{path} read: {:?}",
|
|
api.call(|r| r.read(&path, None))
|
|
));
|
|
if !info.shape.is_empty() && info.shape.iter().all(|&d| d > 1) {
|
|
let slab = Hyperslab {
|
|
start: info.shape.iter().map(|_| 1).collect(),
|
|
count: info.shape.iter().map(|&d| d / 2).collect(),
|
|
stride: None,
|
|
block: None,
|
|
};
|
|
let part = api.call(|r| r.read(&path, Some(&slab)));
|
|
out.push(format!("{path} slab: {part:?}"));
|
|
}
|
|
}
|
|
Err(_) => {}
|
|
}
|
|
}
|
|
out
|
|
}
|
|
|
|
/// The lazy transcript of `data` at `block` bytes per block equals the
|
|
/// transcript of the same file through a range storage that has every byte
|
|
/// (`CountingStorage`: the facade's `Storage` path, the one the lazy reader
|
|
/// takes), and agrees with the in-memory one: the same values, and an error
|
|
/// wherever it has one (a malformed file can fail at a different check,
|
|
/// with a different message, when read by ranges). Returns what the lazy
|
|
/// reader fetched and its transcript.
|
|
fn check_equal(name: &str, data: &[u8], block: u64) -> (u64, u64, Vec<String>) {
|
|
let ctx = format!("{name} (blocks of {block} B)");
|
|
let ranged = Reader::open_storage(Arc::new(CountingStorage::new(data.to_vec())));
|
|
let local = Reader::open(data.to_vec());
|
|
let lazy = Lazy::open(data.to_vec(), config(block));
|
|
let (ranged, local, lazy) = match (ranged, local, lazy) {
|
|
(Ok(r), Ok(l), Ok(z)) => (r, l, z),
|
|
(Err(r), Err(_), Err(z)) => {
|
|
assert_eq!(z, r, "{ctx}: open error");
|
|
return (0, 0, Vec::new());
|
|
}
|
|
(r, l, z) => panic!(
|
|
"{ctx}: opens differently: ranged {:?}, in memory {:?}, lazily {:?}",
|
|
r.err(),
|
|
l.err(),
|
|
z.err()
|
|
),
|
|
};
|
|
let got = transcript(&lazy);
|
|
let want = transcript(&Local(ranged));
|
|
for (i, (w, g)) in want.iter().zip(&got).enumerate() {
|
|
assert_eq!(g, w, "{ctx}, line {i}");
|
|
}
|
|
assert_eq!(got.len(), want.len(), "{ctx}: transcript length");
|
|
let local = transcript(&Local(local));
|
|
for (i, (l, g)) in local.iter().zip(&got).enumerate() {
|
|
let both_errors = match (l.split_once("Err("), g.split_once("Err(")) {
|
|
(Some((a, _)), Some((b, _))) => a == b,
|
|
_ => false,
|
|
};
|
|
assert!(
|
|
l == g || both_errors,
|
|
"{ctx}, line {i}: in memory\n {l}\nlazily\n {g}"
|
|
);
|
|
}
|
|
assert_eq!(got.len(), local.len(), "{ctx}: transcript length");
|
|
let st = lazy.storage.stats();
|
|
(st.requests, st.bytes_fetched, got)
|
|
}
|
|
|
|
fn config(block: u64) -> LazyConfig {
|
|
LazyConfig {
|
|
block_size: block,
|
|
// A small budget, so eviction between operations is exercised.
|
|
capacity: 16 * block,
|
|
max_request: 8 * block,
|
|
..LazyConfig::default()
|
|
}
|
|
}
|
|
|
|
fn builder_file() -> Vec<u8> {
|
|
let mut b = FileBuilder::new();
|
|
b.create_dataset("grid")
|
|
.with_f64_data(&(0..20_000).map(f64::from).collect::<Vec<_>>())
|
|
.with_shape(&[100, 200])
|
|
.with_chunks(&[10, 25])
|
|
.with_deflate(4);
|
|
b.create_dataset("contiguous")
|
|
.with_i32_data(&(0..50_000).collect::<Vec<_>>());
|
|
b.create_dataset("bytes").with_u8_data(&[1, 2, 250]);
|
|
let mut g = b.create_group("sensors");
|
|
for i in 0..40 {
|
|
g.create_dataset(&format!("t{i}"))
|
|
.with_f32_data(&[i as f32, 1.5, -2.25]);
|
|
}
|
|
g.set_attr("location", AttrValue::String("lab".into()));
|
|
b.add_group(g.finish());
|
|
b.set_attr("version", AttrValue::I64(3));
|
|
b.set_attr("scale", AttrValue::F64Array(vec![0.5, 2.0]));
|
|
b.finish().unwrap()
|
|
}
|
|
|
|
#[test]
|
|
fn builder_files_read_the_same_at_every_block_size() {
|
|
let data = builder_file();
|
|
for block in [512, 4096, 1 << 20] {
|
|
let (requests, _, lines) = check_equal("builder", &data, block);
|
|
assert!(requests > 0);
|
|
// The transcript covers every object, values included.
|
|
assert!(lines.iter().any(|l| l.starts_with("/grid read: Ok")));
|
|
assert!(lines.iter().any(|l| l.starts_with("/grid slab: Ok")));
|
|
assert!(lines.iter().any(|l| l.starts_with("/sensors/t39 read: Ok")));
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn garbage_fails_to_open_as_in_memory() {
|
|
check_equal("zeros", &[0u8; 5000], 512);
|
|
check_equal("empty", &[], 512);
|
|
let mut cut = builder_file();
|
|
cut.truncate(cut.len() / 3);
|
|
check_equal("truncated", &cut, 512);
|
|
}
|
|
|
|
/// Listing a large file and reading one small dataset fetches a few blocks,
|
|
/// not the file.
|
|
#[test]
|
|
fn a_small_read_of_a_large_file_fetches_a_few_blocks() {
|
|
let mut b = FileBuilder::new();
|
|
b.create_dataset("small").with_f64_data(&[1.0, 2.0, 3.0]);
|
|
// 48 MB of raw data, written after the small dataset's metadata.
|
|
b.create_dataset("big")
|
|
.with_f64_data(&(0..6_000_000).map(f64::from).collect::<Vec<_>>());
|
|
let mut g = b.create_group("group");
|
|
g.create_dataset("inner").with_i32_data(&[7, 8]);
|
|
b.add_group(g.finish());
|
|
let data = b.finish().unwrap();
|
|
let lazy = Lazy::open(data.clone(), LazyConfig::default()).unwrap();
|
|
let list = lazy.call(|r| r.list("/")).unwrap();
|
|
assert_eq!(list.len(), 3);
|
|
assert_eq!(
|
|
format!("{:?}", lazy.call(|r| r.read("/small", None)).unwrap().data),
|
|
"F64([1.0, 2.0, 3.0])"
|
|
);
|
|
assert_eq!(
|
|
format!(
|
|
"{:?}",
|
|
lazy.call(|r| r.read("/group/inner", None)).unwrap().data
|
|
),
|
|
"I32([7, 8])"
|
|
);
|
|
// A window of the big dataset reads only its block(s).
|
|
let slab = Hyperslab {
|
|
start: vec![3_000_000],
|
|
count: vec![4],
|
|
stride: None,
|
|
block: None,
|
|
};
|
|
assert_eq!(
|
|
format!(
|
|
"{:?}",
|
|
lazy.call(|r| r.read("/big", Some(&slab))).unwrap().data
|
|
),
|
|
"F64([3000000.0, 3000001.0, 3000002.0, 3000003.0])"
|
|
);
|
|
let st = lazy.storage.stats();
|
|
eprintln!("{} bytes: {st:?}", data.len());
|
|
assert!(st.requests <= 6, "{st:?}");
|
|
assert!(st.bytes_fetched <= 6 << 20, "{st:?}");
|
|
assert!(st.bytes_fetched * 8 < data.len() as u64, "{st:?}");
|
|
}
|
|
|
|
fn python() -> String {
|
|
std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string())
|
|
}
|
|
|
|
fn python_available() -> bool {
|
|
Command::new(python())
|
|
.args(["-c", "import h5py, netCDF4, numpy"])
|
|
.output()
|
|
.is_ok_and(|o| o.status.success())
|
|
}
|
|
|
|
#[test]
|
|
fn h5py_and_netcdf4_files_read_the_same_lazily() {
|
|
if !python_available() {
|
|
assert!(
|
|
!std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1"),
|
|
"CLAWHDF5_REQUIRE_INTEROP=1 but {} lacks h5py/netCDF4/numpy",
|
|
python()
|
|
);
|
|
eprintln!("skipping: {} lacks h5py/netCDF4/numpy", python());
|
|
return;
|
|
}
|
|
let dir = fixture_dir();
|
|
for name in ["fixture.h5", "fixture.nc"] {
|
|
let data = std::fs::read(dir.path().join(name)).unwrap();
|
|
for block in [512, 64 * 1024] {
|
|
let (_, _, lines) = check_equal(name, &data, block);
|
|
assert!(lines.iter().filter(|l| l.contains(" read: Ok")).count() >= 2);
|
|
}
|
|
}
|
|
}
|
|
|
|
/// The bytes of `data` at `r`, zero past its end: a server that claims
|
|
/// the file is longer than it is.
|
|
fn fetch_padded(data: &[u8], r: Range<u64>) -> Result<Vec<u8>, String> {
|
|
let mut out = vec![0u8; (r.end - r.start) as usize];
|
|
let len = data.len() as u64;
|
|
if r.start < len {
|
|
let end = r.end.min(len);
|
|
out[..(end - r.start) as usize].copy_from_slice(&data[r.start as usize..end as usize]);
|
|
}
|
|
Ok(out)
|
|
}
|
|
|
|
/// Sizes a hostile server or a large dataset can name are errors, never
|
|
/// allocations that abort the wasm module: make_fixture.py's limits.h5 and
|
|
/// hostile_vl.h5 (see write_limits there).
|
|
#[test]
|
|
fn size_limits_are_errors_not_aborts() {
|
|
if !python_available() {
|
|
assert!(
|
|
!std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1"),
|
|
"CLAWHDF5_REQUIRE_INTEROP=1 but {} lacks h5py/netCDF4/numpy",
|
|
python()
|
|
);
|
|
eprintln!("skipping: {} lacks h5py/netCDF4/numpy", python());
|
|
return;
|
|
}
|
|
let dir = fixture_dir();
|
|
|
|
// Read whole, /huge_u8 would widen 2^28 values to 64 bits (2 GiB): an
|
|
// error naming readHyperslab, before its chunks are read. A window of
|
|
// it reads.
|
|
let data = std::fs::read(dir.path().join("limits.h5")).unwrap();
|
|
let n = (1u64 << 28) + 1024;
|
|
let window = Hyperslab {
|
|
start: vec![n - 4],
|
|
count: vec![4],
|
|
stride: None,
|
|
block: None,
|
|
};
|
|
let local = Reader::open(data.clone()).unwrap();
|
|
let lazy = Lazy::open(data, LazyConfig::default()).unwrap();
|
|
let before = lazy.storage.stats().requests;
|
|
for e in [
|
|
local.read("/huge_u8", None).unwrap_err(),
|
|
lazy.call(|r| r.read("/huge_u8", None)).unwrap_err(),
|
|
] {
|
|
assert!(e.contains("readHyperslab"), "{e}");
|
|
}
|
|
assert_eq!(lazy.storage.stats().requests, before, "nothing fetched");
|
|
for part in [
|
|
local.read("/huge_u8", Some(&window)).unwrap(),
|
|
lazy.call(|r| r.read("/huge_u8", Some(&window))).unwrap(),
|
|
] {
|
|
assert_eq!(format!("{:?}", part.data), "U8([0, 0, 0, 7])");
|
|
}
|
|
|
|
// A server that claims 3 GiB and a heap collection of 2 GiB + 4 KiB:
|
|
// reading the strings fails at once, fetching a few blocks.
|
|
let data = std::fs::read(dir.path().join("hostile_vl.h5")).unwrap();
|
|
let storage = Arc::new(LazyStorage::new(3 << 30, LazyConfig::default()));
|
|
let s = storage.clone();
|
|
let reader = storage
|
|
.run_blocking(
|
|
|| Reader::open_storage(s.clone()),
|
|
|r| fetch_padded(&data, r),
|
|
)
|
|
.unwrap()
|
|
.unwrap();
|
|
let e = storage
|
|
.run_blocking(|| reader.read("/a", None), |r| fetch_padded(&data, r))
|
|
.unwrap()
|
|
.unwrap_err();
|
|
assert!(e.contains("maxFetch"), "{e}");
|
|
let st = storage.stats();
|
|
assert!(st.requests <= 4 && st.bytes_fetched <= 4 << 20, "{st:?}");
|
|
}
|
|
|
|
/// make_fixture.py's files, written to a temporary directory.
|
|
fn fixture_dir() -> tempfile::TempDir {
|
|
let dir = tempfile::tempdir().unwrap();
|
|
let generator = Path::new(env!("CARGO_MANIFEST_DIR"))
|
|
.join("../../examples/wasm-viewer/test/make_fixture.py");
|
|
let out = Command::new(python())
|
|
.arg(&generator)
|
|
.arg(dir.path())
|
|
.output()
|
|
.unwrap();
|
|
assert!(
|
|
out.status.success(),
|
|
"{}",
|
|
String::from_utf8_lossy(&out.stderr)
|
|
);
|
|
dir
|
|
}
|
|
|
|
fn hdf5_files(dir: &Path, out: &mut Vec<PathBuf>) {
|
|
let Ok(entries) = std::fs::read_dir(dir) else {
|
|
return;
|
|
};
|
|
for e in entries.flatten() {
|
|
let p = e.path();
|
|
if p.is_dir() {
|
|
hdf5_files(&p, out);
|
|
} else if std::fs::read(&p)
|
|
.ok()
|
|
.is_some_and(|b| b.len() <= 64 << 20 && is_hdf5(&b))
|
|
{
|
|
out.push(p);
|
|
}
|
|
}
|
|
}
|
|
|
|
/// The HDF5 signature at 0 or a power-of-two user-block offset.
|
|
fn is_hdf5(b: &[u8]) -> bool {
|
|
const SIG: &[u8] = b"\x89HDF\r\n\x1a\n";
|
|
let mut at = 0usize;
|
|
loop {
|
|
if b.get(at..at + 8) == Some(SIG) {
|
|
return true;
|
|
}
|
|
at = if at == 0 { 512 } else { at * 2 };
|
|
if at >= b.len() {
|
|
return false;
|
|
}
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn corpus_files_read_the_same_lazily() {
|
|
let Ok(dirs) = std::env::var("CLAWHDF5_WASM_CORPUS") else {
|
|
eprintln!("CLAWHDF5_WASM_CORPUS not set; skipping the corpus");
|
|
return;
|
|
};
|
|
let mut files = Vec::new();
|
|
for d in std::env::split_paths(&dirs) {
|
|
hdf5_files(&d, &mut files);
|
|
}
|
|
files.sort();
|
|
assert!(!files.is_empty(), "no HDF5 files under {dirs}");
|
|
let (mut requests, mut bytes, mut total) = (0u64, 0u64, 0u64);
|
|
for f in &files {
|
|
let data = std::fs::read(f).unwrap();
|
|
total += data.len() as u64;
|
|
let (r, b, _) = check_equal(&f.display().to_string(), &data, 64 * 1024);
|
|
requests += r;
|
|
bytes += b;
|
|
}
|
|
eprintln!(
|
|
"{} files ({total} bytes): {requests} requests, {bytes} bytes fetched",
|
|
files.len()
|
|
);
|
|
}
|