From dbafa952ac78f95825bca6f3d95f25d1a715746e Mon Sep 17 00:00:00 2001 From: osobh Date: Sun, 27 Sep 2026 07:36:21 -0500 Subject: [PATCH] wasm: sizes a server or a dataset names are errors, not aborts A read longer than isize::MAX (2 GiB on wasm32) aborted the module in LazyStorage::assemble (capacity_overflow), taking every open file on the page with it, and a hostile server only had to claim a large length and serve a heap collection of 2 GiB + 4 KiB to get there (after fetching 2 GiB). Reading a large u8 dataset whole aborted the same way when its values were widened to 64 bits. - LazyConfig::max_fetch (openUrl option maxFetch, default 512 MiB, at most 1 GiB): a read longer than it fails at once, before anything is fetched, and an operation whose passes would fetch more than it fails before fetching (Operation::charge). assemble reserves fallibly. - Reader::read refuses a read that would use more than 1 GiB while decoding (core::MAX_READ_BYTES: stored bytes + 64-bit values + result) with an error naming readHyperslab, before reading. - openUrl refuses a file of 4 GiB or more at open on wasm32: the format code turns offsets into usize, so nothing past 4 GiB can be read there (shown by a new test: data at 3 GiB reads, a 4 GiB file is refused). maxDownload is bounded to 1 GiB. Tests: make_fixture.py writes limits.h5 (a sparse 2^28 + 1024 byte u8 dataset), hostile_vl.h5 (the reviewer's collection) and far.h5 (data at 3 GiB); test.mjs (wasm32) and tests/lazy.rs (native) check each is an error or reads, and that the module survives. Before: RuntimeError: unreachable in Node; the native test read the huge dataset and fetched 2 GiB. Co-Authored-By: Claude Opus 5.5 (1M context) --- crates/clawhdf5-wasm/js/remote.js | 6 +- crates/clawhdf5-wasm/src/core.rs | 48 +++++++- crates/clawhdf5-wasm/src/lazy.rs | 144 ++++++++++++++++++++-- crates/clawhdf5-wasm/src/lib.rs | 47 ++++++- crates/clawhdf5-wasm/tests/lazy.rs | 98 +++++++++++++-- examples/wasm-viewer/test/make_fixture.py | 70 ++++++++++- examples/wasm-viewer/test/test.mjs | 83 +++++++++++++ 7 files changed, 470 insertions(+), 26 deletions(-) diff --git a/crates/clawhdf5-wasm/js/remote.js b/crates/clawhdf5-wasm/js/remote.js index fdbe619..fcf3bad 100644 --- a/crates/clawhdf5-wasm/js/remote.js +++ b/crates/clawhdf5-wasm/js/remote.js @@ -117,7 +117,11 @@ export async function probe(url, firstLen, opts) { } length = Number(cl); } - if (!Number.isSafeInteger(length) || first.length !== Math.min(firstLen, length)) { + if (!Number.isSafeInteger(length) || length < 0) { + throw new Error(`${url}: the server gave a file size of ${length} bytes; openUrl reads files ` + + "of up to 2^53 - 1 bytes (the largest offset a JavaScript number holds exactly)"); + } + if (first.length !== Math.min(firstLen, length)) { throw new Error(`${url}: asked for the first ${firstLen} bytes of ${length}, got ${first.length}`); } return { length, first, validator: validatorOf(resp), requests }; diff --git a/crates/clawhdf5-wasm/src/core.rs b/crates/clawhdf5-wasm/src/core.rs index 23b4b1d..7922c66 100644 --- a/crates/clawhdf5-wasm/src/core.rs +++ b/crates/clawhdf5-wasm/src/core.rs @@ -14,6 +14,15 @@ use clawhdf5_format::datatype::{Datatype, DatatypeByteOrder}; use clawhdf5_format::storage::Storage; use clawhdf5_format::vl_data::{VlResolver, check_element_size}; +/// The most memory one read may use while it decodes: the stored bytes, +/// the values at 64 bits (integers are widened first) and the values +/// returned. A larger read fails with an error naming `readHyperslab`, +/// before anything is read: on wasm32 a buffer past 2 GiB cannot be +/// allocated at all, and failing to allocate aborts the module (every open +/// file on the page with it). 1 GiB leaves room in wasm32's 4 GiB for the +/// file's cached blocks and the JavaScript copy of the result. +pub const MAX_READ_BYTES: u64 = 1 << 30; + /// Errors are reported to JavaScript as messages. pub type Result = std::result::Result; @@ -241,13 +250,27 @@ impl Reader { if let Datatype::VariableLength { size, .. } = array_base(&dt) { check_element_size(*size, self.file.superblock().offset_size).map_err(err)?; } - let raw = ds.read_selection(&selection).map_err(err)?; - let data = self.decode(&raw, &dt)?; out_shape.extend(element_shape(&dt)); let expected = out_shape .iter() .try_fold(1u64, |acc, &d| acc.checked_mul(d)) .ok_or("selection size overflows")?; + let cost = expected.saturating_mul(bytes_per_value(&dt)); + if cost > MAX_READ_BYTES { + return Err(format!( + "reading {path}{} would take about {} MiB of memory, more than the {} MiB \ + one read may use; read it in parts (readHyperslab)", + if slab.is_some() { + " (this selection)" + } else { + " whole" + }, + cost >> 20, + MAX_READ_BYTES >> 20 + )); + } + let raw = ds.read_selection(&selection).map_err(err)?; + let data = self.decode(&raw, &dt)?; if data.len() as u64 != expected { return Err(format!( "read {} values for shape {out_shape:?} ({expected} expected)", @@ -316,6 +339,27 @@ impl Reader { } } +/// Memory one value of type `dt` takes while [`Reader::read`] decodes it +/// (an array type's elements count as values): its stored bytes, plus what +/// [`Reader::decode`] builds from them. A string counts its `String` (24 +/// bytes on 64-bit targets, less on wasm32) and, for a fixed-length one, +/// its text; a variable-length string's text lives in the heap and is +/// bounded by the storage's own read limit. +fn bytes_per_value(dt: &Datatype) -> u64 { + let base = array_base(dt); + let stored = u64::from(base.type_size()); + stored + + match base { + Datatype::FloatingPoint { size, .. } if *size <= 4 => 4, + Datatype::FloatingPoint { .. } => 8, + // Widened to 64 bits, then narrowed to a new vector. + Datatype::FixedPoint { .. } => 8 + stored, + Datatype::String { .. } => 24 + stored, + Datatype::VariableLength { .. } | Datatype::Enumeration { .. } => 24, + _ => 0, + } +} + /// Narrow integers read at 64 bits to the dataset's own width. The source is /// that width, so this cannot fail on correct input; it is checked anyway. fn narrow>(v: Vec) -> Result> { diff --git a/crates/clawhdf5-wasm/src/lazy.rs b/crates/clawhdf5-wasm/src/lazy.rs index 193f61f..5ef01be 100644 --- a/crates/clawhdf5-wasm/src/lazy.rs +++ b/crates/clawhdf5-wasm/src/lazy.rs @@ -42,6 +42,10 @@ use clawhdf5_format::storage::Storage; /// `docs/design/range-reads.md` §2 measured). pub const DEFAULT_BLOCK_SIZE: u64 = 1 << 20; +/// Default of [`LazyConfig::max_fetch`]: 512 MiB, the same as `openUrl`'s +/// `maxDownload` for a server without range support. +pub const DEFAULT_MAX_FETCH: u64 = 512 << 20; + /// The message of the error a read that misses returns. It never reaches /// the caller of [`LazyStorage::attempt`]: a pass that missed is re-run. pub const NEED_BYTES: &str = "bytes not fetched yet (restartable read)"; @@ -58,6 +62,14 @@ pub struct LazyConfig { /// Largest single range asked for, in bytes (whole blocks, at least /// one); longer runs are split so they can be fetched in parallel. pub max_request: u64, + /// Most bytes one operation may fetch (at least one block), and so the + /// longest single read: a read longer than this fails at once, before + /// anything is fetched, and so does an operation whose passes would + /// fetch more. The file's length comes from the server, so without + /// this a hostile file (a heap "collection" claiming 2 GiB) makes the + /// reader fetch and hold whatever it names; on wasm32 a buffer past + /// 2 GiB cannot even be allocated. + pub max_fetch: u64, } impl Default for LazyConfig { @@ -66,6 +78,7 @@ impl Default for LazyConfig { block_size: DEFAULT_BLOCK_SIZE, capacity: 64 << 20, max_request: 8 << 20, + max_fetch: DEFAULT_MAX_FETCH, } } } @@ -137,6 +150,28 @@ fn lock(m: &Mutex) -> MutexGuard<'_, State> { /// [`LazyStorage::operation`]. pub struct Operation<'a> { storage: &'a LazyStorage, + /// Bytes fetched for this operation so far. + fetched: std::cell::Cell, +} + +impl Operation<'_> { + /// Count `ranges` against the operation's budget + /// ([`LazyConfig::max_fetch`]) before they are fetched: an error, and + /// nothing counted, if they would take it past the budget. + pub fn charge(&self, ranges: &[Range]) -> Result<(), String> { + let max = self.storage.config.max_fetch; + let total = ranges.iter().fold(self.fetched.get(), |n, r| { + n.saturating_add(r.end.saturating_sub(r.start)) + }); + if total > max { + return Err(format!( + "this call would fetch more than {max} bytes of the file (the maxFetch limit); \ + read less at a time (readHyperslab) or raise maxFetch" + )); + } + self.fetched.set(total); + Ok(()) + } } impl Drop for Operation<'_> { @@ -154,6 +189,7 @@ impl LazyStorage { pub fn new(len: u64, mut config: LazyConfig) -> Self { config.block_size = config.block_size.max(512); config.max_request = (config.max_request / config.block_size).max(1) * config.block_size; + config.max_fetch = config.max_fetch.max(config.block_size); LazyStorage { len, config, @@ -180,7 +216,10 @@ impl LazyStorage { /// Hold it across every pass of one operation. pub fn operation(&self) -> Operation<'_> { lock(&self.state).active += 1; - Operation { storage: self } + Operation { + storage: self, + fetched: std::cell::Cell::new(0), + } } /// Run one pass of `f` over this storage. `Done` when `f` read nothing @@ -268,11 +307,12 @@ impl LazyStorage { mut f: impl FnMut() -> T, mut fetch: impl FnMut(Range) -> Result, String>, ) -> Result { - let _op = self.operation(); + let op = self.operation(); loop { match self.attempt(&mut f) { Step::Done(v) => return Ok(v), Step::Need(ranges) => { + op.charge(&ranges)?; for r in ranges { let bytes = fetch(r.clone())?; self.supply_range(&r, &bytes)?; @@ -395,9 +435,35 @@ impl LazyStorage { Ok(have) } - fn assemble(&self, offset: u64, end: u64, blocks: &HashMap>) -> Vec { + /// Refuse a read of `n` bytes longer than an operation may fetch + /// ([`LazyConfig::max_fetch`]), before its blocks are asked for. + fn check_len(&self, n: u64) -> Result<(), FormatError> { + let max = self.config.max_fetch; + if n > max { + return Err(FormatError::Storage(format!( + "a read of {n} bytes is more than one call may fetch ({max} bytes, the maxFetch limit)" + ))); + } + Ok(()) + } + + /// The bytes `offset..end` from `blocks`, which hold every block of + /// that span. The buffer is reserved fallibly: a length the address + /// space cannot hold (past `isize::MAX` on wasm32) is an error, never + /// an abort. + fn assemble( + &self, + offset: u64, + end: u64, + blocks: &HashMap>, + ) -> Result, FormatError> { let bs = self.config.block_size; - let mut out = Vec::with_capacity(usize::try_from(end - offset).unwrap_or(0)); + let n = end - offset; + let too_long = + || FormatError::Storage(format!("cannot hold a read of {n} bytes in memory")); + let mut out = Vec::new(); + out.try_reserve_exact(usize::try_from(n).map_err(|_| too_long())?) + .map_err(|_| too_long())?; let mut pos = offset; while pos < end { let i = pos / bs; @@ -407,7 +473,7 @@ impl LazyStorage { out.extend_from_slice(&block[from..to]); pos = i * bs + to as u64; } - out + Ok(out) } } @@ -416,10 +482,11 @@ impl Storage for LazyStorage { let Some(span) = self.span(offset, len as u64) else { return Ok(Cow::Owned(Vec::new())); }; + let end = offset.saturating_add(len as u64).min(self.len); + self.check_len(end - offset)?; let metadata = len as u64 <= self.config.block_size; let blocks = self.blocks(std::slice::from_ref(&span), metadata)?; - let end = offset.saturating_add(len as u64).min(self.len); - Ok(Cow::Owned(self.assemble(offset, end, &blocks))) + Ok(Cow::Owned(self.assemble(offset, end, &blocks)?)) } fn len(&self) -> u64 { @@ -428,26 +495,29 @@ impl Storage for LazyStorage { fn read_ranges(&self, ranges: &[Range]) -> Result>, FormatError> { let mut spans = Vec::with_capacity(ranges.len()); + let mut total = 0u64; for r in ranges { if r.end < r.start { return Err(FormatError::Storage( "read range ends before it starts".into(), )); } + total = total.saturating_add(r.end.min(self.len).saturating_sub(r.start)); spans.extend(self.span(r.start, r.end - r.start)); } + self.check_len(total)?; let blocks = self.blocks(&spans, false)?; - Ok(ranges + ranges .iter() .map(|r| { let end = r.end.min(self.len); if r.start >= end { - Cow::Owned(Vec::new()) + Ok(Cow::Owned(Vec::new())) } else { - Cow::Owned(self.assemble(r.start, end, &blocks)) + self.assemble(r.start, end, &blocks).map(Cow::Owned) } }) - .collect()) + .collect() } } @@ -465,6 +535,7 @@ mod tests { block_size: block, capacity, max_request: 4 * block, + max_fetch: DEFAULT_MAX_FETCH, } } @@ -654,6 +725,57 @@ mod tests { assert_eq!(s.stats().requests, before.requests); } + #[test] + fn a_read_longer_than_max_fetch_fails_without_fetching() { + // A hostile file names a 2 GiB heap collection in a "file" the + // server claims is 1 TiB: the read is refused before any block is + // asked for (on wasm32 its buffer could not even be allocated). + let s = LazyStorage::new(1 << 40, LazyConfig::default()); + let step = s.attempt(|| s.read_at(4096, (1usize << 31) + 4096).map(|b| b.len())); + match step { + Step::Done(Err(e)) => assert!(e.to_string().contains("maxFetch"), "{e}"), + other => panic!("expected a refusal, got {other:?}"), + } + let step = s.attempt(|| s.read_ranges(&[0..(600 << 20)]).map(|v| v.len())); + assert!(matches!(step, Step::Done(Err(_))), "{step:?}"); + assert_eq!(s.stats().requests, 0); + // At the limit it is an ordinary miss. + let s = LazyStorage::new(1 << 40, config(1024, 1 << 20)); + let step = s.attempt(|| s.read_at(0, DEFAULT_MAX_FETCH as usize).map(|b| b.len())); + assert!(matches!(step, Step::Need(_)), "{step:?}"); + } + + #[test] + fn an_operation_stops_at_its_fetch_budget() { + // Many small reads, none over the limit, that together would fetch + // more than the budget: the operation fails before fetching past it. + let data = file(64 * 1024); + let mut c = config(1024, 1 << 20); + c.max_fetch = 8 * 1024; + let s = LazyStorage::new(data.len() as u64, c); + let e = s + .run_blocking( + || { + (0..64) + .map(|i| owned(s.read_at(i * 1024, 8))) + .collect::, _>>() + }, + |r| Ok(data[r.start as usize..r.end as usize].to_vec()), + ) + .unwrap_err(); + assert!(e.contains("maxFetch"), "{e}"); + assert!(s.stats().bytes_fetched <= 8 * 1024, "{:?}", s.stats()); + // Within the budget it completes, and the budget is per operation. + for _ in 0..3 { + s.run_blocking( + || owned(s.read_at(10 * 1024, 3000)), + |r| Ok(data[r.start as usize..r.end as usize].to_vec()), + ) + .unwrap() + .unwrap(); + } + } + #[test] fn a_failed_fetch_is_an_error_not_data() { let data = file(4096); diff --git a/crates/clawhdf5-wasm/src/lib.rs b/crates/clawhdf5-wasm/src/lib.rs index a421573..1fc2521 100644 --- a/crates/clawhdf5-wasm/src/lib.rs +++ b/crates/clawhdf5-wasm/src/lib.rs @@ -389,11 +389,14 @@ impl Http { storage: &LazyStorage, mut f: impl FnMut() -> T, ) -> Result { - let _op = storage.operation(); + let op = storage.operation(); loop { match storage.attempt(&mut f) { Step::Done(v) => return Ok(v), - Step::Need(ranges) => self.fetch(storage, &ranges).await?, + Step::Need(ranges) => { + op.charge(&ranges).map_err(js_err)?; + self.fetch(storage, &ranges).await? + } } } } @@ -426,6 +429,24 @@ fn int_opt(opts: &JsValue, key: &str, min: f64, max: f64) -> Result, } } +/// Most bytes `maxFetch` and `maxDownload` may allow: 1 GiB. wasm32 has +/// 4 GiB of memory and no buffer past 2 GiB, and what is fetched is held +/// while it is decoded. +const MAX_FETCH_LIMIT: u64 = 1 << 30; + +/// The largest file `openUrl` reads by ranges: on wasm32, 4 GiB - 1 bytes. +/// The format code turns file offsets into `usize` to use them (with a +/// clean error past it, see scripts/check-32bit-casts.sh), so on a 32-bit +/// target nothing at 4 GiB or beyond can be read; a larger file is refused +/// at open rather than failing on whichever read reaches past 4 GiB. On +/// 64-bit targets it is 2^53 - 1, the largest offset a JavaScript number +/// holds exactly. +const MAX_REMOTE_LENGTH: u64 = if (usize::MAX as u64) < MAX_SAFE_INTEGER as u64 { + usize::MAX as u64 +} else { + MAX_SAFE_INTEGER as u64 +}; + fn config_from(opts: &JsValue) -> Result { let mut c = LazyConfig::default(); if let Some(b) = int_opt(opts, "blockSize", 512.0, (64u64 << 20) as f64)? { @@ -434,6 +455,12 @@ fn config_from(opts: &JsValue) -> Result { if let Some(n) = int_opt(opts, "cacheSize", 0.0, MAX_SAFE_INTEGER)? { c.capacity = n; } + if let Some(n) = int_opt(opts, "maxFetch", 512.0, MAX_FETCH_LIMIT as f64)? { + c.max_fetch = n; + } + // Read by remote.js; checked here so a value wasm32 cannot hold is an + // option error rather than a download that cannot be kept. + int_opt(opts, "maxDownload", 0.0, MAX_FETCH_LIMIT as f64)?; Ok(c) } @@ -444,15 +471,21 @@ fn config_from(opts: &JsValue) -> Result { /// `opts` (all optional): /// - `blockSize` — bytes per request block, 512 to 64 MiB (default 1 MiB); /// - `cacheSize` — bytes of blocks kept between calls (default 64 MiB); +/// - `maxFetch` — most bytes one call may fetch, and so the longest single +/// read, up to 1 GiB (default 512 MiB): a call that would fetch more +/// fails before fetching it; /// - `fallback` — `"download"` (default) reads the whole file when the /// server ignores `Range` (answers 200), up to `maxDownload` bytes -/// (default 512 MiB); `"error"` refuses such a server; +/// (default 512 MiB, at most 1 GiB); `"error"` refuses such a server; /// - `headers`, `credentials` — passed to every `fetch`; /// - `parallel` — range requests in flight at once (default 6); /// - `fetch` — a `fetch`-compatible function to use instead of the global. /// /// Cross-origin servers must allow CORS and expose `Content-Range` (or -/// answer `HEAD` with `Content-Length`). +/// answer `HEAD` with `Content-Length`). A file may be up to 4 GiB - 1 +/// bytes long (wasm32 offsets); a longer one is refused at open. A whole-dataset `read` that would use more than 1 GiB of +/// memory ([`core::MAX_READ_BYTES`]) is refused: read it in parts with +/// `readHyperslab`. #[wasm_bindgen(js_name = openUrl)] pub async fn open_url(url: String, opts: JsValue) -> Result { let config = config_from(&opts)?; @@ -477,6 +510,12 @@ pub async fn open_url(url: String, opts: JsValue) -> Result .as_f64() .filter(|x| x.fract() == 0.0 && (0.0..=MAX_SAFE_INTEGER).contains(x)) .ok_or_else(|| js_err(format!("{url}: the server gave no usable file size")))?; + if length as u64 > MAX_REMOTE_LENGTH { + return Err(js_err(format!( + "{url} is {length} bytes; openUrl reads files of up to {MAX_REMOTE_LENGTH} bytes \ + (4 GiB - 1: the WebAssembly reader addresses a file with 32-bit offsets)" + ))); + } let http = Http { url, opts, diff --git a/crates/clawhdf5-wasm/tests/lazy.rs b/crates/clawhdf5-wasm/tests/lazy.rs index eed0ed5..4084d68 100644 --- a/crates/clawhdf5-wasm/tests/lazy.rs +++ b/crates/clawhdf5-wasm/tests/lazy.rs @@ -193,6 +193,7 @@ fn config(block: u64) -> LazyConfig { // A small budget, so eviction between operations is exercised. capacity: 16 * block, max_request: 8 * block, + ..LazyConfig::default() } } @@ -310,6 +311,95 @@ fn h5py_and_netcdf4_files_read_the_same_lazily() { eprintln!("skipping: {} lacks h5py/netCDF4/numpy", python()); return; } + let dir = fixture_dir(); + for name in ["fixture.h5", "fixture.nc"] { + let data = std::fs::read(dir.path().join(name)).unwrap(); + for block in [512, 64 * 1024] { + let (_, _, lines) = check_equal(name, &data, block); + assert!(lines.iter().filter(|l| l.contains(" read: Ok")).count() >= 2); + } + } +} + +/// The bytes of `data` at `r`, zero past its end: a server that claims +/// the file is longer than it is. +fn fetch_padded(data: &[u8], r: Range) -> Result, String> { + let mut out = vec![0u8; (r.end - r.start) as usize]; + let len = data.len() as u64; + if r.start < len { + let end = r.end.min(len); + out[..(end - r.start) as usize].copy_from_slice(&data[r.start as usize..end as usize]); + } + Ok(out) +} + +/// Sizes a hostile server or a large dataset can name are errors, never +/// allocations that abort the wasm module: make_fixture.py's limits.h5 and +/// hostile_vl.h5 (see write_limits there). +#[test] +fn size_limits_are_errors_not_aborts() { + if !python_available() { + assert!( + !std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1"), + "CLAWHDF5_REQUIRE_INTEROP=1 but {} lacks h5py/netCDF4/numpy", + python() + ); + eprintln!("skipping: {} lacks h5py/netCDF4/numpy", python()); + return; + } + let dir = fixture_dir(); + + // Read whole, /huge_u8 would widen 2^28 values to 64 bits (2 GiB): an + // error naming readHyperslab, before its chunks are read. A window of + // it reads. + let data = std::fs::read(dir.path().join("limits.h5")).unwrap(); + let n = (1u64 << 28) + 1024; + let window = Hyperslab { + start: vec![n - 4], + count: vec![4], + stride: None, + block: None, + }; + let local = Reader::open(data.clone()).unwrap(); + let lazy = Lazy::open(data, LazyConfig::default()).unwrap(); + let before = lazy.storage.stats().requests; + for e in [ + local.read("/huge_u8", None).unwrap_err(), + lazy.call(|r| r.read("/huge_u8", None)).unwrap_err(), + ] { + assert!(e.contains("readHyperslab"), "{e}"); + } + assert_eq!(lazy.storage.stats().requests, before, "nothing fetched"); + for part in [ + local.read("/huge_u8", Some(&window)).unwrap(), + lazy.call(|r| r.read("/huge_u8", Some(&window))).unwrap(), + ] { + assert_eq!(format!("{:?}", part.data), "U8([0, 0, 0, 7])"); + } + + // A server that claims 3 GiB and a heap collection of 2 GiB + 4 KiB: + // reading the strings fails at once, fetching a few blocks. + let data = std::fs::read(dir.path().join("hostile_vl.h5")).unwrap(); + let storage = Arc::new(LazyStorage::new(3 << 30, LazyConfig::default())); + let s = storage.clone(); + let reader = storage + .run_blocking( + || Reader::open_storage(s.clone()), + |r| fetch_padded(&data, r), + ) + .unwrap() + .unwrap(); + let e = storage + .run_blocking(|| reader.read("/a", None), |r| fetch_padded(&data, r)) + .unwrap() + .unwrap_err(); + assert!(e.contains("maxFetch"), "{e}"); + let st = storage.stats(); + assert!(st.requests <= 4 && st.bytes_fetched <= 4 << 20, "{st:?}"); +} + +/// make_fixture.py's files, written to a temporary directory. +fn fixture_dir() -> tempfile::TempDir { let dir = tempfile::tempdir().unwrap(); let generator = Path::new(env!("CARGO_MANIFEST_DIR")) .join("../../examples/wasm-viewer/test/make_fixture.py"); @@ -323,13 +413,7 @@ fn h5py_and_netcdf4_files_read_the_same_lazily() { "{}", String::from_utf8_lossy(&out.stderr) ); - for name in ["fixture.h5", "fixture.nc"] { - let data = std::fs::read(dir.path().join(name)).unwrap(); - for block in [512, 64 * 1024] { - let (_, _, lines) = check_equal(name, &data, block); - assert!(lines.iter().filter(|l| l.contains(" read: Ok")).count() >= 2); - } - } + dir } fn hdf5_files(dir: &Path, out: &mut Vec) { diff --git a/examples/wasm-viewer/test/make_fixture.py b/examples/wasm-viewer/test/make_fixture.py index 1e79065..2a2af72 100644 --- a/examples/wasm-viewer/test/make_fixture.py +++ b/examples/wasm-viewer/test/make_fixture.py @@ -3,7 +3,9 @@ reads back from them, for the clawhdf5-wasm tests. python make_fixture.py OUT_DIR -writes OUT_DIR/fixture.h5, OUT_DIR/fixture.nc and OUT_DIR/expected.json. +writes OUT_DIR/fixture.h5, OUT_DIR/fixture.nc and OUT_DIR/expected.json, +and the limit-test files OUT_DIR/limits.h5 and OUT_DIR/hostile_vl.h5 (see +write_limits). Both the Rust test (crates/clawhdf5-wasm/tests/h5py_interop.rs, native) and the Node test (test.mjs, the built wasm package) compare against the same expected.json, so the two check the same values. @@ -204,6 +206,72 @@ json.dump({"fixture.h5": describe(h5), "fixture.nc": describe(nc)}, open(out / "expected.json", "w"), indent=1, ensure_ascii=False) +HUGE_U8 = 2**28 + 1024 +# The collection size hostile_vl.h5 claims: past 2 GiB, which a wasm32 +# buffer cannot hold. +HOSTILE_GCOL_SIZE = 2**31 + 4096 +# The file length a server claims for hostile_vl.h5 (the tests' mock fetch +# answers every range with zeros past the real bytes): 3 GiB, within what +# wasm32 opens, and room for the collection. +HOSTILE_LENGTH = 3 << 30 + + +def write_limits(out): + """Files for the size limits (the tests must get errors, not aborts): + + - limits.h5: /huge_u8, 2^28 + 1024 bytes of u8 in compressed chunks + (a small file): read whole it would take over 2 GiB while decoding; + its last value is 7. + - hostile_vl.h5: a variable-length string dataset /a whose global heap + collection claims HOSTILE_GCOL_SIZE bytes, with the superblock's end + of file set to HOSTILE_LENGTH (libhdf5 cannot read it; it is only + served by a mock that claims that length). + - far.h5 and far.json: /x, 16 float64 values, whose contiguous data + address is moved FAR_SHIFT bytes on (past 2 GiB, the sign bit of a + wasm32 isize) in a file whose end of file is moved as far; the tests' + mock serves the data there, to show offsets up to 4 GiB work on + wasm32. + """ + with h5py.File(out / "limits.h5", "w") as f: + d = f.create_dataset("huge_u8", shape=(HUGE_U8,), dtype="u1", + chunks=(1 << 20,), compression="gzip") + d[-1] = 7 + path = out / "hostile_vl.h5" + with h5py.File(path, "w", libver="earliest") as f: + f.create_dataset("a", data=["x", "yy"], dtype=h5py.string_dtype()) + b = bytearray(path.read_bytes()) + assert b[8] == 0, "a version 0 superblock" + b[40:48] = HOSTILE_LENGTH.to_bytes(8, "little") # end of file address + at = b.index(b"GCOL") + b[at + 8:at + 16] = HOSTILE_GCOL_SIZE.to_bytes(8, "little") + path.write_bytes(bytes(b)) + + path = out / "far.h5" + values = np.arange(16, dtype=" { + f.calls++; + if (init.method === "HEAD") { + return new Response(null, { status: 200, headers: { "Content-Length": String(total) } }); + } + const m = /^bytes=(\d+)-(\d+)$/.exec(new Headers(init.headers).get("Range")); + const a = Number(m[1]); + const b = Math.min(Number(m[2]) + 1, total); + const out = new Uint8Array(b - a); + for (const [at, bytes] of [[0, buf], ...extra]) { + const from = Math.max(a, at); + const to = Math.min(b, at + bytes.length); + if (from < to) out.set(bytes.subarray(from - at, to - at), from - a); + } + const headers = { "Content-Length": String(b - a) }; + if (!hide) headers["Content-Range"] = `bytes ${a}-${b - 1}/${total}`; + return new Response(out, { status: 206, headers }); + }; + f.calls = 0; + return f; +} + +// Sizes a hostile server or a large dataset can name are errors, never an +// allocation that aborts the module (which would take every open file on +// the page with it); see write_limits in make_fixture.py. +async function limitTests() { + // Read whole, /huge_u8 (2^28 + 1024 bytes) would take over 2 GiB while + // decoding: refused before its chunks are fetched; a window reads. + const n = 2 ** 28 + 1024; + const limits = readFileSync(join(fixDir, "limits.h5")); + const local = pkg.open(new Uint8Array(limits)); + await fails(() => local.read("/huge_u8"), /readHyperslab/, "huge read, in memory"); + const remote = await pkg.openUrl(`${base}/fix/limits.h5`); + const before = remote.stats().requests; + await fails(() => remote.read("/huge_u8"), /readHyperslab/, "huge read, openUrl"); + eq(remote.stats().requests, before, "huge read fetched nothing"); + eq(Array.from((await remote.readHyperslab("/huge_u8", [n - 4], [4])).data), [0, 0, 0, 7], "huge window"); + eq(Array.from(local.readHyperslab("/huge_u8", [n - 4], [4]).data), [0, 0, 0, 7], "huge window, in memory"); + local.free(); + + // A server claiming 3 GiB, a heap collection claiming 2 GiB + 4 KiB: an + // error after a few requests (it used to fetch 2 GiB, then abort). + const hostile = mockFetch(new Uint8Array(readFileSync(join(fixDir, "hostile_vl.h5"))), { total: 3 * 2 ** 30 }); + const h = await pkg.openUrl("http://hostile.invalid/h.h5", { fetch: hostile }); + await fails(() => h.read("/a"), /maxFetch/, "hostile collection size"); + assert.ok(hostile.calls <= 4, `hostile: ${hostile.calls} requests`); + // The module survived: files open and read. + eq((await (await pkg.openUrl(`${base}/fix/fixture.h5`)).read("/sensors/temp")).data[0], 21.5, "alive after hostile"); + + // A call that would fetch more than maxFetch fails before fetching it. + await fails(async () => { + const f = await pkg.openUrl(`${base}/fix/fixture.h5`, { blockSize: 512, maxFetch: 1024 }); + await f.read("/grid"); + }, /maxFetch/, "maxFetch"); + await fails(() => pkg.openUrl(`${base}/fix/fixture.h5`, { maxFetch: 2 ** 31 }), /maxFetch/, "maxFetch range"); + await fails(() => pkg.openUrl(`${base}/fix/fixture.h5`, { maxDownload: 2 ** 31 }), /maxDownload/, "maxDownload range"); + + // Offsets past 2 GiB work on wasm32: far.h5's data sits at 3 GiB. + const far = JSON.parse(readFileSync(join(fixDir, "far.json"), "utf8")); + const farBytes = new Uint8Array(readFileSync(join(fixDir, "far.h5"))); + const farData = farBytes.subarray(far.data_at, far.data_at + far.nbytes); + const ff = await pkg.openUrl("http://far.invalid/far.h5", { + fetch: mockFetch(farBytes, { total: far.length, extra: [[far.far_at, farData]] }), + }); + eq(Array.from((await ff.read("/x")).data), far.values, "data past 2 GiB"); + eq(ff.stats().size, far.length, "size past 2 GiB"); + + // wasm32's reader holds file offsets in 32 bits: a file of 4 GiB or more + // is refused at open, with the limit in the message (past 2^53 - 1 bytes + // a JavaScript number is not even exact). + for (const total of [2 ** 32, 2 ** 40, 2 ** 53 + 2]) { + await fails(() => pkg.openUrl("http://huge.invalid/x.h5", { fetch: mockFetch(farBytes, { total }) }), + total > 2 ** 53 ? /2\^53 - 1/ : /4 GiB/, `length ${total}`); + } +} + function hdf5Files(dir, out) { for (const e of readdirSync(dir, { withFileTypes: true })) { const p = join(dir, e.name);