wasm: sizes a server or a dataset names are errors, not aborts

A read longer than isize::MAX (2 GiB on wasm32) aborted the module in
LazyStorage::assemble (capacity_overflow), taking every open file on the
page with it, and a hostile server only had to claim a large length and
serve a heap collection of 2 GiB + 4 KiB to get there (after fetching
2 GiB). Reading a large u8 dataset whole aborted the same way when its
values were widened to 64 bits.

- LazyConfig::max_fetch (openUrl option maxFetch, default 512 MiB, at
  most 1 GiB): a read longer than it fails at once, before anything is
  fetched, and an operation whose passes would fetch more than it fails
  before fetching (Operation::charge). assemble reserves fallibly.
- Reader::read refuses a read that would use more than 1 GiB while
  decoding (core::MAX_READ_BYTES: stored bytes + 64-bit values + result)
  with an error naming readHyperslab, before reading.
- openUrl refuses a file of 4 GiB or more at open on wasm32: the format
  code turns offsets into usize, so nothing past 4 GiB can be read there
  (shown by a new test: data at 3 GiB reads, a 4 GiB file is refused).
  maxDownload is bounded to 1 GiB.

Tests: make_fixture.py writes limits.h5 (a sparse 2^28 + 1024 byte u8
dataset), hostile_vl.h5 (the reviewer's collection) and far.h5 (data at
3 GiB); test.mjs (wasm32) and tests/lazy.rs (native) check each is an
error or reads, and that the module survives. Before: RuntimeError:
unreachable in Node; the native test read the huge dataset and fetched
2 GiB.

Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
osobh
2026-09-27 07:36:21 -05:00
co-authored by Claude Opus 5.5
parent f825a89e23
commit dbafa952ac
7 changed files with 470 additions and 26 deletions
+5 -1
View File
@@ -117,7 +117,11 @@ export async function probe(url, firstLen, opts) {
} }
length = Number(cl); length = Number(cl);
} }
if (!Number.isSafeInteger(length) || first.length !== Math.min(firstLen, length)) { if (!Number.isSafeInteger(length) || length < 0) {
throw new Error(`${url}: the server gave a file size of ${length} bytes; openUrl reads files ` +
"of up to 2^53 - 1 bytes (the largest offset a JavaScript number holds exactly)");
}
if (first.length !== Math.min(firstLen, length)) {
throw new Error(`${url}: asked for the first ${firstLen} bytes of ${length}, got ${first.length}`); throw new Error(`${url}: asked for the first ${firstLen} bytes of ${length}, got ${first.length}`);
} }
return { length, first, validator: validatorOf(resp), requests }; return { length, first, validator: validatorOf(resp), requests };
+46 -2
View File
@@ -14,6 +14,15 @@ use clawhdf5_format::datatype::{Datatype, DatatypeByteOrder};
use clawhdf5_format::storage::Storage; use clawhdf5_format::storage::Storage;
use clawhdf5_format::vl_data::{VlResolver, check_element_size}; use clawhdf5_format::vl_data::{VlResolver, check_element_size};
/// The most memory one read may use while it decodes: the stored bytes,
/// the values at 64 bits (integers are widened first) and the values
/// returned. A larger read fails with an error naming `readHyperslab`,
/// before anything is read: on wasm32 a buffer past 2 GiB cannot be
/// allocated at all, and failing to allocate aborts the module (every open
/// file on the page with it). 1 GiB leaves room in wasm32's 4 GiB for the
/// file's cached blocks and the JavaScript copy of the result.
pub const MAX_READ_BYTES: u64 = 1 << 30;
/// Errors are reported to JavaScript as messages. /// Errors are reported to JavaScript as messages.
pub type Result<T> = std::result::Result<T, String>; pub type Result<T> = std::result::Result<T, String>;
@@ -241,13 +250,27 @@ impl Reader {
if let Datatype::VariableLength { size, .. } = array_base(&dt) { if let Datatype::VariableLength { size, .. } = array_base(&dt) {
check_element_size(*size, self.file.superblock().offset_size).map_err(err)?; check_element_size(*size, self.file.superblock().offset_size).map_err(err)?;
} }
let raw = ds.read_selection(&selection).map_err(err)?;
let data = self.decode(&raw, &dt)?;
out_shape.extend(element_shape(&dt)); out_shape.extend(element_shape(&dt));
let expected = out_shape let expected = out_shape
.iter() .iter()
.try_fold(1u64, |acc, &d| acc.checked_mul(d)) .try_fold(1u64, |acc, &d| acc.checked_mul(d))
.ok_or("selection size overflows")?; .ok_or("selection size overflows")?;
let cost = expected.saturating_mul(bytes_per_value(&dt));
if cost > MAX_READ_BYTES {
return Err(format!(
"reading {path}{} would take about {} MiB of memory, more than the {} MiB \
one read may use; read it in parts (readHyperslab)",
if slab.is_some() {
" (this selection)"
} else {
" whole"
},
cost >> 20,
MAX_READ_BYTES >> 20
));
}
let raw = ds.read_selection(&selection).map_err(err)?;
let data = self.decode(&raw, &dt)?;
if data.len() as u64 != expected { if data.len() as u64 != expected {
return Err(format!( return Err(format!(
"read {} values for shape {out_shape:?} ({expected} expected)", "read {} values for shape {out_shape:?} ({expected} expected)",
@@ -316,6 +339,27 @@ impl Reader {
} }
} }
/// Memory one value of type `dt` takes while [`Reader::read`] decodes it
/// (an array type's elements count as values): its stored bytes, plus what
/// [`Reader::decode`] builds from them. A string counts its `String` (24
/// bytes on 64-bit targets, less on wasm32) and, for a fixed-length one,
/// its text; a variable-length string's text lives in the heap and is
/// bounded by the storage's own read limit.
fn bytes_per_value(dt: &Datatype) -> u64 {
let base = array_base(dt);
let stored = u64::from(base.type_size());
stored
+ match base {
Datatype::FloatingPoint { size, .. } if *size <= 4 => 4,
Datatype::FloatingPoint { .. } => 8,
// Widened to 64 bits, then narrowed to a new vector.
Datatype::FixedPoint { .. } => 8 + stored,
Datatype::String { .. } => 24 + stored,
Datatype::VariableLength { .. } | Datatype::Enumeration { .. } => 24,
_ => 0,
}
}
/// Narrow integers read at 64 bits to the dataset's own width. The source is /// Narrow integers read at 64 bits to the dataset's own width. The source is
/// that width, so this cannot fail on correct input; it is checked anyway. /// that width, so this cannot fail on correct input; it is checked anyway.
fn narrow<S: Copy + std::fmt::Display, T: TryFrom<S>>(v: Vec<S>) -> Result<Vec<T>> { fn narrow<S: Copy + std::fmt::Display, T: TryFrom<S>>(v: Vec<S>) -> Result<Vec<T>> {
+133 -11
View File
@@ -42,6 +42,10 @@ use clawhdf5_format::storage::Storage;
/// `docs/design/range-reads.md` §2 measured). /// `docs/design/range-reads.md` §2 measured).
pub const DEFAULT_BLOCK_SIZE: u64 = 1 << 20; pub const DEFAULT_BLOCK_SIZE: u64 = 1 << 20;
/// Default of [`LazyConfig::max_fetch`]: 512 MiB, the same as `openUrl`'s
/// `maxDownload` for a server without range support.
pub const DEFAULT_MAX_FETCH: u64 = 512 << 20;
/// The message of the error a read that misses returns. It never reaches /// The message of the error a read that misses returns. It never reaches
/// the caller of [`LazyStorage::attempt`]: a pass that missed is re-run. /// the caller of [`LazyStorage::attempt`]: a pass that missed is re-run.
pub const NEED_BYTES: &str = "bytes not fetched yet (restartable read)"; pub const NEED_BYTES: &str = "bytes not fetched yet (restartable read)";
@@ -58,6 +62,14 @@ pub struct LazyConfig {
/// Largest single range asked for, in bytes (whole blocks, at least /// Largest single range asked for, in bytes (whole blocks, at least
/// one); longer runs are split so they can be fetched in parallel. /// one); longer runs are split so they can be fetched in parallel.
pub max_request: u64, pub max_request: u64,
/// Most bytes one operation may fetch (at least one block), and so the
/// longest single read: a read longer than this fails at once, before
/// anything is fetched, and so does an operation whose passes would
/// fetch more. The file's length comes from the server, so without
/// this a hostile file (a heap "collection" claiming 2 GiB) makes the
/// reader fetch and hold whatever it names; on wasm32 a buffer past
/// 2 GiB cannot even be allocated.
pub max_fetch: u64,
} }
impl Default for LazyConfig { impl Default for LazyConfig {
@@ -66,6 +78,7 @@ impl Default for LazyConfig {
block_size: DEFAULT_BLOCK_SIZE, block_size: DEFAULT_BLOCK_SIZE,
capacity: 64 << 20, capacity: 64 << 20,
max_request: 8 << 20, max_request: 8 << 20,
max_fetch: DEFAULT_MAX_FETCH,
} }
} }
} }
@@ -137,6 +150,28 @@ fn lock(m: &Mutex<State>) -> MutexGuard<'_, State> {
/// [`LazyStorage::operation`]. /// [`LazyStorage::operation`].
pub struct Operation<'a> { pub struct Operation<'a> {
storage: &'a LazyStorage, storage: &'a LazyStorage,
/// Bytes fetched for this operation so far.
fetched: std::cell::Cell<u64>,
}
impl Operation<'_> {
/// Count `ranges` against the operation's budget
/// ([`LazyConfig::max_fetch`]) before they are fetched: an error, and
/// nothing counted, if they would take it past the budget.
pub fn charge(&self, ranges: &[Range<u64>]) -> Result<(), String> {
let max = self.storage.config.max_fetch;
let total = ranges.iter().fold(self.fetched.get(), |n, r| {
n.saturating_add(r.end.saturating_sub(r.start))
});
if total > max {
return Err(format!(
"this call would fetch more than {max} bytes of the file (the maxFetch limit); \
read less at a time (readHyperslab) or raise maxFetch"
));
}
self.fetched.set(total);
Ok(())
}
} }
impl Drop for Operation<'_> { impl Drop for Operation<'_> {
@@ -154,6 +189,7 @@ impl LazyStorage {
pub fn new(len: u64, mut config: LazyConfig) -> Self { pub fn new(len: u64, mut config: LazyConfig) -> Self {
config.block_size = config.block_size.max(512); config.block_size = config.block_size.max(512);
config.max_request = (config.max_request / config.block_size).max(1) * config.block_size; config.max_request = (config.max_request / config.block_size).max(1) * config.block_size;
config.max_fetch = config.max_fetch.max(config.block_size);
LazyStorage { LazyStorage {
len, len,
config, config,
@@ -180,7 +216,10 @@ impl LazyStorage {
/// Hold it across every pass of one operation. /// Hold it across every pass of one operation.
pub fn operation(&self) -> Operation<'_> { pub fn operation(&self) -> Operation<'_> {
lock(&self.state).active += 1; lock(&self.state).active += 1;
Operation { storage: self } Operation {
storage: self,
fetched: std::cell::Cell::new(0),
}
} }
/// Run one pass of `f` over this storage. `Done` when `f` read nothing /// Run one pass of `f` over this storage. `Done` when `f` read nothing
@@ -268,11 +307,12 @@ impl LazyStorage {
mut f: impl FnMut() -> T, mut f: impl FnMut() -> T,
mut fetch: impl FnMut(Range<u64>) -> Result<Vec<u8>, String>, mut fetch: impl FnMut(Range<u64>) -> Result<Vec<u8>, String>,
) -> Result<T, String> { ) -> Result<T, String> {
let _op = self.operation(); let op = self.operation();
loop { loop {
match self.attempt(&mut f) { match self.attempt(&mut f) {
Step::Done(v) => return Ok(v), Step::Done(v) => return Ok(v),
Step::Need(ranges) => { Step::Need(ranges) => {
op.charge(&ranges)?;
for r in ranges { for r in ranges {
let bytes = fetch(r.clone())?; let bytes = fetch(r.clone())?;
self.supply_range(&r, &bytes)?; self.supply_range(&r, &bytes)?;
@@ -395,9 +435,35 @@ impl LazyStorage {
Ok(have) Ok(have)
} }
fn assemble(&self, offset: u64, end: u64, blocks: &HashMap<u64, Arc<[u8]>>) -> Vec<u8> { /// Refuse a read of `n` bytes longer than an operation may fetch
/// ([`LazyConfig::max_fetch`]), before its blocks are asked for.
fn check_len(&self, n: u64) -> Result<(), FormatError> {
let max = self.config.max_fetch;
if n > max {
return Err(FormatError::Storage(format!(
"a read of {n} bytes is more than one call may fetch ({max} bytes, the maxFetch limit)"
)));
}
Ok(())
}
/// The bytes `offset..end` from `blocks`, which hold every block of
/// that span. The buffer is reserved fallibly: a length the address
/// space cannot hold (past `isize::MAX` on wasm32) is an error, never
/// an abort.
fn assemble(
&self,
offset: u64,
end: u64,
blocks: &HashMap<u64, Arc<[u8]>>,
) -> Result<Vec<u8>, FormatError> {
let bs = self.config.block_size; let bs = self.config.block_size;
let mut out = Vec::with_capacity(usize::try_from(end - offset).unwrap_or(0)); let n = end - offset;
let too_long =
|| FormatError::Storage(format!("cannot hold a read of {n} bytes in memory"));
let mut out = Vec::new();
out.try_reserve_exact(usize::try_from(n).map_err(|_| too_long())?)
.map_err(|_| too_long())?;
let mut pos = offset; let mut pos = offset;
while pos < end { while pos < end {
let i = pos / bs; let i = pos / bs;
@@ -407,7 +473,7 @@ impl LazyStorage {
out.extend_from_slice(&block[from..to]); out.extend_from_slice(&block[from..to]);
pos = i * bs + to as u64; pos = i * bs + to as u64;
} }
out Ok(out)
} }
} }
@@ -416,10 +482,11 @@ impl Storage for LazyStorage {
let Some(span) = self.span(offset, len as u64) else { let Some(span) = self.span(offset, len as u64) else {
return Ok(Cow::Owned(Vec::new())); return Ok(Cow::Owned(Vec::new()));
}; };
let end = offset.saturating_add(len as u64).min(self.len);
self.check_len(end - offset)?;
let metadata = len as u64 <= self.config.block_size; let metadata = len as u64 <= self.config.block_size;
let blocks = self.blocks(std::slice::from_ref(&span), metadata)?; let blocks = self.blocks(std::slice::from_ref(&span), metadata)?;
let end = offset.saturating_add(len as u64).min(self.len); Ok(Cow::Owned(self.assemble(offset, end, &blocks)?))
Ok(Cow::Owned(self.assemble(offset, end, &blocks)))
} }
fn len(&self) -> u64 { fn len(&self) -> u64 {
@@ -428,26 +495,29 @@ impl Storage for LazyStorage {
fn read_ranges(&self, ranges: &[Range<u64>]) -> Result<Vec<Cow<'_, [u8]>>, FormatError> { fn read_ranges(&self, ranges: &[Range<u64>]) -> Result<Vec<Cow<'_, [u8]>>, FormatError> {
let mut spans = Vec::with_capacity(ranges.len()); let mut spans = Vec::with_capacity(ranges.len());
let mut total = 0u64;
for r in ranges { for r in ranges {
if r.end < r.start { if r.end < r.start {
return Err(FormatError::Storage( return Err(FormatError::Storage(
"read range ends before it starts".into(), "read range ends before it starts".into(),
)); ));
} }
total = total.saturating_add(r.end.min(self.len).saturating_sub(r.start));
spans.extend(self.span(r.start, r.end - r.start)); spans.extend(self.span(r.start, r.end - r.start));
} }
self.check_len(total)?;
let blocks = self.blocks(&spans, false)?; let blocks = self.blocks(&spans, false)?;
Ok(ranges ranges
.iter() .iter()
.map(|r| { .map(|r| {
let end = r.end.min(self.len); let end = r.end.min(self.len);
if r.start >= end { if r.start >= end {
Cow::Owned(Vec::new()) Ok(Cow::Owned(Vec::new()))
} else { } else {
Cow::Owned(self.assemble(r.start, end, &blocks)) self.assemble(r.start, end, &blocks).map(Cow::Owned)
} }
}) })
.collect()) .collect()
} }
} }
@@ -465,6 +535,7 @@ mod tests {
block_size: block, block_size: block,
capacity, capacity,
max_request: 4 * block, max_request: 4 * block,
max_fetch: DEFAULT_MAX_FETCH,
} }
} }
@@ -654,6 +725,57 @@ mod tests {
assert_eq!(s.stats().requests, before.requests); assert_eq!(s.stats().requests, before.requests);
} }
#[test]
fn a_read_longer_than_max_fetch_fails_without_fetching() {
// A hostile file names a 2 GiB heap collection in a "file" the
// server claims is 1 TiB: the read is refused before any block is
// asked for (on wasm32 its buffer could not even be allocated).
let s = LazyStorage::new(1 << 40, LazyConfig::default());
let step = s.attempt(|| s.read_at(4096, (1usize << 31) + 4096).map(|b| b.len()));
match step {
Step::Done(Err(e)) => assert!(e.to_string().contains("maxFetch"), "{e}"),
other => panic!("expected a refusal, got {other:?}"),
}
let step = s.attempt(|| s.read_ranges(&[0..(600 << 20)]).map(|v| v.len()));
assert!(matches!(step, Step::Done(Err(_))), "{step:?}");
assert_eq!(s.stats().requests, 0);
// At the limit it is an ordinary miss.
let s = LazyStorage::new(1 << 40, config(1024, 1 << 20));
let step = s.attempt(|| s.read_at(0, DEFAULT_MAX_FETCH as usize).map(|b| b.len()));
assert!(matches!(step, Step::Need(_)), "{step:?}");
}
#[test]
fn an_operation_stops_at_its_fetch_budget() {
// Many small reads, none over the limit, that together would fetch
// more than the budget: the operation fails before fetching past it.
let data = file(64 * 1024);
let mut c = config(1024, 1 << 20);
c.max_fetch = 8 * 1024;
let s = LazyStorage::new(data.len() as u64, c);
let e = s
.run_blocking(
|| {
(0..64)
.map(|i| owned(s.read_at(i * 1024, 8)))
.collect::<Result<Vec<_>, _>>()
},
|r| Ok(data[r.start as usize..r.end as usize].to_vec()),
)
.unwrap_err();
assert!(e.contains("maxFetch"), "{e}");
assert!(s.stats().bytes_fetched <= 8 * 1024, "{:?}", s.stats());
// Within the budget it completes, and the budget is per operation.
for _ in 0..3 {
s.run_blocking(
|| owned(s.read_at(10 * 1024, 3000)),
|r| Ok(data[r.start as usize..r.end as usize].to_vec()),
)
.unwrap()
.unwrap();
}
}
#[test] #[test]
fn a_failed_fetch_is_an_error_not_data() { fn a_failed_fetch_is_an_error_not_data() {
let data = file(4096); let data = file(4096);
+43 -4
View File
@@ -389,11 +389,14 @@ impl Http {
storage: &LazyStorage, storage: &LazyStorage,
mut f: impl FnMut() -> T, mut f: impl FnMut() -> T,
) -> Result<T, JsError> { ) -> Result<T, JsError> {
let _op = storage.operation(); let op = storage.operation();
loop { loop {
match storage.attempt(&mut f) { match storage.attempt(&mut f) {
Step::Done(v) => return Ok(v), Step::Done(v) => return Ok(v),
Step::Need(ranges) => self.fetch(storage, &ranges).await?, Step::Need(ranges) => {
op.charge(&ranges).map_err(js_err)?;
self.fetch(storage, &ranges).await?
}
} }
} }
} }
@@ -426,6 +429,24 @@ fn int_opt(opts: &JsValue, key: &str, min: f64, max: f64) -> Result<Option<u64>,
} }
} }
/// Most bytes `maxFetch` and `maxDownload` may allow: 1 GiB. wasm32 has
/// 4 GiB of memory and no buffer past 2 GiB, and what is fetched is held
/// while it is decoded.
const MAX_FETCH_LIMIT: u64 = 1 << 30;
/// The largest file `openUrl` reads by ranges: on wasm32, 4 GiB - 1 bytes.
/// The format code turns file offsets into `usize` to use them (with a
/// clean error past it, see scripts/check-32bit-casts.sh), so on a 32-bit
/// target nothing at 4 GiB or beyond can be read; a larger file is refused
/// at open rather than failing on whichever read reaches past 4 GiB. On
/// 64-bit targets it is 2^53 - 1, the largest offset a JavaScript number
/// holds exactly.
const MAX_REMOTE_LENGTH: u64 = if (usize::MAX as u64) < MAX_SAFE_INTEGER as u64 {
usize::MAX as u64
} else {
MAX_SAFE_INTEGER as u64
};
fn config_from(opts: &JsValue) -> Result<LazyConfig, JsError> { fn config_from(opts: &JsValue) -> Result<LazyConfig, JsError> {
let mut c = LazyConfig::default(); let mut c = LazyConfig::default();
if let Some(b) = int_opt(opts, "blockSize", 512.0, (64u64 << 20) as f64)? { if let Some(b) = int_opt(opts, "blockSize", 512.0, (64u64 << 20) as f64)? {
@@ -434,6 +455,12 @@ fn config_from(opts: &JsValue) -> Result<LazyConfig, JsError> {
if let Some(n) = int_opt(opts, "cacheSize", 0.0, MAX_SAFE_INTEGER)? { if let Some(n) = int_opt(opts, "cacheSize", 0.0, MAX_SAFE_INTEGER)? {
c.capacity = n; c.capacity = n;
} }
if let Some(n) = int_opt(opts, "maxFetch", 512.0, MAX_FETCH_LIMIT as f64)? {
c.max_fetch = n;
}
// Read by remote.js; checked here so a value wasm32 cannot hold is an
// option error rather than a download that cannot be kept.
int_opt(opts, "maxDownload", 0.0, MAX_FETCH_LIMIT as f64)?;
Ok(c) Ok(c)
} }
@@ -444,15 +471,21 @@ fn config_from(opts: &JsValue) -> Result<LazyConfig, JsError> {
/// `opts` (all optional): /// `opts` (all optional):
/// - `blockSize` — bytes per request block, 512 to 64 MiB (default 1 MiB); /// - `blockSize` — bytes per request block, 512 to 64 MiB (default 1 MiB);
/// - `cacheSize` — bytes of blocks kept between calls (default 64 MiB); /// - `cacheSize` — bytes of blocks kept between calls (default 64 MiB);
/// - `maxFetch` — most bytes one call may fetch, and so the longest single
/// read, up to 1 GiB (default 512 MiB): a call that would fetch more
/// fails before fetching it;
/// - `fallback` — `"download"` (default) reads the whole file when the /// - `fallback` — `"download"` (default) reads the whole file when the
/// server ignores `Range` (answers 200), up to `maxDownload` bytes /// server ignores `Range` (answers 200), up to `maxDownload` bytes
/// (default 512 MiB); `"error"` refuses such a server; /// (default 512 MiB, at most 1 GiB); `"error"` refuses such a server;
/// - `headers`, `credentials` — passed to every `fetch`; /// - `headers`, `credentials` — passed to every `fetch`;
/// - `parallel` — range requests in flight at once (default 6); /// - `parallel` — range requests in flight at once (default 6);
/// - `fetch` — a `fetch`-compatible function to use instead of the global. /// - `fetch` — a `fetch`-compatible function to use instead of the global.
/// ///
/// Cross-origin servers must allow CORS and expose `Content-Range` (or /// Cross-origin servers must allow CORS and expose `Content-Range` (or
/// answer `HEAD` with `Content-Length`). /// answer `HEAD` with `Content-Length`). A file may be up to 4 GiB - 1
/// bytes long (wasm32 offsets); a longer one is refused at open. A whole-dataset `read` that would use more than 1 GiB of
/// memory ([`core::MAX_READ_BYTES`]) is refused: read it in parts with
/// `readHyperslab`.
#[wasm_bindgen(js_name = openUrl)] #[wasm_bindgen(js_name = openUrl)]
pub async fn open_url(url: String, opts: JsValue) -> Result<RemoteFile, JsError> { pub async fn open_url(url: String, opts: JsValue) -> Result<RemoteFile, JsError> {
let config = config_from(&opts)?; let config = config_from(&opts)?;
@@ -477,6 +510,12 @@ pub async fn open_url(url: String, opts: JsValue) -> Result<RemoteFile, JsError>
.as_f64() .as_f64()
.filter(|x| x.fract() == 0.0 && (0.0..=MAX_SAFE_INTEGER).contains(x)) .filter(|x| x.fract() == 0.0 && (0.0..=MAX_SAFE_INTEGER).contains(x))
.ok_or_else(|| js_err(format!("{url}: the server gave no usable file size")))?; .ok_or_else(|| js_err(format!("{url}: the server gave no usable file size")))?;
if length as u64 > MAX_REMOTE_LENGTH {
return Err(js_err(format!(
"{url} is {length} bytes; openUrl reads files of up to {MAX_REMOTE_LENGTH} bytes \
(4 GiB - 1: the WebAssembly reader addresses a file with 32-bit offsets)"
)));
}
let http = Http { let http = Http {
url, url,
opts, opts,
+91 -7
View File
@@ -193,6 +193,7 @@ fn config(block: u64) -> LazyConfig {
// A small budget, so eviction between operations is exercised. // A small budget, so eviction between operations is exercised.
capacity: 16 * block, capacity: 16 * block,
max_request: 8 * block, max_request: 8 * block,
..LazyConfig::default()
} }
} }
@@ -310,6 +311,95 @@ fn h5py_and_netcdf4_files_read_the_same_lazily() {
eprintln!("skipping: {} lacks h5py/netCDF4/numpy", python()); eprintln!("skipping: {} lacks h5py/netCDF4/numpy", python());
return; return;
} }
let dir = fixture_dir();
for name in ["fixture.h5", "fixture.nc"] {
let data = std::fs::read(dir.path().join(name)).unwrap();
for block in [512, 64 * 1024] {
let (_, _, lines) = check_equal(name, &data, block);
assert!(lines.iter().filter(|l| l.contains(" read: Ok")).count() >= 2);
}
}
}
/// The bytes of `data` at `r`, zero past its end: a server that claims
/// the file is longer than it is.
fn fetch_padded(data: &[u8], r: Range<u64>) -> Result<Vec<u8>, String> {
let mut out = vec![0u8; (r.end - r.start) as usize];
let len = data.len() as u64;
if r.start < len {
let end = r.end.min(len);
out[..(end - r.start) as usize].copy_from_slice(&data[r.start as usize..end as usize]);
}
Ok(out)
}
/// Sizes a hostile server or a large dataset can name are errors, never
/// allocations that abort the wasm module: make_fixture.py's limits.h5 and
/// hostile_vl.h5 (see write_limits there).
#[test]
fn size_limits_are_errors_not_aborts() {
if !python_available() {
assert!(
!std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1"),
"CLAWHDF5_REQUIRE_INTEROP=1 but {} lacks h5py/netCDF4/numpy",
python()
);
eprintln!("skipping: {} lacks h5py/netCDF4/numpy", python());
return;
}
let dir = fixture_dir();
// Read whole, /huge_u8 would widen 2^28 values to 64 bits (2 GiB): an
// error naming readHyperslab, before its chunks are read. A window of
// it reads.
let data = std::fs::read(dir.path().join("limits.h5")).unwrap();
let n = (1u64 << 28) + 1024;
let window = Hyperslab {
start: vec![n - 4],
count: vec![4],
stride: None,
block: None,
};
let local = Reader::open(data.clone()).unwrap();
let lazy = Lazy::open(data, LazyConfig::default()).unwrap();
let before = lazy.storage.stats().requests;
for e in [
local.read("/huge_u8", None).unwrap_err(),
lazy.call(|r| r.read("/huge_u8", None)).unwrap_err(),
] {
assert!(e.contains("readHyperslab"), "{e}");
}
assert_eq!(lazy.storage.stats().requests, before, "nothing fetched");
for part in [
local.read("/huge_u8", Some(&window)).unwrap(),
lazy.call(|r| r.read("/huge_u8", Some(&window))).unwrap(),
] {
assert_eq!(format!("{:?}", part.data), "U8([0, 0, 0, 7])");
}
// A server that claims 3 GiB and a heap collection of 2 GiB + 4 KiB:
// reading the strings fails at once, fetching a few blocks.
let data = std::fs::read(dir.path().join("hostile_vl.h5")).unwrap();
let storage = Arc::new(LazyStorage::new(3 << 30, LazyConfig::default()));
let s = storage.clone();
let reader = storage
.run_blocking(
|| Reader::open_storage(s.clone()),
|r| fetch_padded(&data, r),
)
.unwrap()
.unwrap();
let e = storage
.run_blocking(|| reader.read("/a", None), |r| fetch_padded(&data, r))
.unwrap()
.unwrap_err();
assert!(e.contains("maxFetch"), "{e}");
let st = storage.stats();
assert!(st.requests <= 4 && st.bytes_fetched <= 4 << 20, "{st:?}");
}
/// make_fixture.py's files, written to a temporary directory.
fn fixture_dir() -> tempfile::TempDir {
let dir = tempfile::tempdir().unwrap(); let dir = tempfile::tempdir().unwrap();
let generator = Path::new(env!("CARGO_MANIFEST_DIR")) let generator = Path::new(env!("CARGO_MANIFEST_DIR"))
.join("../../examples/wasm-viewer/test/make_fixture.py"); .join("../../examples/wasm-viewer/test/make_fixture.py");
@@ -323,13 +413,7 @@ fn h5py_and_netcdf4_files_read_the_same_lazily() {
"{}", "{}",
String::from_utf8_lossy(&out.stderr) String::from_utf8_lossy(&out.stderr)
); );
for name in ["fixture.h5", "fixture.nc"] { dir
let data = std::fs::read(dir.path().join(name)).unwrap();
for block in [512, 64 * 1024] {
let (_, _, lines) = check_equal(name, &data, block);
assert!(lines.iter().filter(|l| l.contains(" read: Ok")).count() >= 2);
}
}
} }
fn hdf5_files(dir: &Path, out: &mut Vec<PathBuf>) { fn hdf5_files(dir: &Path, out: &mut Vec<PathBuf>) {
+69 -1
View File
@@ -3,7 +3,9 @@ reads back from them, for the clawhdf5-wasm tests.
python make_fixture.py OUT_DIR python make_fixture.py OUT_DIR
writes OUT_DIR/fixture.h5, OUT_DIR/fixture.nc and OUT_DIR/expected.json. writes OUT_DIR/fixture.h5, OUT_DIR/fixture.nc and OUT_DIR/expected.json,
and the limit-test files OUT_DIR/limits.h5 and OUT_DIR/hostile_vl.h5 (see
write_limits).
Both the Rust test (crates/clawhdf5-wasm/tests/h5py_interop.rs, native) and Both the Rust test (crates/clawhdf5-wasm/tests/h5py_interop.rs, native) and
the Node test (test.mjs, the built wasm package) compare against the same the Node test (test.mjs, the built wasm package) compare against the same
expected.json, so the two check the same values. expected.json, so the two check the same values.
@@ -204,6 +206,72 @@ json.dump({"fixture.h5": describe(h5), "fixture.nc": describe(nc)},
open(out / "expected.json", "w"), indent=1, ensure_ascii=False) open(out / "expected.json", "w"), indent=1, ensure_ascii=False)
HUGE_U8 = 2**28 + 1024
# The collection size hostile_vl.h5 claims: past 2 GiB, which a wasm32
# buffer cannot hold.
HOSTILE_GCOL_SIZE = 2**31 + 4096
# The file length a server claims for hostile_vl.h5 (the tests' mock fetch
# answers every range with zeros past the real bytes): 3 GiB, within what
# wasm32 opens, and room for the collection.
HOSTILE_LENGTH = 3 << 30
def write_limits(out):
"""Files for the size limits (the tests must get errors, not aborts):
- limits.h5: /huge_u8, 2^28 + 1024 bytes of u8 in compressed chunks
(a small file): read whole it would take over 2 GiB while decoding;
its last value is 7.
- hostile_vl.h5: a variable-length string dataset /a whose global heap
collection claims HOSTILE_GCOL_SIZE bytes, with the superblock's end
of file set to HOSTILE_LENGTH (libhdf5 cannot read it; it is only
served by a mock that claims that length).
- far.h5 and far.json: /x, 16 float64 values, whose contiguous data
address is moved FAR_SHIFT bytes on (past 2 GiB, the sign bit of a
wasm32 isize) in a file whose end of file is moved as far; the tests'
mock serves the data there, to show offsets up to 4 GiB work on
wasm32.
"""
with h5py.File(out / "limits.h5", "w") as f:
d = f.create_dataset("huge_u8", shape=(HUGE_U8,), dtype="u1",
chunks=(1 << 20,), compression="gzip")
d[-1] = 7
path = out / "hostile_vl.h5"
with h5py.File(path, "w", libver="earliest") as f:
f.create_dataset("a", data=["x", "yy"], dtype=h5py.string_dtype())
b = bytearray(path.read_bytes())
assert b[8] == 0, "a version 0 superblock"
b[40:48] = HOSTILE_LENGTH.to_bytes(8, "little") # end of file address
at = b.index(b"GCOL")
b[at + 8:at + 16] = HOSTILE_GCOL_SIZE.to_bytes(8, "little")
path.write_bytes(bytes(b))
path = out / "far.h5"
values = np.arange(16, dtype="<f8") * 1.5
with h5py.File(path, "w", libver="earliest") as f:
f.create_dataset("x", data=values)
data_at = f["x"].id.get_offset()
b = bytearray(path.read_bytes())
# The layout message: the data's address, then its size.
old = data_at.to_bytes(8, "little") + (values.nbytes).to_bytes(8, "little")
assert b.count(old) == 1
at = b.index(old)
b[at:at + 8] = (data_at + FAR_SHIFT).to_bytes(8, "little")
length = len(b) + FAR_SHIFT
b[40:48] = length.to_bytes(8, "little")
path.write_bytes(bytes(b))
json.dump({"data_at": data_at, "far_at": data_at + FAR_SHIFT,
"nbytes": values.nbytes, "length": length,
"values": [float(x) for x in values]},
open(out / "far.json", "w"))
# far.h5's data moves this far: past 2 GiB, below 4 GiB.
FAR_SHIFT = 3 << 30
write_limits(out)
def write_big(path, megabytes): def write_big(path, megabytes):
"""A large file for the range-request tests (`openUrl`): `/big`, about """A large file for the range-request tests (`openUrl`): `/big`, about
`megabytes` MB of float64 in 1 MiB chunks, written after a small `megabytes` MB of float64 in 1 MiB chunks, written after a small
+83
View File
@@ -321,6 +321,8 @@ async function remoteTests() {
await f.read("/grid"); await f.read("/grid");
}, /unusable Content-Range/, "unusable Content-Range"); }, /unusable Content-Range/, "unusable Content-Range");
await limitTests();
// Corpus files: what the viewer can show of each is the same read by // Corpus files: what the viewer can show of each is the same read by
// ranges as in memory (an error wherever it gives one). // ranges as in memory (an error wherever it gives one).
const corpus = process.env.CLAWHDF5_WASM_CORPUS; const corpus = process.env.CLAWHDF5_WASM_CORPUS;
@@ -328,6 +330,87 @@ async function remoteTests() {
console.log(`openUrl: ${checks - before} checks passed`); console.log(`openUrl: ${checks - before} checks passed`);
} }
// A fetch that serves `buf` as a file of `total` bytes (zeros past the end
// of `buf`, and `extra` = [[offset, bytes], ...] laid over them), counting
// its calls. `hide` leaves out Content-Range and the validators, as a
// cross-origin server that exposes neither does; HEAD then gives the length.
function mockFetch(buf, { total = buf.length, extra = [], hide = false } = {}) {
const f = async (url, init) => {
f.calls++;
if (init.method === "HEAD") {
return new Response(null, { status: 200, headers: { "Content-Length": String(total) } });
}
const m = /^bytes=(\d+)-(\d+)$/.exec(new Headers(init.headers).get("Range"));
const a = Number(m[1]);
const b = Math.min(Number(m[2]) + 1, total);
const out = new Uint8Array(b - a);
for (const [at, bytes] of [[0, buf], ...extra]) {
const from = Math.max(a, at);
const to = Math.min(b, at + bytes.length);
if (from < to) out.set(bytes.subarray(from - at, to - at), from - a);
}
const headers = { "Content-Length": String(b - a) };
if (!hide) headers["Content-Range"] = `bytes ${a}-${b - 1}/${total}`;
return new Response(out, { status: 206, headers });
};
f.calls = 0;
return f;
}
// Sizes a hostile server or a large dataset can name are errors, never an
// allocation that aborts the module (which would take every open file on
// the page with it); see write_limits in make_fixture.py.
async function limitTests() {
// Read whole, /huge_u8 (2^28 + 1024 bytes) would take over 2 GiB while
// decoding: refused before its chunks are fetched; a window reads.
const n = 2 ** 28 + 1024;
const limits = readFileSync(join(fixDir, "limits.h5"));
const local = pkg.open(new Uint8Array(limits));
await fails(() => local.read("/huge_u8"), /readHyperslab/, "huge read, in memory");
const remote = await pkg.openUrl(`${base}/fix/limits.h5`);
const before = remote.stats().requests;
await fails(() => remote.read("/huge_u8"), /readHyperslab/, "huge read, openUrl");
eq(remote.stats().requests, before, "huge read fetched nothing");
eq(Array.from((await remote.readHyperslab("/huge_u8", [n - 4], [4])).data), [0, 0, 0, 7], "huge window");
eq(Array.from(local.readHyperslab("/huge_u8", [n - 4], [4]).data), [0, 0, 0, 7], "huge window, in memory");
local.free();
// A server claiming 3 GiB, a heap collection claiming 2 GiB + 4 KiB: an
// error after a few requests (it used to fetch 2 GiB, then abort).
const hostile = mockFetch(new Uint8Array(readFileSync(join(fixDir, "hostile_vl.h5"))), { total: 3 * 2 ** 30 });
const h = await pkg.openUrl("http://hostile.invalid/h.h5", { fetch: hostile });
await fails(() => h.read("/a"), /maxFetch/, "hostile collection size");
assert.ok(hostile.calls <= 4, `hostile: ${hostile.calls} requests`);
// The module survived: files open and read.
eq((await (await pkg.openUrl(`${base}/fix/fixture.h5`)).read("/sensors/temp")).data[0], 21.5, "alive after hostile");
// A call that would fetch more than maxFetch fails before fetching it.
await fails(async () => {
const f = await pkg.openUrl(`${base}/fix/fixture.h5`, { blockSize: 512, maxFetch: 1024 });
await f.read("/grid");
}, /maxFetch/, "maxFetch");
await fails(() => pkg.openUrl(`${base}/fix/fixture.h5`, { maxFetch: 2 ** 31 }), /maxFetch/, "maxFetch range");
await fails(() => pkg.openUrl(`${base}/fix/fixture.h5`, { maxDownload: 2 ** 31 }), /maxDownload/, "maxDownload range");
// Offsets past 2 GiB work on wasm32: far.h5's data sits at 3 GiB.
const far = JSON.parse(readFileSync(join(fixDir, "far.json"), "utf8"));
const farBytes = new Uint8Array(readFileSync(join(fixDir, "far.h5")));
const farData = farBytes.subarray(far.data_at, far.data_at + far.nbytes);
const ff = await pkg.openUrl("http://far.invalid/far.h5", {
fetch: mockFetch(farBytes, { total: far.length, extra: [[far.far_at, farData]] }),
});
eq(Array.from((await ff.read("/x")).data), far.values, "data past 2 GiB");
eq(ff.stats().size, far.length, "size past 2 GiB");
// wasm32's reader holds file offsets in 32 bits: a file of 4 GiB or more
// is refused at open, with the limit in the message (past 2^53 - 1 bytes
// a JavaScript number is not even exact).
for (const total of [2 ** 32, 2 ** 40, 2 ** 53 + 2]) {
await fails(() => pkg.openUrl("http://huge.invalid/x.h5", { fetch: mockFetch(farBytes, { total }) }),
total > 2 ** 53 ? /2\^53 - 1/ : /4 GiB/, `length ${total}`);
}
}
function hdf5Files(dir, out) { function hdf5Files(dir, out) {
for (const e of readdirSync(dir, { withFileTypes: true })) { for (const e of readdirSync(dir, { withFileTypes: true })) {
const p = join(dir, e.name); const p = join(dir, e.name);