wasm: sizes a server or a dataset names are errors, not aborts

A read longer than isize::MAX (2 GiB on wasm32) aborted the module in
LazyStorage::assemble (capacity_overflow), taking every open file on the
page with it, and a hostile server only had to claim a large length and
serve a heap collection of 2 GiB + 4 KiB to get there (after fetching
2 GiB). Reading a large u8 dataset whole aborted the same way when its
values were widened to 64 bits.

- LazyConfig::max_fetch (openUrl option maxFetch, default 512 MiB, at
  most 1 GiB): a read longer than it fails at once, before anything is
  fetched, and an operation whose passes would fetch more than it fails
  before fetching (Operation::charge). assemble reserves fallibly.
- Reader::read refuses a read that would use more than 1 GiB while
  decoding (core::MAX_READ_BYTES: stored bytes + 64-bit values + result)
  with an error naming readHyperslab, before reading.
- openUrl refuses a file of 4 GiB or more at open on wasm32: the format
  code turns offsets into usize, so nothing past 4 GiB can be read there
  (shown by a new test: data at 3 GiB reads, a 4 GiB file is refused).
  maxDownload is bounded to 1 GiB.

Tests: make_fixture.py writes limits.h5 (a sparse 2^28 + 1024 byte u8
dataset), hostile_vl.h5 (the reviewer's collection) and far.h5 (data at
3 GiB); test.mjs (wasm32) and tests/lazy.rs (native) check each is an
error or reads, and that the module survives. Before: RuntimeError:
unreachable in Node; the native test read the huge dataset and fetched
2 GiB.

Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
osobh
2026-09-27 07:36:21 -05:00
co-authored by Claude Opus 5.5
parent f825a89e23
commit dbafa952ac
7 changed files with 470 additions and 26 deletions
+5 -1
View File
@@ -117,7 +117,11 @@ export async function probe(url, firstLen, opts) {
}
length = Number(cl);
}
if (!Number.isSafeInteger(length) || first.length !== Math.min(firstLen, length)) {
if (!Number.isSafeInteger(length) || length < 0) {
throw new Error(`${url}: the server gave a file size of ${length} bytes; openUrl reads files ` +
"of up to 2^53 - 1 bytes (the largest offset a JavaScript number holds exactly)");
}
if (first.length !== Math.min(firstLen, length)) {
throw new Error(`${url}: asked for the first ${firstLen} bytes of ${length}, got ${first.length}`);
}
return { length, first, validator: validatorOf(resp), requests };
+46 -2
View File
@@ -14,6 +14,15 @@ use clawhdf5_format::datatype::{Datatype, DatatypeByteOrder};
use clawhdf5_format::storage::Storage;
use clawhdf5_format::vl_data::{VlResolver, check_element_size};
/// The most memory one read may use while it decodes: the stored bytes,
/// the values at 64 bits (integers are widened first) and the values
/// returned. A larger read fails with an error naming `readHyperslab`,
/// before anything is read: on wasm32 a buffer past 2 GiB cannot be
/// allocated at all, and failing to allocate aborts the module (every open
/// file on the page with it). 1 GiB leaves room in wasm32's 4 GiB for the
/// file's cached blocks and the JavaScript copy of the result.
pub const MAX_READ_BYTES: u64 = 1 << 30;
/// Errors are reported to JavaScript as messages.
pub type Result<T> = std::result::Result<T, String>;
@@ -241,13 +250,27 @@ impl Reader {
if let Datatype::VariableLength { size, .. } = array_base(&dt) {
check_element_size(*size, self.file.superblock().offset_size).map_err(err)?;
}
let raw = ds.read_selection(&selection).map_err(err)?;
let data = self.decode(&raw, &dt)?;
out_shape.extend(element_shape(&dt));
let expected = out_shape
.iter()
.try_fold(1u64, |acc, &d| acc.checked_mul(d))
.ok_or("selection size overflows")?;
let cost = expected.saturating_mul(bytes_per_value(&dt));
if cost > MAX_READ_BYTES {
return Err(format!(
"reading {path}{} would take about {} MiB of memory, more than the {} MiB \
one read may use; read it in parts (readHyperslab)",
if slab.is_some() {
" (this selection)"
} else {
" whole"
},
cost >> 20,
MAX_READ_BYTES >> 20
));
}
let raw = ds.read_selection(&selection).map_err(err)?;
let data = self.decode(&raw, &dt)?;
if data.len() as u64 != expected {
return Err(format!(
"read {} values for shape {out_shape:?} ({expected} expected)",
@@ -316,6 +339,27 @@ impl Reader {
}
}
/// Memory one value of type `dt` takes while [`Reader::read`] decodes it
/// (an array type's elements count as values): its stored bytes, plus what
/// [`Reader::decode`] builds from them. A string counts its `String` (24
/// bytes on 64-bit targets, less on wasm32) and, for a fixed-length one,
/// its text; a variable-length string's text lives in the heap and is
/// bounded by the storage's own read limit.
fn bytes_per_value(dt: &Datatype) -> u64 {
let base = array_base(dt);
let stored = u64::from(base.type_size());
stored
+ match base {
Datatype::FloatingPoint { size, .. } if *size <= 4 => 4,
Datatype::FloatingPoint { .. } => 8,
// Widened to 64 bits, then narrowed to a new vector.
Datatype::FixedPoint { .. } => 8 + stored,
Datatype::String { .. } => 24 + stored,
Datatype::VariableLength { .. } | Datatype::Enumeration { .. } => 24,
_ => 0,
}
}
/// Narrow integers read at 64 bits to the dataset's own width. The source is
/// that width, so this cannot fail on correct input; it is checked anyway.
fn narrow<S: Copy + std::fmt::Display, T: TryFrom<S>>(v: Vec<S>) -> Result<Vec<T>> {
+133 -11
View File
@@ -42,6 +42,10 @@ use clawhdf5_format::storage::Storage;
/// `docs/design/range-reads.md` §2 measured).
pub const DEFAULT_BLOCK_SIZE: u64 = 1 << 20;
/// Default of [`LazyConfig::max_fetch`]: 512 MiB, the same as `openUrl`'s
/// `maxDownload` for a server without range support.
pub const DEFAULT_MAX_FETCH: u64 = 512 << 20;
/// The message of the error a read that misses returns. It never reaches
/// the caller of [`LazyStorage::attempt`]: a pass that missed is re-run.
pub const NEED_BYTES: &str = "bytes not fetched yet (restartable read)";
@@ -58,6 +62,14 @@ pub struct LazyConfig {
/// Largest single range asked for, in bytes (whole blocks, at least
/// one); longer runs are split so they can be fetched in parallel.
pub max_request: u64,
/// Most bytes one operation may fetch (at least one block), and so the
/// longest single read: a read longer than this fails at once, before
/// anything is fetched, and so does an operation whose passes would
/// fetch more. The file's length comes from the server, so without
/// this a hostile file (a heap "collection" claiming 2 GiB) makes the
/// reader fetch and hold whatever it names; on wasm32 a buffer past
/// 2 GiB cannot even be allocated.
pub max_fetch: u64,
}
impl Default for LazyConfig {
@@ -66,6 +78,7 @@ impl Default for LazyConfig {
block_size: DEFAULT_BLOCK_SIZE,
capacity: 64 << 20,
max_request: 8 << 20,
max_fetch: DEFAULT_MAX_FETCH,
}
}
}
@@ -137,6 +150,28 @@ fn lock(m: &Mutex<State>) -> MutexGuard<'_, State> {
/// [`LazyStorage::operation`].
pub struct Operation<'a> {
storage: &'a LazyStorage,
/// Bytes fetched for this operation so far.
fetched: std::cell::Cell<u64>,
}
impl Operation<'_> {
/// Count `ranges` against the operation's budget
/// ([`LazyConfig::max_fetch`]) before they are fetched: an error, and
/// nothing counted, if they would take it past the budget.
pub fn charge(&self, ranges: &[Range<u64>]) -> Result<(), String> {
let max = self.storage.config.max_fetch;
let total = ranges.iter().fold(self.fetched.get(), |n, r| {
n.saturating_add(r.end.saturating_sub(r.start))
});
if total > max {
return Err(format!(
"this call would fetch more than {max} bytes of the file (the maxFetch limit); \
read less at a time (readHyperslab) or raise maxFetch"
));
}
self.fetched.set(total);
Ok(())
}
}
impl Drop for Operation<'_> {
@@ -154,6 +189,7 @@ impl LazyStorage {
pub fn new(len: u64, mut config: LazyConfig) -> Self {
config.block_size = config.block_size.max(512);
config.max_request = (config.max_request / config.block_size).max(1) * config.block_size;
config.max_fetch = config.max_fetch.max(config.block_size);
LazyStorage {
len,
config,
@@ -180,7 +216,10 @@ impl LazyStorage {
/// Hold it across every pass of one operation.
pub fn operation(&self) -> Operation<'_> {
lock(&self.state).active += 1;
Operation { storage: self }
Operation {
storage: self,
fetched: std::cell::Cell::new(0),
}
}
/// Run one pass of `f` over this storage. `Done` when `f` read nothing
@@ -268,11 +307,12 @@ impl LazyStorage {
mut f: impl FnMut() -> T,
mut fetch: impl FnMut(Range<u64>) -> Result<Vec<u8>, String>,
) -> Result<T, String> {
let _op = self.operation();
let op = self.operation();
loop {
match self.attempt(&mut f) {
Step::Done(v) => return Ok(v),
Step::Need(ranges) => {
op.charge(&ranges)?;
for r in ranges {
let bytes = fetch(r.clone())?;
self.supply_range(&r, &bytes)?;
@@ -395,9 +435,35 @@ impl LazyStorage {
Ok(have)
}
fn assemble(&self, offset: u64, end: u64, blocks: &HashMap<u64, Arc<[u8]>>) -> Vec<u8> {
/// Refuse a read of `n` bytes longer than an operation may fetch
/// ([`LazyConfig::max_fetch`]), before its blocks are asked for.
fn check_len(&self, n: u64) -> Result<(), FormatError> {
let max = self.config.max_fetch;
if n > max {
return Err(FormatError::Storage(format!(
"a read of {n} bytes is more than one call may fetch ({max} bytes, the maxFetch limit)"
)));
}
Ok(())
}
/// The bytes `offset..end` from `blocks`, which hold every block of
/// that span. The buffer is reserved fallibly: a length the address
/// space cannot hold (past `isize::MAX` on wasm32) is an error, never
/// an abort.
fn assemble(
&self,
offset: u64,
end: u64,
blocks: &HashMap<u64, Arc<[u8]>>,
) -> Result<Vec<u8>, FormatError> {
let bs = self.config.block_size;
let mut out = Vec::with_capacity(usize::try_from(end - offset).unwrap_or(0));
let n = end - offset;
let too_long =
|| FormatError::Storage(format!("cannot hold a read of {n} bytes in memory"));
let mut out = Vec::new();
out.try_reserve_exact(usize::try_from(n).map_err(|_| too_long())?)
.map_err(|_| too_long())?;
let mut pos = offset;
while pos < end {
let i = pos / bs;
@@ -407,7 +473,7 @@ impl LazyStorage {
out.extend_from_slice(&block[from..to]);
pos = i * bs + to as u64;
}
out
Ok(out)
}
}
@@ -416,10 +482,11 @@ impl Storage for LazyStorage {
let Some(span) = self.span(offset, len as u64) else {
return Ok(Cow::Owned(Vec::new()));
};
let end = offset.saturating_add(len as u64).min(self.len);
self.check_len(end - offset)?;
let metadata = len as u64 <= self.config.block_size;
let blocks = self.blocks(std::slice::from_ref(&span), metadata)?;
let end = offset.saturating_add(len as u64).min(self.len);
Ok(Cow::Owned(self.assemble(offset, end, &blocks)))
Ok(Cow::Owned(self.assemble(offset, end, &blocks)?))
}
fn len(&self) -> u64 {
@@ -428,26 +495,29 @@ impl Storage for LazyStorage {
fn read_ranges(&self, ranges: &[Range<u64>]) -> Result<Vec<Cow<'_, [u8]>>, FormatError> {
let mut spans = Vec::with_capacity(ranges.len());
let mut total = 0u64;
for r in ranges {
if r.end < r.start {
return Err(FormatError::Storage(
"read range ends before it starts".into(),
));
}
total = total.saturating_add(r.end.min(self.len).saturating_sub(r.start));
spans.extend(self.span(r.start, r.end - r.start));
}
self.check_len(total)?;
let blocks = self.blocks(&spans, false)?;
Ok(ranges
ranges
.iter()
.map(|r| {
let end = r.end.min(self.len);
if r.start >= end {
Cow::Owned(Vec::new())
Ok(Cow::Owned(Vec::new()))
} else {
Cow::Owned(self.assemble(r.start, end, &blocks))
self.assemble(r.start, end, &blocks).map(Cow::Owned)
}
})
.collect())
.collect()
}
}
@@ -465,6 +535,7 @@ mod tests {
block_size: block,
capacity,
max_request: 4 * block,
max_fetch: DEFAULT_MAX_FETCH,
}
}
@@ -654,6 +725,57 @@ mod tests {
assert_eq!(s.stats().requests, before.requests);
}
#[test]
fn a_read_longer_than_max_fetch_fails_without_fetching() {
// A hostile file names a 2 GiB heap collection in a "file" the
// server claims is 1 TiB: the read is refused before any block is
// asked for (on wasm32 its buffer could not even be allocated).
let s = LazyStorage::new(1 << 40, LazyConfig::default());
let step = s.attempt(|| s.read_at(4096, (1usize << 31) + 4096).map(|b| b.len()));
match step {
Step::Done(Err(e)) => assert!(e.to_string().contains("maxFetch"), "{e}"),
other => panic!("expected a refusal, got {other:?}"),
}
let step = s.attempt(|| s.read_ranges(&[0..(600 << 20)]).map(|v| v.len()));
assert!(matches!(step, Step::Done(Err(_))), "{step:?}");
assert_eq!(s.stats().requests, 0);
// At the limit it is an ordinary miss.
let s = LazyStorage::new(1 << 40, config(1024, 1 << 20));
let step = s.attempt(|| s.read_at(0, DEFAULT_MAX_FETCH as usize).map(|b| b.len()));
assert!(matches!(step, Step::Need(_)), "{step:?}");
}
#[test]
fn an_operation_stops_at_its_fetch_budget() {
// Many small reads, none over the limit, that together would fetch
// more than the budget: the operation fails before fetching past it.
let data = file(64 * 1024);
let mut c = config(1024, 1 << 20);
c.max_fetch = 8 * 1024;
let s = LazyStorage::new(data.len() as u64, c);
let e = s
.run_blocking(
|| {
(0..64)
.map(|i| owned(s.read_at(i * 1024, 8)))
.collect::<Result<Vec<_>, _>>()
},
|r| Ok(data[r.start as usize..r.end as usize].to_vec()),
)
.unwrap_err();
assert!(e.contains("maxFetch"), "{e}");
assert!(s.stats().bytes_fetched <= 8 * 1024, "{:?}", s.stats());
// Within the budget it completes, and the budget is per operation.
for _ in 0..3 {
s.run_blocking(
|| owned(s.read_at(10 * 1024, 3000)),
|r| Ok(data[r.start as usize..r.end as usize].to_vec()),
)
.unwrap()
.unwrap();
}
}
#[test]
fn a_failed_fetch_is_an_error_not_data() {
let data = file(4096);
+43 -4
View File
@@ -389,11 +389,14 @@ impl Http {
storage: &LazyStorage,
mut f: impl FnMut() -> T,
) -> Result<T, JsError> {
let _op = storage.operation();
let op = storage.operation();
loop {
match storage.attempt(&mut f) {
Step::Done(v) => return Ok(v),
Step::Need(ranges) => self.fetch(storage, &ranges).await?,
Step::Need(ranges) => {
op.charge(&ranges).map_err(js_err)?;
self.fetch(storage, &ranges).await?
}
}
}
}
@@ -426,6 +429,24 @@ fn int_opt(opts: &JsValue, key: &str, min: f64, max: f64) -> Result<Option<u64>,
}
}
/// Most bytes `maxFetch` and `maxDownload` may allow: 1 GiB. wasm32 has
/// 4 GiB of memory and no buffer past 2 GiB, and what is fetched is held
/// while it is decoded.
const MAX_FETCH_LIMIT: u64 = 1 << 30;
/// The largest file `openUrl` reads by ranges: on wasm32, 4 GiB - 1 bytes.
/// The format code turns file offsets into `usize` to use them (with a
/// clean error past it, see scripts/check-32bit-casts.sh), so on a 32-bit
/// target nothing at 4 GiB or beyond can be read; a larger file is refused
/// at open rather than failing on whichever read reaches past 4 GiB. On
/// 64-bit targets it is 2^53 - 1, the largest offset a JavaScript number
/// holds exactly.
const MAX_REMOTE_LENGTH: u64 = if (usize::MAX as u64) < MAX_SAFE_INTEGER as u64 {
usize::MAX as u64
} else {
MAX_SAFE_INTEGER as u64
};
fn config_from(opts: &JsValue) -> Result<LazyConfig, JsError> {
let mut c = LazyConfig::default();
if let Some(b) = int_opt(opts, "blockSize", 512.0, (64u64 << 20) as f64)? {
@@ -434,6 +455,12 @@ fn config_from(opts: &JsValue) -> Result<LazyConfig, JsError> {
if let Some(n) = int_opt(opts, "cacheSize", 0.0, MAX_SAFE_INTEGER)? {
c.capacity = n;
}
if let Some(n) = int_opt(opts, "maxFetch", 512.0, MAX_FETCH_LIMIT as f64)? {
c.max_fetch = n;
}
// Read by remote.js; checked here so a value wasm32 cannot hold is an
// option error rather than a download that cannot be kept.
int_opt(opts, "maxDownload", 0.0, MAX_FETCH_LIMIT as f64)?;
Ok(c)
}
@@ -444,15 +471,21 @@ fn config_from(opts: &JsValue) -> Result<LazyConfig, JsError> {
/// `opts` (all optional):
/// - `blockSize` — bytes per request block, 512 to 64 MiB (default 1 MiB);
/// - `cacheSize` — bytes of blocks kept between calls (default 64 MiB);
/// - `maxFetch` — most bytes one call may fetch, and so the longest single
/// read, up to 1 GiB (default 512 MiB): a call that would fetch more
/// fails before fetching it;
/// - `fallback` — `"download"` (default) reads the whole file when the
/// server ignores `Range` (answers 200), up to `maxDownload` bytes
/// (default 512 MiB); `"error"` refuses such a server;
/// (default 512 MiB, at most 1 GiB); `"error"` refuses such a server;
/// - `headers`, `credentials` — passed to every `fetch`;
/// - `parallel` — range requests in flight at once (default 6);
/// - `fetch` — a `fetch`-compatible function to use instead of the global.
///
/// Cross-origin servers must allow CORS and expose `Content-Range` (or
/// answer `HEAD` with `Content-Length`).
/// answer `HEAD` with `Content-Length`). A file may be up to 4 GiB - 1
/// bytes long (wasm32 offsets); a longer one is refused at open. A whole-dataset `read` that would use more than 1 GiB of
/// memory ([`core::MAX_READ_BYTES`]) is refused: read it in parts with
/// `readHyperslab`.
#[wasm_bindgen(js_name = openUrl)]
pub async fn open_url(url: String, opts: JsValue) -> Result<RemoteFile, JsError> {
let config = config_from(&opts)?;
@@ -477,6 +510,12 @@ pub async fn open_url(url: String, opts: JsValue) -> Result<RemoteFile, JsError>
.as_f64()
.filter(|x| x.fract() == 0.0 && (0.0..=MAX_SAFE_INTEGER).contains(x))
.ok_or_else(|| js_err(format!("{url}: the server gave no usable file size")))?;
if length as u64 > MAX_REMOTE_LENGTH {
return Err(js_err(format!(
"{url} is {length} bytes; openUrl reads files of up to {MAX_REMOTE_LENGTH} bytes \
(4 GiB - 1: the WebAssembly reader addresses a file with 32-bit offsets)"
)));
}
let http = Http {
url,
opts,
+91 -7
View File
@@ -193,6 +193,7 @@ fn config(block: u64) -> LazyConfig {
// A small budget, so eviction between operations is exercised.
capacity: 16 * block,
max_request: 8 * block,
..LazyConfig::default()
}
}
@@ -310,6 +311,95 @@ fn h5py_and_netcdf4_files_read_the_same_lazily() {
eprintln!("skipping: {} lacks h5py/netCDF4/numpy", python());
return;
}
let dir = fixture_dir();
for name in ["fixture.h5", "fixture.nc"] {
let data = std::fs::read(dir.path().join(name)).unwrap();
for block in [512, 64 * 1024] {
let (_, _, lines) = check_equal(name, &data, block);
assert!(lines.iter().filter(|l| l.contains(" read: Ok")).count() >= 2);
}
}
}
/// The bytes of `data` at `r`, zero past its end: a server that claims
/// the file is longer than it is.
fn fetch_padded(data: &[u8], r: Range<u64>) -> Result<Vec<u8>, String> {
let mut out = vec![0u8; (r.end - r.start) as usize];
let len = data.len() as u64;
if r.start < len {
let end = r.end.min(len);
out[..(end - r.start) as usize].copy_from_slice(&data[r.start as usize..end as usize]);
}
Ok(out)
}
/// Sizes a hostile server or a large dataset can name are errors, never
/// allocations that abort the wasm module: make_fixture.py's limits.h5 and
/// hostile_vl.h5 (see write_limits there).
#[test]
fn size_limits_are_errors_not_aborts() {
if !python_available() {
assert!(
!std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1"),
"CLAWHDF5_REQUIRE_INTEROP=1 but {} lacks h5py/netCDF4/numpy",
python()
);
eprintln!("skipping: {} lacks h5py/netCDF4/numpy", python());
return;
}
let dir = fixture_dir();
// Read whole, /huge_u8 would widen 2^28 values to 64 bits (2 GiB): an
// error naming readHyperslab, before its chunks are read. A window of
// it reads.
let data = std::fs::read(dir.path().join("limits.h5")).unwrap();
let n = (1u64 << 28) + 1024;
let window = Hyperslab {
start: vec![n - 4],
count: vec![4],
stride: None,
block: None,
};
let local = Reader::open(data.clone()).unwrap();
let lazy = Lazy::open(data, LazyConfig::default()).unwrap();
let before = lazy.storage.stats().requests;
for e in [
local.read("/huge_u8", None).unwrap_err(),
lazy.call(|r| r.read("/huge_u8", None)).unwrap_err(),
] {
assert!(e.contains("readHyperslab"), "{e}");
}
assert_eq!(lazy.storage.stats().requests, before, "nothing fetched");
for part in [
local.read("/huge_u8", Some(&window)).unwrap(),
lazy.call(|r| r.read("/huge_u8", Some(&window))).unwrap(),
] {
assert_eq!(format!("{:?}", part.data), "U8([0, 0, 0, 7])");
}
// A server that claims 3 GiB and a heap collection of 2 GiB + 4 KiB:
// reading the strings fails at once, fetching a few blocks.
let data = std::fs::read(dir.path().join("hostile_vl.h5")).unwrap();
let storage = Arc::new(LazyStorage::new(3 << 30, LazyConfig::default()));
let s = storage.clone();
let reader = storage
.run_blocking(
|| Reader::open_storage(s.clone()),
|r| fetch_padded(&data, r),
)
.unwrap()
.unwrap();
let e = storage
.run_blocking(|| reader.read("/a", None), |r| fetch_padded(&data, r))
.unwrap()
.unwrap_err();
assert!(e.contains("maxFetch"), "{e}");
let st = storage.stats();
assert!(st.requests <= 4 && st.bytes_fetched <= 4 << 20, "{st:?}");
}
/// make_fixture.py's files, written to a temporary directory.
fn fixture_dir() -> tempfile::TempDir {
let dir = tempfile::tempdir().unwrap();
let generator = Path::new(env!("CARGO_MANIFEST_DIR"))
.join("../../examples/wasm-viewer/test/make_fixture.py");
@@ -323,13 +413,7 @@ fn h5py_and_netcdf4_files_read_the_same_lazily() {
"{}",
String::from_utf8_lossy(&out.stderr)
);
for name in ["fixture.h5", "fixture.nc"] {
let data = std::fs::read(dir.path().join(name)).unwrap();
for block in [512, 64 * 1024] {
let (_, _, lines) = check_equal(name, &data, block);
assert!(lines.iter().filter(|l| l.contains(" read: Ok")).count() >= 2);
}
}
dir
}
fn hdf5_files(dir: &Path, out: &mut Vec<PathBuf>) {
+69 -1
View File
@@ -3,7 +3,9 @@ reads back from them, for the clawhdf5-wasm tests.
python make_fixture.py OUT_DIR
writes OUT_DIR/fixture.h5, OUT_DIR/fixture.nc and OUT_DIR/expected.json.
writes OUT_DIR/fixture.h5, OUT_DIR/fixture.nc and OUT_DIR/expected.json,
and the limit-test files OUT_DIR/limits.h5 and OUT_DIR/hostile_vl.h5 (see
write_limits).
Both the Rust test (crates/clawhdf5-wasm/tests/h5py_interop.rs, native) and
the Node test (test.mjs, the built wasm package) compare against the same
expected.json, so the two check the same values.
@@ -204,6 +206,72 @@ json.dump({"fixture.h5": describe(h5), "fixture.nc": describe(nc)},
open(out / "expected.json", "w"), indent=1, ensure_ascii=False)
HUGE_U8 = 2**28 + 1024
# The collection size hostile_vl.h5 claims: past 2 GiB, which a wasm32
# buffer cannot hold.
HOSTILE_GCOL_SIZE = 2**31 + 4096
# The file length a server claims for hostile_vl.h5 (the tests' mock fetch
# answers every range with zeros past the real bytes): 3 GiB, within what
# wasm32 opens, and room for the collection.
HOSTILE_LENGTH = 3 << 30
def write_limits(out):
"""Files for the size limits (the tests must get errors, not aborts):
- limits.h5: /huge_u8, 2^28 + 1024 bytes of u8 in compressed chunks
(a small file): read whole it would take over 2 GiB while decoding;
its last value is 7.
- hostile_vl.h5: a variable-length string dataset /a whose global heap
collection claims HOSTILE_GCOL_SIZE bytes, with the superblock's end
of file set to HOSTILE_LENGTH (libhdf5 cannot read it; it is only
served by a mock that claims that length).
- far.h5 and far.json: /x, 16 float64 values, whose contiguous data
address is moved FAR_SHIFT bytes on (past 2 GiB, the sign bit of a
wasm32 isize) in a file whose end of file is moved as far; the tests'
mock serves the data there, to show offsets up to 4 GiB work on
wasm32.
"""
with h5py.File(out / "limits.h5", "w") as f:
d = f.create_dataset("huge_u8", shape=(HUGE_U8,), dtype="u1",
chunks=(1 << 20,), compression="gzip")
d[-1] = 7
path = out / "hostile_vl.h5"
with h5py.File(path, "w", libver="earliest") as f:
f.create_dataset("a", data=["x", "yy"], dtype=h5py.string_dtype())
b = bytearray(path.read_bytes())
assert b[8] == 0, "a version 0 superblock"
b[40:48] = HOSTILE_LENGTH.to_bytes(8, "little") # end of file address
at = b.index(b"GCOL")
b[at + 8:at + 16] = HOSTILE_GCOL_SIZE.to_bytes(8, "little")
path.write_bytes(bytes(b))
path = out / "far.h5"
values = np.arange(16, dtype="<f8") * 1.5
with h5py.File(path, "w", libver="earliest") as f:
f.create_dataset("x", data=values)
data_at = f["x"].id.get_offset()
b = bytearray(path.read_bytes())
# The layout message: the data's address, then its size.
old = data_at.to_bytes(8, "little") + (values.nbytes).to_bytes(8, "little")
assert b.count(old) == 1
at = b.index(old)
b[at:at + 8] = (data_at + FAR_SHIFT).to_bytes(8, "little")
length = len(b) + FAR_SHIFT
b[40:48] = length.to_bytes(8, "little")
path.write_bytes(bytes(b))
json.dump({"data_at": data_at, "far_at": data_at + FAR_SHIFT,
"nbytes": values.nbytes, "length": length,
"values": [float(x) for x in values]},
open(out / "far.json", "w"))
# far.h5's data moves this far: past 2 GiB, below 4 GiB.
FAR_SHIFT = 3 << 30
write_limits(out)
def write_big(path, megabytes):
"""A large file for the range-request tests (`openUrl`): `/big`, about
`megabytes` MB of float64 in 1 MiB chunks, written after a small
+83
View File
@@ -321,6 +321,8 @@ async function remoteTests() {
await f.read("/grid");
}, /unusable Content-Range/, "unusable Content-Range");
await limitTests();
// Corpus files: what the viewer can show of each is the same read by
// ranges as in memory (an error wherever it gives one).
const corpus = process.env.CLAWHDF5_WASM_CORPUS;
@@ -328,6 +330,87 @@ async function remoteTests() {
console.log(`openUrl: ${checks - before} checks passed`);
}
// A fetch that serves `buf` as a file of `total` bytes (zeros past the end
// of `buf`, and `extra` = [[offset, bytes], ...] laid over them), counting
// its calls. `hide` leaves out Content-Range and the validators, as a
// cross-origin server that exposes neither does; HEAD then gives the length.
function mockFetch(buf, { total = buf.length, extra = [], hide = false } = {}) {
const f = async (url, init) => {
f.calls++;
if (init.method === "HEAD") {
return new Response(null, { status: 200, headers: { "Content-Length": String(total) } });
}
const m = /^bytes=(\d+)-(\d+)$/.exec(new Headers(init.headers).get("Range"));
const a = Number(m[1]);
const b = Math.min(Number(m[2]) + 1, total);
const out = new Uint8Array(b - a);
for (const [at, bytes] of [[0, buf], ...extra]) {
const from = Math.max(a, at);
const to = Math.min(b, at + bytes.length);
if (from < to) out.set(bytes.subarray(from - at, to - at), from - a);
}
const headers = { "Content-Length": String(b - a) };
if (!hide) headers["Content-Range"] = `bytes ${a}-${b - 1}/${total}`;
return new Response(out, { status: 206, headers });
};
f.calls = 0;
return f;
}
// Sizes a hostile server or a large dataset can name are errors, never an
// allocation that aborts the module (which would take every open file on
// the page with it); see write_limits in make_fixture.py.
async function limitTests() {
// Read whole, /huge_u8 (2^28 + 1024 bytes) would take over 2 GiB while
// decoding: refused before its chunks are fetched; a window reads.
const n = 2 ** 28 + 1024;
const limits = readFileSync(join(fixDir, "limits.h5"));
const local = pkg.open(new Uint8Array(limits));
await fails(() => local.read("/huge_u8"), /readHyperslab/, "huge read, in memory");
const remote = await pkg.openUrl(`${base}/fix/limits.h5`);
const before = remote.stats().requests;
await fails(() => remote.read("/huge_u8"), /readHyperslab/, "huge read, openUrl");
eq(remote.stats().requests, before, "huge read fetched nothing");
eq(Array.from((await remote.readHyperslab("/huge_u8", [n - 4], [4])).data), [0, 0, 0, 7], "huge window");
eq(Array.from(local.readHyperslab("/huge_u8", [n - 4], [4]).data), [0, 0, 0, 7], "huge window, in memory");
local.free();
// A server claiming 3 GiB, a heap collection claiming 2 GiB + 4 KiB: an
// error after a few requests (it used to fetch 2 GiB, then abort).
const hostile = mockFetch(new Uint8Array(readFileSync(join(fixDir, "hostile_vl.h5"))), { total: 3 * 2 ** 30 });
const h = await pkg.openUrl("http://hostile.invalid/h.h5", { fetch: hostile });
await fails(() => h.read("/a"), /maxFetch/, "hostile collection size");
assert.ok(hostile.calls <= 4, `hostile: ${hostile.calls} requests`);
// The module survived: files open and read.
eq((await (await pkg.openUrl(`${base}/fix/fixture.h5`)).read("/sensors/temp")).data[0], 21.5, "alive after hostile");
// A call that would fetch more than maxFetch fails before fetching it.
await fails(async () => {
const f = await pkg.openUrl(`${base}/fix/fixture.h5`, { blockSize: 512, maxFetch: 1024 });
await f.read("/grid");
}, /maxFetch/, "maxFetch");
await fails(() => pkg.openUrl(`${base}/fix/fixture.h5`, { maxFetch: 2 ** 31 }), /maxFetch/, "maxFetch range");
await fails(() => pkg.openUrl(`${base}/fix/fixture.h5`, { maxDownload: 2 ** 31 }), /maxDownload/, "maxDownload range");
// Offsets past 2 GiB work on wasm32: far.h5's data sits at 3 GiB.
const far = JSON.parse(readFileSync(join(fixDir, "far.json"), "utf8"));
const farBytes = new Uint8Array(readFileSync(join(fixDir, "far.h5")));
const farData = farBytes.subarray(far.data_at, far.data_at + far.nbytes);
const ff = await pkg.openUrl("http://far.invalid/far.h5", {
fetch: mockFetch(farBytes, { total: far.length, extra: [[far.far_at, farData]] }),
});
eq(Array.from((await ff.read("/x")).data), far.values, "data past 2 GiB");
eq(ff.stats().size, far.length, "size past 2 GiB");
// wasm32's reader holds file offsets in 32 bits: a file of 4 GiB or more
// is refused at open, with the limit in the message (past 2^53 - 1 bytes
// a JavaScript number is not even exact).
for (const total of [2 ** 32, 2 ** 40, 2 ** 53 + 2]) {
await fails(() => pkg.openUrl("http://huge.invalid/x.h5", { fetch: mockFetch(farBytes, { total }) }),
total > 2 ** 53 ? /2\^53 - 1/ : /4 GiB/, `length ${total}`);
}
}
function hdf5Files(dir, out) {
for (const e of readdirSync(dir, { withFileTypes: true })) {
const p = join(dir, e.name);