wasm: sizes a server or a dataset names are errors, not aborts
A read longer than isize::MAX (2 GiB on wasm32) aborted the module in LazyStorage::assemble (capacity_overflow), taking every open file on the page with it, and a hostile server only had to claim a large length and serve a heap collection of 2 GiB + 4 KiB to get there (after fetching 2 GiB). Reading a large u8 dataset whole aborted the same way when its values were widened to 64 bits. - LazyConfig::max_fetch (openUrl option maxFetch, default 512 MiB, at most 1 GiB): a read longer than it fails at once, before anything is fetched, and an operation whose passes would fetch more than it fails before fetching (Operation::charge). assemble reserves fallibly. - Reader::read refuses a read that would use more than 1 GiB while decoding (core::MAX_READ_BYTES: stored bytes + 64-bit values + result) with an error naming readHyperslab, before reading. - openUrl refuses a file of 4 GiB or more at open on wasm32: the format code turns offsets into usize, so nothing past 4 GiB can be read there (shown by a new test: data at 3 GiB reads, a 4 GiB file is refused). maxDownload is bounded to 1 GiB. Tests: make_fixture.py writes limits.h5 (a sparse 2^28 + 1024 byte u8 dataset), hostile_vl.h5 (the reviewer's collection) and far.h5 (data at 3 GiB); test.mjs (wasm32) and tests/lazy.rs (native) check each is an error or reads, and that the module survives. Before: RuntimeError: unreachable in Node; the native test read the huge dataset and fetched 2 GiB. Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
@@ -117,7 +117,11 @@ export async function probe(url, firstLen, opts) {
|
||||
}
|
||||
length = Number(cl);
|
||||
}
|
||||
if (!Number.isSafeInteger(length) || first.length !== Math.min(firstLen, length)) {
|
||||
if (!Number.isSafeInteger(length) || length < 0) {
|
||||
throw new Error(`${url}: the server gave a file size of ${length} bytes; openUrl reads files ` +
|
||||
"of up to 2^53 - 1 bytes (the largest offset a JavaScript number holds exactly)");
|
||||
}
|
||||
if (first.length !== Math.min(firstLen, length)) {
|
||||
throw new Error(`${url}: asked for the first ${firstLen} bytes of ${length}, got ${first.length}`);
|
||||
}
|
||||
return { length, first, validator: validatorOf(resp), requests };
|
||||
|
||||
@@ -14,6 +14,15 @@ use clawhdf5_format::datatype::{Datatype, DatatypeByteOrder};
|
||||
use clawhdf5_format::storage::Storage;
|
||||
use clawhdf5_format::vl_data::{VlResolver, check_element_size};
|
||||
|
||||
/// The most memory one read may use while it decodes: the stored bytes,
|
||||
/// the values at 64 bits (integers are widened first) and the values
|
||||
/// returned. A larger read fails with an error naming `readHyperslab`,
|
||||
/// before anything is read: on wasm32 a buffer past 2 GiB cannot be
|
||||
/// allocated at all, and failing to allocate aborts the module (every open
|
||||
/// file on the page with it). 1 GiB leaves room in wasm32's 4 GiB for the
|
||||
/// file's cached blocks and the JavaScript copy of the result.
|
||||
pub const MAX_READ_BYTES: u64 = 1 << 30;
|
||||
|
||||
/// Errors are reported to JavaScript as messages.
|
||||
pub type Result<T> = std::result::Result<T, String>;
|
||||
|
||||
@@ -241,13 +250,27 @@ impl Reader {
|
||||
if let Datatype::VariableLength { size, .. } = array_base(&dt) {
|
||||
check_element_size(*size, self.file.superblock().offset_size).map_err(err)?;
|
||||
}
|
||||
let raw = ds.read_selection(&selection).map_err(err)?;
|
||||
let data = self.decode(&raw, &dt)?;
|
||||
out_shape.extend(element_shape(&dt));
|
||||
let expected = out_shape
|
||||
.iter()
|
||||
.try_fold(1u64, |acc, &d| acc.checked_mul(d))
|
||||
.ok_or("selection size overflows")?;
|
||||
let cost = expected.saturating_mul(bytes_per_value(&dt));
|
||||
if cost > MAX_READ_BYTES {
|
||||
return Err(format!(
|
||||
"reading {path}{} would take about {} MiB of memory, more than the {} MiB \
|
||||
one read may use; read it in parts (readHyperslab)",
|
||||
if slab.is_some() {
|
||||
" (this selection)"
|
||||
} else {
|
||||
" whole"
|
||||
},
|
||||
cost >> 20,
|
||||
MAX_READ_BYTES >> 20
|
||||
));
|
||||
}
|
||||
let raw = ds.read_selection(&selection).map_err(err)?;
|
||||
let data = self.decode(&raw, &dt)?;
|
||||
if data.len() as u64 != expected {
|
||||
return Err(format!(
|
||||
"read {} values for shape {out_shape:?} ({expected} expected)",
|
||||
@@ -316,6 +339,27 @@ impl Reader {
|
||||
}
|
||||
}
|
||||
|
||||
/// Memory one value of type `dt` takes while [`Reader::read`] decodes it
|
||||
/// (an array type's elements count as values): its stored bytes, plus what
|
||||
/// [`Reader::decode`] builds from them. A string counts its `String` (24
|
||||
/// bytes on 64-bit targets, less on wasm32) and, for a fixed-length one,
|
||||
/// its text; a variable-length string's text lives in the heap and is
|
||||
/// bounded by the storage's own read limit.
|
||||
fn bytes_per_value(dt: &Datatype) -> u64 {
|
||||
let base = array_base(dt);
|
||||
let stored = u64::from(base.type_size());
|
||||
stored
|
||||
+ match base {
|
||||
Datatype::FloatingPoint { size, .. } if *size <= 4 => 4,
|
||||
Datatype::FloatingPoint { .. } => 8,
|
||||
// Widened to 64 bits, then narrowed to a new vector.
|
||||
Datatype::FixedPoint { .. } => 8 + stored,
|
||||
Datatype::String { .. } => 24 + stored,
|
||||
Datatype::VariableLength { .. } | Datatype::Enumeration { .. } => 24,
|
||||
_ => 0,
|
||||
}
|
||||
}
|
||||
|
||||
/// Narrow integers read at 64 bits to the dataset's own width. The source is
|
||||
/// that width, so this cannot fail on correct input; it is checked anyway.
|
||||
fn narrow<S: Copy + std::fmt::Display, T: TryFrom<S>>(v: Vec<S>) -> Result<Vec<T>> {
|
||||
|
||||
@@ -42,6 +42,10 @@ use clawhdf5_format::storage::Storage;
|
||||
/// `docs/design/range-reads.md` §2 measured).
|
||||
pub const DEFAULT_BLOCK_SIZE: u64 = 1 << 20;
|
||||
|
||||
/// Default of [`LazyConfig::max_fetch`]: 512 MiB, the same as `openUrl`'s
|
||||
/// `maxDownload` for a server without range support.
|
||||
pub const DEFAULT_MAX_FETCH: u64 = 512 << 20;
|
||||
|
||||
/// The message of the error a read that misses returns. It never reaches
|
||||
/// the caller of [`LazyStorage::attempt`]: a pass that missed is re-run.
|
||||
pub const NEED_BYTES: &str = "bytes not fetched yet (restartable read)";
|
||||
@@ -58,6 +62,14 @@ pub struct LazyConfig {
|
||||
/// Largest single range asked for, in bytes (whole blocks, at least
|
||||
/// one); longer runs are split so they can be fetched in parallel.
|
||||
pub max_request: u64,
|
||||
/// Most bytes one operation may fetch (at least one block), and so the
|
||||
/// longest single read: a read longer than this fails at once, before
|
||||
/// anything is fetched, and so does an operation whose passes would
|
||||
/// fetch more. The file's length comes from the server, so without
|
||||
/// this a hostile file (a heap "collection" claiming 2 GiB) makes the
|
||||
/// reader fetch and hold whatever it names; on wasm32 a buffer past
|
||||
/// 2 GiB cannot even be allocated.
|
||||
pub max_fetch: u64,
|
||||
}
|
||||
|
||||
impl Default for LazyConfig {
|
||||
@@ -66,6 +78,7 @@ impl Default for LazyConfig {
|
||||
block_size: DEFAULT_BLOCK_SIZE,
|
||||
capacity: 64 << 20,
|
||||
max_request: 8 << 20,
|
||||
max_fetch: DEFAULT_MAX_FETCH,
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -137,6 +150,28 @@ fn lock(m: &Mutex<State>) -> MutexGuard<'_, State> {
|
||||
/// [`LazyStorage::operation`].
|
||||
pub struct Operation<'a> {
|
||||
storage: &'a LazyStorage,
|
||||
/// Bytes fetched for this operation so far.
|
||||
fetched: std::cell::Cell<u64>,
|
||||
}
|
||||
|
||||
impl Operation<'_> {
|
||||
/// Count `ranges` against the operation's budget
|
||||
/// ([`LazyConfig::max_fetch`]) before they are fetched: an error, and
|
||||
/// nothing counted, if they would take it past the budget.
|
||||
pub fn charge(&self, ranges: &[Range<u64>]) -> Result<(), String> {
|
||||
let max = self.storage.config.max_fetch;
|
||||
let total = ranges.iter().fold(self.fetched.get(), |n, r| {
|
||||
n.saturating_add(r.end.saturating_sub(r.start))
|
||||
});
|
||||
if total > max {
|
||||
return Err(format!(
|
||||
"this call would fetch more than {max} bytes of the file (the maxFetch limit); \
|
||||
read less at a time (readHyperslab) or raise maxFetch"
|
||||
));
|
||||
}
|
||||
self.fetched.set(total);
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
impl Drop for Operation<'_> {
|
||||
@@ -154,6 +189,7 @@ impl LazyStorage {
|
||||
pub fn new(len: u64, mut config: LazyConfig) -> Self {
|
||||
config.block_size = config.block_size.max(512);
|
||||
config.max_request = (config.max_request / config.block_size).max(1) * config.block_size;
|
||||
config.max_fetch = config.max_fetch.max(config.block_size);
|
||||
LazyStorage {
|
||||
len,
|
||||
config,
|
||||
@@ -180,7 +216,10 @@ impl LazyStorage {
|
||||
/// Hold it across every pass of one operation.
|
||||
pub fn operation(&self) -> Operation<'_> {
|
||||
lock(&self.state).active += 1;
|
||||
Operation { storage: self }
|
||||
Operation {
|
||||
storage: self,
|
||||
fetched: std::cell::Cell::new(0),
|
||||
}
|
||||
}
|
||||
|
||||
/// Run one pass of `f` over this storage. `Done` when `f` read nothing
|
||||
@@ -268,11 +307,12 @@ impl LazyStorage {
|
||||
mut f: impl FnMut() -> T,
|
||||
mut fetch: impl FnMut(Range<u64>) -> Result<Vec<u8>, String>,
|
||||
) -> Result<T, String> {
|
||||
let _op = self.operation();
|
||||
let op = self.operation();
|
||||
loop {
|
||||
match self.attempt(&mut f) {
|
||||
Step::Done(v) => return Ok(v),
|
||||
Step::Need(ranges) => {
|
||||
op.charge(&ranges)?;
|
||||
for r in ranges {
|
||||
let bytes = fetch(r.clone())?;
|
||||
self.supply_range(&r, &bytes)?;
|
||||
@@ -395,9 +435,35 @@ impl LazyStorage {
|
||||
Ok(have)
|
||||
}
|
||||
|
||||
fn assemble(&self, offset: u64, end: u64, blocks: &HashMap<u64, Arc<[u8]>>) -> Vec<u8> {
|
||||
/// Refuse a read of `n` bytes longer than an operation may fetch
|
||||
/// ([`LazyConfig::max_fetch`]), before its blocks are asked for.
|
||||
fn check_len(&self, n: u64) -> Result<(), FormatError> {
|
||||
let max = self.config.max_fetch;
|
||||
if n > max {
|
||||
return Err(FormatError::Storage(format!(
|
||||
"a read of {n} bytes is more than one call may fetch ({max} bytes, the maxFetch limit)"
|
||||
)));
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// The bytes `offset..end` from `blocks`, which hold every block of
|
||||
/// that span. The buffer is reserved fallibly: a length the address
|
||||
/// space cannot hold (past `isize::MAX` on wasm32) is an error, never
|
||||
/// an abort.
|
||||
fn assemble(
|
||||
&self,
|
||||
offset: u64,
|
||||
end: u64,
|
||||
blocks: &HashMap<u64, Arc<[u8]>>,
|
||||
) -> Result<Vec<u8>, FormatError> {
|
||||
let bs = self.config.block_size;
|
||||
let mut out = Vec::with_capacity(usize::try_from(end - offset).unwrap_or(0));
|
||||
let n = end - offset;
|
||||
let too_long =
|
||||
|| FormatError::Storage(format!("cannot hold a read of {n} bytes in memory"));
|
||||
let mut out = Vec::new();
|
||||
out.try_reserve_exact(usize::try_from(n).map_err(|_| too_long())?)
|
||||
.map_err(|_| too_long())?;
|
||||
let mut pos = offset;
|
||||
while pos < end {
|
||||
let i = pos / bs;
|
||||
@@ -407,7 +473,7 @@ impl LazyStorage {
|
||||
out.extend_from_slice(&block[from..to]);
|
||||
pos = i * bs + to as u64;
|
||||
}
|
||||
out
|
||||
Ok(out)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -416,10 +482,11 @@ impl Storage for LazyStorage {
|
||||
let Some(span) = self.span(offset, len as u64) else {
|
||||
return Ok(Cow::Owned(Vec::new()));
|
||||
};
|
||||
let end = offset.saturating_add(len as u64).min(self.len);
|
||||
self.check_len(end - offset)?;
|
||||
let metadata = len as u64 <= self.config.block_size;
|
||||
let blocks = self.blocks(std::slice::from_ref(&span), metadata)?;
|
||||
let end = offset.saturating_add(len as u64).min(self.len);
|
||||
Ok(Cow::Owned(self.assemble(offset, end, &blocks)))
|
||||
Ok(Cow::Owned(self.assemble(offset, end, &blocks)?))
|
||||
}
|
||||
|
||||
fn len(&self) -> u64 {
|
||||
@@ -428,26 +495,29 @@ impl Storage for LazyStorage {
|
||||
|
||||
fn read_ranges(&self, ranges: &[Range<u64>]) -> Result<Vec<Cow<'_, [u8]>>, FormatError> {
|
||||
let mut spans = Vec::with_capacity(ranges.len());
|
||||
let mut total = 0u64;
|
||||
for r in ranges {
|
||||
if r.end < r.start {
|
||||
return Err(FormatError::Storage(
|
||||
"read range ends before it starts".into(),
|
||||
));
|
||||
}
|
||||
total = total.saturating_add(r.end.min(self.len).saturating_sub(r.start));
|
||||
spans.extend(self.span(r.start, r.end - r.start));
|
||||
}
|
||||
self.check_len(total)?;
|
||||
let blocks = self.blocks(&spans, false)?;
|
||||
Ok(ranges
|
||||
ranges
|
||||
.iter()
|
||||
.map(|r| {
|
||||
let end = r.end.min(self.len);
|
||||
if r.start >= end {
|
||||
Cow::Owned(Vec::new())
|
||||
Ok(Cow::Owned(Vec::new()))
|
||||
} else {
|
||||
Cow::Owned(self.assemble(r.start, end, &blocks))
|
||||
self.assemble(r.start, end, &blocks).map(Cow::Owned)
|
||||
}
|
||||
})
|
||||
.collect())
|
||||
.collect()
|
||||
}
|
||||
}
|
||||
|
||||
@@ -465,6 +535,7 @@ mod tests {
|
||||
block_size: block,
|
||||
capacity,
|
||||
max_request: 4 * block,
|
||||
max_fetch: DEFAULT_MAX_FETCH,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -654,6 +725,57 @@ mod tests {
|
||||
assert_eq!(s.stats().requests, before.requests);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_read_longer_than_max_fetch_fails_without_fetching() {
|
||||
// A hostile file names a 2 GiB heap collection in a "file" the
|
||||
// server claims is 1 TiB: the read is refused before any block is
|
||||
// asked for (on wasm32 its buffer could not even be allocated).
|
||||
let s = LazyStorage::new(1 << 40, LazyConfig::default());
|
||||
let step = s.attempt(|| s.read_at(4096, (1usize << 31) + 4096).map(|b| b.len()));
|
||||
match step {
|
||||
Step::Done(Err(e)) => assert!(e.to_string().contains("maxFetch"), "{e}"),
|
||||
other => panic!("expected a refusal, got {other:?}"),
|
||||
}
|
||||
let step = s.attempt(|| s.read_ranges(&[0..(600 << 20)]).map(|v| v.len()));
|
||||
assert!(matches!(step, Step::Done(Err(_))), "{step:?}");
|
||||
assert_eq!(s.stats().requests, 0);
|
||||
// At the limit it is an ordinary miss.
|
||||
let s = LazyStorage::new(1 << 40, config(1024, 1 << 20));
|
||||
let step = s.attempt(|| s.read_at(0, DEFAULT_MAX_FETCH as usize).map(|b| b.len()));
|
||||
assert!(matches!(step, Step::Need(_)), "{step:?}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_operation_stops_at_its_fetch_budget() {
|
||||
// Many small reads, none over the limit, that together would fetch
|
||||
// more than the budget: the operation fails before fetching past it.
|
||||
let data = file(64 * 1024);
|
||||
let mut c = config(1024, 1 << 20);
|
||||
c.max_fetch = 8 * 1024;
|
||||
let s = LazyStorage::new(data.len() as u64, c);
|
||||
let e = s
|
||||
.run_blocking(
|
||||
|| {
|
||||
(0..64)
|
||||
.map(|i| owned(s.read_at(i * 1024, 8)))
|
||||
.collect::<Result<Vec<_>, _>>()
|
||||
},
|
||||
|r| Ok(data[r.start as usize..r.end as usize].to_vec()),
|
||||
)
|
||||
.unwrap_err();
|
||||
assert!(e.contains("maxFetch"), "{e}");
|
||||
assert!(s.stats().bytes_fetched <= 8 * 1024, "{:?}", s.stats());
|
||||
// Within the budget it completes, and the budget is per operation.
|
||||
for _ in 0..3 {
|
||||
s.run_blocking(
|
||||
|| owned(s.read_at(10 * 1024, 3000)),
|
||||
|r| Ok(data[r.start as usize..r.end as usize].to_vec()),
|
||||
)
|
||||
.unwrap()
|
||||
.unwrap();
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_failed_fetch_is_an_error_not_data() {
|
||||
let data = file(4096);
|
||||
|
||||
@@ -389,11 +389,14 @@ impl Http {
|
||||
storage: &LazyStorage,
|
||||
mut f: impl FnMut() -> T,
|
||||
) -> Result<T, JsError> {
|
||||
let _op = storage.operation();
|
||||
let op = storage.operation();
|
||||
loop {
|
||||
match storage.attempt(&mut f) {
|
||||
Step::Done(v) => return Ok(v),
|
||||
Step::Need(ranges) => self.fetch(storage, &ranges).await?,
|
||||
Step::Need(ranges) => {
|
||||
op.charge(&ranges).map_err(js_err)?;
|
||||
self.fetch(storage, &ranges).await?
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -426,6 +429,24 @@ fn int_opt(opts: &JsValue, key: &str, min: f64, max: f64) -> Result<Option<u64>,
|
||||
}
|
||||
}
|
||||
|
||||
/// Most bytes `maxFetch` and `maxDownload` may allow: 1 GiB. wasm32 has
|
||||
/// 4 GiB of memory and no buffer past 2 GiB, and what is fetched is held
|
||||
/// while it is decoded.
|
||||
const MAX_FETCH_LIMIT: u64 = 1 << 30;
|
||||
|
||||
/// The largest file `openUrl` reads by ranges: on wasm32, 4 GiB - 1 bytes.
|
||||
/// The format code turns file offsets into `usize` to use them (with a
|
||||
/// clean error past it, see scripts/check-32bit-casts.sh), so on a 32-bit
|
||||
/// target nothing at 4 GiB or beyond can be read; a larger file is refused
|
||||
/// at open rather than failing on whichever read reaches past 4 GiB. On
|
||||
/// 64-bit targets it is 2^53 - 1, the largest offset a JavaScript number
|
||||
/// holds exactly.
|
||||
const MAX_REMOTE_LENGTH: u64 = if (usize::MAX as u64) < MAX_SAFE_INTEGER as u64 {
|
||||
usize::MAX as u64
|
||||
} else {
|
||||
MAX_SAFE_INTEGER as u64
|
||||
};
|
||||
|
||||
fn config_from(opts: &JsValue) -> Result<LazyConfig, JsError> {
|
||||
let mut c = LazyConfig::default();
|
||||
if let Some(b) = int_opt(opts, "blockSize", 512.0, (64u64 << 20) as f64)? {
|
||||
@@ -434,6 +455,12 @@ fn config_from(opts: &JsValue) -> Result<LazyConfig, JsError> {
|
||||
if let Some(n) = int_opt(opts, "cacheSize", 0.0, MAX_SAFE_INTEGER)? {
|
||||
c.capacity = n;
|
||||
}
|
||||
if let Some(n) = int_opt(opts, "maxFetch", 512.0, MAX_FETCH_LIMIT as f64)? {
|
||||
c.max_fetch = n;
|
||||
}
|
||||
// Read by remote.js; checked here so a value wasm32 cannot hold is an
|
||||
// option error rather than a download that cannot be kept.
|
||||
int_opt(opts, "maxDownload", 0.0, MAX_FETCH_LIMIT as f64)?;
|
||||
Ok(c)
|
||||
}
|
||||
|
||||
@@ -444,15 +471,21 @@ fn config_from(opts: &JsValue) -> Result<LazyConfig, JsError> {
|
||||
/// `opts` (all optional):
|
||||
/// - `blockSize` — bytes per request block, 512 to 64 MiB (default 1 MiB);
|
||||
/// - `cacheSize` — bytes of blocks kept between calls (default 64 MiB);
|
||||
/// - `maxFetch` — most bytes one call may fetch, and so the longest single
|
||||
/// read, up to 1 GiB (default 512 MiB): a call that would fetch more
|
||||
/// fails before fetching it;
|
||||
/// - `fallback` — `"download"` (default) reads the whole file when the
|
||||
/// server ignores `Range` (answers 200), up to `maxDownload` bytes
|
||||
/// (default 512 MiB); `"error"` refuses such a server;
|
||||
/// (default 512 MiB, at most 1 GiB); `"error"` refuses such a server;
|
||||
/// - `headers`, `credentials` — passed to every `fetch`;
|
||||
/// - `parallel` — range requests in flight at once (default 6);
|
||||
/// - `fetch` — a `fetch`-compatible function to use instead of the global.
|
||||
///
|
||||
/// Cross-origin servers must allow CORS and expose `Content-Range` (or
|
||||
/// answer `HEAD` with `Content-Length`).
|
||||
/// answer `HEAD` with `Content-Length`). A file may be up to 4 GiB - 1
|
||||
/// bytes long (wasm32 offsets); a longer one is refused at open. A whole-dataset `read` that would use more than 1 GiB of
|
||||
/// memory ([`core::MAX_READ_BYTES`]) is refused: read it in parts with
|
||||
/// `readHyperslab`.
|
||||
#[wasm_bindgen(js_name = openUrl)]
|
||||
pub async fn open_url(url: String, opts: JsValue) -> Result<RemoteFile, JsError> {
|
||||
let config = config_from(&opts)?;
|
||||
@@ -477,6 +510,12 @@ pub async fn open_url(url: String, opts: JsValue) -> Result<RemoteFile, JsError>
|
||||
.as_f64()
|
||||
.filter(|x| x.fract() == 0.0 && (0.0..=MAX_SAFE_INTEGER).contains(x))
|
||||
.ok_or_else(|| js_err(format!("{url}: the server gave no usable file size")))?;
|
||||
if length as u64 > MAX_REMOTE_LENGTH {
|
||||
return Err(js_err(format!(
|
||||
"{url} is {length} bytes; openUrl reads files of up to {MAX_REMOTE_LENGTH} bytes \
|
||||
(4 GiB - 1: the WebAssembly reader addresses a file with 32-bit offsets)"
|
||||
)));
|
||||
}
|
||||
let http = Http {
|
||||
url,
|
||||
opts,
|
||||
|
||||
@@ -193,6 +193,7 @@ fn config(block: u64) -> LazyConfig {
|
||||
// A small budget, so eviction between operations is exercised.
|
||||
capacity: 16 * block,
|
||||
max_request: 8 * block,
|
||||
..LazyConfig::default()
|
||||
}
|
||||
}
|
||||
|
||||
@@ -310,6 +311,95 @@ fn h5py_and_netcdf4_files_read_the_same_lazily() {
|
||||
eprintln!("skipping: {} lacks h5py/netCDF4/numpy", python());
|
||||
return;
|
||||
}
|
||||
let dir = fixture_dir();
|
||||
for name in ["fixture.h5", "fixture.nc"] {
|
||||
let data = std::fs::read(dir.path().join(name)).unwrap();
|
||||
for block in [512, 64 * 1024] {
|
||||
let (_, _, lines) = check_equal(name, &data, block);
|
||||
assert!(lines.iter().filter(|l| l.contains(" read: Ok")).count() >= 2);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The bytes of `data` at `r`, zero past its end: a server that claims
|
||||
/// the file is longer than it is.
|
||||
fn fetch_padded(data: &[u8], r: Range<u64>) -> Result<Vec<u8>, String> {
|
||||
let mut out = vec![0u8; (r.end - r.start) as usize];
|
||||
let len = data.len() as u64;
|
||||
if r.start < len {
|
||||
let end = r.end.min(len);
|
||||
out[..(end - r.start) as usize].copy_from_slice(&data[r.start as usize..end as usize]);
|
||||
}
|
||||
Ok(out)
|
||||
}
|
||||
|
||||
/// Sizes a hostile server or a large dataset can name are errors, never
|
||||
/// allocations that abort the wasm module: make_fixture.py's limits.h5 and
|
||||
/// hostile_vl.h5 (see write_limits there).
|
||||
#[test]
|
||||
fn size_limits_are_errors_not_aborts() {
|
||||
if !python_available() {
|
||||
assert!(
|
||||
!std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1"),
|
||||
"CLAWHDF5_REQUIRE_INTEROP=1 but {} lacks h5py/netCDF4/numpy",
|
||||
python()
|
||||
);
|
||||
eprintln!("skipping: {} lacks h5py/netCDF4/numpy", python());
|
||||
return;
|
||||
}
|
||||
let dir = fixture_dir();
|
||||
|
||||
// Read whole, /huge_u8 would widen 2^28 values to 64 bits (2 GiB): an
|
||||
// error naming readHyperslab, before its chunks are read. A window of
|
||||
// it reads.
|
||||
let data = std::fs::read(dir.path().join("limits.h5")).unwrap();
|
||||
let n = (1u64 << 28) + 1024;
|
||||
let window = Hyperslab {
|
||||
start: vec![n - 4],
|
||||
count: vec![4],
|
||||
stride: None,
|
||||
block: None,
|
||||
};
|
||||
let local = Reader::open(data.clone()).unwrap();
|
||||
let lazy = Lazy::open(data, LazyConfig::default()).unwrap();
|
||||
let before = lazy.storage.stats().requests;
|
||||
for e in [
|
||||
local.read("/huge_u8", None).unwrap_err(),
|
||||
lazy.call(|r| r.read("/huge_u8", None)).unwrap_err(),
|
||||
] {
|
||||
assert!(e.contains("readHyperslab"), "{e}");
|
||||
}
|
||||
assert_eq!(lazy.storage.stats().requests, before, "nothing fetched");
|
||||
for part in [
|
||||
local.read("/huge_u8", Some(&window)).unwrap(),
|
||||
lazy.call(|r| r.read("/huge_u8", Some(&window))).unwrap(),
|
||||
] {
|
||||
assert_eq!(format!("{:?}", part.data), "U8([0, 0, 0, 7])");
|
||||
}
|
||||
|
||||
// A server that claims 3 GiB and a heap collection of 2 GiB + 4 KiB:
|
||||
// reading the strings fails at once, fetching a few blocks.
|
||||
let data = std::fs::read(dir.path().join("hostile_vl.h5")).unwrap();
|
||||
let storage = Arc::new(LazyStorage::new(3 << 30, LazyConfig::default()));
|
||||
let s = storage.clone();
|
||||
let reader = storage
|
||||
.run_blocking(
|
||||
|| Reader::open_storage(s.clone()),
|
||||
|r| fetch_padded(&data, r),
|
||||
)
|
||||
.unwrap()
|
||||
.unwrap();
|
||||
let e = storage
|
||||
.run_blocking(|| reader.read("/a", None), |r| fetch_padded(&data, r))
|
||||
.unwrap()
|
||||
.unwrap_err();
|
||||
assert!(e.contains("maxFetch"), "{e}");
|
||||
let st = storage.stats();
|
||||
assert!(st.requests <= 4 && st.bytes_fetched <= 4 << 20, "{st:?}");
|
||||
}
|
||||
|
||||
/// make_fixture.py's files, written to a temporary directory.
|
||||
fn fixture_dir() -> tempfile::TempDir {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let generator = Path::new(env!("CARGO_MANIFEST_DIR"))
|
||||
.join("../../examples/wasm-viewer/test/make_fixture.py");
|
||||
@@ -323,13 +413,7 @@ fn h5py_and_netcdf4_files_read_the_same_lazily() {
|
||||
"{}",
|
||||
String::from_utf8_lossy(&out.stderr)
|
||||
);
|
||||
for name in ["fixture.h5", "fixture.nc"] {
|
||||
let data = std::fs::read(dir.path().join(name)).unwrap();
|
||||
for block in [512, 64 * 1024] {
|
||||
let (_, _, lines) = check_equal(name, &data, block);
|
||||
assert!(lines.iter().filter(|l| l.contains(" read: Ok")).count() >= 2);
|
||||
}
|
||||
}
|
||||
dir
|
||||
}
|
||||
|
||||
fn hdf5_files(dir: &Path, out: &mut Vec<PathBuf>) {
|
||||
|
||||
@@ -3,7 +3,9 @@ reads back from them, for the clawhdf5-wasm tests.
|
||||
|
||||
python make_fixture.py OUT_DIR
|
||||
|
||||
writes OUT_DIR/fixture.h5, OUT_DIR/fixture.nc and OUT_DIR/expected.json.
|
||||
writes OUT_DIR/fixture.h5, OUT_DIR/fixture.nc and OUT_DIR/expected.json,
|
||||
and the limit-test files OUT_DIR/limits.h5 and OUT_DIR/hostile_vl.h5 (see
|
||||
write_limits).
|
||||
Both the Rust test (crates/clawhdf5-wasm/tests/h5py_interop.rs, native) and
|
||||
the Node test (test.mjs, the built wasm package) compare against the same
|
||||
expected.json, so the two check the same values.
|
||||
@@ -204,6 +206,72 @@ json.dump({"fixture.h5": describe(h5), "fixture.nc": describe(nc)},
|
||||
open(out / "expected.json", "w"), indent=1, ensure_ascii=False)
|
||||
|
||||
|
||||
HUGE_U8 = 2**28 + 1024
|
||||
# The collection size hostile_vl.h5 claims: past 2 GiB, which a wasm32
|
||||
# buffer cannot hold.
|
||||
HOSTILE_GCOL_SIZE = 2**31 + 4096
|
||||
# The file length a server claims for hostile_vl.h5 (the tests' mock fetch
|
||||
# answers every range with zeros past the real bytes): 3 GiB, within what
|
||||
# wasm32 opens, and room for the collection.
|
||||
HOSTILE_LENGTH = 3 << 30
|
||||
|
||||
|
||||
def write_limits(out):
|
||||
"""Files for the size limits (the tests must get errors, not aborts):
|
||||
|
||||
- limits.h5: /huge_u8, 2^28 + 1024 bytes of u8 in compressed chunks
|
||||
(a small file): read whole it would take over 2 GiB while decoding;
|
||||
its last value is 7.
|
||||
- hostile_vl.h5: a variable-length string dataset /a whose global heap
|
||||
collection claims HOSTILE_GCOL_SIZE bytes, with the superblock's end
|
||||
of file set to HOSTILE_LENGTH (libhdf5 cannot read it; it is only
|
||||
served by a mock that claims that length).
|
||||
- far.h5 and far.json: /x, 16 float64 values, whose contiguous data
|
||||
address is moved FAR_SHIFT bytes on (past 2 GiB, the sign bit of a
|
||||
wasm32 isize) in a file whose end of file is moved as far; the tests'
|
||||
mock serves the data there, to show offsets up to 4 GiB work on
|
||||
wasm32.
|
||||
"""
|
||||
with h5py.File(out / "limits.h5", "w") as f:
|
||||
d = f.create_dataset("huge_u8", shape=(HUGE_U8,), dtype="u1",
|
||||
chunks=(1 << 20,), compression="gzip")
|
||||
d[-1] = 7
|
||||
path = out / "hostile_vl.h5"
|
||||
with h5py.File(path, "w", libver="earliest") as f:
|
||||
f.create_dataset("a", data=["x", "yy"], dtype=h5py.string_dtype())
|
||||
b = bytearray(path.read_bytes())
|
||||
assert b[8] == 0, "a version 0 superblock"
|
||||
b[40:48] = HOSTILE_LENGTH.to_bytes(8, "little") # end of file address
|
||||
at = b.index(b"GCOL")
|
||||
b[at + 8:at + 16] = HOSTILE_GCOL_SIZE.to_bytes(8, "little")
|
||||
path.write_bytes(bytes(b))
|
||||
|
||||
path = out / "far.h5"
|
||||
values = np.arange(16, dtype="<f8") * 1.5
|
||||
with h5py.File(path, "w", libver="earliest") as f:
|
||||
f.create_dataset("x", data=values)
|
||||
data_at = f["x"].id.get_offset()
|
||||
b = bytearray(path.read_bytes())
|
||||
# The layout message: the data's address, then its size.
|
||||
old = data_at.to_bytes(8, "little") + (values.nbytes).to_bytes(8, "little")
|
||||
assert b.count(old) == 1
|
||||
at = b.index(old)
|
||||
b[at:at + 8] = (data_at + FAR_SHIFT).to_bytes(8, "little")
|
||||
length = len(b) + FAR_SHIFT
|
||||
b[40:48] = length.to_bytes(8, "little")
|
||||
path.write_bytes(bytes(b))
|
||||
json.dump({"data_at": data_at, "far_at": data_at + FAR_SHIFT,
|
||||
"nbytes": values.nbytes, "length": length,
|
||||
"values": [float(x) for x in values]},
|
||||
open(out / "far.json", "w"))
|
||||
|
||||
|
||||
# far.h5's data moves this far: past 2 GiB, below 4 GiB.
|
||||
FAR_SHIFT = 3 << 30
|
||||
|
||||
write_limits(out)
|
||||
|
||||
|
||||
def write_big(path, megabytes):
|
||||
"""A large file for the range-request tests (`openUrl`): `/big`, about
|
||||
`megabytes` MB of float64 in 1 MiB chunks, written after a small
|
||||
|
||||
@@ -321,6 +321,8 @@ async function remoteTests() {
|
||||
await f.read("/grid");
|
||||
}, /unusable Content-Range/, "unusable Content-Range");
|
||||
|
||||
await limitTests();
|
||||
|
||||
// Corpus files: what the viewer can show of each is the same read by
|
||||
// ranges as in memory (an error wherever it gives one).
|
||||
const corpus = process.env.CLAWHDF5_WASM_CORPUS;
|
||||
@@ -328,6 +330,87 @@ async function remoteTests() {
|
||||
console.log(`openUrl: ${checks - before} checks passed`);
|
||||
}
|
||||
|
||||
// A fetch that serves `buf` as a file of `total` bytes (zeros past the end
|
||||
// of `buf`, and `extra` = [[offset, bytes], ...] laid over them), counting
|
||||
// its calls. `hide` leaves out Content-Range and the validators, as a
|
||||
// cross-origin server that exposes neither does; HEAD then gives the length.
|
||||
function mockFetch(buf, { total = buf.length, extra = [], hide = false } = {}) {
|
||||
const f = async (url, init) => {
|
||||
f.calls++;
|
||||
if (init.method === "HEAD") {
|
||||
return new Response(null, { status: 200, headers: { "Content-Length": String(total) } });
|
||||
}
|
||||
const m = /^bytes=(\d+)-(\d+)$/.exec(new Headers(init.headers).get("Range"));
|
||||
const a = Number(m[1]);
|
||||
const b = Math.min(Number(m[2]) + 1, total);
|
||||
const out = new Uint8Array(b - a);
|
||||
for (const [at, bytes] of [[0, buf], ...extra]) {
|
||||
const from = Math.max(a, at);
|
||||
const to = Math.min(b, at + bytes.length);
|
||||
if (from < to) out.set(bytes.subarray(from - at, to - at), from - a);
|
||||
}
|
||||
const headers = { "Content-Length": String(b - a) };
|
||||
if (!hide) headers["Content-Range"] = `bytes ${a}-${b - 1}/${total}`;
|
||||
return new Response(out, { status: 206, headers });
|
||||
};
|
||||
f.calls = 0;
|
||||
return f;
|
||||
}
|
||||
|
||||
// Sizes a hostile server or a large dataset can name are errors, never an
|
||||
// allocation that aborts the module (which would take every open file on
|
||||
// the page with it); see write_limits in make_fixture.py.
|
||||
async function limitTests() {
|
||||
// Read whole, /huge_u8 (2^28 + 1024 bytes) would take over 2 GiB while
|
||||
// decoding: refused before its chunks are fetched; a window reads.
|
||||
const n = 2 ** 28 + 1024;
|
||||
const limits = readFileSync(join(fixDir, "limits.h5"));
|
||||
const local = pkg.open(new Uint8Array(limits));
|
||||
await fails(() => local.read("/huge_u8"), /readHyperslab/, "huge read, in memory");
|
||||
const remote = await pkg.openUrl(`${base}/fix/limits.h5`);
|
||||
const before = remote.stats().requests;
|
||||
await fails(() => remote.read("/huge_u8"), /readHyperslab/, "huge read, openUrl");
|
||||
eq(remote.stats().requests, before, "huge read fetched nothing");
|
||||
eq(Array.from((await remote.readHyperslab("/huge_u8", [n - 4], [4])).data), [0, 0, 0, 7], "huge window");
|
||||
eq(Array.from(local.readHyperslab("/huge_u8", [n - 4], [4]).data), [0, 0, 0, 7], "huge window, in memory");
|
||||
local.free();
|
||||
|
||||
// A server claiming 3 GiB, a heap collection claiming 2 GiB + 4 KiB: an
|
||||
// error after a few requests (it used to fetch 2 GiB, then abort).
|
||||
const hostile = mockFetch(new Uint8Array(readFileSync(join(fixDir, "hostile_vl.h5"))), { total: 3 * 2 ** 30 });
|
||||
const h = await pkg.openUrl("http://hostile.invalid/h.h5", { fetch: hostile });
|
||||
await fails(() => h.read("/a"), /maxFetch/, "hostile collection size");
|
||||
assert.ok(hostile.calls <= 4, `hostile: ${hostile.calls} requests`);
|
||||
// The module survived: files open and read.
|
||||
eq((await (await pkg.openUrl(`${base}/fix/fixture.h5`)).read("/sensors/temp")).data[0], 21.5, "alive after hostile");
|
||||
|
||||
// A call that would fetch more than maxFetch fails before fetching it.
|
||||
await fails(async () => {
|
||||
const f = await pkg.openUrl(`${base}/fix/fixture.h5`, { blockSize: 512, maxFetch: 1024 });
|
||||
await f.read("/grid");
|
||||
}, /maxFetch/, "maxFetch");
|
||||
await fails(() => pkg.openUrl(`${base}/fix/fixture.h5`, { maxFetch: 2 ** 31 }), /maxFetch/, "maxFetch range");
|
||||
await fails(() => pkg.openUrl(`${base}/fix/fixture.h5`, { maxDownload: 2 ** 31 }), /maxDownload/, "maxDownload range");
|
||||
|
||||
// Offsets past 2 GiB work on wasm32: far.h5's data sits at 3 GiB.
|
||||
const far = JSON.parse(readFileSync(join(fixDir, "far.json"), "utf8"));
|
||||
const farBytes = new Uint8Array(readFileSync(join(fixDir, "far.h5")));
|
||||
const farData = farBytes.subarray(far.data_at, far.data_at + far.nbytes);
|
||||
const ff = await pkg.openUrl("http://far.invalid/far.h5", {
|
||||
fetch: mockFetch(farBytes, { total: far.length, extra: [[far.far_at, farData]] }),
|
||||
});
|
||||
eq(Array.from((await ff.read("/x")).data), far.values, "data past 2 GiB");
|
||||
eq(ff.stats().size, far.length, "size past 2 GiB");
|
||||
|
||||
// wasm32's reader holds file offsets in 32 bits: a file of 4 GiB or more
|
||||
// is refused at open, with the limit in the message (past 2^53 - 1 bytes
|
||||
// a JavaScript number is not even exact).
|
||||
for (const total of [2 ** 32, 2 ** 40, 2 ** 53 + 2]) {
|
||||
await fails(() => pkg.openUrl("http://huge.invalid/x.h5", { fetch: mockFetch(farBytes, { total }) }),
|
||||
total > 2 ** 53 ? /2\^53 - 1/ : /4 GiB/, `length ${total}`);
|
||||
}
|
||||
}
|
||||
|
||||
function hdf5Files(dir, out) {
|
||||
for (const e of readdirSync(dir, { withFileTypes: true })) {
|
||||
const p = join(dir, e.name);
|
||||
|
||||
Reference in New Issue
Block a user