wasm: range answers are read with a hard limit, not buffered whole

remote.js read every 206 body (the probe and each range) with
resp.arrayBuffer() and checked its length afterwards, so a hostile
server could make the page buffer gigabytes before the check failed.

Every body is now piped through a TransformStream that errors as soon as
the count passes the limit, which cancels the body (and the request):
the requested range length for a 206, maxDownload for the 200 fallback.
A declared Content-Length past the limit is refused before reading. The
200 path now always streams (it read a declared length at once, because
a reader loop stalls on small bodies in headless Chromium under
--virtual-time-budget; a pipe does not).

Test (test.mjs): a probe and a range answered with a 64 MiB body read at
most the range + one 64 KiB piece; before, all 64 MiB were read ("asked
for the first 1048576 bytes of 2000000, got 67108864").

Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
osobh
2026-09-27 07:39:50 -05:00
co-authored by Claude Opus 5.5
parent dbafa952ac
commit 1d065adf8b
2 changed files with 95 additions and 29 deletions
+40 -26
View File
@@ -45,44 +45,58 @@ async function discard(resp) {
}
}
// The whole body, refusing more than `limit` bytes as they arrive.
async function readAll(resp, limit, url) {
const tooBig = (n) =>
new Error(`${url} is ${n} bytes, more than maxDownload (${limit}); ` +
"the server does not support range requests, so the whole file would have to be downloaded");
// The body, refusing more than `limit` bytes as they arrive: it is piped
// through a TransformStream that errors the moment the count passes the
// limit, which cancels the body and so aborts the request. Whatever the
// server declares or sends, the page never holds more than `limit` bytes of
// it. (`tooBig(n)` makes the error; n is the count so far.) A reader loop
// would do the same, but it stalls on small bodies in headless Chromium
// under --virtual-time-budget, which the page test uses; a pipe does not.
async function readCapped(resp, limit, tooBig) {
const declared = resp.headers.get("Content-Length");
if (declared !== null && Number(declared) > limit) {
await discard(resp);
throw tooBig(declared);
}
// A declared length was checked above: read the body at once. (Only an
// undeclared length is streamed, to stop at the limit; stream reads
// also stalled in the headless Chromium test under --virtual-time-budget.)
if (!resp.body || declared !== null) {
if (!resp.body) {
// No stream to read from (some fetch implementations): all at once.
const all = new Uint8Array(await resp.arrayBuffer());
if (all.length > limit) throw tooBig(all.length);
return all;
}
const reader = resp.body.getReader();
const parts = [];
let n = 0;
for (;;) {
const { done, value } = await reader.read();
if (done) break;
n += value.length;
let over = null;
const capped = resp.body.pipeThrough(new TransformStream({
transform(chunk, ctl) {
n += chunk.length;
if (n > limit) {
await reader.cancel();
throw tooBig(`over ${limit}`);
over = tooBig(`over ${limit}`);
ctl.error(over);
return;
}
parts.push(value);
ctl.enqueue(chunk);
},
}));
try {
return new Uint8Array(await new Response(capped).arrayBuffer());
} catch (e) {
throw over ?? e;
}
const all = new Uint8Array(n);
let at = 0;
for (const p of parts) {
all.set(p, at);
at += p.length;
}
return all;
// The whole body of a 200 answer (a server without range support), at
// most `limit` (maxDownload) bytes.
function readAll(resp, limit, url) {
return readCapped(resp, limit, (n) =>
new Error(`${url} is ${n} bytes, more than maxDownload (${limit}); ` +
"the server does not support range requests, so the whole file would have to be downloaded"));
}
// The body of a 206 answer, which may not be longer than the `limit` bytes
// asked for at `start` (the caller checks the exact length).
function readLimited(resp, limit, url, start) {
return readCapped(resp, limit, () =>
new Error(`${url}: asked for ${limit} bytes at offset ${start}, the server sent more`));
}
/**
@@ -101,7 +115,7 @@ export async function probe(url, firstLen, opts) {
await discard(resp);
throw new Error(`${url}: asked for bytes from 0, the server sent bytes from ${cr.start}`);
}
const first = new Uint8Array(await resp.arrayBuffer());
const first = await readLimited(resp, firstLen, url, 0);
let length = cr?.total ?? null;
let requests = 1;
if (length === null) {
@@ -173,7 +187,7 @@ export async function fetchRanges(url, ranges, opts, validator, length) {
await discard(resp);
throw new Error(`${url}: asked for bytes ${start}-${end - 1}, the server sent ${cr.start}-${cr.end - 1}`);
}
const body = new Uint8Array(await resp.arrayBuffer());
const body = await readLimited(resp, end - start, url, start);
if (body.length !== end - start) {
throw new Error(`${url}: asked for ${end - start} bytes at offset ${start}, got ${body.length}`);
}
+52
View File
@@ -322,6 +322,7 @@ async function remoteTests() {
}, /unusable Content-Range/, "unusable Content-Range");
await limitTests();
await floodTests();
// Corpus files: what the viewer can show of each is the same read by
// ranges as in memory (an error wherever it gives one).
@@ -411,6 +412,57 @@ async function limitTests() {
}
}
// A body of `total` bytes in 64 KiB pieces, made as they are read; `pulled()`
// tells how many were.
function flood(total) {
let pulled = 0;
const body = new ReadableStream({
pull(c) {
if (pulled >= total) return c.close();
const n = Math.min(65536, total - pulled);
pulled += n;
c.enqueue(new Uint8Array(n));
},
}, { highWaterMark: 0 });
return { body, pulled: () => pulled };
}
// A 206 whose body runs past the range asked for is cut off as it arrives:
// the page never reads (or holds) more than it asked for, whatever the
// server sends. 64 MiB stands for "gigabytes".
async function floodTests() {
const FLOOD = 64 << 20;
// The probe: asked for the first 1 MiB.
let f = flood(FLOOD);
await fails(() => pkg.openUrl("http://flood.invalid/x.h5", {
fetch: async () => new Response(f.body, { status: 206, headers: { "Content-Range": "bytes 0-1048575/2000000" } }),
}), /the server sent more/, "flooded probe");
assert.ok(f.pulled() <= (1 << 20) + 65536, `probe: read ${f.pulled()} bytes`);
// A range request after a correct probe.
f = flood(FLOOD);
const file = await pkg.openUrl(`${base}/fix/fixture.h5`, {
blockSize: 512,
fetch: async (url, init) => {
const range = new Headers(init.headers).get("Range");
if (range === "bytes=0-511") return fetch(url, init);
const m = /^bytes=(\d+)-(\d+)$/.exec(range);
return new Response(f.body, { status: 206, headers: { "Content-Range": `bytes ${m[1]}-${m[2]}/25752` } });
},
}).catch((e) => e);
// The open itself may need a second range: flooded either way.
const read = file instanceof Error ? Promise.reject(file) : file.read("/grid");
await fails(() => read, /the server sent more/, "flooded range");
assert.ok(f.pulled() <= 8 * 512 + 65536, `range: read ${f.pulled()} bytes`);
// A declared Content-Length past the range is refused before reading.
f = flood(FLOOD);
await fails(() => pkg.openUrl("http://flood.invalid/x.h5", {
fetch: async () => new Response(f.body, {
status: 206, headers: { "Content-Range": "bytes 0-1048575/2000000", "Content-Length": String(FLOOD) },
}),
}), /the server sent more/, "declared too long");
assert.ok(f.pulled() <= 65536, `declared: read ${f.pulled()} bytes`);
}
function hdf5Files(dir, out) {
for (const e of readdirSync(dir, { withFileTypes: true })) {
const p = join(dir, e.name);