remote.js read every 206 body (the probe and each range) with
resp.arrayBuffer() and checked its length afterwards, so a hostile
server could make the page buffer gigabytes before the check failed.
Every body is now piped through a TransformStream that errors as soon as
the count passes the limit, which cancels the body (and the request):
the requested range length for a 206, maxDownload for the 200 fallback.
A declared Content-Length past the limit is refused before reading. The
200 path now always streams (it read a declared length at once, because
a reader loop stalls on small bodies in headless Chromium under
--virtual-time-budget; a pipe does not).
Test (test.mjs): a probe and a range answered with a 64 MiB body read at
most the range + one 64 KiB piece; before, all 64 MiB were read ("asked
for the first 1048576 bytes of 2000000, got 67108864").
Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
201 lines
8.0 KiB
JavaScript
201 lines
8.0 KiB
JavaScript
// HTTP for clawhdf5-wasm's openUrl (see src/lib.rs and src/lazy.rs).
|
|
//
|
|
// The Rust side decides which byte ranges a read needs; this file fetches
|
|
// them with `fetch` and `Range` headers and checks every answer, so a server
|
|
// that ignores the range, answers with other bytes, or serves a file that
|
|
// changed since it was opened is an error, never data. wasm-bindgen copies
|
|
// it into the package (pkg/snippets/...).
|
|
|
|
const DEFAULT_MAX_DOWNLOAD = 512 * 1024 * 1024;
|
|
const DEFAULT_PARALLEL = 6;
|
|
|
|
function fetcher(opts) {
|
|
const f = opts?.fetch ?? globalThis.fetch;
|
|
if (typeof f !== "function") {
|
|
throw new Error("openUrl: no fetch() in this environment (pass opts.fetch)");
|
|
}
|
|
return f;
|
|
}
|
|
|
|
function init(opts, extra, method = "GET") {
|
|
return { method, headers: { ...(opts?.headers ?? {}), ...extra }, credentials: opts?.credentials };
|
|
}
|
|
|
|
// "bytes a-b/total" -> { start, end (exclusive), total | null }; null when
|
|
// the page cannot see the header (cross-origin, not exposed).
|
|
function contentRange(resp, url) {
|
|
const v = resp.headers.get("Content-Range");
|
|
if (v === null) return null;
|
|
const m = /^bytes (\d+)-(\d+)\/(\d+|\*)$/.exec(v.trim());
|
|
if (!m) throw new Error(`${url}: the server sent an unusable Content-Range: ${v}`);
|
|
return { start: Number(m[1]), end: Number(m[2]) + 1, total: m[3] === "*" ? null : Number(m[3]) };
|
|
}
|
|
|
|
// What pins the file: its ETag, else its Last-Modified (null if neither is
|
|
// visible to this page).
|
|
function validatorOf(resp) {
|
|
return resp.headers.get("ETag") ?? resp.headers.get("Last-Modified");
|
|
}
|
|
|
|
async function discard(resp) {
|
|
try {
|
|
await resp.body?.cancel();
|
|
} catch {
|
|
// Nothing to release.
|
|
}
|
|
}
|
|
|
|
// The body, refusing more than `limit` bytes as they arrive: it is piped
|
|
// through a TransformStream that errors the moment the count passes the
|
|
// limit, which cancels the body and so aborts the request. Whatever the
|
|
// server declares or sends, the page never holds more than `limit` bytes of
|
|
// it. (`tooBig(n)` makes the error; n is the count so far.) A reader loop
|
|
// would do the same, but it stalls on small bodies in headless Chromium
|
|
// under --virtual-time-budget, which the page test uses; a pipe does not.
|
|
async function readCapped(resp, limit, tooBig) {
|
|
const declared = resp.headers.get("Content-Length");
|
|
if (declared !== null && Number(declared) > limit) {
|
|
await discard(resp);
|
|
throw tooBig(declared);
|
|
}
|
|
if (!resp.body) {
|
|
// No stream to read from (some fetch implementations): all at once.
|
|
const all = new Uint8Array(await resp.arrayBuffer());
|
|
if (all.length > limit) throw tooBig(all.length);
|
|
return all;
|
|
}
|
|
let n = 0;
|
|
let over = null;
|
|
const capped = resp.body.pipeThrough(new TransformStream({
|
|
transform(chunk, ctl) {
|
|
n += chunk.length;
|
|
if (n > limit) {
|
|
over = tooBig(`over ${limit}`);
|
|
ctl.error(over);
|
|
return;
|
|
}
|
|
ctl.enqueue(chunk);
|
|
},
|
|
}));
|
|
try {
|
|
return new Uint8Array(await new Response(capped).arrayBuffer());
|
|
} catch (e) {
|
|
throw over ?? e;
|
|
}
|
|
}
|
|
|
|
// The whole body of a 200 answer (a server without range support), at
|
|
// most `limit` (maxDownload) bytes.
|
|
function readAll(resp, limit, url) {
|
|
return readCapped(resp, limit, (n) =>
|
|
new Error(`${url} is ${n} bytes, more than maxDownload (${limit}); ` +
|
|
"the server does not support range requests, so the whole file would have to be downloaded"));
|
|
}
|
|
|
|
// The body of a 206 answer, which may not be longer than the `limit` bytes
|
|
// asked for at `start` (the caller checks the exact length).
|
|
function readLimited(resp, limit, url, start) {
|
|
return readCapped(resp, limit, () =>
|
|
new Error(`${url}: asked for ${limit} bytes at offset ${start}, the server sent more`));
|
|
}
|
|
|
|
/**
|
|
* Ask for the file's first `firstLen` bytes. A server that honours the
|
|
* range (206) gives `{ length, first, validator, requests }`; one that
|
|
* answers 200 sends the whole file, which is kept (`{ whole, requests }`)
|
|
* when `opts.fallback` is "download" (the default) and the file is at most
|
|
* `opts.maxDownload` bytes, and is an error otherwise.
|
|
*/
|
|
export async function probe(url, firstLen, opts) {
|
|
const f = fetcher(opts);
|
|
const resp = await f(url, init(opts, { Range: `bytes=0-${firstLen - 1}` }));
|
|
if (resp.status === 206) {
|
|
const cr = contentRange(resp, url);
|
|
if (cr && cr.start !== 0) {
|
|
await discard(resp);
|
|
throw new Error(`${url}: asked for bytes from 0, the server sent bytes from ${cr.start}`);
|
|
}
|
|
const first = await readLimited(resp, firstLen, url, 0);
|
|
let length = cr?.total ?? null;
|
|
let requests = 1;
|
|
if (length === null) {
|
|
// Content-Range is not readable here: a cross-origin server that does
|
|
// not list it in Access-Control-Expose-Headers. Content-Length of a
|
|
// HEAD request is always readable.
|
|
const head = await f(url, init(opts, {}, "HEAD"));
|
|
requests++;
|
|
const cl = head.headers.get("Content-Length");
|
|
if (!head.ok || cl === null) {
|
|
throw new Error(`${url}: cannot learn the file's size (a cross-origin server must send ` +
|
|
"Access-Control-Expose-Headers: Content-Range, or answer HEAD with Content-Length)");
|
|
}
|
|
length = Number(cl);
|
|
}
|
|
if (!Number.isSafeInteger(length) || length < 0) {
|
|
throw new Error(`${url}: the server gave a file size of ${length} bytes; openUrl reads files ` +
|
|
"of up to 2^53 - 1 bytes (the largest offset a JavaScript number holds exactly)");
|
|
}
|
|
if (first.length !== Math.min(firstLen, length)) {
|
|
throw new Error(`${url}: asked for the first ${firstLen} bytes of ${length}, got ${first.length}`);
|
|
}
|
|
return { length, first, validator: validatorOf(resp), requests };
|
|
}
|
|
if (resp.status === 200) {
|
|
if ((opts?.fallback ?? "download") !== "download") {
|
|
await discard(resp);
|
|
throw new Error(`${url}: the server does not support HTTP range requests (it answered 200 ` +
|
|
"to a Range request); open it with { fallback: \"download\" } to download the whole file");
|
|
}
|
|
const whole = await readAll(resp, opts?.maxDownload ?? DEFAULT_MAX_DOWNLOAD, url);
|
|
return { whole, requests: 1 };
|
|
}
|
|
await discard(resp);
|
|
throw new Error(`${url}: HTTP ${resp.status} ${resp.statusText ?? ""}`.trim());
|
|
}
|
|
|
|
/**
|
|
* Fetch `ranges` ([start0, end0, start1, end1, ...], ends exclusive) of a
|
|
* file opened by `probe`, at most `opts.parallel` (default 6) at a time.
|
|
* Every answer must be a 206 with exactly the bytes asked for, from the same
|
|
* file (validator and length).
|
|
*/
|
|
export async function fetchRanges(url, ranges, opts, validator, length) {
|
|
const f = fetcher(opts);
|
|
const n = ranges.length / 2;
|
|
const out = new Array(n);
|
|
let next = 0;
|
|
async function worker() {
|
|
while (next < n) {
|
|
const i = next++;
|
|
const start = ranges[2 * i];
|
|
const end = ranges[2 * i + 1];
|
|
const resp = await f(url, init(opts, { Range: `bytes=${start}-${end - 1}` }));
|
|
if (resp.status !== 206) {
|
|
await discard(resp);
|
|
throw new Error(resp.status === 200
|
|
? `${url}: the server stopped honouring range requests`
|
|
: `${url}: HTTP ${resp.status} ${resp.statusText ?? ""}`.trim());
|
|
}
|
|
const cr = contentRange(resp, url);
|
|
const v = validatorOf(resp);
|
|
if ((validator !== null && v !== null && v !== validator) ||
|
|
(cr?.total != null && cr.total !== length)) {
|
|
await discard(resp);
|
|
throw new Error(`${url} changed on the server since it was opened`);
|
|
}
|
|
if (cr && (cr.start !== start || cr.end !== end)) {
|
|
await discard(resp);
|
|
throw new Error(`${url}: asked for bytes ${start}-${end - 1}, the server sent ${cr.start}-${cr.end - 1}`);
|
|
}
|
|
const body = await readLimited(resp, end - start, url, start);
|
|
if (body.length !== end - start) {
|
|
throw new Error(`${url}: asked for ${end - start} bytes at offset ${start}, got ${body.length}`);
|
|
}
|
|
out[i] = body;
|
|
}
|
|
}
|
|
const workers = Math.max(1, Math.min(opts?.parallel ?? DEFAULT_PARALLEL, n));
|
|
await Promise.all(Array.from({ length: workers }, worker));
|
|
return out;
|
|
}
|