Files
clawhdf5/crates/clawhdf5-wasm/js/remote.js
T
osobhandClaude Opus 5.5 5107583b97 wasm: openUrl reads remote files by HTTP range requests (range-read M4)
openUrl(url, opts) returns a RemoteFile with the methods of H5File
(kind, list, info, attrs, attrErrors, read, readHyperslab), each a
promise, and stats(). It runs every call through the restartable
LazyStorage: a pass that misses reports the byte ranges, js/remote.js
fetches them with fetch() and Range headers (six at a time), and the
pass is re-run. This keeps the main thread free without a Worker or
synchronous XHR (h5wasm's lazy files need both), as the design doc
recommends; the cost is re-running a pass per wave of misses.

Every answer is checked: a 206 with exactly the bytes asked for, and
the same ETag/Last-Modified and length as at open, else an error (never
data). A server that ignores Range (200) is downloaded whole, up to
maxDownload (512 MiB), unless fallback: "error". Options: blockSize,
cacheSize, headers, credentials, parallel, fetch.

test/serve.py is a range-capable static server with request counting
(and /norange/ for a server without range support). test.mjs repeats
every fixture check on files opened by URL (1 MiB and 512 B blocks),
checks the request budget on a 200 MB h5py file (list, three small
reads and a window of the big dataset: 5 requests, 6 MiB), the
download fallback, and HTTP errors, changed files and wrong answers;
with CLAWHDF5_WASM_CORPUS every corpus file is compared with open(bytes).

Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
2026-09-27 06:44:45 -05:00

183 lines
7.0 KiB
JavaScript

// HTTP for clawhdf5-wasm's openUrl (see src/lib.rs and src/lazy.rs).
//
// The Rust side decides which byte ranges a read needs; this file fetches
// them with `fetch` and `Range` headers and checks every answer, so a server
// that ignores the range, answers with other bytes, or serves a file that
// changed since it was opened is an error, never data. wasm-bindgen copies
// it into the package (pkg/snippets/...).
const DEFAULT_MAX_DOWNLOAD = 512 * 1024 * 1024;
const DEFAULT_PARALLEL = 6;
function fetcher(opts) {
const f = opts?.fetch ?? globalThis.fetch;
if (typeof f !== "function") {
throw new Error("openUrl: no fetch() in this environment (pass opts.fetch)");
}
return f;
}
function init(opts, extra, method = "GET") {
return { method, headers: { ...(opts?.headers ?? {}), ...extra }, credentials: opts?.credentials };
}
// "bytes a-b/total" -> { start, end (exclusive), total | null }; null when
// the page cannot see the header (cross-origin, not exposed).
function contentRange(resp, url) {
const v = resp.headers.get("Content-Range");
if (v === null) return null;
const m = /^bytes (\d+)-(\d+)\/(\d+|\*)$/.exec(v.trim());
if (!m) throw new Error(`${url}: the server sent an unusable Content-Range: ${v}`);
return { start: Number(m[1]), end: Number(m[2]) + 1, total: m[3] === "*" ? null : Number(m[3]) };
}
// What pins the file: its ETag, else its Last-Modified (null if neither is
// visible to this page).
function validatorOf(resp) {
return resp.headers.get("ETag") ?? resp.headers.get("Last-Modified");
}
async function discard(resp) {
try {
await resp.body?.cancel();
} catch {
// Nothing to release.
}
}
// The whole body, refusing more than `limit` bytes as they arrive.
async function readAll(resp, limit, url) {
const tooBig = (n) =>
new Error(`${url} is ${n} bytes, more than maxDownload (${limit}); ` +
"the server does not support range requests, so the whole file would have to be downloaded");
const declared = resp.headers.get("Content-Length");
if (declared !== null && Number(declared) > limit) {
await discard(resp);
throw tooBig(declared);
}
// A declared length was checked above: read the body at once. (Only an
// undeclared length is streamed, to stop at the limit; stream reads
// also stalled in the headless Chromium test under --virtual-time-budget.)
if (!resp.body || declared !== null) {
const all = new Uint8Array(await resp.arrayBuffer());
if (all.length > limit) throw tooBig(all.length);
return all;
}
const reader = resp.body.getReader();
const parts = [];
let n = 0;
for (;;) {
const { done, value } = await reader.read();
if (done) break;
n += value.length;
if (n > limit) {
await reader.cancel();
throw tooBig(`over ${limit}`);
}
parts.push(value);
}
const all = new Uint8Array(n);
let at = 0;
for (const p of parts) {
all.set(p, at);
at += p.length;
}
return all;
}
/**
* Ask for the file's first `firstLen` bytes. A server that honours the
* range (206) gives `{ length, first, validator, requests }`; one that
* answers 200 sends the whole file, which is kept (`{ whole, requests }`)
* when `opts.fallback` is "download" (the default) and the file is at most
* `opts.maxDownload` bytes, and is an error otherwise.
*/
export async function probe(url, firstLen, opts) {
const f = fetcher(opts);
const resp = await f(url, init(opts, { Range: `bytes=0-${firstLen - 1}` }));
if (resp.status === 206) {
const cr = contentRange(resp, url);
if (cr && cr.start !== 0) {
await discard(resp);
throw new Error(`${url}: asked for bytes from 0, the server sent bytes from ${cr.start}`);
}
const first = new Uint8Array(await resp.arrayBuffer());
let length = cr?.total ?? null;
let requests = 1;
if (length === null) {
// Content-Range is not readable here: a cross-origin server that does
// not list it in Access-Control-Expose-Headers. Content-Length of a
// HEAD request is always readable.
const head = await f(url, init(opts, {}, "HEAD"));
requests++;
const cl = head.headers.get("Content-Length");
if (!head.ok || cl === null) {
throw new Error(`${url}: cannot learn the file's size (a cross-origin server must send ` +
"Access-Control-Expose-Headers: Content-Range, or answer HEAD with Content-Length)");
}
length = Number(cl);
}
if (!Number.isSafeInteger(length) || first.length !== Math.min(firstLen, length)) {
throw new Error(`${url}: asked for the first ${firstLen} bytes of ${length}, got ${first.length}`);
}
return { length, first, validator: validatorOf(resp), requests };
}
if (resp.status === 200) {
if ((opts?.fallback ?? "download") !== "download") {
await discard(resp);
throw new Error(`${url}: the server does not support HTTP range requests (it answered 200 ` +
"to a Range request); open it with { fallback: \"download\" } to download the whole file");
}
const whole = await readAll(resp, opts?.maxDownload ?? DEFAULT_MAX_DOWNLOAD, url);
return { whole, requests: 1 };
}
await discard(resp);
throw new Error(`${url}: HTTP ${resp.status} ${resp.statusText ?? ""}`.trim());
}
/**
* Fetch `ranges` ([start0, end0, start1, end1, ...], ends exclusive) of a
* file opened by `probe`, at most `opts.parallel` (default 6) at a time.
* Every answer must be a 206 with exactly the bytes asked for, from the same
* file (validator and length).
*/
export async function fetchRanges(url, ranges, opts, validator, length) {
const f = fetcher(opts);
const n = ranges.length / 2;
const out = new Array(n);
let next = 0;
async function worker() {
while (next < n) {
const i = next++;
const start = ranges[2 * i];
const end = ranges[2 * i + 1];
const resp = await f(url, init(opts, { Range: `bytes=${start}-${end - 1}` }));
if (resp.status !== 206) {
await discard(resp);
throw new Error(resp.status === 200
? `${url}: the server stopped honouring range requests`
: `${url}: HTTP ${resp.status} ${resp.statusText ?? ""}`.trim());
}
const cr = contentRange(resp, url);
const v = validatorOf(resp);
if ((validator !== null && v !== null && v !== validator) ||
(cr?.total != null && cr.total !== length)) {
await discard(resp);
throw new Error(`${url} changed on the server since it was opened`);
}
if (cr && (cr.start !== start || cr.end !== end)) {
await discard(resp);
throw new Error(`${url}: asked for bytes ${start}-${end - 1}, the server sent ${cr.start}-${cr.end - 1}`);
}
const body = new Uint8Array(await resp.arrayBuffer());
if (body.length !== end - start) {
throw new Error(`${url}: asked for ${end - start} bytes at offset ${start}, got ${body.length}`);
}
out[i] = body;
}
}
const workers = Math.max(1, Math.min(opts?.parallel ?? DEFAULT_PARALLEL, n));
await Promise.all(Array.from({ length: workers }, worker));
return out;
}