Files
clawhdf5/crates/clawhdf5-wasm/src/core.rs
T
osobhandClaude Opus 5.5 dbafa952ac wasm: sizes a server or a dataset names are errors, not aborts
A read longer than isize::MAX (2 GiB on wasm32) aborted the module in
LazyStorage::assemble (capacity_overflow), taking every open file on the
page with it, and a hostile server only had to claim a large length and
serve a heap collection of 2 GiB + 4 KiB to get there (after fetching
2 GiB). Reading a large u8 dataset whole aborted the same way when its
values were widened to 64 bits.

- LazyConfig::max_fetch (openUrl option maxFetch, default 512 MiB, at
  most 1 GiB): a read longer than it fails at once, before anything is
  fetched, and an operation whose passes would fetch more than it fails
  before fetching (Operation::charge). assemble reserves fallibly.
- Reader::read refuses a read that would use more than 1 GiB while
  decoding (core::MAX_READ_BYTES: stored bytes + 64-bit values + result)
  with an error naming readHyperslab, before reading.
- openUrl refuses a file of 4 GiB or more at open on wasm32: the format
  code turns offsets into usize, so nothing past 4 GiB can be read there
  (shown by a new test: data at 3 GiB reads, a 4 GiB file is refused).
  maxDownload is bounded to 1 GiB.

Tests: make_fixture.py writes limits.h5 (a sparse 2^28 + 1024 byte u8
dataset), hostile_vl.h5 (the reviewer's collection) and far.h5 (data at
3 GiB); test.mjs (wasm32) and tests/lazy.rs (native) check each is an
error or reads, and that the module survives. Before: RuntimeError:
unreachable in Node; the native test read the huge dataset and fetched
2 GiB.

Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
2026-09-27 07:36:21 -05:00

662 lines
22 KiB
Rust

//! The reader behind the JavaScript API, in plain Rust so it is tested
//! natively. The `wasm_bindgen` layer in `lib.rs` only converts these types
//! to JavaScript values.
//!
//! Every read either returns the dataset's values or an error: a datatype
//! with no typed-array mapping (compound, reference, opaque, ...) is refused
//! with a message naming it, never returned as reinterpreted bytes.
use std::sync::Arc;
use clawhdf5::{AttrValue, File, Selection};
use clawhdf5_format::data_read;
use clawhdf5_format::datatype::{Datatype, DatatypeByteOrder};
use clawhdf5_format::storage::Storage;
use clawhdf5_format::vl_data::{VlResolver, check_element_size};
/// The most memory one read may use while it decodes: the stored bytes,
/// the values at 64 bits (integers are widened first) and the values
/// returned. A larger read fails with an error naming `readHyperslab`,
/// before anything is read: on wasm32 a buffer past 2 GiB cannot be
/// allocated at all, and failing to allocate aborts the module (every open
/// file on the page with it). 1 GiB leaves room in wasm32's 4 GiB for the
/// file's cached blocks and the JavaScript copy of the result.
pub const MAX_READ_BYTES: u64 = 1 << 30;
/// Errors are reported to JavaScript as messages.
pub type Result<T> = std::result::Result<T, String>;
fn err(e: impl std::fmt::Display) -> String {
e.to_string()
}
/// What a path names.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum Kind {
Group,
Dataset,
}
impl Kind {
pub fn as_str(self) -> &'static str {
match self {
Kind::Group => "group",
Kind::Dataset => "dataset",
}
}
}
/// One entry of a group listing.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct Child {
pub name: String,
pub kind: Kind,
}
/// A dataset's metadata.
#[derive(Debug, Clone, PartialEq)]
pub struct DatasetInfo {
/// Dataspace dimensions (empty for a scalar).
pub shape: Vec<u64>,
/// Maximum dimensions, `None` per unlimited dimension; `None` overall
/// when the dataspace records none.
pub maxshape: Option<Vec<Option<u64>>>,
/// Human-readable datatype, e.g. `f64`, `i16 (big-endian)`, `string[8]`.
pub dtype: String,
/// Dimensions of an array datatype's elements, appended to the shape of
/// what [`Reader::read`] returns (empty otherwise).
pub element_shape: Vec<u64>,
}
/// Decoded values, one variant per JavaScript typed array.
#[derive(Debug, Clone, PartialEq)]
pub enum Data {
F32(Vec<f32>),
F64(Vec<f64>),
I8(Vec<i8>),
I16(Vec<i16>),
I32(Vec<i32>),
I64(Vec<i64>),
U8(Vec<u8>),
U16(Vec<u16>),
U32(Vec<u32>),
U64(Vec<u64>),
/// Fixed- and variable-length strings, and enumeration member names.
Strings(Vec<String>),
}
impl Data {
pub fn len(&self) -> usize {
match self {
Data::F32(v) => v.len(),
Data::F64(v) => v.len(),
Data::I8(v) => v.len(),
Data::I16(v) => v.len(),
Data::I32(v) => v.len(),
Data::I64(v) => v.len(),
Data::U8(v) => v.len(),
Data::U16(v) => v.len(),
Data::U32(v) => v.len(),
Data::U64(v) => v.len(),
Data::Strings(v) => v.len(),
}
}
pub fn is_empty(&self) -> bool {
self.len() == 0
}
}
/// Values in row-major order with their shape.
#[derive(Debug, Clone, PartialEq)]
pub struct Array {
pub shape: Vec<u64>,
pub data: Data,
}
/// A regular hyperslab, as in `H5Sselect_hyperslab`. `stride` and `block`
/// default to 1 in every dimension.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct Hyperslab {
pub start: Vec<u64>,
pub count: Vec<u64>,
pub stride: Option<Vec<u64>>,
pub block: Option<Vec<u64>>,
}
/// An attribute: its value, or why it has none.
#[derive(Debug, Clone)]
pub struct Attr {
pub name: String,
pub value: AttrValue,
}
/// An open file: held in memory ([`Reader::open`]) or read through a
/// [`Storage`] ([`Reader::open_storage`], such as a
/// [`LazyStorage`](crate::lazy::LazyStorage)).
pub struct Reader {
file: File,
}
impl Reader {
/// Parse a file from its bytes (the browser hands over the whole file).
pub fn open(bytes: Vec<u8>) -> Result<Self> {
Ok(Self {
file: File::from_bytes(bytes).map_err(err)?,
})
}
/// Open a file read through `storage` (the file's bytes from offset 0,
/// user block included, as [`File::open_storage`] takes them).
pub fn open_storage(storage: Arc<dyn Storage + Send + Sync>) -> Result<Self> {
Ok(Self {
file: File::open_storage(storage).map_err(err)?,
})
}
/// Whether `path` names a group or a dataset.
pub fn kind(&self, path: &str) -> Result<Kind> {
match self.file.dataset(path) {
Ok(_) => Ok(Kind::Dataset),
Err(clawhdf5::Error::NotADataset(_)) => Ok(Kind::Group),
Err(e) => Err(err(e)),
}
}
/// The groups, then the datasets, in the group at `path` (`/` is the
/// root). Soft links are listed as their targets; external and dangling
/// links, and named datatypes, are left out.
pub fn list(&self, path: &str) -> Result<Vec<Child>> {
if self.kind(path)? != Kind::Group {
return Err(format!("not a group: {path}"));
}
let group = self.file.group(path).map_err(err)?;
let mut out: Vec<Child> = group
.groups()
.map_err(err)?
.into_iter()
.map(|name| Child {
name,
kind: Kind::Group,
})
.collect();
out.extend(
group
.datasets()
.map_err(err)?
.into_iter()
.map(|name| Child {
name,
kind: Kind::Dataset,
}),
);
Ok(out)
}
/// Shape, max shape and datatype of the dataset at `path`.
pub fn info(&self, path: &str) -> Result<DatasetInfo> {
let ds = self.file.dataset(path).map_err(err)?;
let dt = ds.raw_datatype().map_err(err)?;
let maxshape = ds.max_dimensions().map_err(err)?.map(|dims| {
dims.into_iter()
.map(|d| (d != u64::MAX).then_some(d))
.collect()
});
Ok(DatasetInfo {
shape: ds.shape().map_err(err)?,
maxshape,
dtype: describe(&dt),
element_shape: element_shape(&dt),
})
}
/// The attributes of the group or dataset at `path`, sorted by name, and
/// one message per attribute that could not be read at all. An attribute
/// whose type has no plain JavaScript form is returned as
/// [`AttrValue::Raw`].
pub fn attrs(&self, path: &str) -> Result<(Vec<Attr>, Vec<String>)> {
let (map, errors) = match self.kind(path)? {
Kind::Dataset => self
.file
.dataset(path)
.and_then(|d| d.attrs_with_errors())
.map_err(err)?,
Kind::Group => self
.file
.group(path)
.and_then(|g| g.attrs_with_errors())
.map_err(err)?,
};
let mut attrs: Vec<Attr> = map
.into_iter()
.map(|(name, value)| Attr { name, value })
.collect();
attrs.sort_by(|a, b| a.name.cmp(&b.name));
Ok((attrs, errors.into_iter().map(err).collect()))
}
/// Read the dataset at `path`, whole or a hyperslab of it.
pub fn read(&self, path: &str, slab: Option<&Hyperslab>) -> Result<Array> {
let ds = self.file.dataset(path).map_err(err)?;
let dt = ds.raw_datatype().map_err(err)?;
let shape = ds.shape().map_err(err)?;
let (selection, mut out_shape) = match slab {
None => (Selection::All, shape.clone()),
Some(h) => hyperslab_selection(h, &shape)?,
};
// A VL type whose stored element size is not the one the file's
// offset size implies is refused before its data is read, as
// `File::read_string` refuses it.
if let Datatype::VariableLength { size, .. } = array_base(&dt) {
check_element_size(*size, self.file.superblock().offset_size).map_err(err)?;
}
out_shape.extend(element_shape(&dt));
let expected = out_shape
.iter()
.try_fold(1u64, |acc, &d| acc.checked_mul(d))
.ok_or("selection size overflows")?;
let cost = expected.saturating_mul(bytes_per_value(&dt));
if cost > MAX_READ_BYTES {
return Err(format!(
"reading {path}{} would take about {} MiB of memory, more than the {} MiB \
one read may use; read it in parts (readHyperslab)",
if slab.is_some() {
" (this selection)"
} else {
" whole"
},
cost >> 20,
MAX_READ_BYTES >> 20
));
}
let raw = ds.read_selection(&selection).map_err(err)?;
let data = self.decode(&raw, &dt)?;
if data.len() as u64 != expected {
return Err(format!(
"read {} values for shape {out_shape:?} ({expected} expected)",
data.len()
));
}
Ok(Array {
shape: out_shape,
data,
})
}
fn decode(&self, raw: &[u8], dt: &Datatype) -> Result<Data> {
let base = array_base(dt);
let is_array = !std::ptr::eq(base, dt);
Ok(match base {
Datatype::FloatingPoint { size, .. } if *size <= 4 => {
Data::F32(data_read::read_as_f32(raw, dt).map_err(err)?)
}
Datatype::FloatingPoint { .. } => {
Data::F64(data_read::read_as_f64(raw, dt).map_err(err)?)
}
Datatype::FixedPoint { size, signed, .. } => {
let signed_ints = || data_read::read_as_i64(raw, dt).map_err(err);
let unsigned_ints = || data_read::read_as_u64(raw, dt).map_err(err);
match (size, signed) {
(1, true) => Data::I8(narrow(signed_ints()?)?),
(2, true) => Data::I16(narrow(signed_ints()?)?),
(4, true) => Data::I32(narrow(signed_ints()?)?),
(_, true) => Data::I64(signed_ints()?),
(1, false) => Data::U8(narrow(unsigned_ints()?)?),
(2, false) => Data::U16(narrow(unsigned_ints()?)?),
(4, false) => Data::U32(narrow(unsigned_ints()?)?),
(_, false) => Data::U64(unsigned_ints()?),
}
}
Datatype::String { .. } if !is_array => {
Data::Strings(data_read::read_as_strings(raw, dt).map_err(err)?)
}
Datatype::VariableLength {
is_string: true, ..
} if !is_array => {
// The library's resolver, as File::read_string uses: a
// string ends at its first NUL and a heap object of the
// wrong size is an error, as in libhdf5 and h5py.
let sb = self.file.superblock();
let strings = match self.file.contiguous_bytes() {
Some(bytes) => {
VlResolver::new(bytes, sb.offset_size, sb.length_size).strings(raw)
}
None => VlResolver::new_in(self.file.storage(), sb.offset_size, sb.length_size)
.strings(raw),
};
Data::Strings(strings.map_err(err)?)
}
Datatype::Enumeration { .. } if !is_array => {
Data::Strings(data_read::read_enum_names(raw, dt).map_err(err)?)
}
_ => {
return Err(format!(
"reading {} datasets is not supported",
describe(dt)
));
}
})
}
}
/// Memory one value of type `dt` takes while [`Reader::read`] decodes it
/// (an array type's elements count as values): its stored bytes, plus what
/// [`Reader::decode`] builds from them. A string counts its `String` (24
/// bytes on 64-bit targets, less on wasm32) and, for a fixed-length one,
/// its text; a variable-length string's text lives in the heap and is
/// bounded by the storage's own read limit.
fn bytes_per_value(dt: &Datatype) -> u64 {
let base = array_base(dt);
let stored = u64::from(base.type_size());
stored
+ match base {
Datatype::FloatingPoint { size, .. } if *size <= 4 => 4,
Datatype::FloatingPoint { .. } => 8,
// Widened to 64 bits, then narrowed to a new vector.
Datatype::FixedPoint { .. } => 8 + stored,
Datatype::String { .. } => 24 + stored,
Datatype::VariableLength { .. } | Datatype::Enumeration { .. } => 24,
_ => 0,
}
}
/// Narrow integers read at 64 bits to the dataset's own width. The source is
/// that width, so this cannot fail on correct input; it is checked anyway.
fn narrow<S: Copy + std::fmt::Display, T: TryFrom<S>>(v: Vec<S>) -> Result<Vec<T>> {
v.into_iter()
.map(|x| T::try_from(x).map_err(|_| format!("value {x} out of range")))
.collect()
}
/// Innermost element type of (possibly nested) array datatypes.
fn array_base(dt: &Datatype) -> &Datatype {
match dt {
Datatype::Array { base_type, .. } => array_base(base_type),
_ => dt,
}
}
fn element_shape(dt: &Datatype) -> Vec<u64> {
match dt {
Datatype::Array {
base_type,
dimensions,
} => {
let mut dims: Vec<u64> = dimensions.iter().map(|&d| u64::from(d)).collect();
dims.extend(element_shape(base_type));
dims
}
_ => Vec::new(),
}
}
fn hyperslab_selection(h: &Hyperslab, shape: &[u64]) -> Result<(Selection, Vec<u64>)> {
let rank = shape.len();
let ones = vec![1u64; rank];
let stride = h.stride.clone().unwrap_or_else(|| ones.clone());
let block = h.block.clone().unwrap_or(ones);
for (what, v) in [
("start", &h.start),
("count", &h.count),
("stride", &stride),
("block", &block),
] {
if v.len() != rank {
return Err(format!(
"hyperslab {what} has {} dimensions, the dataset has {rank}",
v.len()
));
}
}
let mut out = Vec::with_capacity(rank);
for d in 0..rank {
if stride[d] == 0 || block[d] == 0 {
return Err(format!("hyperslab stride and block must be >= 1 (dim {d})"));
}
if h.count[d] > 1 && block[d] > stride[d] {
return Err(format!(
"hyperslab blocks overlap in dim {d}: block {} > stride {}",
block[d], stride[d]
));
}
// Last element selected: start + (count-1)*stride + block - 1.
if h.count[d] > 0 {
let last = (h.count[d] - 1)
.checked_mul(stride[d])
.and_then(|x| x.checked_add(h.start[d]))
.and_then(|x| x.checked_add(block[d] - 1));
match last {
Some(l) if l < shape[d] => {}
_ => {
return Err(format!(
"hyperslab exceeds dimension {d} (extent {})",
shape[d]
));
}
}
}
out.push(
h.count[d]
.checked_mul(block[d])
.ok_or("selection size overflows")?,
);
}
Ok((
Selection::Hyperslab {
start: h.start.clone(),
stride,
count: h.count.clone(),
block,
},
out,
))
}
/// A short, human-readable datatype name.
pub fn describe(dt: &Datatype) -> String {
fn endian(order: &DatatypeByteOrder) -> &'static str {
match order {
DatatypeByteOrder::BigEndian => " (big-endian)",
DatatypeByteOrder::Vax => " (VAX)",
_ => "",
}
}
match dt {
Datatype::FixedPoint {
size,
signed,
byte_order,
..
} => format!(
"{}{}{}",
if *signed { "i" } else { "u" },
size * 8,
endian(byte_order)
),
Datatype::FloatingPoint {
size, byte_order, ..
} => format!("f{}{}", size * 8, endian(byte_order)),
Datatype::Time { size, .. } => format!("time{}", size * 8),
Datatype::String { size, .. } => format!("string[{size}]"),
Datatype::BitField { size, .. } => format!("bitfield{}", size * 8),
Datatype::Opaque { size, .. } => format!("opaque[{size}]"),
Datatype::Compound { members, .. } => {
let fields: Vec<String> = members
.iter()
.map(|m| format!("{}: {}", m.name, describe(&m.datatype)))
.collect();
format!("compound{{{}}}", fields.join(", "))
}
Datatype::Reference { .. } => "reference".to_string(),
Datatype::Enumeration {
base_type, members, ..
} => {
let names: Vec<&str> = members.iter().map(|m| m.name.as_str()).collect();
format!("enum<{}>{{{}}}", describe(base_type), names.join(", "))
}
Datatype::VariableLength {
is_string: true, ..
} => "vlen string".to_string(),
Datatype::VariableLength { base_type, .. } => {
format!("vlen<{}>", describe(base_type))
}
Datatype::Array {
base_type,
dimensions,
} => format!("array{dimensions:?}<{}>", describe(base_type)),
}
}
#[cfg(test)]
mod tests {
use super::*;
use clawhdf5::FileBuilder;
fn sample() -> Reader {
let mut b = FileBuilder::new();
b.create_dataset("grid")
.with_f64_data(&(0..12).map(f64::from).collect::<Vec<_>>())
.with_shape(&[3, 4])
.with_chunks(&[2, 2])
.with_deflate(4);
b.create_dataset("bytes").with_u8_data(&[1, 2, 250]);
let mut g = b.create_group("sensors");
g.create_dataset("temp").with_f32_data(&[1.5, -2.25]);
g.set_attr("location", AttrValue::String("lab".into()));
b.add_group(g.finish());
b.set_attr("version", AttrValue::I64(3));
b.set_attr("scale", AttrValue::F64Array(vec![0.5, 2.0]));
Reader::open(b.finish().unwrap()).unwrap()
}
#[test]
fn lists_groups_then_datasets() {
let r = sample();
let names: Vec<(String, Kind)> = r
.list("/")
.unwrap()
.into_iter()
.map(|c| (c.name, c.kind))
.collect();
assert_eq!(names[0], ("sensors".to_string(), Kind::Group));
let mut ds: Vec<&str> = names[1..].iter().map(|(n, _)| n.as_str()).collect();
ds.sort();
assert_eq!(ds, ["bytes", "grid"]);
assert_eq!(
r.list("sensors").unwrap(),
vec![Child {
name: "temp".into(),
kind: Kind::Dataset
}]
);
assert!(r.list("grid").unwrap_err().contains("not a group"));
assert!(r.list("missing").is_err());
}
#[test]
fn info_reports_shape_and_dtype() {
let r = sample();
let i = r.info("grid").unwrap();
assert_eq!(i.shape, vec![3, 4]);
assert_eq!(i.dtype, "f64");
assert!(i.element_shape.is_empty());
assert_eq!(r.info("sensors/temp").unwrap().dtype, "f32");
assert_eq!(r.kind("/sensors").unwrap(), Kind::Group);
assert_eq!(r.kind("/sensors/temp").unwrap(), Kind::Dataset);
}
#[test]
fn reads_whole_and_hyperslab() {
let r = sample();
let all = r.read("grid", None).unwrap();
assert_eq!(all.shape, vec![3, 4]);
assert_eq!(all.data, Data::F64((0..12).map(f64::from).collect()));
let slab = Hyperslab {
start: vec![1, 0],
count: vec![2, 2],
stride: Some(vec![1, 2]),
block: None,
};
let part = r.read("grid", Some(&slab)).unwrap();
assert_eq!(part.shape, vec![2, 2]);
assert_eq!(part.data, Data::F64(vec![4.0, 6.0, 8.0, 10.0]));
assert_eq!(
r.read("bytes", None).unwrap().data,
Data::U8(vec![1, 2, 250])
);
assert_eq!(
r.read("sensors/temp", None).unwrap().data,
Data::F32(vec![1.5, -2.25])
);
}
#[test]
fn bad_hyperslabs_are_refused() {
let r = sample();
let mk = |start: Vec<u64>, count: Vec<u64>| Hyperslab {
start,
count,
stride: None,
block: None,
};
assert!(
r.read("grid", Some(&mk(vec![0], vec![1])))
.unwrap_err()
.contains("dimensions")
);
assert!(
r.read("grid", Some(&mk(vec![2, 0], vec![2, 1])))
.unwrap_err()
.contains("exceeds")
);
let overlap = Hyperslab {
start: vec![0, 0],
count: vec![2, 1],
stride: Some(vec![1, 1]),
block: Some(vec![2, 1]),
};
assert!(
r.read("grid", Some(&overlap))
.unwrap_err()
.contains("overlap")
);
}
#[test]
fn attrs_are_sorted() {
let r = sample();
let (attrs, errors) = r.attrs("/").unwrap();
assert!(errors.is_empty());
let names: Vec<&str> = attrs.iter().map(|a| a.name.as_str()).collect();
assert_eq!(names, ["scale", "version"]);
let (g, _) = r.attrs("sensors").unwrap();
assert!(matches!(&g[0].value, AttrValue::String(s) if s == "lab"));
}
#[test]
fn compound_is_refused_not_reinterpreted() {
use clawhdf5::CompoundTypeBuilder;
let ct = CompoundTypeBuilder::new()
.f64_field("x")
.i32_field("n")
.build();
let mut rec = Vec::new();
rec.extend_from_slice(&1.0f64.to_le_bytes());
rec.extend_from_slice(&7i32.to_le_bytes());
let mut b = FileBuilder::new();
b.create_dataset("table").with_compound_data(ct, rec, 1);
let r = Reader::open(b.finish().unwrap()).unwrap();
let e = r.read("table", None).unwrap_err();
assert!(e.contains("compound{x: f64, n: i32}"), "{e}");
assert!(e.contains("not supported"), "{e}");
}
#[test]
fn garbage_is_an_error() {
assert!(Reader::open(vec![0u8; 64]).is_err());
assert!(Reader::open(Vec::new()).is_err());
}
}