For a metadata cache image libhdf5 cannot load, libhdf5 opens the file and fails the first metadata read (the image loads on the first H5C_protect after open); h5py reports the error on the root group. The conformance probe reported it that way, but File::open refused the file, so the gate counted cve-2025-6269-1..4 and cve-2025-6516 as agreeing with h5py for behaviour the library did not have. The library now behaves as the probe reports: File (mmap, buffered and from_bytes) and MmapFile open the file and every object lookup (dataset, dataset_at, group, group listings and attributes, VL decoding) fails with the image's error; LazyFile reads the root group's header at open, so its open is that first read and fails. Probe and library take the three-way decision (refuse at open / image loads / image cannot load) from the same clawhdf5_format::superblock_ext::cache_image_state. One deliberate difference from libhdf5 remains, documented: after the failed first read libhdf5 reads the file's own metadata, which the image was meant to replace and may be stale; here every lookup keeps failing. File::cache_image_error / MmapFile::cache_image_error expose the error to code that parses as_bytes() itself; h5rs checks it before reading any object header (h5rs ls on cve-2025-6269-1 said "invalid object header version: 0" from the stale bytes). Test: metadata_cache_image.rs an_image_libhdf5_cannot_load_fails_every_object (the fixture with its image signature broken; h5py opens that file and fails the first read with "Bad metadata cache image header signature"). It fails on the previous commit, where File::open refuses the file. Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
961 lines
35 KiB
Rust
961 lines
35 KiB
Rust
//! Conformance probe: walks an HDF5 file with clawhdf5-format (the same calls
|
|
//! the `clawhdf5` facade makes) and prints a canonical JSON description:
|
|
//! every hard-linked object (sorted-name DFS, deduplicated by header address),
|
|
//! and for each dataset / attribute its shape plus the SHA-256 of its values
|
|
//! in a canonical encoding shared with `ref.py`.
|
|
//!
|
|
//! Canonical value encoding (per element, concatenated, row-major):
|
|
//! int / float / bitfield / enum / time : element bytes, little-endian
|
|
//! non-IEEE-layout float (e.g. N-Bit) : the IEEE float of the same size it converts to
|
|
//! int with bit offset / short precision: the full-width integer it converts to
|
|
//! opaque : raw bytes
|
|
//! compound : members in declaration order (padding dropped)
|
|
//! array : base elements row-major
|
|
//! string (fixed or VL) : b'S' + u32le len + bytes (cut at first NUL, trailing spaces stripped)
|
|
//! VL sequence : b'V' + u32le count + base elements
|
|
//! reference : b'R' (payload not compared)
|
|
//!
|
|
//! Every object is processed inside catch_unwind; a caught panic is recorded
|
|
//! with its message, location and the clawhdf5 frames of its backtrace.
|
|
|
|
use std::cell::RefCell;
|
|
use std::collections::HashSet;
|
|
use std::panic::{self, AssertUnwindSafe};
|
|
|
|
use clawhdf5_format::attribute::extract_attributes_full;
|
|
use clawhdf5_format::data_layout::DataLayout;
|
|
use clawhdf5_format::data_read;
|
|
use clawhdf5_format::dataspace::{Dataspace, DataspaceType};
|
|
use clawhdf5_format::datatype::{Datatype, DatatypeByteOrder};
|
|
use clawhdf5_format::filter_pipeline::FilterPipeline;
|
|
use clawhdf5_format::group_v1::{self, GroupEntry};
|
|
use clawhdf5_format::group_v2;
|
|
use clawhdf5_format::message_type::MessageType;
|
|
use clawhdf5_format::object_header::{ObjectClass, ObjectHeader};
|
|
use clawhdf5_format::signature;
|
|
use clawhdf5_format::superblock::Superblock;
|
|
use clawhdf5_format::symbol_table::SymbolTableMessage;
|
|
use clawhdf5_format::vl_data::{VlResolver, check_element_size};
|
|
use serde_json::{Map, Value, json};
|
|
use sha2::{Digest, Sha256};
|
|
|
|
const MAX_BYTES: u64 = 200 * 1024 * 1024;
|
|
const MAX_OBJECTS: usize = 200_000;
|
|
|
|
thread_local! {
|
|
static LAST_PANIC: RefCell<Option<String>> = const { RefCell::new(None) };
|
|
}
|
|
|
|
fn install_hook() {
|
|
panic::set_hook(Box::new(|info| {
|
|
let msg = if let Some(s) = info.payload().downcast_ref::<&str>() {
|
|
s.to_string()
|
|
} else if let Some(s) = info.payload().downcast_ref::<String>() {
|
|
s.clone()
|
|
} else {
|
|
"<non-string panic>".into()
|
|
};
|
|
let loc = info
|
|
.location()
|
|
.map(|l| format!("{}:{}", l.file(), l.line()))
|
|
.unwrap_or_default();
|
|
let bt = std::backtrace::Backtrace::force_capture().to_string();
|
|
// keep only frames from clawhdf5 code
|
|
let mut frames = Vec::new();
|
|
let lines: Vec<&str> = bt.lines().collect();
|
|
for (i, l) in lines.iter().enumerate() {
|
|
let t = l.trim();
|
|
if t.contains("clawhdf5_format::") || t.contains("conformance_probe::") {
|
|
let at = lines
|
|
.get(i + 1)
|
|
.map(|n| n.trim())
|
|
.filter(|n| n.starts_with("at "))
|
|
.map(|n| {
|
|
let n = n.trim_start_matches("at ");
|
|
match n.find("/crates/") {
|
|
Some(p) => n[p + 1..].to_string(),
|
|
None => n.to_string(),
|
|
}
|
|
})
|
|
.unwrap_or_default();
|
|
let name = t.split_once(": ").map(|x| x.1).unwrap_or(t);
|
|
frames.push(format!("{name} ({at})"));
|
|
if frames.len() >= 12 {
|
|
break;
|
|
}
|
|
}
|
|
}
|
|
let full = format!("PANIC: {msg} @ {loc}\n {}", frames.join("\n "));
|
|
eprintln!("{full}");
|
|
LAST_PANIC.with(|p| *p.borrow_mut() = Some(full));
|
|
}));
|
|
}
|
|
|
|
/// Run `f`, turning a panic into Err("PANIC: ...").
|
|
fn guarded<T>(f: impl FnOnce() -> Result<T, String>) -> Result<T, String> {
|
|
match panic::catch_unwind(AssertUnwindSafe(f)) {
|
|
Ok(r) => r,
|
|
Err(_) => Err(LAST_PANIC
|
|
.with(|p| p.borrow_mut().take())
|
|
.unwrap_or_else(|| "PANIC: <unknown>".into())),
|
|
}
|
|
}
|
|
|
|
fn e<E: std::fmt::Debug>(x: E) -> String {
|
|
format!("{x:?}")
|
|
}
|
|
|
|
struct Ctx<'a> {
|
|
data: &'a [u8],
|
|
os: u8,
|
|
ls: u8,
|
|
base_dir: std::path::PathBuf,
|
|
/// Resolves variable-length elements as the library does (null
|
|
/// elements, strings cut at a NUL, heap objects of the wrong size
|
|
/// refused), caching each heap collection.
|
|
vl: RefCell<VlResolver<'a>>,
|
|
}
|
|
|
|
impl<'a> Ctx<'a> {
|
|
fn header(&self, addr: u64) -> Result<ObjectHeader, String> {
|
|
ObjectHeader::parse(self.data, addr as usize, self.os, self.ls).map_err(e)
|
|
}
|
|
|
|
fn payload(&self, h: &ObjectHeader, t: MessageType) -> Result<Option<Vec<u8>>, String> {
|
|
match h.messages.iter().find(|m| m.msg_type == t) {
|
|
None => Ok(None),
|
|
Some(m) => {
|
|
clawhdf5_format::shared_message::message_data(self.data, m, self.os, self.ls)
|
|
.map(|c| Some(c.into_owned()))
|
|
.map_err(e)
|
|
}
|
|
}
|
|
}
|
|
|
|
fn canon(&self, dt: &Datatype, b: &[u8], out: &mut Vec<u8>) -> Result<(), String> {
|
|
let size = dt.type_size() as usize;
|
|
if b.len() < size {
|
|
return Err(format!(
|
|
"canon: element slice {} < type size {size}",
|
|
b.len()
|
|
));
|
|
}
|
|
match dt {
|
|
Datatype::FloatingPoint { .. } if !ieee_layout(dt) => {
|
|
canon_custom_float(dt, &b[..size], out)?
|
|
}
|
|
Datatype::FixedPoint { .. } if partial_int(dt) => {
|
|
canon_partial_int(dt, &b[..size], out)?
|
|
}
|
|
Datatype::FixedPoint { byte_order, .. }
|
|
| Datatype::BitField { byte_order, .. }
|
|
| Datatype::FloatingPoint { byte_order, .. } => match byte_order {
|
|
DatatypeByteOrder::LittleEndian => out.extend_from_slice(&b[..size]),
|
|
DatatypeByteOrder::BigEndian => out.extend(b[..size].iter().rev()),
|
|
DatatypeByteOrder::Vax => return Err("canon: VAX byte order".into()),
|
|
},
|
|
Datatype::Time { .. } | Datatype::Opaque { .. } => out.extend_from_slice(&b[..size]),
|
|
Datatype::String { .. } => canon_str(&b[..size], out),
|
|
Datatype::Compound { members, .. } => {
|
|
for m in members {
|
|
let off = m.byte_offset as usize;
|
|
let ms = m.datatype.type_size() as usize;
|
|
if off.checked_add(ms).is_none_or(|end| end > size) {
|
|
return Err(format!("canon: member {} out of bounds", m.name));
|
|
}
|
|
self.canon(&m.datatype, &b[off..off + ms], out)?;
|
|
}
|
|
}
|
|
Datatype::Reference { .. } => out.push(b'R'),
|
|
Datatype::Enumeration { base_type, .. } => self.canon(base_type, b, out)?,
|
|
Datatype::Array {
|
|
base_type,
|
|
dimensions,
|
|
} => {
|
|
let n: usize = dimensions.iter().map(|d| *d as usize).product();
|
|
let bs = base_type.type_size() as usize;
|
|
for i in 0..n {
|
|
self.canon(base_type, &b[i * bs..], out)?;
|
|
}
|
|
}
|
|
Datatype::VariableLength {
|
|
size: vl_size,
|
|
is_string,
|
|
base_type,
|
|
..
|
|
} => {
|
|
check_element_size(*vl_size, self.os).map_err(e)?;
|
|
let el = &b[..size];
|
|
if *is_string {
|
|
let s = self.vl.borrow_mut().string_bytes(el).map_err(e)?;
|
|
canon_str(&s[0], out);
|
|
} else {
|
|
let bs = base_type.type_size() as usize;
|
|
// The borrow ends here: the base type may itself be
|
|
// variable-length.
|
|
let seq = self.vl.borrow_mut().sequences(el, bs).map_err(e)?;
|
|
let seq = &seq[0];
|
|
let len = seq.len() / bs;
|
|
out.push(b'V');
|
|
out.extend_from_slice(&(len as u32).to_le_bytes());
|
|
for i in 0..len {
|
|
self.canon(base_type, &seq[i * bs..], out)?;
|
|
}
|
|
}
|
|
}
|
|
}
|
|
Ok(())
|
|
}
|
|
|
|
/// Returns (shape json, n_elements)
|
|
fn shape(ds: &Dataspace) -> (Value, u64) {
|
|
match ds.space_type {
|
|
DataspaceType::Null => (Value::String("null".into()), 0),
|
|
DataspaceType::Scalar => (json!([]), 1),
|
|
DataspaceType::Simple => {
|
|
let n = ds.dimensions.iter().fold(1u64, |a, d| a.saturating_mul(*d));
|
|
(json!(ds.dimensions), n)
|
|
}
|
|
}
|
|
}
|
|
|
|
fn hash_values(
|
|
&self,
|
|
dt: &Datatype,
|
|
raw: &[u8],
|
|
n: u64,
|
|
rec: &mut Map<String, Value>,
|
|
) -> Result<(), String> {
|
|
let size = dt.type_size() as usize;
|
|
let need = (n as usize).checked_mul(size).ok_or("n*size overflow")?;
|
|
if raw.len() != need {
|
|
return Err(format!(
|
|
"raw length {} != n_elements {n} * type_size {size}",
|
|
raw.len()
|
|
));
|
|
}
|
|
let mut canon = Vec::with_capacity(need);
|
|
for i in 0..n as usize {
|
|
self.canon(dt, &raw[i * size..(i + 1) * size], &mut canon)?;
|
|
}
|
|
let h = Sha256::digest(&canon);
|
|
rec.insert("hash".into(), Value::String(hex(&h)));
|
|
rec.insert(
|
|
"head".into(),
|
|
Value::String(hex(&canon[..canon.len().min(48)])),
|
|
);
|
|
Ok(())
|
|
}
|
|
|
|
/// VDS source files resolve next to the virtual file; like the library,
|
|
/// refuse absolute paths and `..`.
|
|
fn vds_resolver(
|
|
&self,
|
|
) -> impl Fn(&str) -> Result<Option<Vec<u8>>, clawhdf5_format::error::FormatError> + use<> {
|
|
let base = self.base_dir.clone();
|
|
move |name: &str| {
|
|
use clawhdf5_format::error::FormatError;
|
|
let p = std::path::Path::new(name);
|
|
if p.is_absolute()
|
|
|| p.components()
|
|
.any(|c| matches!(c, std::path::Component::ParentDir))
|
|
{
|
|
return Err(FormatError::ChunkedReadError(format!("refused {name}")));
|
|
}
|
|
match std::fs::read(base.join(p)) {
|
|
Ok(b) => Ok(Some(b)),
|
|
Err(err) if err.kind() == std::io::ErrorKind::NotFound => Ok(None),
|
|
Err(err) => Err(FormatError::ChunkedReadError(err.to_string())),
|
|
}
|
|
}
|
|
}
|
|
|
|
fn read_named_datatype(&self, h: &ObjectHeader) -> Result<(), String> {
|
|
let dtb = self
|
|
.payload(h, MessageType::Datatype)?
|
|
.ok_or("MissingMessage(Datatype)")?;
|
|
Datatype::parse_in_header(&dtb, h.version).map_err(e)?;
|
|
Ok(())
|
|
}
|
|
|
|
fn read_dataset(&self, h: &ObjectHeader, rec: &mut Map<String, Value>) -> Result<(), String> {
|
|
let dtb = self
|
|
.payload(h, MessageType::Datatype)?
|
|
.ok_or("MissingMessage(Datatype)")?;
|
|
let (dt, _) = Datatype::parse_in_header(&dtb, h.version).map_err(e)?;
|
|
rec.insert("dtype".into(), Value::String(dtype_str(&dt)));
|
|
let dsb = self
|
|
.payload(h, MessageType::Dataspace)?
|
|
.ok_or("MissingMessage(Dataspace)")?;
|
|
let mut ds = Dataspace::parse(&dsb, self.ls).map_err(e)?;
|
|
// A virtual dataset's extent can come from its sources (unlimited /
|
|
// printf mappings), as h5py reports it, rather than the stored one.
|
|
if let Some(lm) = h
|
|
.messages
|
|
.iter()
|
|
.find(|m| m.msg_type == MessageType::DataLayout)
|
|
&& let Ok(dl @ DataLayout::Virtual { .. }) =
|
|
DataLayout::parse(&lm.data, self.os, self.ls)
|
|
{
|
|
let resolver = self.vds_resolver();
|
|
ds.dimensions = clawhdf5_format::vds::virtual_dataset_extent(
|
|
self.data,
|
|
&dl,
|
|
&ds,
|
|
self.os,
|
|
self.ls,
|
|
Some(&resolver),
|
|
)
|
|
.map_err(e)?;
|
|
}
|
|
let lm = h
|
|
.messages
|
|
.iter()
|
|
.find(|m| m.msg_type == MessageType::DataLayout)
|
|
.ok_or("MissingMessage(DataLayout)")?;
|
|
let dl = DataLayout::parse(&lm.data, self.os, self.ls).map_err(e)?;
|
|
// What libhdf5 checks when it opens the dataset (as File::dataset).
|
|
data_read::check_dataset_storage(&dl, &ds, &dt, self.data.len() as u64).map_err(e)?;
|
|
let (shape, n) = Self::shape(&ds);
|
|
rec.insert("shape".into(), shape);
|
|
if n.saturating_mul(dt.type_size() as u64) > MAX_BYTES {
|
|
rec.insert("skipped".into(), Value::String("too large".into()));
|
|
return Ok(());
|
|
}
|
|
rec.insert(
|
|
"layout".into(),
|
|
Value::String(
|
|
match &dl {
|
|
DataLayout::Compact { .. } => "compact",
|
|
DataLayout::Contiguous { .. } => "contiguous",
|
|
DataLayout::Chunked { .. } => "chunked",
|
|
DataLayout::Virtual { .. } => "virtual",
|
|
}
|
|
.into(),
|
|
),
|
|
);
|
|
let pipeline = match self.payload(h, MessageType::FilterPipeline)? {
|
|
Some(p) => Some(FilterPipeline::parse(&p).map_err(e)?),
|
|
None => None,
|
|
};
|
|
if let Some(p) = &pipeline {
|
|
rec.insert(
|
|
"filters".into(),
|
|
json!(p.filters.iter().map(|f| f.filter_id).collect::<Vec<_>>()),
|
|
);
|
|
}
|
|
let raw = if matches!(dl, DataLayout::Virtual { .. }) {
|
|
let resolver = self.vds_resolver();
|
|
let fill = clawhdf5_format::fill_value::dataset_fill_value_in(
|
|
self.data,
|
|
&h.messages,
|
|
self.os,
|
|
self.ls,
|
|
)
|
|
.map_err(e)?;
|
|
clawhdf5_format::vds::read_virtual_dataset(
|
|
self.data,
|
|
&dl,
|
|
&ds,
|
|
&dt,
|
|
fill.as_deref(),
|
|
self.os,
|
|
self.ls,
|
|
Some(&resolver),
|
|
)
|
|
.map_err(e)?
|
|
.data
|
|
} else {
|
|
let cache = clawhdf5_format::chunk_cache::ChunkCache::new();
|
|
clawhdf5_format::fill_value::read_full_with_fill::<clawhdf5_format::error::FormatError>(
|
|
&h.messages,
|
|
self.data,
|
|
&dl,
|
|
&ds,
|
|
dt.type_size() as usize,
|
|
self.os,
|
|
self.ls,
|
|
|| {
|
|
data_read::read_raw_data_cached(
|
|
self.data,
|
|
&dl,
|
|
&ds,
|
|
&dt,
|
|
pipeline.as_ref(),
|
|
self.os,
|
|
self.ls,
|
|
&cache,
|
|
)
|
|
},
|
|
)
|
|
.map_err(e)?
|
|
};
|
|
self.hash_values(&dt, &raw, n, rec)
|
|
}
|
|
|
|
fn attrs(&self, h: &ObjectHeader) -> Result<Map<String, Value>, String> {
|
|
let msgs = extract_attributes_full(self.data, h, self.os, self.ls).map_err(e)?;
|
|
let mut out = Map::new();
|
|
for a in &msgs {
|
|
let r = guarded(|| {
|
|
let mut rec = Map::new();
|
|
rec.insert("dtype".into(), Value::String(dtype_str(&a.datatype)));
|
|
let (shape, n) = Self::shape(&a.dataspace);
|
|
rec.insert("shape".into(), shape);
|
|
self.hash_values(&a.datatype, &a.raw_data, n, &mut rec)?;
|
|
Ok(rec)
|
|
});
|
|
let v = match r {
|
|
Ok(rec) => Value::Object(rec),
|
|
Err(msg) => json!({ "error": msg }),
|
|
};
|
|
out.insert(a.name.clone(), v);
|
|
}
|
|
Ok(out)
|
|
}
|
|
|
|
fn entries(&self, h: &ObjectHeader) -> Result<Vec<GroupEntry>, String> {
|
|
let v1 = h
|
|
.messages
|
|
.iter()
|
|
.find(|m| m.msg_type == MessageType::SymbolTable);
|
|
if let Some(m) = v1 {
|
|
let stm = SymbolTableMessage::parse(&m.data, self.os).map_err(e)?;
|
|
group_v1::resolve_v1_group_entries(self.data, &stm, self.os, self.ls).map_err(e)
|
|
} else if h
|
|
.messages
|
|
.iter()
|
|
.any(|m| m.msg_type == MessageType::LinkInfo || m.msg_type == MessageType::Link)
|
|
{
|
|
group_v2::resolve_v2_group_entries(self.data, h, self.os, self.ls).map_err(e)
|
|
} else {
|
|
Ok(Vec::new())
|
|
}
|
|
}
|
|
}
|
|
|
|
/// Element bytes as an unsigned integer (at most 16 bytes), honouring byte order.
|
|
fn element_bits(b: &[u8], byte_order: &DatatypeByteOrder) -> Result<u128, String> {
|
|
if b.len() > 16 {
|
|
return Err(format!("canon: {}-byte numeric element", b.len()));
|
|
}
|
|
let mut v = 0u128;
|
|
match byte_order {
|
|
DatatypeByteOrder::LittleEndian => {
|
|
for (i, x) in b.iter().enumerate() {
|
|
v |= u128::from(*x) << (8 * i);
|
|
}
|
|
}
|
|
DatatypeByteOrder::BigEndian => {
|
|
for x in b {
|
|
v = (v << 8) | u128::from(*x);
|
|
}
|
|
}
|
|
DatatypeByteOrder::Vax => return Err("canon: VAX byte order".into()),
|
|
}
|
|
Ok(v)
|
|
}
|
|
|
|
fn field(v: u128, pos: u32, len: u32) -> u128 {
|
|
if len == 0 || pos >= 128 {
|
|
return 0;
|
|
}
|
|
let v = v >> pos;
|
|
if len >= 128 {
|
|
v
|
|
} else {
|
|
v & ((1u128 << len) - 1)
|
|
}
|
|
}
|
|
|
|
/// True when a float's bit fields are exactly IEEE 754 binary16/32/64 for its
|
|
/// size. h5py hands back such a type's bytes untouched; any other layout (an
|
|
/// N-Bit `H5Tset_precision` float, say) is *converted* by libhdf5 into the
|
|
/// numpy float of the same size, so comparing raw bytes would be meaningless.
|
|
fn ieee_layout(dt: &Datatype) -> bool {
|
|
let Datatype::FloatingPoint {
|
|
size,
|
|
bit_offset,
|
|
bit_precision,
|
|
exponent_location,
|
|
exponent_size,
|
|
mantissa_location,
|
|
mantissa_size,
|
|
exponent_bias,
|
|
..
|
|
} = dt
|
|
else {
|
|
return true;
|
|
};
|
|
let std = match size {
|
|
2 => (16, 10, 5, 10, 15),
|
|
4 => (32, 23, 8, 23, 127),
|
|
8 => (64, 52, 11, 52, 1023),
|
|
_ => return true, // no same-size numpy float to convert to: compare raw
|
|
};
|
|
*bit_offset == 0
|
|
&& (
|
|
*bit_precision,
|
|
*exponent_location,
|
|
*exponent_size,
|
|
*mantissa_size,
|
|
*exponent_bias,
|
|
) == (std.0, std.1, std.2, std.3, std.4)
|
|
&& *mantissa_location == 0
|
|
}
|
|
|
|
/// Canonicalise a non-IEEE-layout float the way libhdf5's float->float
|
|
/// conversion presents it to h5py: as the IEEE float of the same size.
|
|
/// Assumes the implied-leading-one normalisation and the sign bit at the top
|
|
/// of the precision (what `H5Tset_precision` produces; the parser does not
|
|
/// keep either field).
|
|
fn canon_custom_float(dt: &Datatype, b: &[u8], out: &mut Vec<u8>) -> Result<(), String> {
|
|
let Datatype::FloatingPoint {
|
|
size,
|
|
byte_order,
|
|
bit_offset,
|
|
bit_precision,
|
|
exponent_location,
|
|
exponent_size,
|
|
mantissa_location,
|
|
mantissa_size,
|
|
exponent_bias,
|
|
} = dt
|
|
else {
|
|
unreachable!()
|
|
};
|
|
let (esize, msize) = (u32::from(*exponent_size), u32::from(*mantissa_size));
|
|
if esize == 0 || esize > 30 || msize > 64 {
|
|
return Err(format!("canon: unsupported float layout e{esize} m{msize}"));
|
|
}
|
|
let v = element_bits(b, byte_order)?;
|
|
let sign_pos = (u32::from(*bit_offset) + u32::from(*bit_precision)).saturating_sub(1);
|
|
let neg = field(v, sign_pos, 1) == 1;
|
|
let e = field(v, u32::from(*exponent_location), esize) as i64;
|
|
let m = field(v, u32::from(*mantissa_location), msize);
|
|
let emax = (1i64 << esize) - 1;
|
|
let bias = i64::from(*exponent_bias);
|
|
let mag = if e == emax {
|
|
if m == 0 { f64::INFINITY } else { f64::NAN }
|
|
} else if e == 0 {
|
|
(m as f64) * 2f64.powi((1 - bias - msize as i64) as i32)
|
|
} else {
|
|
((1u128 << msize) as f64 + m as f64) * 2f64.powi((e - bias - msize as i64) as i32)
|
|
};
|
|
let x = if neg { -mag } else { mag };
|
|
match size {
|
|
2 => out
|
|
.extend_from_slice(&clawhdf5_format::float16::f32_to_f16_bits(x as f32).to_le_bytes()),
|
|
4 => out.extend_from_slice(&(x as f32).to_le_bytes()),
|
|
8 => out.extend_from_slice(&x.to_le_bytes()),
|
|
_ => unreachable!("ieee_layout keeps other sizes raw"),
|
|
}
|
|
Ok(())
|
|
}
|
|
|
|
/// Integers stored with a bit offset or reduced precision (N-Bit): libhdf5
|
|
/// converts them to the full-width integer of the same size, shifting the
|
|
/// value down and sign-extending from the top precision bit.
|
|
fn canon_partial_int(dt: &Datatype, b: &[u8], out: &mut Vec<u8>) -> Result<(), String> {
|
|
let Datatype::FixedPoint {
|
|
size,
|
|
byte_order,
|
|
signed,
|
|
bit_offset,
|
|
bit_precision,
|
|
} = dt
|
|
else {
|
|
unreachable!()
|
|
};
|
|
let prec = u32::from(*bit_precision);
|
|
let v = element_bits(b, byte_order)?;
|
|
let mut x = field(v, u32::from(*bit_offset), prec);
|
|
if *signed && prec > 0 && prec < 128 && field(x, prec - 1, 1) == 1 {
|
|
x |= !0u128 << prec;
|
|
}
|
|
out.extend_from_slice(&x.to_le_bytes()[..*size as usize]);
|
|
Ok(())
|
|
}
|
|
|
|
fn partial_int(dt: &Datatype) -> bool {
|
|
matches!(dt, Datatype::FixedPoint { size, bit_offset, bit_precision, .. }
|
|
if *bit_offset != 0 || u32::from(*bit_precision) != size * 8)
|
|
}
|
|
|
|
fn canon_str(b: &[u8], out: &mut Vec<u8>) {
|
|
let cut = b.iter().position(|&c| c == 0).unwrap_or(b.len());
|
|
let mut s = &b[..cut];
|
|
while let [rest @ .., b' '] = s {
|
|
s = rest;
|
|
}
|
|
out.push(b'S');
|
|
out.extend_from_slice(&(s.len() as u32).to_le_bytes());
|
|
out.extend_from_slice(s);
|
|
}
|
|
|
|
fn hex(b: &[u8]) -> String {
|
|
b.iter().map(|x| format!("{x:02x}")).collect()
|
|
}
|
|
|
|
fn dtype_str(dt: &Datatype) -> String {
|
|
match dt {
|
|
Datatype::FixedPoint {
|
|
size,
|
|
signed,
|
|
byte_order,
|
|
..
|
|
} => {
|
|
format!(
|
|
"{}{}{}",
|
|
bo(byte_order),
|
|
if *signed { "i" } else { "u" },
|
|
size
|
|
)
|
|
}
|
|
Datatype::FloatingPoint {
|
|
size, byte_order, ..
|
|
} => format!("{}f{}", bo(byte_order), size),
|
|
Datatype::BitField {
|
|
size, byte_order, ..
|
|
} => format!("{}b{}", bo(byte_order), size),
|
|
Datatype::Time { size, .. } => format!("time{size}"),
|
|
Datatype::String { size, .. } => format!("S{size}"),
|
|
Datatype::Opaque { size, .. } => format!("V{size}"),
|
|
Datatype::Compound { size, members } => format!(
|
|
"{{{}}}{size}",
|
|
members
|
|
.iter()
|
|
.map(|m| format!("{}:{}", m.name, dtype_str(&m.datatype)))
|
|
.collect::<Vec<_>>()
|
|
.join(",")
|
|
),
|
|
Datatype::Reference { ref_type, .. } => format!("ref({ref_type:?})"),
|
|
Datatype::Enumeration { base_type, .. } => format!("enum({})", dtype_str(base_type)),
|
|
Datatype::VariableLength {
|
|
is_string: true, ..
|
|
} => "vlstr".into(),
|
|
Datatype::VariableLength { base_type, .. } => format!("vlen({})", dtype_str(base_type)),
|
|
Datatype::Array {
|
|
base_type,
|
|
dimensions,
|
|
} => format!("({}){dimensions:?}", dtype_str(base_type)),
|
|
}
|
|
}
|
|
|
|
fn bo(b: &DatatypeByteOrder) -> &'static str {
|
|
match b {
|
|
DatatypeByteOrder::LittleEndian => "<",
|
|
DatatypeByteOrder::BigEndian => ">",
|
|
DatatypeByteOrder::Vax => "vax",
|
|
}
|
|
}
|
|
|
|
fn is_group(h: &ObjectHeader) -> bool {
|
|
h.messages.iter().any(|m| {
|
|
matches!(
|
|
m.msg_type,
|
|
MessageType::LinkInfo | MessageType::Link | MessageType::SymbolTable
|
|
)
|
|
})
|
|
}
|
|
|
|
/// The probe's kind for an object header: libhdf5's object class
|
|
/// ([`ObjectHeader::object_class`]: group, then dataset — a datatype *and* a
|
|
/// dataspace — then named datatype), which is what h5py opens the object as.
|
|
/// The root group, and a header with only link messages, count as groups.
|
|
fn kind_of(h: &ObjectHeader, is_root: bool) -> &'static str {
|
|
match h.object_class() {
|
|
Some(ObjectClass::Group) => "group",
|
|
Some(ObjectClass::Dataset) => "dataset",
|
|
_ if is_root || is_group(h) => "group",
|
|
Some(ObjectClass::NamedDatatype) => "datatype",
|
|
None => "unknown",
|
|
}
|
|
}
|
|
|
|
fn main() {
|
|
install_hook();
|
|
let path = std::env::args().nth(1).expect("usage: probe <file>");
|
|
let mut top = Map::new();
|
|
top.insert("file".into(), Value::String(path.clone()));
|
|
let data = match std::fs::read(&path) {
|
|
Ok(d) => d,
|
|
Err(err) => {
|
|
top.insert("open_error".into(), Value::String(format!("Io({err})")));
|
|
println!("{}", Value::Object(top));
|
|
return;
|
|
}
|
|
};
|
|
// Every address is relative to the superblock: look at the file from
|
|
// there on (past any user block), as libhdf5 does.
|
|
let hdf5: &[u8] = match signature::find_signature(&data) {
|
|
Ok(off) => &data[off..],
|
|
Err(_) => &data,
|
|
};
|
|
let sb = guarded(|| Superblock::parse(hdf5, 0).map_err(e));
|
|
let sb = match sb {
|
|
Ok(sb) => sb,
|
|
Err(msg) => {
|
|
top.insert("open_error".into(), Value::String(msg));
|
|
println!("{}", Value::Object(top));
|
|
return;
|
|
}
|
|
};
|
|
// libhdf5 refuses a truncated file and reads nothing past the recorded
|
|
// end of file.
|
|
let base = (data.len() - hdf5.len()) as u64;
|
|
let hdf5 = match sb.data_end(base, data.len() as u64) {
|
|
Ok(end) => &hdf5[..end as usize],
|
|
Err(err) => {
|
|
top.insert("open_error".into(), Value::String(e(err)));
|
|
println!("{}", Value::Object(top));
|
|
return;
|
|
}
|
|
};
|
|
// libhdf5 decodes the superblock extension at open (an error refuses
|
|
// the file), and loads a metadata cache image over the file's own
|
|
// metadata. It loads the image only when it first reads metadata — the
|
|
// root group — so a file whose image it cannot load still opens and
|
|
// that read fails. The library decides all three cases with the same
|
|
// `cache_image_state`: `File` and `MmapFile` open such a file and fail
|
|
// every object lookup with the image's error, which is what the probe
|
|
// records here (on the root group, where libhdf5 reports it).
|
|
use clawhdf5_format::superblock_ext::{self, CacheImageState};
|
|
let state = match guarded(|| superblock_ext::cache_image_state(hdf5, &sb).map_err(e)) {
|
|
Ok(x) => x,
|
|
Err(msg) => {
|
|
top.insert("open_error".into(), Value::String(msg));
|
|
println!("{}", Value::Object(top));
|
|
return;
|
|
}
|
|
};
|
|
let mut image_error = None;
|
|
let view = match state {
|
|
CacheImageState::Absent => None,
|
|
CacheImageState::Unloadable(err) => {
|
|
image_error = Some(e(err));
|
|
None
|
|
}
|
|
CacheImageState::Loaded(image) => {
|
|
let mut v = hdf5.to_vec();
|
|
match image.block(hdf5).and_then(|b| image.apply(b, &mut v)) {
|
|
Ok(()) => Some(v),
|
|
Err(err) => {
|
|
image_error = Some(e(err));
|
|
None
|
|
}
|
|
}
|
|
}
|
|
};
|
|
let hdf5: &[u8] = view.as_deref().unwrap_or(hdf5);
|
|
top.insert("superblock_version".into(), json!(sb.version));
|
|
let ctx = Ctx {
|
|
data: hdf5,
|
|
os: sb.offset_size,
|
|
ls: sb.length_size,
|
|
base_dir: std::path::Path::new(&path)
|
|
.parent()
|
|
.map(|p| p.to_path_buf())
|
|
.unwrap_or_default(),
|
|
vl: RefCell::new(VlResolver::new(hdf5, sb.offset_size, sb.length_size)),
|
|
};
|
|
let mut objects: Vec<Value> = Vec::new();
|
|
let mut visited = HashSet::new();
|
|
let mut soft_v1 = 0u64;
|
|
// explicit DFS stack: (address, path)
|
|
let mut stack: Vec<(u64, String)> = vec![(sb.root_group_address, "/".to_string())];
|
|
while let Some((addr, p)) = stack.pop() {
|
|
if objects.len() >= MAX_OBJECTS {
|
|
top.insert("truncated".into(), json!(true));
|
|
break;
|
|
}
|
|
if !visited.insert(addr) {
|
|
continue;
|
|
}
|
|
let mut rec = Map::new();
|
|
rec.insert("path".into(), Value::String(p.clone()));
|
|
let r = guarded(|| {
|
|
if let Some(msg) = &image_error {
|
|
return Err(msg.clone());
|
|
}
|
|
let h = ctx.header(addr)?;
|
|
Ok(h)
|
|
});
|
|
let h = match r {
|
|
Ok(h) => h,
|
|
Err(msg) => {
|
|
rec.insert("kind".into(), Value::String("unknown".into()));
|
|
rec.insert("error".into(), Value::String(msg));
|
|
objects.push(Value::Object(rec));
|
|
continue;
|
|
}
|
|
};
|
|
let kind = kind_of(&h, addr == sb.root_group_address);
|
|
rec.insert("kind".into(), Value::String(kind.into()));
|
|
if kind == "dataset"
|
|
&& let Err(msg) = guarded(|| ctx.read_dataset(&h, &mut rec))
|
|
{
|
|
rec.insert("error".into(), Value::String(msg));
|
|
}
|
|
// Opening a committed datatype decodes it (h5py's `f[name]` fails on
|
|
// one libhdf5 cannot decode), so decode it here too.
|
|
if kind == "datatype"
|
|
&& let Err(msg) = guarded(|| ctx.read_named_datatype(&h))
|
|
{
|
|
rec.insert("error".into(), Value::String(msg));
|
|
}
|
|
if kind != "datatype" {
|
|
match guarded(|| ctx.attrs(&h)) {
|
|
Ok(m) => {
|
|
rec.insert("attrs".into(), Value::Object(m));
|
|
}
|
|
Err(msg) => {
|
|
rec.insert("attrs_error".into(), Value::String(msg));
|
|
}
|
|
}
|
|
}
|
|
if kind == "group" {
|
|
match guarded(|| ctx.entries(&h)) {
|
|
Ok(mut ents) => {
|
|
ents.retain(|en| {
|
|
if en.cache_type == 2 {
|
|
soft_v1 += 1;
|
|
false
|
|
} else {
|
|
true
|
|
}
|
|
});
|
|
ents.sort_by(|a, b| a.name.cmp(&b.name));
|
|
let base = if p == "/" { String::new() } else { p.clone() };
|
|
for en in ents.into_iter().rev() {
|
|
stack.push((en.object_header_address, format!("{base}/{}", en.name)));
|
|
}
|
|
}
|
|
Err(msg) => {
|
|
rec.insert("list_error".into(), Value::String(msg));
|
|
}
|
|
}
|
|
}
|
|
objects.push(Value::Object(rec));
|
|
}
|
|
if soft_v1 > 0 {
|
|
top.insert("v1_soft_link_entries".into(), json!(soft_v1));
|
|
}
|
|
top.insert("objects".into(), Value::Array(objects));
|
|
println!("{}", Value::Object(top));
|
|
}
|
|
|
|
#[cfg(test)]
|
|
mod tests {
|
|
use super::*;
|
|
|
|
/// The N-Bit float of libhdf5's `test/testfiles/le_data.h5`
|
|
/// (`Nbit_float_data_le`): offset 7, precision 20, sign bit 26, exponent
|
|
/// 20+6 (bias 31), mantissa 7+13.
|
|
fn nbit_f32(byte_order: DatatypeByteOrder) -> Datatype {
|
|
Datatype::FloatingPoint {
|
|
size: 4,
|
|
byte_order,
|
|
bit_offset: 7,
|
|
bit_precision: 20,
|
|
exponent_location: 20,
|
|
exponent_size: 6,
|
|
mantissa_location: 7,
|
|
mantissa_size: 13,
|
|
exponent_bias: 31,
|
|
}
|
|
}
|
|
|
|
fn canon_one(dt: &Datatype, bytes: &[u8]) -> Vec<u8> {
|
|
let mut out = Vec::new();
|
|
canon_custom_float(dt, bytes, &mut out).unwrap();
|
|
out
|
|
}
|
|
|
|
#[test]
|
|
fn nbit_float_canonicalises_to_the_value_libhdf5_returns() {
|
|
let le = nbit_f32(DatatypeByteOrder::LittleEndian);
|
|
let be = nbit_f32(DatatypeByteOrder::BigEndian);
|
|
assert!(!ieee_layout(&le));
|
|
// 1.0: exponent = bias, mantissa 0
|
|
let one: u32 = 31 << 20;
|
|
assert_eq!(canon_one(&le, &one.to_le_bytes()), 1.0f32.to_le_bytes());
|
|
assert_eq!(canon_one(&be, &one.to_be_bytes()), 1.0f32.to_le_bytes());
|
|
// -2.1999512 (h5py's reading of the file's -2.2): sign, e = 32, m = 819
|
|
let v: u32 = (1 << 26) | (32 << 20) | (819 << 7);
|
|
assert_eq!(
|
|
canon_one(&le, &v.to_le_bytes()),
|
|
(-2.199_951_2f32).to_le_bytes()
|
|
);
|
|
assert_eq!(canon_one(&le, &[0; 4]), 0.0f32.to_le_bytes());
|
|
}
|
|
|
|
#[test]
|
|
fn ieee_floats_keep_their_raw_bytes() {
|
|
let f32le = Datatype::FloatingPoint {
|
|
size: 4,
|
|
byte_order: DatatypeByteOrder::LittleEndian,
|
|
bit_offset: 0,
|
|
bit_precision: 32,
|
|
exponent_location: 23,
|
|
exponent_size: 8,
|
|
mantissa_location: 0,
|
|
mantissa_size: 23,
|
|
exponent_bias: 127,
|
|
};
|
|
assert!(ieee_layout(&f32le));
|
|
}
|
|
|
|
#[test]
|
|
fn kind_follows_libhdf5_object_class() {
|
|
use clawhdf5_format::object_header::HeaderMessage;
|
|
let header = |types: &[MessageType]| ObjectHeader {
|
|
version: 2,
|
|
messages: types
|
|
.iter()
|
|
.map(|&msg_type| HeaderMessage {
|
|
msg_type,
|
|
size: 0,
|
|
flags: 0,
|
|
creation_order: None,
|
|
data: Vec::new(),
|
|
})
|
|
.collect(),
|
|
reference_count: None,
|
|
flags: 0,
|
|
access_time: None,
|
|
modification_time: None,
|
|
change_time: None,
|
|
birth_time: None,
|
|
};
|
|
use MessageType::*;
|
|
// cve-2024-33874 `/Dset1`: a datatype and a layout but no dataspace
|
|
// is a named datatype to libhdf5 (h5py opens it as one).
|
|
assert_eq!(kind_of(&header(&[Datatype, DataLayout]), false), "datatype");
|
|
assert_eq!(
|
|
kind_of(&header(&[Datatype, Dataspace, DataLayout]), false),
|
|
"dataset"
|
|
);
|
|
assert_eq!(kind_of(&header(&[SymbolTable]), false), "group");
|
|
assert_eq!(kind_of(&header(&[Link]), false), "group");
|
|
assert_eq!(kind_of(&header(&[]), true), "group");
|
|
assert_eq!(kind_of(&header(&[]), false), "unknown");
|
|
}
|
|
|
|
#[test]
|
|
fn partial_precision_int_is_shifted_and_sign_extended() {
|
|
let dt = Datatype::FixedPoint {
|
|
size: 4,
|
|
byte_order: DatatypeByteOrder::BigEndian,
|
|
signed: true,
|
|
bit_offset: 4,
|
|
bit_precision: 17,
|
|
};
|
|
assert!(partial_int(&dt));
|
|
let stored = (((-5i32) as u32) & 0x1_FFFF) << 4;
|
|
let mut out = Vec::new();
|
|
canon_partial_int(&dt, &stored.to_be_bytes(), &mut out).unwrap();
|
|
assert_eq!(out, (-5i32).to_le_bytes());
|
|
}
|
|
}
|