Datatype::parse now makes the checks of libhdf5's H5O__dtype_decode_helper
and fails with InvalidDatatype (libhdf5's own error text) instead of
decoding a corrupt type:
- size 0 ("invalid datatype size"), for every class;
- integer bit offset/precision outside the type, or precision 0;
- float sign/exponent/mantissa outside the type, empty, or overlapping;
normalization 3; bit 6 without bit 0 from version 3;
- compound with no members, a member outside the compound, a duplicate
name, or a member overlapping an earlier one;
- enum whose size differs from its base type's, or an empty member name;
- array of more than 32 dimensions or with a zero-sized one (v1 compound
array members now say so rather than InvalidDatatypeVersion);
- opaque tag length that is not a multiple of 8.
Bit 6 of a version-1/2 float's class bits used to be read as VAX order,
byte-swapping values; libhdf5 ignores it before version 3, and so does
this now.
Only checks HDF5 2.0 (h5py 3.16) makes are added: newer libhdf5 also
checks bit fields, the variable-length kind and array sizes, but h5py
opens files that fail those, so they are left out. Each check was
confirmed against h5py by corrupting a file it wrote.
The conformance probe now decodes committed datatypes, as h5py's f[name]
does. Conformance (cached corpus, tank): 570 ok, unchanged. Objects
libhdf5 refuses that clawhdf5 used to read: cve-2016-4332-mtime (/cmpnd),
cve-2017-17508, cve-2024-32616 (/type1), cve-2024-32618, cve-2026-34734,
bad_compound.h5 (/cmpnd, /dataset); eight more that already failed now
fail with libhdf5's reason (e.g. cve-2024-29163 "mantissa range out of
bounds").
Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
914 lines
32 KiB
Rust
914 lines
32 KiB
Rust
//! Conformance probe: walks an HDF5 file with clawhdf5-format (the same calls
|
|
//! the `clawhdf5` facade makes) and prints a canonical JSON description:
|
|
//! every hard-linked object (sorted-name DFS, deduplicated by header address),
|
|
//! and for each dataset / attribute its shape plus the SHA-256 of its values
|
|
//! in a canonical encoding shared with `ref.py`.
|
|
//!
|
|
//! Canonical value encoding (per element, concatenated, row-major):
|
|
//! int / float / bitfield / enum / time : element bytes, little-endian
|
|
//! non-IEEE-layout float (e.g. N-Bit) : the IEEE float of the same size it converts to
|
|
//! int with bit offset / short precision: the full-width integer it converts to
|
|
//! opaque : raw bytes
|
|
//! compound : members in declaration order (padding dropped)
|
|
//! array : base elements row-major
|
|
//! string (fixed or VL) : b'S' + u32le len + bytes (cut at first NUL, trailing spaces stripped)
|
|
//! VL sequence : b'V' + u32le count + base elements
|
|
//! reference : b'R' (payload not compared)
|
|
//!
|
|
//! Every object is processed inside catch_unwind; a caught panic is recorded
|
|
//! with its message, location and the clawhdf5 frames of its backtrace.
|
|
|
|
use std::cell::RefCell;
|
|
use std::collections::{HashMap, HashSet};
|
|
use std::panic::{self, AssertUnwindSafe};
|
|
use std::rc::Rc;
|
|
|
|
use clawhdf5_format::attribute::extract_attributes_full;
|
|
use clawhdf5_format::data_layout::DataLayout;
|
|
use clawhdf5_format::data_read;
|
|
use clawhdf5_format::dataspace::{Dataspace, DataspaceType};
|
|
use clawhdf5_format::datatype::{Datatype, DatatypeByteOrder};
|
|
use clawhdf5_format::filter_pipeline::FilterPipeline;
|
|
use clawhdf5_format::global_heap::GlobalHeapCollection;
|
|
use clawhdf5_format::group_v1::{self, GroupEntry};
|
|
use clawhdf5_format::group_v2;
|
|
use clawhdf5_format::message_type::MessageType;
|
|
use clawhdf5_format::object_header::ObjectHeader;
|
|
use clawhdf5_format::signature;
|
|
use clawhdf5_format::superblock::Superblock;
|
|
use clawhdf5_format::symbol_table::SymbolTableMessage;
|
|
use serde_json::{Map, Value, json};
|
|
use sha2::{Digest, Sha256};
|
|
|
|
const MAX_BYTES: u64 = 200 * 1024 * 1024;
|
|
const MAX_OBJECTS: usize = 200_000;
|
|
|
|
thread_local! {
|
|
static LAST_PANIC: RefCell<Option<String>> = const { RefCell::new(None) };
|
|
}
|
|
|
|
fn install_hook() {
|
|
panic::set_hook(Box::new(|info| {
|
|
let msg = if let Some(s) = info.payload().downcast_ref::<&str>() {
|
|
s.to_string()
|
|
} else if let Some(s) = info.payload().downcast_ref::<String>() {
|
|
s.clone()
|
|
} else {
|
|
"<non-string panic>".into()
|
|
};
|
|
let loc = info
|
|
.location()
|
|
.map(|l| format!("{}:{}", l.file(), l.line()))
|
|
.unwrap_or_default();
|
|
let bt = std::backtrace::Backtrace::force_capture().to_string();
|
|
// keep only frames from clawhdf5 code
|
|
let mut frames = Vec::new();
|
|
let lines: Vec<&str> = bt.lines().collect();
|
|
for (i, l) in lines.iter().enumerate() {
|
|
let t = l.trim();
|
|
if t.contains("clawhdf5_format::") || t.contains("conformance_probe::") {
|
|
let at = lines
|
|
.get(i + 1)
|
|
.map(|n| n.trim())
|
|
.filter(|n| n.starts_with("at "))
|
|
.map(|n| {
|
|
let n = n.trim_start_matches("at ");
|
|
match n.find("/crates/") {
|
|
Some(p) => n[p + 1..].to_string(),
|
|
None => n.to_string(),
|
|
}
|
|
})
|
|
.unwrap_or_default();
|
|
let name = t.split_once(": ").map(|x| x.1).unwrap_or(t);
|
|
frames.push(format!("{name} ({at})"));
|
|
if frames.len() >= 12 {
|
|
break;
|
|
}
|
|
}
|
|
}
|
|
let full = format!("PANIC: {msg} @ {loc}\n {}", frames.join("\n "));
|
|
eprintln!("{full}");
|
|
LAST_PANIC.with(|p| *p.borrow_mut() = Some(full));
|
|
}));
|
|
}
|
|
|
|
/// Run `f`, turning a panic into Err("PANIC: ...").
|
|
fn guarded<T>(f: impl FnOnce() -> Result<T, String>) -> Result<T, String> {
|
|
match panic::catch_unwind(AssertUnwindSafe(f)) {
|
|
Ok(r) => r,
|
|
Err(_) => Err(LAST_PANIC
|
|
.with(|p| p.borrow_mut().take())
|
|
.unwrap_or_else(|| "PANIC: <unknown>".into())),
|
|
}
|
|
}
|
|
|
|
fn e<E: std::fmt::Debug>(x: E) -> String {
|
|
format!("{x:?}")
|
|
}
|
|
|
|
struct Ctx<'a> {
|
|
data: &'a [u8],
|
|
os: u8,
|
|
ls: u8,
|
|
base_dir: std::path::PathBuf,
|
|
heaps: RefCell<HashMap<u64, Result<Rc<GlobalHeapCollection>, String>>>,
|
|
}
|
|
|
|
impl<'a> Ctx<'a> {
|
|
fn header(&self, addr: u64) -> Result<ObjectHeader, String> {
|
|
ObjectHeader::parse(self.data, addr as usize, self.os, self.ls).map_err(e)
|
|
}
|
|
|
|
fn payload(&self, h: &ObjectHeader, t: MessageType) -> Result<Option<Vec<u8>>, String> {
|
|
match h.messages.iter().find(|m| m.msg_type == t) {
|
|
None => Ok(None),
|
|
Some(m) => {
|
|
clawhdf5_format::shared_message::message_data(self.data, m, self.os, self.ls)
|
|
.map(|c| Some(c.into_owned()))
|
|
.map_err(e)
|
|
}
|
|
}
|
|
}
|
|
|
|
fn heap_obj(&self, addr: u64, idx: u32) -> Result<Vec<u8>, String> {
|
|
let coll = {
|
|
let mut cache = self.heaps.borrow_mut();
|
|
cache
|
|
.entry(addr)
|
|
.or_insert_with(|| {
|
|
GlobalHeapCollection::parse(self.data, addr as usize, self.ls)
|
|
.map(Rc::new)
|
|
.map_err(e)
|
|
})
|
|
.clone()?
|
|
};
|
|
coll.get_object(idx as u16)
|
|
.map(|o| o.data.clone())
|
|
.ok_or_else(|| {
|
|
format!("GlobalHeapObjectNotFound {{ collection_address: {addr}, index: {idx} }}")
|
|
})
|
|
}
|
|
|
|
fn read_offset(&self, b: &[u8]) -> u64 {
|
|
let mut v = 0u64;
|
|
for (i, x) in b.iter().take(self.os as usize).enumerate() {
|
|
v |= (*x as u64) << (8 * i);
|
|
}
|
|
v
|
|
}
|
|
|
|
fn canon(&self, dt: &Datatype, b: &[u8], out: &mut Vec<u8>) -> Result<(), String> {
|
|
let size = dt.type_size() as usize;
|
|
if b.len() < size {
|
|
return Err(format!(
|
|
"canon: element slice {} < type size {size}",
|
|
b.len()
|
|
));
|
|
}
|
|
match dt {
|
|
Datatype::FloatingPoint { .. } if !ieee_layout(dt) => {
|
|
canon_custom_float(dt, &b[..size], out)?
|
|
}
|
|
Datatype::FixedPoint { .. } if partial_int(dt) => {
|
|
canon_partial_int(dt, &b[..size], out)?
|
|
}
|
|
Datatype::FixedPoint { byte_order, .. }
|
|
| Datatype::BitField { byte_order, .. }
|
|
| Datatype::FloatingPoint { byte_order, .. } => match byte_order {
|
|
DatatypeByteOrder::LittleEndian => out.extend_from_slice(&b[..size]),
|
|
DatatypeByteOrder::BigEndian => out.extend(b[..size].iter().rev()),
|
|
DatatypeByteOrder::Vax => return Err("canon: VAX byte order".into()),
|
|
},
|
|
Datatype::Time { .. } | Datatype::Opaque { .. } => out.extend_from_slice(&b[..size]),
|
|
Datatype::String { .. } => canon_str(&b[..size], out),
|
|
Datatype::Compound { members, .. } => {
|
|
for m in members {
|
|
let off = m.byte_offset as usize;
|
|
let ms = m.datatype.type_size() as usize;
|
|
if off.checked_add(ms).is_none_or(|end| end > size) {
|
|
return Err(format!("canon: member {} out of bounds", m.name));
|
|
}
|
|
self.canon(&m.datatype, &b[off..off + ms], out)?;
|
|
}
|
|
}
|
|
Datatype::Reference { .. } => out.push(b'R'),
|
|
Datatype::Enumeration { base_type, .. } => self.canon(base_type, b, out)?,
|
|
Datatype::Array {
|
|
base_type,
|
|
dimensions,
|
|
} => {
|
|
let n: usize = dimensions.iter().map(|d| *d as usize).product();
|
|
let bs = base_type.type_size() as usize;
|
|
for i in 0..n {
|
|
self.canon(base_type, &b[i * bs..], out)?;
|
|
}
|
|
}
|
|
Datatype::VariableLength {
|
|
is_string,
|
|
base_type,
|
|
..
|
|
} => {
|
|
let len = u32::from_le_bytes([b[0], b[1], b[2], b[3]]) as usize;
|
|
let addr = self.read_offset(&b[4..]);
|
|
let idx_off = 4 + self.os as usize;
|
|
let idx = u32::from_le_bytes([
|
|
b[idx_off],
|
|
b[idx_off + 1],
|
|
b[idx_off + 2],
|
|
b[idx_off + 3],
|
|
]);
|
|
let obj = if len == 0 || addr == 0 || addr == u64::MAX >> (64 - 8 * self.os as u32)
|
|
{
|
|
Vec::new()
|
|
} else {
|
|
self.heap_obj(addr, idx)?
|
|
};
|
|
if *is_string {
|
|
let l = len.min(obj.len());
|
|
canon_str(&obj[..l], out);
|
|
} else {
|
|
let bs = base_type.type_size() as usize;
|
|
if bs == 0 {
|
|
return Err("canon: VL base size 0".into());
|
|
}
|
|
let need = len.checked_mul(bs).ok_or("canon: VL overflow")?;
|
|
if len > 0 && obj.len() < need {
|
|
return Err(format!("canon: VL object {} < {need}", obj.len()));
|
|
}
|
|
out.push(b'V');
|
|
out.extend_from_slice(&(len as u32).to_le_bytes());
|
|
for i in 0..len {
|
|
self.canon(base_type, &obj[i * bs..], out)?;
|
|
}
|
|
}
|
|
}
|
|
}
|
|
Ok(())
|
|
}
|
|
|
|
/// Returns (shape json, n_elements)
|
|
fn shape(ds: &Dataspace) -> (Value, u64) {
|
|
match ds.space_type {
|
|
DataspaceType::Null => (Value::String("null".into()), 0),
|
|
DataspaceType::Scalar => (json!([]), 1),
|
|
DataspaceType::Simple => {
|
|
let n = ds.dimensions.iter().fold(1u64, |a, d| a.saturating_mul(*d));
|
|
(json!(ds.dimensions), n)
|
|
}
|
|
}
|
|
}
|
|
|
|
fn hash_values(
|
|
&self,
|
|
dt: &Datatype,
|
|
raw: &[u8],
|
|
n: u64,
|
|
rec: &mut Map<String, Value>,
|
|
) -> Result<(), String> {
|
|
let size = dt.type_size() as usize;
|
|
let need = (n as usize).checked_mul(size).ok_or("n*size overflow")?;
|
|
if raw.len() != need {
|
|
return Err(format!(
|
|
"raw length {} != n_elements {n} * type_size {size}",
|
|
raw.len()
|
|
));
|
|
}
|
|
let mut canon = Vec::with_capacity(need);
|
|
for i in 0..n as usize {
|
|
self.canon(dt, &raw[i * size..(i + 1) * size], &mut canon)?;
|
|
}
|
|
let h = Sha256::digest(&canon);
|
|
rec.insert("hash".into(), Value::String(hex(&h)));
|
|
rec.insert(
|
|
"head".into(),
|
|
Value::String(hex(&canon[..canon.len().min(48)])),
|
|
);
|
|
Ok(())
|
|
}
|
|
|
|
/// VDS source files resolve next to the virtual file; like the library,
|
|
/// refuse absolute paths and `..`.
|
|
fn vds_resolver(
|
|
&self,
|
|
) -> impl Fn(&str) -> Result<Option<Vec<u8>>, clawhdf5_format::error::FormatError> + use<> {
|
|
let base = self.base_dir.clone();
|
|
move |name: &str| {
|
|
use clawhdf5_format::error::FormatError;
|
|
let p = std::path::Path::new(name);
|
|
if p.is_absolute()
|
|
|| p.components()
|
|
.any(|c| matches!(c, std::path::Component::ParentDir))
|
|
{
|
|
return Err(FormatError::ChunkedReadError(format!("refused {name}")));
|
|
}
|
|
match std::fs::read(base.join(p)) {
|
|
Ok(b) => Ok(Some(b)),
|
|
Err(err) if err.kind() == std::io::ErrorKind::NotFound => Ok(None),
|
|
Err(err) => Err(FormatError::ChunkedReadError(err.to_string())),
|
|
}
|
|
}
|
|
}
|
|
|
|
fn read_named_datatype(&self, h: &ObjectHeader) -> Result<(), String> {
|
|
let dtb = self
|
|
.payload(h, MessageType::Datatype)?
|
|
.ok_or("MissingMessage(Datatype)")?;
|
|
Datatype::parse(&dtb).map_err(e)?;
|
|
Ok(())
|
|
}
|
|
|
|
fn read_dataset(&self, h: &ObjectHeader, rec: &mut Map<String, Value>) -> Result<(), String> {
|
|
let dtb = self
|
|
.payload(h, MessageType::Datatype)?
|
|
.ok_or("MissingMessage(Datatype)")?;
|
|
let (dt, _) = Datatype::parse(&dtb).map_err(e)?;
|
|
rec.insert("dtype".into(), Value::String(dtype_str(&dt)));
|
|
let dsb = self
|
|
.payload(h, MessageType::Dataspace)?
|
|
.ok_or("MissingMessage(Dataspace)")?;
|
|
let mut ds = Dataspace::parse(&dsb, self.ls).map_err(e)?;
|
|
// A virtual dataset's extent can come from its sources (unlimited /
|
|
// printf mappings), as h5py reports it, rather than the stored one.
|
|
if let Some(lm) = h
|
|
.messages
|
|
.iter()
|
|
.find(|m| m.msg_type == MessageType::DataLayout)
|
|
&& let Ok(dl @ DataLayout::Virtual { .. }) =
|
|
DataLayout::parse(&lm.data, self.os, self.ls)
|
|
{
|
|
let resolver = self.vds_resolver();
|
|
ds.dimensions = clawhdf5_format::vds::virtual_dataset_extent(
|
|
self.data,
|
|
&dl,
|
|
&ds,
|
|
self.os,
|
|
self.ls,
|
|
Some(&resolver),
|
|
)
|
|
.map_err(e)?;
|
|
}
|
|
let (shape, n) = Self::shape(&ds);
|
|
rec.insert("shape".into(), shape);
|
|
if n.saturating_mul(dt.type_size() as u64) > MAX_BYTES {
|
|
rec.insert("skipped".into(), Value::String("too large".into()));
|
|
return Ok(());
|
|
}
|
|
let lm = h
|
|
.messages
|
|
.iter()
|
|
.find(|m| m.msg_type == MessageType::DataLayout)
|
|
.ok_or("MissingMessage(DataLayout)")?;
|
|
let dl = DataLayout::parse(&lm.data, self.os, self.ls).map_err(e)?;
|
|
rec.insert(
|
|
"layout".into(),
|
|
Value::String(
|
|
match &dl {
|
|
DataLayout::Compact { .. } => "compact",
|
|
DataLayout::Contiguous { .. } => "contiguous",
|
|
DataLayout::Chunked { .. } => "chunked",
|
|
DataLayout::Virtual { .. } => "virtual",
|
|
}
|
|
.into(),
|
|
),
|
|
);
|
|
let pipeline = match self.payload(h, MessageType::FilterPipeline)? {
|
|
Some(p) => Some(FilterPipeline::parse(&p).map_err(e)?),
|
|
None => None,
|
|
};
|
|
if let Some(p) = &pipeline {
|
|
rec.insert(
|
|
"filters".into(),
|
|
json!(p.filters.iter().map(|f| f.filter_id).collect::<Vec<_>>()),
|
|
);
|
|
}
|
|
let raw = if matches!(dl, DataLayout::Virtual { .. }) {
|
|
let resolver = self.vds_resolver();
|
|
let fill = clawhdf5_format::fill_value::dataset_fill_value_in(
|
|
self.data,
|
|
&h.messages,
|
|
self.os,
|
|
self.ls,
|
|
)
|
|
.map_err(e)?;
|
|
clawhdf5_format::vds::read_virtual_dataset(
|
|
self.data,
|
|
&dl,
|
|
&ds,
|
|
&dt,
|
|
fill.as_deref(),
|
|
self.os,
|
|
self.ls,
|
|
Some(&resolver),
|
|
)
|
|
.map_err(e)?
|
|
.data
|
|
} else {
|
|
let cache = clawhdf5_format::chunk_cache::ChunkCache::new();
|
|
clawhdf5_format::fill_value::read_full_with_fill::<clawhdf5_format::error::FormatError>(
|
|
&h.messages,
|
|
self.data,
|
|
&dl,
|
|
&ds,
|
|
dt.type_size() as usize,
|
|
self.os,
|
|
self.ls,
|
|
|| {
|
|
data_read::read_raw_data_cached(
|
|
self.data,
|
|
&dl,
|
|
&ds,
|
|
&dt,
|
|
pipeline.as_ref(),
|
|
self.os,
|
|
self.ls,
|
|
&cache,
|
|
)
|
|
},
|
|
)
|
|
.map_err(e)?
|
|
};
|
|
self.hash_values(&dt, &raw, n, rec)
|
|
}
|
|
|
|
fn attrs(&self, h: &ObjectHeader) -> Result<Map<String, Value>, String> {
|
|
let msgs = extract_attributes_full(self.data, h, self.os, self.ls).map_err(e)?;
|
|
let mut out = Map::new();
|
|
for a in &msgs {
|
|
let r = guarded(|| {
|
|
let mut rec = Map::new();
|
|
rec.insert("dtype".into(), Value::String(dtype_str(&a.datatype)));
|
|
let (shape, n) = Self::shape(&a.dataspace);
|
|
rec.insert("shape".into(), shape);
|
|
self.hash_values(&a.datatype, &a.raw_data, n, &mut rec)?;
|
|
Ok(rec)
|
|
});
|
|
let v = match r {
|
|
Ok(rec) => Value::Object(rec),
|
|
Err(msg) => json!({ "error": msg }),
|
|
};
|
|
out.insert(a.name.clone(), v);
|
|
}
|
|
Ok(out)
|
|
}
|
|
|
|
fn entries(&self, h: &ObjectHeader) -> Result<Vec<GroupEntry>, String> {
|
|
let v1 = h
|
|
.messages
|
|
.iter()
|
|
.find(|m| m.msg_type == MessageType::SymbolTable);
|
|
if let Some(m) = v1 {
|
|
let stm = SymbolTableMessage::parse(&m.data, self.os).map_err(e)?;
|
|
group_v1::resolve_v1_group_entries(self.data, &stm, self.os, self.ls).map_err(e)
|
|
} else if h
|
|
.messages
|
|
.iter()
|
|
.any(|m| m.msg_type == MessageType::LinkInfo || m.msg_type == MessageType::Link)
|
|
{
|
|
group_v2::resolve_v2_group_entries(self.data, h, self.os, self.ls).map_err(e)
|
|
} else {
|
|
Ok(Vec::new())
|
|
}
|
|
}
|
|
}
|
|
|
|
/// Element bytes as an unsigned integer (at most 16 bytes), honouring byte order.
|
|
fn element_bits(b: &[u8], byte_order: &DatatypeByteOrder) -> Result<u128, String> {
|
|
if b.len() > 16 {
|
|
return Err(format!("canon: {}-byte numeric element", b.len()));
|
|
}
|
|
let mut v = 0u128;
|
|
match byte_order {
|
|
DatatypeByteOrder::LittleEndian => {
|
|
for (i, x) in b.iter().enumerate() {
|
|
v |= u128::from(*x) << (8 * i);
|
|
}
|
|
}
|
|
DatatypeByteOrder::BigEndian => {
|
|
for x in b {
|
|
v = (v << 8) | u128::from(*x);
|
|
}
|
|
}
|
|
DatatypeByteOrder::Vax => return Err("canon: VAX byte order".into()),
|
|
}
|
|
Ok(v)
|
|
}
|
|
|
|
fn field(v: u128, pos: u32, len: u32) -> u128 {
|
|
if len == 0 || pos >= 128 {
|
|
return 0;
|
|
}
|
|
let v = v >> pos;
|
|
if len >= 128 {
|
|
v
|
|
} else {
|
|
v & ((1u128 << len) - 1)
|
|
}
|
|
}
|
|
|
|
/// True when a float's bit fields are exactly IEEE 754 binary16/32/64 for its
|
|
/// size. h5py hands back such a type's bytes untouched; any other layout (an
|
|
/// N-Bit `H5Tset_precision` float, say) is *converted* by libhdf5 into the
|
|
/// numpy float of the same size, so comparing raw bytes would be meaningless.
|
|
fn ieee_layout(dt: &Datatype) -> bool {
|
|
let Datatype::FloatingPoint {
|
|
size,
|
|
bit_offset,
|
|
bit_precision,
|
|
exponent_location,
|
|
exponent_size,
|
|
mantissa_location,
|
|
mantissa_size,
|
|
exponent_bias,
|
|
..
|
|
} = dt
|
|
else {
|
|
return true;
|
|
};
|
|
let std = match size {
|
|
2 => (16, 10, 5, 10, 15),
|
|
4 => (32, 23, 8, 23, 127),
|
|
8 => (64, 52, 11, 52, 1023),
|
|
_ => return true, // no same-size numpy float to convert to: compare raw
|
|
};
|
|
*bit_offset == 0
|
|
&& (
|
|
*bit_precision,
|
|
*exponent_location,
|
|
*exponent_size,
|
|
*mantissa_size,
|
|
*exponent_bias,
|
|
) == (std.0, std.1, std.2, std.3, std.4)
|
|
&& *mantissa_location == 0
|
|
}
|
|
|
|
/// Canonicalise a non-IEEE-layout float the way libhdf5's float->float
|
|
/// conversion presents it to h5py: as the IEEE float of the same size.
|
|
/// Assumes the implied-leading-one normalisation and the sign bit at the top
|
|
/// of the precision (what `H5Tset_precision` produces; the parser does not
|
|
/// keep either field).
|
|
fn canon_custom_float(dt: &Datatype, b: &[u8], out: &mut Vec<u8>) -> Result<(), String> {
|
|
let Datatype::FloatingPoint {
|
|
size,
|
|
byte_order,
|
|
bit_offset,
|
|
bit_precision,
|
|
exponent_location,
|
|
exponent_size,
|
|
mantissa_location,
|
|
mantissa_size,
|
|
exponent_bias,
|
|
} = dt
|
|
else {
|
|
unreachable!()
|
|
};
|
|
let (esize, msize) = (u32::from(*exponent_size), u32::from(*mantissa_size));
|
|
if esize == 0 || esize > 30 || msize > 64 {
|
|
return Err(format!("canon: unsupported float layout e{esize} m{msize}"));
|
|
}
|
|
let v = element_bits(b, byte_order)?;
|
|
let sign_pos = (u32::from(*bit_offset) + u32::from(*bit_precision)).saturating_sub(1);
|
|
let neg = field(v, sign_pos, 1) == 1;
|
|
let e = field(v, u32::from(*exponent_location), esize) as i64;
|
|
let m = field(v, u32::from(*mantissa_location), msize);
|
|
let emax = (1i64 << esize) - 1;
|
|
let bias = i64::from(*exponent_bias);
|
|
let mag = if e == emax {
|
|
if m == 0 { f64::INFINITY } else { f64::NAN }
|
|
} else if e == 0 {
|
|
(m as f64) * 2f64.powi((1 - bias - msize as i64) as i32)
|
|
} else {
|
|
((1u128 << msize) as f64 + m as f64) * 2f64.powi((e - bias - msize as i64) as i32)
|
|
};
|
|
let x = if neg { -mag } else { mag };
|
|
match size {
|
|
2 => out
|
|
.extend_from_slice(&clawhdf5_format::float16::f32_to_f16_bits(x as f32).to_le_bytes()),
|
|
4 => out.extend_from_slice(&(x as f32).to_le_bytes()),
|
|
8 => out.extend_from_slice(&x.to_le_bytes()),
|
|
_ => unreachable!("ieee_layout keeps other sizes raw"),
|
|
}
|
|
Ok(())
|
|
}
|
|
|
|
/// Integers stored with a bit offset or reduced precision (N-Bit): libhdf5
|
|
/// converts them to the full-width integer of the same size, shifting the
|
|
/// value down and sign-extending from the top precision bit.
|
|
fn canon_partial_int(dt: &Datatype, b: &[u8], out: &mut Vec<u8>) -> Result<(), String> {
|
|
let Datatype::FixedPoint {
|
|
size,
|
|
byte_order,
|
|
signed,
|
|
bit_offset,
|
|
bit_precision,
|
|
} = dt
|
|
else {
|
|
unreachable!()
|
|
};
|
|
let prec = u32::from(*bit_precision);
|
|
let v = element_bits(b, byte_order)?;
|
|
let mut x = field(v, u32::from(*bit_offset), prec);
|
|
if *signed && prec > 0 && prec < 128 && field(x, prec - 1, 1) == 1 {
|
|
x |= !0u128 << prec;
|
|
}
|
|
out.extend_from_slice(&x.to_le_bytes()[..*size as usize]);
|
|
Ok(())
|
|
}
|
|
|
|
fn partial_int(dt: &Datatype) -> bool {
|
|
matches!(dt, Datatype::FixedPoint { size, bit_offset, bit_precision, .. }
|
|
if *bit_offset != 0 || u32::from(*bit_precision) != size * 8)
|
|
}
|
|
|
|
fn canon_str(b: &[u8], out: &mut Vec<u8>) {
|
|
let cut = b.iter().position(|&c| c == 0).unwrap_or(b.len());
|
|
let mut s = &b[..cut];
|
|
while let [rest @ .., b' '] = s {
|
|
s = rest;
|
|
}
|
|
out.push(b'S');
|
|
out.extend_from_slice(&(s.len() as u32).to_le_bytes());
|
|
out.extend_from_slice(s);
|
|
}
|
|
|
|
fn hex(b: &[u8]) -> String {
|
|
b.iter().map(|x| format!("{x:02x}")).collect()
|
|
}
|
|
|
|
fn dtype_str(dt: &Datatype) -> String {
|
|
match dt {
|
|
Datatype::FixedPoint {
|
|
size,
|
|
signed,
|
|
byte_order,
|
|
..
|
|
} => {
|
|
format!(
|
|
"{}{}{}",
|
|
bo(byte_order),
|
|
if *signed { "i" } else { "u" },
|
|
size
|
|
)
|
|
}
|
|
Datatype::FloatingPoint {
|
|
size, byte_order, ..
|
|
} => format!("{}f{}", bo(byte_order), size),
|
|
Datatype::BitField {
|
|
size, byte_order, ..
|
|
} => format!("{}b{}", bo(byte_order), size),
|
|
Datatype::Time { size, .. } => format!("time{size}"),
|
|
Datatype::String { size, .. } => format!("S{size}"),
|
|
Datatype::Opaque { size, .. } => format!("V{size}"),
|
|
Datatype::Compound { size, members } => format!(
|
|
"{{{}}}{size}",
|
|
members
|
|
.iter()
|
|
.map(|m| format!("{}:{}", m.name, dtype_str(&m.datatype)))
|
|
.collect::<Vec<_>>()
|
|
.join(",")
|
|
),
|
|
Datatype::Reference { ref_type, .. } => format!("ref({ref_type:?})"),
|
|
Datatype::Enumeration { base_type, .. } => format!("enum({})", dtype_str(base_type)),
|
|
Datatype::VariableLength {
|
|
is_string: true, ..
|
|
} => "vlstr".into(),
|
|
Datatype::VariableLength { base_type, .. } => format!("vlen({})", dtype_str(base_type)),
|
|
Datatype::Array {
|
|
base_type,
|
|
dimensions,
|
|
} => format!("({}){dimensions:?}", dtype_str(base_type)),
|
|
}
|
|
}
|
|
|
|
fn bo(b: &DatatypeByteOrder) -> &'static str {
|
|
match b {
|
|
DatatypeByteOrder::LittleEndian => "<",
|
|
DatatypeByteOrder::BigEndian => ">",
|
|
DatatypeByteOrder::Vax => "vax",
|
|
}
|
|
}
|
|
|
|
fn is_group(h: &ObjectHeader) -> bool {
|
|
h.messages.iter().any(|m| {
|
|
matches!(
|
|
m.msg_type,
|
|
MessageType::LinkInfo | MessageType::Link | MessageType::SymbolTable
|
|
)
|
|
})
|
|
}
|
|
|
|
fn main() {
|
|
install_hook();
|
|
let path = std::env::args().nth(1).expect("usage: probe <file>");
|
|
let mut top = Map::new();
|
|
top.insert("file".into(), Value::String(path.clone()));
|
|
let data = match std::fs::read(&path) {
|
|
Ok(d) => d,
|
|
Err(err) => {
|
|
top.insert("open_error".into(), Value::String(format!("Io({err})")));
|
|
println!("{}", Value::Object(top));
|
|
return;
|
|
}
|
|
};
|
|
// Every address is relative to the superblock: look at the file from
|
|
// there on (past any user block), as libhdf5 does.
|
|
let hdf5: &[u8] = match signature::find_signature(&data) {
|
|
Ok(off) => &data[off..],
|
|
Err(_) => &data,
|
|
};
|
|
let sb = guarded(|| Superblock::parse(hdf5, 0).map_err(e));
|
|
let sb = match sb {
|
|
Ok(sb) => sb,
|
|
Err(msg) => {
|
|
top.insert("open_error".into(), Value::String(msg));
|
|
println!("{}", Value::Object(top));
|
|
return;
|
|
}
|
|
};
|
|
top.insert("superblock_version".into(), json!(sb.version));
|
|
let ctx = Ctx {
|
|
data: hdf5,
|
|
os: sb.offset_size,
|
|
ls: sb.length_size,
|
|
base_dir: std::path::Path::new(&path)
|
|
.parent()
|
|
.map(|p| p.to_path_buf())
|
|
.unwrap_or_default(),
|
|
heaps: RefCell::new(HashMap::new()),
|
|
};
|
|
let mut objects: Vec<Value> = Vec::new();
|
|
let mut visited = HashSet::new();
|
|
let mut soft_v1 = 0u64;
|
|
// explicit DFS stack: (address, path)
|
|
let mut stack: Vec<(u64, String)> = vec![(sb.root_group_address, "/".to_string())];
|
|
while let Some((addr, p)) = stack.pop() {
|
|
if objects.len() >= MAX_OBJECTS {
|
|
top.insert("truncated".into(), json!(true));
|
|
break;
|
|
}
|
|
if !visited.insert(addr) {
|
|
continue;
|
|
}
|
|
let mut rec = Map::new();
|
|
rec.insert("path".into(), Value::String(p.clone()));
|
|
let r = guarded(|| {
|
|
let h = ctx.header(addr)?;
|
|
Ok(h)
|
|
});
|
|
let h = match r {
|
|
Ok(h) => h,
|
|
Err(msg) => {
|
|
rec.insert("kind".into(), Value::String("unknown".into()));
|
|
rec.insert("error".into(), Value::String(msg));
|
|
objects.push(Value::Object(rec));
|
|
continue;
|
|
}
|
|
};
|
|
let is_ds = h
|
|
.messages
|
|
.iter()
|
|
.any(|m| m.msg_type == MessageType::DataLayout);
|
|
let kind = if is_ds {
|
|
"dataset"
|
|
} else if is_group(&h) || addr == sb.root_group_address {
|
|
"group"
|
|
} else if h
|
|
.messages
|
|
.iter()
|
|
.any(|m| m.msg_type == MessageType::Datatype)
|
|
{
|
|
"datatype"
|
|
} else {
|
|
"unknown"
|
|
};
|
|
rec.insert("kind".into(), Value::String(kind.into()));
|
|
if kind == "dataset"
|
|
&& let Err(msg) = guarded(|| ctx.read_dataset(&h, &mut rec))
|
|
{
|
|
rec.insert("error".into(), Value::String(msg));
|
|
}
|
|
// Opening a committed datatype decodes it (h5py's `f[name]` fails on
|
|
// one libhdf5 cannot decode), so decode it here too.
|
|
if kind == "datatype"
|
|
&& let Err(msg) = guarded(|| ctx.read_named_datatype(&h))
|
|
{
|
|
rec.insert("error".into(), Value::String(msg));
|
|
}
|
|
if kind != "datatype" {
|
|
match guarded(|| ctx.attrs(&h)) {
|
|
Ok(m) => {
|
|
rec.insert("attrs".into(), Value::Object(m));
|
|
}
|
|
Err(msg) => {
|
|
rec.insert("attrs_error".into(), Value::String(msg));
|
|
}
|
|
}
|
|
}
|
|
if kind == "group" {
|
|
match guarded(|| ctx.entries(&h)) {
|
|
Ok(mut ents) => {
|
|
ents.retain(|en| {
|
|
if en.cache_type == 2 {
|
|
soft_v1 += 1;
|
|
false
|
|
} else {
|
|
true
|
|
}
|
|
});
|
|
ents.sort_by(|a, b| a.name.cmp(&b.name));
|
|
let base = if p == "/" { String::new() } else { p.clone() };
|
|
for en in ents.into_iter().rev() {
|
|
stack.push((en.object_header_address, format!("{base}/{}", en.name)));
|
|
}
|
|
}
|
|
Err(msg) => {
|
|
rec.insert("list_error".into(), Value::String(msg));
|
|
}
|
|
}
|
|
}
|
|
objects.push(Value::Object(rec));
|
|
}
|
|
if soft_v1 > 0 {
|
|
top.insert("v1_soft_link_entries".into(), json!(soft_v1));
|
|
}
|
|
top.insert("objects".into(), Value::Array(objects));
|
|
println!("{}", Value::Object(top));
|
|
}
|
|
|
|
#[cfg(test)]
|
|
mod tests {
|
|
use super::*;
|
|
|
|
/// The N-Bit float of libhdf5's `test/testfiles/le_data.h5`
|
|
/// (`Nbit_float_data_le`): offset 7, precision 20, sign bit 26, exponent
|
|
/// 20+6 (bias 31), mantissa 7+13.
|
|
fn nbit_f32(byte_order: DatatypeByteOrder) -> Datatype {
|
|
Datatype::FloatingPoint {
|
|
size: 4,
|
|
byte_order,
|
|
bit_offset: 7,
|
|
bit_precision: 20,
|
|
exponent_location: 20,
|
|
exponent_size: 6,
|
|
mantissa_location: 7,
|
|
mantissa_size: 13,
|
|
exponent_bias: 31,
|
|
}
|
|
}
|
|
|
|
fn canon_one(dt: &Datatype, bytes: &[u8]) -> Vec<u8> {
|
|
let mut out = Vec::new();
|
|
canon_custom_float(dt, bytes, &mut out).unwrap();
|
|
out
|
|
}
|
|
|
|
#[test]
|
|
fn nbit_float_canonicalises_to_the_value_libhdf5_returns() {
|
|
let le = nbit_f32(DatatypeByteOrder::LittleEndian);
|
|
let be = nbit_f32(DatatypeByteOrder::BigEndian);
|
|
assert!(!ieee_layout(&le));
|
|
// 1.0: exponent = bias, mantissa 0
|
|
let one: u32 = 31 << 20;
|
|
assert_eq!(canon_one(&le, &one.to_le_bytes()), 1.0f32.to_le_bytes());
|
|
assert_eq!(canon_one(&be, &one.to_be_bytes()), 1.0f32.to_le_bytes());
|
|
// -2.1999512 (h5py's reading of the file's -2.2): sign, e = 32, m = 819
|
|
let v: u32 = (1 << 26) | (32 << 20) | (819 << 7);
|
|
assert_eq!(
|
|
canon_one(&le, &v.to_le_bytes()),
|
|
(-2.199_951_2f32).to_le_bytes()
|
|
);
|
|
assert_eq!(canon_one(&le, &[0; 4]), 0.0f32.to_le_bytes());
|
|
}
|
|
|
|
#[test]
|
|
fn ieee_floats_keep_their_raw_bytes() {
|
|
let f32le = Datatype::FloatingPoint {
|
|
size: 4,
|
|
byte_order: DatatypeByteOrder::LittleEndian,
|
|
bit_offset: 0,
|
|
bit_precision: 32,
|
|
exponent_location: 23,
|
|
exponent_size: 8,
|
|
mantissa_location: 0,
|
|
mantissa_size: 23,
|
|
exponent_bias: 127,
|
|
};
|
|
assert!(ieee_layout(&f32le));
|
|
}
|
|
|
|
#[test]
|
|
fn partial_precision_int_is_shifted_and_sign_extended() {
|
|
let dt = Datatype::FixedPoint {
|
|
size: 4,
|
|
byte_order: DatatypeByteOrder::BigEndian,
|
|
signed: true,
|
|
bit_offset: 4,
|
|
bit_precision: 17,
|
|
};
|
|
assert!(partial_int(&dt));
|
|
let stored = (((-5i32) as u32) & 0x1_FFFF) << 4;
|
|
let mut out = Vec::new();
|
|
canon_partial_int(&dt, &stored.to_be_bytes(), &mut out).unwrap();
|
|
assert_eq!(out, (-5i32).to_le_bytes());
|
|
}
|
|
}
|