feat: check HDF5 2.x small floats against libhdf5 2.2.0
CI / test-arm64 (pull_request) Successful in 1m38s
CI / test (pull_request) Successful in 31m22s

Fixture written by libhdf5 2.2.0 (built from tag 2.2.0) through ctypes:
every bit pattern of FP4 E2M1, FP6 E2M3/E3M2, FP8 E4M3/E5M2 and a
bfloat16 LE/BE set, as datasets and attributes, with what H5Dread/H5Aread
return into double and float and the conversion exceptions libhdf5
raises. clawhdf5 already decoded every value as libhdf5 does, including
an all-ones exponent as inf/NaN in the OCP formats that have none
(documented as a deliberate match in known-issues).

- data_read: NaNs of non-native float layouts get libhdf5's bits (sign
  kept, every mantissa bit set) in f64 and f32.
- h5rs dump/ls name these types as h5dump/h5ls 2.x do
  (H5T_FLOAT_F4E2M1, "FP4 E2M1 4-bit float", float4-e2m1 ...), checked
  against h5dump 2.2.0's output of the fixture.
- Python bindings read them as h5py 3.16 does (float32 for bfloat16,
  float16 for the 1-byte formats, file byte order, same bytes as h5py);
  writing them in 'r+' is refused.

Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
osobh
2026-09-28 21:25:03 -05:00
co-authored by Claude Opus 5.5
parent 31efac1ae1
commit bce07e9cb9
16 changed files with 1207 additions and 13 deletions
+117 -4
View File
@@ -8,10 +8,16 @@
//! returns become the array's buffer as they are: the `Vec<u8>` is handed to
//! numpy without a copy and viewed as the dtype.
//!
//! Anything this mapping cannot describe exactly — non-IEEE floats, integers
//! with padding bits, VAX byte order, references, bitfields, time, and
//! variable-length members inside compounds or arrays — is a `TypeError`,
//! never a best-effort guess.
//! The one exception is a dataset or attribute of a non-IEEE float
//! (bfloat16, FP8 E4M3/E5M2, FP6, FP4, ...): like h5py, it reads as the
//! narrowest of `float16`/`float32`/`float64` that holds every value of the
//! format exactly, in the file's byte order, and the values are converted
//! (as libhdf5 converts them, NaN bits included).
//!
//! Anything this mapping cannot describe exactly — non-IEEE floats inside
//! compounds or arrays, integers with padding bits, VAX byte order,
//! references, bitfields, time, and variable-length members inside compounds
//! or arrays — is a `TypeError`, never a best-effort guess.
use std::collections::HashMap;
@@ -35,6 +41,13 @@ pub(crate) enum Layout {
VlString { utf8: bool },
/// Variable-length sequence of a fixed-size base type.
VlSequence,
/// A non-IEEE float (`source`), converted to the IEEE float of `size`
/// bytes numpy reports (see [`widened_float`]).
Float {
source: Datatype,
size: usize,
big_endian: bool,
},
}
/// Everything needed to turn a dataset's or attribute's bytes into numpy.
@@ -139,6 +152,74 @@ fn float_format(dt: &Datatype) -> PyResult<String> {
Ok(format!("{}f{size}", byte_order_char(byte_order, *size)?))
}
/// Whether `dt` is an IEEE 754 binary16/32/64 float numpy reads as it is.
pub(crate) fn is_ieee_float(dt: &Datatype) -> bool {
float_format(dt).is_ok()
}
/// The IEEE float h5py reads a non-IEEE float as (`TypeFloatID.py_dtype` in
/// h5py 3.16): the first of `float16`, `float32`, `float64` at least as wide
/// whose mantissa holds the format's and whose normal exponent range covers
/// it — bfloat16 as `float32`, FP8/FP6/FP4 as `float16` — in the file's byte
/// order. Returns `(size in bytes, big endian)`, or `None` when no IEEE type
/// holds it (or the byte order is VAX).
fn widened_float(dt: &Datatype) -> Option<(usize, bool)> {
let Datatype::FloatingPoint {
size,
byte_order,
exponent_size,
mantissa_size,
exponent_bias,
..
} = dt
else {
return None;
};
let big_endian = match byte_order {
DatatypeByteOrder::LittleEndian => false,
DatatypeByteOrder::BigEndian => true,
DatatypeByteOrder::Vax => return None,
};
if *exponent_size >= 32 {
return None;
}
let max_exp = (1i64 << exponent_size) - i64::from(*exponent_bias) - 1;
let min_exp = 1 - i64::from(*exponent_bias);
// numpy's finfo: (itemsize, nmant, maxexp, minexp).
[(2, 10, 16, -14), (4, 23, 128, -126), (8, 52, 1024, -1022)]
.into_iter()
.find(|&(bytes, nmant, maxexp, minexp)| {
bytes >= *size && *mantissa_size <= nmant && max_exp <= maxexp && min_exp >= minexp
})
.map(|(bytes, ..)| (bytes as usize, big_endian))
}
/// Encode `values` as IEEE floats of `size` bytes. Every value is exact in
/// the target (`widened_float` chose it so); a NaN keeps its sign and gets
/// every mantissa bit set, as libhdf5 converts NaNs.
fn encode_floats(values: &[f64], size: usize, big_endian: bool) -> Vec<u8> {
let mut out = Vec::with_capacity(values.len() * size);
for &v in values {
let bits: u64 = if v.is_nan() {
let sign = u64::from(v.is_sign_negative()) << (size * 8 - 1);
sign | ((1u64 << (size * 8 - 1)) - 1)
} else {
match size {
2 => u64::from(clawhdf5_format::float16::f32_to_f16_bits(v as f32)),
4 => u64::from((v as f32).to_bits()),
_ => v.to_bits(),
}
};
let bytes = bits.to_le_bytes();
if big_endian {
out.extend(bytes[..size].iter().rev());
} else {
out.extend_from_slice(&bytes[..size]);
}
}
out
}
/// `r`/`i` compounds of two identical IEEE floats are complex numbers in h5py.
fn complex_format(
size: u32,
@@ -366,6 +447,25 @@ impl Converter {
vl_unit: 0,
})
}
Datatype::FloatingPoint { .. } if !is_ieee_float(dt) => {
let Some((size, big_endian)) = widened_float(dt) else {
// The error names the layout.
return Err(float_format(dt).expect_err("not IEEE"));
};
let order = if big_endian { '>' } else { '<' };
let dtype = np_dtype(py, format!("{order}f{size}"))?;
Ok(Self {
view: dtype.clone().unbind(),
dtype: dtype.unbind(),
layout: Layout::Float {
source: dt.clone(),
size,
big_endian,
},
elem_size: dt.type_size() as usize,
vl_unit: 0,
})
}
_ => {
let dtype = fixed_dtype(py, dt)?;
Ok(Self {
@@ -416,6 +516,19 @@ impl Converter {
(Elements::Bytes(bytes), Layout::Fixed) => {
bytes_as_array(py, bytes, self.view.bind(py), shape)
}
(
Elements::Bytes(bytes),
Layout::Float {
source,
size,
big_endian,
},
) => {
let values = clawhdf5_format::data_read::read_as_f64(&bytes, source)
.map_err(|e| unsupported(e.to_string()))?;
let converted = encode_floats(&values, *size, *big_endian);
bytes_as_array(py, converted, self.view.bind(py), shape)
}
(Elements::Bytes(bytes), Layout::Subarray(dims)) => {
let mut full = shape.to_vec();
full.extend_from_slice(dims);
+6 -1
View File
@@ -48,7 +48,12 @@ fn not_implemented(what: impl std::fmt::Display) -> PyErr {
pub(crate) fn category(dt: &Datatype) -> PyResult<&'static str> {
match dt {
Datatype::FixedPoint { .. } => Ok("int"),
Datatype::FloatingPoint { .. } => Ok("float"),
Datatype::FloatingPoint { .. } if crate::convert::is_ieee_float(dt) => Ok("float"),
// Read as a wider IEEE float; writing would need the reverse
// conversion (rounding into bfloat16, FP8, ...).
Datatype::FloatingPoint { .. } => Err(not_implemented(
"writing non-IEEE floats (bfloat16, FP8, FP6, FP4, ...)",
)),
Datatype::Enumeration {
base_type, members, ..
} => {