feat: check HDF5 2.x small floats against libhdf5 2.2.0
Fixture written by libhdf5 2.2.0 (built from tag 2.2.0) through ctypes: every bit pattern of FP4 E2M1, FP6 E2M3/E3M2, FP8 E4M3/E5M2 and a bfloat16 LE/BE set, as datasets and attributes, with what H5Dread/H5Aread return into double and float and the conversion exceptions libhdf5 raises. clawhdf5 already decoded every value as libhdf5 does, including an all-ones exponent as inf/NaN in the OCP formats that have none (documented as a deliberate match in known-issues). - data_read: NaNs of non-native float layouts get libhdf5's bits (sign kept, every mantissa bit set) in f64 and f32. - h5rs dump/ls name these types as h5dump/h5ls 2.x do (H5T_FLOAT_F4E2M1, "FP4 E2M1 4-bit float", float4-e2m1 ...), checked against h5dump 2.2.0's output of the fixture. - Python bindings read them as h5py 3.16 does (float32 for bfloat16, float16 for the 1-byte formats, file byte order, same bytes as h5py); writing them in 'r+' is refused. Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
@@ -8,10 +8,16 @@
|
||||
//! returns become the array's buffer as they are: the `Vec<u8>` is handed to
|
||||
//! numpy without a copy and viewed as the dtype.
|
||||
//!
|
||||
//! Anything this mapping cannot describe exactly — non-IEEE floats, integers
|
||||
//! with padding bits, VAX byte order, references, bitfields, time, and
|
||||
//! variable-length members inside compounds or arrays — is a `TypeError`,
|
||||
//! never a best-effort guess.
|
||||
//! The one exception is a dataset or attribute of a non-IEEE float
|
||||
//! (bfloat16, FP8 E4M3/E5M2, FP6, FP4, ...): like h5py, it reads as the
|
||||
//! narrowest of `float16`/`float32`/`float64` that holds every value of the
|
||||
//! format exactly, in the file's byte order, and the values are converted
|
||||
//! (as libhdf5 converts them, NaN bits included).
|
||||
//!
|
||||
//! Anything this mapping cannot describe exactly — non-IEEE floats inside
|
||||
//! compounds or arrays, integers with padding bits, VAX byte order,
|
||||
//! references, bitfields, time, and variable-length members inside compounds
|
||||
//! or arrays — is a `TypeError`, never a best-effort guess.
|
||||
|
||||
use std::collections::HashMap;
|
||||
|
||||
@@ -35,6 +41,13 @@ pub(crate) enum Layout {
|
||||
VlString { utf8: bool },
|
||||
/// Variable-length sequence of a fixed-size base type.
|
||||
VlSequence,
|
||||
/// A non-IEEE float (`source`), converted to the IEEE float of `size`
|
||||
/// bytes numpy reports (see [`widened_float`]).
|
||||
Float {
|
||||
source: Datatype,
|
||||
size: usize,
|
||||
big_endian: bool,
|
||||
},
|
||||
}
|
||||
|
||||
/// Everything needed to turn a dataset's or attribute's bytes into numpy.
|
||||
@@ -139,6 +152,74 @@ fn float_format(dt: &Datatype) -> PyResult<String> {
|
||||
Ok(format!("{}f{size}", byte_order_char(byte_order, *size)?))
|
||||
}
|
||||
|
||||
/// Whether `dt` is an IEEE 754 binary16/32/64 float numpy reads as it is.
|
||||
pub(crate) fn is_ieee_float(dt: &Datatype) -> bool {
|
||||
float_format(dt).is_ok()
|
||||
}
|
||||
|
||||
/// The IEEE float h5py reads a non-IEEE float as (`TypeFloatID.py_dtype` in
|
||||
/// h5py 3.16): the first of `float16`, `float32`, `float64` at least as wide
|
||||
/// whose mantissa holds the format's and whose normal exponent range covers
|
||||
/// it — bfloat16 as `float32`, FP8/FP6/FP4 as `float16` — in the file's byte
|
||||
/// order. Returns `(size in bytes, big endian)`, or `None` when no IEEE type
|
||||
/// holds it (or the byte order is VAX).
|
||||
fn widened_float(dt: &Datatype) -> Option<(usize, bool)> {
|
||||
let Datatype::FloatingPoint {
|
||||
size,
|
||||
byte_order,
|
||||
exponent_size,
|
||||
mantissa_size,
|
||||
exponent_bias,
|
||||
..
|
||||
} = dt
|
||||
else {
|
||||
return None;
|
||||
};
|
||||
let big_endian = match byte_order {
|
||||
DatatypeByteOrder::LittleEndian => false,
|
||||
DatatypeByteOrder::BigEndian => true,
|
||||
DatatypeByteOrder::Vax => return None,
|
||||
};
|
||||
if *exponent_size >= 32 {
|
||||
return None;
|
||||
}
|
||||
let max_exp = (1i64 << exponent_size) - i64::from(*exponent_bias) - 1;
|
||||
let min_exp = 1 - i64::from(*exponent_bias);
|
||||
// numpy's finfo: (itemsize, nmant, maxexp, minexp).
|
||||
[(2, 10, 16, -14), (4, 23, 128, -126), (8, 52, 1024, -1022)]
|
||||
.into_iter()
|
||||
.find(|&(bytes, nmant, maxexp, minexp)| {
|
||||
bytes >= *size && *mantissa_size <= nmant && max_exp <= maxexp && min_exp >= minexp
|
||||
})
|
||||
.map(|(bytes, ..)| (bytes as usize, big_endian))
|
||||
}
|
||||
|
||||
/// Encode `values` as IEEE floats of `size` bytes. Every value is exact in
|
||||
/// the target (`widened_float` chose it so); a NaN keeps its sign and gets
|
||||
/// every mantissa bit set, as libhdf5 converts NaNs.
|
||||
fn encode_floats(values: &[f64], size: usize, big_endian: bool) -> Vec<u8> {
|
||||
let mut out = Vec::with_capacity(values.len() * size);
|
||||
for &v in values {
|
||||
let bits: u64 = if v.is_nan() {
|
||||
let sign = u64::from(v.is_sign_negative()) << (size * 8 - 1);
|
||||
sign | ((1u64 << (size * 8 - 1)) - 1)
|
||||
} else {
|
||||
match size {
|
||||
2 => u64::from(clawhdf5_format::float16::f32_to_f16_bits(v as f32)),
|
||||
4 => u64::from((v as f32).to_bits()),
|
||||
_ => v.to_bits(),
|
||||
}
|
||||
};
|
||||
let bytes = bits.to_le_bytes();
|
||||
if big_endian {
|
||||
out.extend(bytes[..size].iter().rev());
|
||||
} else {
|
||||
out.extend_from_slice(&bytes[..size]);
|
||||
}
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
/// `r`/`i` compounds of two identical IEEE floats are complex numbers in h5py.
|
||||
fn complex_format(
|
||||
size: u32,
|
||||
@@ -366,6 +447,25 @@ impl Converter {
|
||||
vl_unit: 0,
|
||||
})
|
||||
}
|
||||
Datatype::FloatingPoint { .. } if !is_ieee_float(dt) => {
|
||||
let Some((size, big_endian)) = widened_float(dt) else {
|
||||
// The error names the layout.
|
||||
return Err(float_format(dt).expect_err("not IEEE"));
|
||||
};
|
||||
let order = if big_endian { '>' } else { '<' };
|
||||
let dtype = np_dtype(py, format!("{order}f{size}"))?;
|
||||
Ok(Self {
|
||||
view: dtype.clone().unbind(),
|
||||
dtype: dtype.unbind(),
|
||||
layout: Layout::Float {
|
||||
source: dt.clone(),
|
||||
size,
|
||||
big_endian,
|
||||
},
|
||||
elem_size: dt.type_size() as usize,
|
||||
vl_unit: 0,
|
||||
})
|
||||
}
|
||||
_ => {
|
||||
let dtype = fixed_dtype(py, dt)?;
|
||||
Ok(Self {
|
||||
@@ -416,6 +516,19 @@ impl Converter {
|
||||
(Elements::Bytes(bytes), Layout::Fixed) => {
|
||||
bytes_as_array(py, bytes, self.view.bind(py), shape)
|
||||
}
|
||||
(
|
||||
Elements::Bytes(bytes),
|
||||
Layout::Float {
|
||||
source,
|
||||
size,
|
||||
big_endian,
|
||||
},
|
||||
) => {
|
||||
let values = clawhdf5_format::data_read::read_as_f64(&bytes, source)
|
||||
.map_err(|e| unsupported(e.to_string()))?;
|
||||
let converted = encode_floats(&values, *size, *big_endian);
|
||||
bytes_as_array(py, converted, self.view.bind(py), shape)
|
||||
}
|
||||
(Elements::Bytes(bytes), Layout::Subarray(dims)) => {
|
||||
let mut full = shape.to_vec();
|
||||
full.extend_from_slice(dims);
|
||||
|
||||
@@ -48,7 +48,12 @@ fn not_implemented(what: impl std::fmt::Display) -> PyErr {
|
||||
pub(crate) fn category(dt: &Datatype) -> PyResult<&'static str> {
|
||||
match dt {
|
||||
Datatype::FixedPoint { .. } => Ok("int"),
|
||||
Datatype::FloatingPoint { .. } => Ok("float"),
|
||||
Datatype::FloatingPoint { .. } if crate::convert::is_ieee_float(dt) => Ok("float"),
|
||||
// Read as a wider IEEE float; writing would need the reverse
|
||||
// conversion (rounding into bfloat16, FP8, ...).
|
||||
Datatype::FloatingPoint { .. } => Err(not_implemented(
|
||||
"writing non-IEEE floats (bfloat16, FP8, FP6, FP4, ...)",
|
||||
)),
|
||||
Datatype::Enumeration {
|
||||
base_type, members, ..
|
||||
} => {
|
||||
|
||||
Reference in New Issue
Block a user