feat: check HDF5 2.x small floats against libhdf5 2.2.0
Fixture written by libhdf5 2.2.0 (built from tag 2.2.0) through ctypes: every bit pattern of FP4 E2M1, FP6 E2M3/E3M2, FP8 E4M3/E5M2 and a bfloat16 LE/BE set, as datasets and attributes, with what H5Dread/H5Aread return into double and float and the conversion exceptions libhdf5 raises. clawhdf5 already decoded every value as libhdf5 does, including an all-ones exponent as inf/NaN in the OCP formats that have none (documented as a deliberate match in known-issues). - data_read: NaNs of non-native float layouts get libhdf5's bits (sign kept, every mantissa bit set) in f64 and f32. - h5rs dump/ls name these types as h5dump/h5ls 2.x do (H5T_FLOAT_F4E2M1, "FP4 E2M1 4-bit float", float4-e2m1 ...), checked against h5dump 2.2.0's output of the fixture. - Python bindings read them as h5py 3.16 does (float32 for bfloat16, float16 for the 1-byte formats, file byte order, same bytes as h5py); writing them in 'r+' is refused. Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
@@ -1419,9 +1419,11 @@ pub fn read_as_f32(raw: &[u8], datatype: &Datatype) -> Result<Vec<f32>, FormatEr
|
||||
result.push(match format {
|
||||
FloatFormat::Single => read_f32_bytes(chunk, &order),
|
||||
FloatFormat::Half => read_f16_bytes(chunk, &order),
|
||||
// Double rounds; every other supported layout (bfloat16, FP8)
|
||||
// is exact in f32.
|
||||
_ => format.decode(chunk, &order) as f32,
|
||||
// Double rounds.
|
||||
FloatFormat::Double => format.decode(chunk, &order) as f32,
|
||||
// Every other supported layout (bfloat16, FP8, FP6, FP4) is
|
||||
// exact in f32.
|
||||
FloatFormat::Other(_) => narrow_decoded(format.decode(chunk, &order)),
|
||||
});
|
||||
}
|
||||
return Ok(result);
|
||||
@@ -1947,7 +1949,12 @@ enum FloatFormat {
|
||||
Double,
|
||||
/// Any other IEEE-style layout (implied leading mantissa bit, all-ones
|
||||
/// exponent for infinity/NaN) whose values are all exact in `f64`:
|
||||
/// bfloat16, the FP8 formats, and similar.
|
||||
/// bfloat16, FP8 E4M3/E5M2, FP6 E2M3/E3M2, FP4 E2M1, and similar.
|
||||
///
|
||||
/// libhdf5 (checked against 2.2.0) decodes all of them this way, also the
|
||||
/// OCP MX formats whose specification has no infinity (FP6, FP4) or a
|
||||
/// single NaN (FP8 E4M3): an all-ones exponent is infinity or NaN, not a
|
||||
/// finite value. clawhdf5 follows libhdf5 so both read a file alike.
|
||||
Other(FloatLayout),
|
||||
}
|
||||
|
||||
@@ -2046,7 +2053,9 @@ impl FloatLayout {
|
||||
if mantissa == 0 {
|
||||
f64::INFINITY
|
||||
} else {
|
||||
f64::NAN
|
||||
// The NaN libhdf5 converts every NaN to: all mantissa bits
|
||||
// set (the sign is applied below).
|
||||
LIBHDF5_NAN
|
||||
}
|
||||
} else {
|
||||
let bias = i64::from(self.exponent_bias);
|
||||
@@ -2067,6 +2076,27 @@ impl FloatLayout {
|
||||
}
|
||||
}
|
||||
|
||||
/// The `f64` NaN libhdf5's conversion (`H5T__conv_f_f`) produces from a NaN
|
||||
/// of a non-native float layout: sign clear, every mantissa bit set.
|
||||
const LIBHDF5_NAN: f64 = f64::from_bits(0x7FFF_FFFF_FFFF_FFFF);
|
||||
|
||||
/// Narrow a value decoded from a non-native float layout to `f32`. Every such
|
||||
/// value is exact in `f32`; a NaN becomes the NaN libhdf5 gives for
|
||||
/// `H5T_NATIVE_FLOAT` (sign kept, every mantissa bit set) rather than
|
||||
/// whatever payload an `as` cast leaves.
|
||||
fn narrow_decoded(value: f64) -> f32 {
|
||||
if value.is_nan() {
|
||||
let sign = if value.is_sign_negative() {
|
||||
1u32 << 31
|
||||
} else {
|
||||
0
|
||||
};
|
||||
f32::from_bits(sign | 0x7FFF_FFFF)
|
||||
} else {
|
||||
value as f32
|
||||
}
|
||||
}
|
||||
|
||||
/// `x * 2^power` without `std` (no `powi`/`libm`). `x` is a non-negative
|
||||
/// integer below 2^53, so it is exact.
|
||||
fn scale_by_pow2(x: f64, power: i64) -> f64 {
|
||||
@@ -2425,6 +2455,37 @@ mod tests {
|
||||
assert!(got[4].is_nan());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn fp4_decodes_as_libhdf5_does() {
|
||||
// H5T_FLOAT_F4E2M1: 4 significant bits in a byte; the high bits are
|
||||
// padding. libhdf5 2.2.0 reads 0b0110 as +inf and 0b0111 as NaN
|
||||
// (IEEE-style, although OCP MX FP4 has neither), and returns NaNs
|
||||
// with every mantissa bit set.
|
||||
let fp4 = Datatype::FloatingPoint {
|
||||
size: 1,
|
||||
byte_order: DatatypeByteOrder::LittleEndian,
|
||||
bit_offset: 0,
|
||||
bit_precision: 4,
|
||||
exponent_location: 1,
|
||||
exponent_size: 2,
|
||||
mantissa_location: 0,
|
||||
mantissa_size: 1,
|
||||
exponent_bias: 1,
|
||||
};
|
||||
let raw = [0x01, 0x05, 0xF5, 0x06, 0x0E, 0x07, 0x0F];
|
||||
let got = read_as_f64(&raw, &fp4).unwrap();
|
||||
assert_eq!(
|
||||
&got[..5],
|
||||
&[0.5, 3.0, 3.0, f64::INFINITY, f64::NEG_INFINITY]
|
||||
);
|
||||
assert_eq!(got[5].to_bits(), 0x7FFF_FFFF_FFFF_FFFF);
|
||||
assert_eq!(got[6].to_bits(), 0xFFFF_FFFF_FFFF_FFFF);
|
||||
let got = read_as_f32(&raw, &fp4).unwrap();
|
||||
assert_eq!(&got[..3], &[0.5, 3.0, 3.0]);
|
||||
assert_eq!(got[5].to_bits(), 0x7FFF_FFFF);
|
||||
assert_eq!(got[6].to_bits(), 0xFFFF_FFFF);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn full_width_signed_unchanged() {
|
||||
// Regression: full-width 32-bit signed must be unaffected.
|
||||
|
||||
Reference in New Issue
Block a user