feat: check HDF5 2.x small floats against libhdf5 2.2.0
Fixture written by libhdf5 2.2.0 (built from tag 2.2.0) through ctypes: every bit pattern of FP4 E2M1, FP6 E2M3/E3M2, FP8 E4M3/E5M2 and a bfloat16 LE/BE set, as datasets and attributes, with what H5Dread/H5Aread return into double and float and the conversion exceptions libhdf5 raises. clawhdf5 already decoded every value as libhdf5 does, including an all-ones exponent as inf/NaN in the OCP formats that have none (documented as a deliberate match in known-issues). - data_read: NaNs of non-native float layouts get libhdf5's bits (sign kept, every mantissa bit set) in f64 and f32. - h5rs dump/ls name these types as h5dump/h5ls 2.x do (H5T_FLOAT_F4E2M1, "FP4 E2M1 4-bit float", float4-e2m1 ...), checked against h5dump 2.2.0's output of the fixture. - Python bindings read them as h5py 3.16 does (float32 for bfloat16, float16 for the 1-byte formats, file byte order, same bytes as h5py); writing them in 'r+' is refused. Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
@@ -59,8 +59,81 @@ pub fn is_ieee(dt: &Datatype) -> bool {
|
||||
) == std
|
||||
}
|
||||
|
||||
/// A float datatype predefined by libhdf5 2.x besides the IEEE ones.
|
||||
struct SmallFloat {
|
||||
/// h5dump's name (`H5T_FLOAT_F8E4M3`).
|
||||
ddl: &'static str,
|
||||
/// h5ls's description (`FP8 E4M3 8-bit float`).
|
||||
long: &'static str,
|
||||
/// The short name `ls` lists (`float8-e4m3`).
|
||||
short: &'static str,
|
||||
}
|
||||
|
||||
/// The libhdf5 2.x predefined float `dt` is equal to (as `H5Tequal` sees it:
|
||||
/// the same size, byte order, precision, offset, fields and bias), if any.
|
||||
/// The 1-byte types are predefined little-endian only.
|
||||
fn small_float(dt: &Datatype) -> Option<SmallFloat> {
|
||||
let Datatype::FloatingPoint {
|
||||
size,
|
||||
byte_order,
|
||||
bit_offset,
|
||||
bit_precision,
|
||||
exponent_location,
|
||||
exponent_size,
|
||||
mantissa_location,
|
||||
mantissa_size,
|
||||
exponent_bias,
|
||||
} = dt
|
||||
else {
|
||||
return None;
|
||||
};
|
||||
if *bit_offset != 0 || *mantissa_location != 0 {
|
||||
return None;
|
||||
}
|
||||
let f = |ddl, long, short| Some(SmallFloat { ddl, long, short });
|
||||
let fields = (
|
||||
*size,
|
||||
*bit_precision,
|
||||
*exponent_location,
|
||||
*exponent_size,
|
||||
*mantissa_size,
|
||||
*exponent_bias,
|
||||
);
|
||||
match (fields, byte_order) {
|
||||
((2, 16, 7, 8, 7, 127), DatatypeByteOrder::LittleEndian) => f(
|
||||
"H5T_FLOAT_BFLOAT16LE",
|
||||
"bfloat16 16-bit little-endian float",
|
||||
"bfloat16",
|
||||
),
|
||||
((2, 16, 7, 8, 7, 127), DatatypeByteOrder::BigEndian) => f(
|
||||
"H5T_FLOAT_BFLOAT16BE",
|
||||
"bfloat16 16-bit big-endian float",
|
||||
"bfloat16-be",
|
||||
),
|
||||
((1, 8, 3, 4, 3, 7), DatatypeByteOrder::LittleEndian) => {
|
||||
f("H5T_FLOAT_F8E4M3", "FP8 E4M3 8-bit float", "float8-e4m3")
|
||||
}
|
||||
((1, 8, 2, 5, 2, 15), DatatypeByteOrder::LittleEndian) => {
|
||||
f("H5T_FLOAT_F8E5M2", "FP8 E5M2 8-bit float", "float8-e5m2")
|
||||
}
|
||||
((1, 6, 3, 2, 3, 1), DatatypeByteOrder::LittleEndian) => {
|
||||
f("H5T_FLOAT_F6E2M3", "FP6 E2M3 6-bit float", "float6-e2m3")
|
||||
}
|
||||
((1, 6, 2, 3, 2, 3), DatatypeByteOrder::LittleEndian) => {
|
||||
f("H5T_FLOAT_F6E3M2", "FP6 E3M2 6-bit float", "float6-e3m2")
|
||||
}
|
||||
((1, 4, 1, 2, 1, 1), DatatypeByteOrder::LittleEndian) => {
|
||||
f("H5T_FLOAT_F4E2M1", "FP4 E2M1 4-bit float", "float4-e2m1")
|
||||
}
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
/// Short name used by `ls`: `int32`, `float64-be`, `string[3]`, ...
|
||||
pub fn short(dt: &Datatype) -> String {
|
||||
if let Some(f) = small_float(dt) {
|
||||
return f.short.into();
|
||||
}
|
||||
match dt {
|
||||
Datatype::FixedPoint {
|
||||
size,
|
||||
@@ -136,6 +209,9 @@ fn cset_word(c: &CharacterSet) -> &'static str {
|
||||
|
||||
/// h5ls -v style description.
|
||||
pub fn long(dt: &Datatype) -> String {
|
||||
if let Some(f) = small_float(dt) {
|
||||
return f.long.into();
|
||||
}
|
||||
match dt {
|
||||
Datatype::FixedPoint {
|
||||
size,
|
||||
@@ -277,6 +353,8 @@ fn atomic_ddl(dt: &Datatype) -> Option<String> {
|
||||
u64::from(*size) * 8,
|
||||
order_suffix(byte_order)
|
||||
)),
|
||||
// As h5dump 2.x names them (checked against h5dump 2.2.0).
|
||||
Datatype::FloatingPoint { .. } => small_float(dt).map(|f| f.ddl.to_string()),
|
||||
Datatype::BitField {
|
||||
size, byte_order, ..
|
||||
} => Some(format!(
|
||||
|
||||
Reference in New Issue
Block a user