feat: check HDF5 2.x small floats against libhdf5 2.2.0
CI / test-arm64 (pull_request) Successful in 1m38s
CI / test (pull_request) Successful in 31m22s

Fixture written by libhdf5 2.2.0 (built from tag 2.2.0) through ctypes:
every bit pattern of FP4 E2M1, FP6 E2M3/E3M2, FP8 E4M3/E5M2 and a
bfloat16 LE/BE set, as datasets and attributes, with what H5Dread/H5Aread
return into double and float and the conversion exceptions libhdf5
raises. clawhdf5 already decoded every value as libhdf5 does, including
an all-ones exponent as inf/NaN in the OCP formats that have none
(documented as a deliberate match in known-issues).

- data_read: NaNs of non-native float layouts get libhdf5's bits (sign
  kept, every mantissa bit set) in f64 and f32.
- h5rs dump/ls name these types as h5dump/h5ls 2.x do
  (H5T_FLOAT_F4E2M1, "FP4 E2M1 4-bit float", float4-e2m1 ...), checked
  against h5dump 2.2.0's output of the fixture.
- Python bindings read them as h5py 3.16 does (float32 for bfloat16,
  float16 for the 1-byte formats, file byte order, same bytes as h5py);
  writing them in 'r+' is refused.

Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
osobh
2026-09-28 21:25:03 -05:00
co-authored by Claude Opus 5.5
parent 31efac1ae1
commit bce07e9cb9
16 changed files with 1207 additions and 13 deletions
+10 -1
View File
@@ -109,7 +109,16 @@ virtual datasets. Known differences from
h5dump:
- Floats print at their own precision (a `float32` 0.1 prints as `0.1`),
which for some values is more digits than h5dump's `%g`.
which for some values is more digits than h5dump's `%g`; non-finite
values print as `Inf`, `-Inf` and `NaN` (h5dump: `inf`, `-inf`, `nan`,
`-nan`).
- HDF5 2.x's predefined small floats print under h5dump 2.x's names
(`H5T_FLOAT_BFLOAT16LE`, `H5T_FLOAT_F8E4M3`, ..., `H5T_FLOAT_F4E2M1`;
`ls -v` as h5ls 2.x does: `FP4 E2M1 4-bit float`).
`tests/mx_floats_dump.rs` checks this, and the values, against the output
of h5dump 2.2.0 stored with the fixture. h5dump 1.14 describes these types
instead (`8-bit floating-point 4-bit precision`); other non-IEEE floats
print as an `H5T_FLOAT { ... }` block.
- A compound nested in a compound prints inline (`{ 1, 2.5 }`) where
h5dump prints it as an indented block, one member per line; only the
outer compound is a block.
+78
View File
@@ -59,8 +59,81 @@ pub fn is_ieee(dt: &Datatype) -> bool {
) == std
}
/// A float datatype predefined by libhdf5 2.x besides the IEEE ones.
struct SmallFloat {
/// h5dump's name (`H5T_FLOAT_F8E4M3`).
ddl: &'static str,
/// h5ls's description (`FP8 E4M3 8-bit float`).
long: &'static str,
/// The short name `ls` lists (`float8-e4m3`).
short: &'static str,
}
/// The libhdf5 2.x predefined float `dt` is equal to (as `H5Tequal` sees it:
/// the same size, byte order, precision, offset, fields and bias), if any.
/// The 1-byte types are predefined little-endian only.
fn small_float(dt: &Datatype) -> Option<SmallFloat> {
let Datatype::FloatingPoint {
size,
byte_order,
bit_offset,
bit_precision,
exponent_location,
exponent_size,
mantissa_location,
mantissa_size,
exponent_bias,
} = dt
else {
return None;
};
if *bit_offset != 0 || *mantissa_location != 0 {
return None;
}
let f = |ddl, long, short| Some(SmallFloat { ddl, long, short });
let fields = (
*size,
*bit_precision,
*exponent_location,
*exponent_size,
*mantissa_size,
*exponent_bias,
);
match (fields, byte_order) {
((2, 16, 7, 8, 7, 127), DatatypeByteOrder::LittleEndian) => f(
"H5T_FLOAT_BFLOAT16LE",
"bfloat16 16-bit little-endian float",
"bfloat16",
),
((2, 16, 7, 8, 7, 127), DatatypeByteOrder::BigEndian) => f(
"H5T_FLOAT_BFLOAT16BE",
"bfloat16 16-bit big-endian float",
"bfloat16-be",
),
((1, 8, 3, 4, 3, 7), DatatypeByteOrder::LittleEndian) => {
f("H5T_FLOAT_F8E4M3", "FP8 E4M3 8-bit float", "float8-e4m3")
}
((1, 8, 2, 5, 2, 15), DatatypeByteOrder::LittleEndian) => {
f("H5T_FLOAT_F8E5M2", "FP8 E5M2 8-bit float", "float8-e5m2")
}
((1, 6, 3, 2, 3, 1), DatatypeByteOrder::LittleEndian) => {
f("H5T_FLOAT_F6E2M3", "FP6 E2M3 6-bit float", "float6-e2m3")
}
((1, 6, 2, 3, 2, 3), DatatypeByteOrder::LittleEndian) => {
f("H5T_FLOAT_F6E3M2", "FP6 E3M2 6-bit float", "float6-e3m2")
}
((1, 4, 1, 2, 1, 1), DatatypeByteOrder::LittleEndian) => {
f("H5T_FLOAT_F4E2M1", "FP4 E2M1 4-bit float", "float4-e2m1")
}
_ => None,
}
}
/// Short name used by `ls`: `int32`, `float64-be`, `string[3]`, ...
pub fn short(dt: &Datatype) -> String {
if let Some(f) = small_float(dt) {
return f.short.into();
}
match dt {
Datatype::FixedPoint {
size,
@@ -136,6 +209,9 @@ fn cset_word(c: &CharacterSet) -> &'static str {
/// h5ls -v style description.
pub fn long(dt: &Datatype) -> String {
if let Some(f) = small_float(dt) {
return f.long.into();
}
match dt {
Datatype::FixedPoint {
size,
@@ -277,6 +353,8 @@ fn atomic_ddl(dt: &Datatype) -> Option<String> {
u64::from(*size) * 8,
order_suffix(byte_order)
)),
// As h5dump 2.x names them (checked against h5dump 2.2.0).
Datatype::FloatingPoint { .. } => small_float(dt).map(|f| f.ddl.to_string()),
Datatype::BitField {
size, byte_order, ..
} => Some(format!(
@@ -0,0 +1,111 @@
//! `h5rs dump` and `h5rs ls` of the small floats libhdf5 2.x predefines
//! (bfloat16, FP8 E4M3/E5M2, FP6 E2M3/E3M2, FP4 E2M1), against the output
//! of h5dump 2.2.0 and h5ls 2.2.0 stored next to the fixture (the Debian
//! h5dump CI installs, 1.14.x, predates these types). See
//! `crates/clawhdf5/tests/fixtures/gen_mx_floats.py`.
use std::path::PathBuf;
use std::process::Command;
fn fixture(name: &str) -> PathBuf {
PathBuf::from(env!("CARGO_MANIFEST_DIR"))
.join("../clawhdf5/tests/fixtures")
.join(name)
}
fn h5rs(args: &[&str]) -> String {
let out = Command::new(env!("CARGO_BIN_EXE_h5rs"))
.args(args)
.output()
.unwrap();
assert!(out.status.success(), "h5rs {args:?}: {out:?}");
String::from_utf8(out.stdout).unwrap()
}
/// Split a dump into the lines outside `DATA { ... }` blocks and the values
/// inside them, per block.
fn split(ddl: &str) -> (Vec<&str>, Vec<Vec<String>>) {
let (mut frame, mut data) = (Vec::new(), Vec::new());
let mut values: Option<Vec<String>> = None;
for line in ddl.lines() {
match &mut values {
None if line.trim() == "DATA {" => values = Some(Vec::new()),
None => frame.push(line),
Some(v) if line.trim() == "}" => {
data.push(std::mem::take(v));
values = None;
}
Some(v) => {
let (_, body) = line.split_once("):").expect("an indexed data line");
v.extend(
body.split(',')
.map(str::trim)
.filter(|s| !s.is_empty())
.map(String::from),
);
}
}
}
(frame, data)
}
#[test]
fn dump_matches_h5dump_2_2() {
let path = fixture("mx_floats_hdf5_2_2.h5");
let ours = h5rs(&["dump", path.to_str().unwrap()]);
let reference = std::fs::read_to_string(fixture("mx_floats_hdf5_2_2.ddl")).unwrap();
let (our_frame, our_data) = split(&ours);
let (ref_frame, ref_data) = split(&reference);
// Everything but the values is h5dump's byte for byte: the datatypes
// print as H5T_FLOAT_BFLOAT16LE, H5T_FLOAT_F4E2M1, ...
assert_eq!(our_frame, ref_frame);
assert_eq!(our_data.len(), 17);
assert_eq!(our_data.len(), ref_data.len());
// Values: h5dump prints `%g` (6 significant digits) and inf/-inf/nan/
// -nan; h5rs prints the shortest string that round-trips and Inf/-Inf/NaN.
for (block, (ours, theirs)) in our_data.iter().zip(&ref_data).enumerate() {
assert_eq!(ours.len(), theirs.len(), "block {block}");
for (i, (o, r)) in ours.iter().zip(theirs).enumerate() {
let o: f64 = o.parse().unwrap();
let r: f64 = r.parse().unwrap();
let same = if r.is_nan() || r.is_infinite() {
o.is_nan() == r.is_nan() && (r.is_nan() || o == r)
} else {
// Both round to the same f32 (a bfloat16 subnormal prints
// as its shortest f32 string, `9.1835e-41`, fewer digits
// than h5dump's), or agree to h5dump's 6 digits.
o as f32 == r as f32 || (o - r).abs() <= 5e-6 * r.abs()
};
assert!(same, "block {block}[{i}]: h5rs {o}, h5dump {r}");
}
}
}
#[test]
fn ls_names_small_floats_like_h5ls_2_2() {
let path = fixture("mx_floats_hdf5_2_2.h5");
// h5ls 2.2.0 -v, `Type:` line of each dataset.
for (name, long, short) in [
("bf16le", "bfloat16 16-bit little-endian float", "bfloat16"),
("bf16be", "bfloat16 16-bit big-endian float", "bfloat16-be"),
("f8e4m3", "FP8 E4M3 8-bit float", "float8-e4m3"),
("f8e5m2", "FP8 E5M2 8-bit float", "float8-e5m2"),
("f6e2m3", "FP6 E2M3 6-bit float", "float6-e2m3"),
("f6e3m2", "FP6 E3M2 6-bit float", "float6-e3m2"),
("f4e2m1", "FP4 E2M1 4-bit float", "float4-e2m1"),
] {
let target = format!("{}/{name}", path.display());
let verbose = h5rs(&["ls", "-v", &target]);
assert!(
verbose.contains(&format!(" Type: {long}\n")),
"{name}:\n{verbose}"
);
let listing = h5rs(&["ls", path.to_str().unwrap()]);
assert!(
listing
.lines()
.any(|l| l.starts_with(&format!("{name} ")) && l.ends_with(&format!(" {short}"))),
"{name}:\n{listing}"
);
}
}