feat: check HDF5 2.x small floats against libhdf5 2.2.0
Fixture written by libhdf5 2.2.0 (built from tag 2.2.0) through ctypes: every bit pattern of FP4 E2M1, FP6 E2M3/E3M2, FP8 E4M3/E5M2 and a bfloat16 LE/BE set, as datasets and attributes, with what H5Dread/H5Aread return into double and float and the conversion exceptions libhdf5 raises. clawhdf5 already decoded every value as libhdf5 does, including an all-ones exponent as inf/NaN in the OCP formats that have none (documented as a deliberate match in known-issues). - data_read: NaNs of non-native float layouts get libhdf5's bits (sign kept, every mantissa bit set) in f64 and f32. - h5rs dump/ls name these types as h5dump/h5ls 2.x do (H5T_FLOAT_F4E2M1, "FP4 E2M1 4-bit float", float4-e2m1 ...), checked against h5dump 2.2.0's output of the fixture. - Python bindings read them as h5py 3.16 does (float32 for bfloat16, float16 for the 1-byte formats, file byte order, same bytes as h5py); writing them in 'r+' is refused. Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
@@ -0,0 +1,111 @@
|
||||
//! `h5rs dump` and `h5rs ls` of the small floats libhdf5 2.x predefines
|
||||
//! (bfloat16, FP8 E4M3/E5M2, FP6 E2M3/E3M2, FP4 E2M1), against the output
|
||||
//! of h5dump 2.2.0 and h5ls 2.2.0 stored next to the fixture (the Debian
|
||||
//! h5dump CI installs, 1.14.x, predates these types). See
|
||||
//! `crates/clawhdf5/tests/fixtures/gen_mx_floats.py`.
|
||||
|
||||
use std::path::PathBuf;
|
||||
use std::process::Command;
|
||||
|
||||
fn fixture(name: &str) -> PathBuf {
|
||||
PathBuf::from(env!("CARGO_MANIFEST_DIR"))
|
||||
.join("../clawhdf5/tests/fixtures")
|
||||
.join(name)
|
||||
}
|
||||
|
||||
fn h5rs(args: &[&str]) -> String {
|
||||
let out = Command::new(env!("CARGO_BIN_EXE_h5rs"))
|
||||
.args(args)
|
||||
.output()
|
||||
.unwrap();
|
||||
assert!(out.status.success(), "h5rs {args:?}: {out:?}");
|
||||
String::from_utf8(out.stdout).unwrap()
|
||||
}
|
||||
|
||||
/// Split a dump into the lines outside `DATA { ... }` blocks and the values
|
||||
/// inside them, per block.
|
||||
fn split(ddl: &str) -> (Vec<&str>, Vec<Vec<String>>) {
|
||||
let (mut frame, mut data) = (Vec::new(), Vec::new());
|
||||
let mut values: Option<Vec<String>> = None;
|
||||
for line in ddl.lines() {
|
||||
match &mut values {
|
||||
None if line.trim() == "DATA {" => values = Some(Vec::new()),
|
||||
None => frame.push(line),
|
||||
Some(v) if line.trim() == "}" => {
|
||||
data.push(std::mem::take(v));
|
||||
values = None;
|
||||
}
|
||||
Some(v) => {
|
||||
let (_, body) = line.split_once("):").expect("an indexed data line");
|
||||
v.extend(
|
||||
body.split(',')
|
||||
.map(str::trim)
|
||||
.filter(|s| !s.is_empty())
|
||||
.map(String::from),
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
(frame, data)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn dump_matches_h5dump_2_2() {
|
||||
let path = fixture("mx_floats_hdf5_2_2.h5");
|
||||
let ours = h5rs(&["dump", path.to_str().unwrap()]);
|
||||
let reference = std::fs::read_to_string(fixture("mx_floats_hdf5_2_2.ddl")).unwrap();
|
||||
let (our_frame, our_data) = split(&ours);
|
||||
let (ref_frame, ref_data) = split(&reference);
|
||||
// Everything but the values is h5dump's byte for byte: the datatypes
|
||||
// print as H5T_FLOAT_BFLOAT16LE, H5T_FLOAT_F4E2M1, ...
|
||||
assert_eq!(our_frame, ref_frame);
|
||||
assert_eq!(our_data.len(), 17);
|
||||
assert_eq!(our_data.len(), ref_data.len());
|
||||
// Values: h5dump prints `%g` (6 significant digits) and inf/-inf/nan/
|
||||
// -nan; h5rs prints the shortest string that round-trips and Inf/-Inf/NaN.
|
||||
for (block, (ours, theirs)) in our_data.iter().zip(&ref_data).enumerate() {
|
||||
assert_eq!(ours.len(), theirs.len(), "block {block}");
|
||||
for (i, (o, r)) in ours.iter().zip(theirs).enumerate() {
|
||||
let o: f64 = o.parse().unwrap();
|
||||
let r: f64 = r.parse().unwrap();
|
||||
let same = if r.is_nan() || r.is_infinite() {
|
||||
o.is_nan() == r.is_nan() && (r.is_nan() || o == r)
|
||||
} else {
|
||||
// Both round to the same f32 (a bfloat16 subnormal prints
|
||||
// as its shortest f32 string, `9.1835e-41`, fewer digits
|
||||
// than h5dump's), or agree to h5dump's 6 digits.
|
||||
o as f32 == r as f32 || (o - r).abs() <= 5e-6 * r.abs()
|
||||
};
|
||||
assert!(same, "block {block}[{i}]: h5rs {o}, h5dump {r}");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ls_names_small_floats_like_h5ls_2_2() {
|
||||
let path = fixture("mx_floats_hdf5_2_2.h5");
|
||||
// h5ls 2.2.0 -v, `Type:` line of each dataset.
|
||||
for (name, long, short) in [
|
||||
("bf16le", "bfloat16 16-bit little-endian float", "bfloat16"),
|
||||
("bf16be", "bfloat16 16-bit big-endian float", "bfloat16-be"),
|
||||
("f8e4m3", "FP8 E4M3 8-bit float", "float8-e4m3"),
|
||||
("f8e5m2", "FP8 E5M2 8-bit float", "float8-e5m2"),
|
||||
("f6e2m3", "FP6 E2M3 6-bit float", "float6-e2m3"),
|
||||
("f6e3m2", "FP6 E3M2 6-bit float", "float6-e3m2"),
|
||||
("f4e2m1", "FP4 E2M1 4-bit float", "float4-e2m1"),
|
||||
] {
|
||||
let target = format!("{}/{name}", path.display());
|
||||
let verbose = h5rs(&["ls", "-v", &target]);
|
||||
assert!(
|
||||
verbose.contains(&format!(" Type: {long}\n")),
|
||||
"{name}:\n{verbose}"
|
||||
);
|
||||
let listing = h5rs(&["ls", path.to_str().unwrap()]);
|
||||
assert!(
|
||||
listing
|
||||
.lines()
|
||||
.any(|l| l.starts_with(&format!("{name} ")) && l.ends_with(&format!(" {short}"))),
|
||||
"{name}:\n{listing}"
|
||||
);
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user