feat: check HDF5 2.x small floats against libhdf5 2.2.0
CI / test-arm64 (pull_request) Successful in 1m38s
CI / test (pull_request) Successful in 31m22s

Fixture written by libhdf5 2.2.0 (built from tag 2.2.0) through ctypes:
every bit pattern of FP4 E2M1, FP6 E2M3/E3M2, FP8 E4M3/E5M2 and a
bfloat16 LE/BE set, as datasets and attributes, with what H5Dread/H5Aread
return into double and float and the conversion exceptions libhdf5
raises. clawhdf5 already decoded every value as libhdf5 does, including
an all-ones exponent as inf/NaN in the OCP formats that have none
(documented as a deliberate match in known-issues).

- data_read: NaNs of non-native float layouts get libhdf5's bits (sign
  kept, every mantissa bit set) in f64 and f32.
- h5rs dump/ls name these types as h5dump/h5ls 2.x do
  (H5T_FLOAT_F4E2M1, "FP4 E2M1 4-bit float", float4-e2m1 ...), checked
  against h5dump 2.2.0's output of the fixture.
- Python bindings read them as h5py 3.16 does (float32 for bfloat16,
  float16 for the 1-byte formats, file byte order, same bytes as h5py);
  writing them in 'r+' is refused.

Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
osobh
2026-09-28 21:25:03 -05:00
co-authored by Claude Opus 5.5
parent 31efac1ae1
commit bce07e9cb9
16 changed files with 1207 additions and 13 deletions
+66 -5
View File
@@ -1419,9 +1419,11 @@ pub fn read_as_f32(raw: &[u8], datatype: &Datatype) -> Result<Vec<f32>, FormatEr
result.push(match format {
FloatFormat::Single => read_f32_bytes(chunk, &order),
FloatFormat::Half => read_f16_bytes(chunk, &order),
// Double rounds; every other supported layout (bfloat16, FP8)
// is exact in f32.
_ => format.decode(chunk, &order) as f32,
// Double rounds.
FloatFormat::Double => format.decode(chunk, &order) as f32,
// Every other supported layout (bfloat16, FP8, FP6, FP4) is
// exact in f32.
FloatFormat::Other(_) => narrow_decoded(format.decode(chunk, &order)),
});
}
return Ok(result);
@@ -1947,7 +1949,12 @@ enum FloatFormat {
Double,
/// Any other IEEE-style layout (implied leading mantissa bit, all-ones
/// exponent for infinity/NaN) whose values are all exact in `f64`:
/// bfloat16, the FP8 formats, and similar.
/// bfloat16, FP8 E4M3/E5M2, FP6 E2M3/E3M2, FP4 E2M1, and similar.
///
/// libhdf5 (checked against 2.2.0) decodes all of them this way, also the
/// OCP MX formats whose specification has no infinity (FP6, FP4) or a
/// single NaN (FP8 E4M3): an all-ones exponent is infinity or NaN, not a
/// finite value. clawhdf5 follows libhdf5 so both read a file alike.
Other(FloatLayout),
}
@@ -2046,7 +2053,9 @@ impl FloatLayout {
if mantissa == 0 {
f64::INFINITY
} else {
f64::NAN
// The NaN libhdf5 converts every NaN to: all mantissa bits
// set (the sign is applied below).
LIBHDF5_NAN
}
} else {
let bias = i64::from(self.exponent_bias);
@@ -2067,6 +2076,27 @@ impl FloatLayout {
}
}
/// The `f64` NaN libhdf5's conversion (`H5T__conv_f_f`) produces from a NaN
/// of a non-native float layout: sign clear, every mantissa bit set.
const LIBHDF5_NAN: f64 = f64::from_bits(0x7FFF_FFFF_FFFF_FFFF);
/// Narrow a value decoded from a non-native float layout to `f32`. Every such
/// value is exact in `f32`; a NaN becomes the NaN libhdf5 gives for
/// `H5T_NATIVE_FLOAT` (sign kept, every mantissa bit set) rather than
/// whatever payload an `as` cast leaves.
fn narrow_decoded(value: f64) -> f32 {
if value.is_nan() {
let sign = if value.is_sign_negative() {
1u32 << 31
} else {
0
};
f32::from_bits(sign | 0x7FFF_FFFF)
} else {
value as f32
}
}
/// `x * 2^power` without `std` (no `powi`/`libm`). `x` is a non-negative
/// integer below 2^53, so it is exact.
fn scale_by_pow2(x: f64, power: i64) -> f64 {
@@ -2425,6 +2455,37 @@ mod tests {
assert!(got[4].is_nan());
}
#[test]
fn fp4_decodes_as_libhdf5_does() {
// H5T_FLOAT_F4E2M1: 4 significant bits in a byte; the high bits are
// padding. libhdf5 2.2.0 reads 0b0110 as +inf and 0b0111 as NaN
// (IEEE-style, although OCP MX FP4 has neither), and returns NaNs
// with every mantissa bit set.
let fp4 = Datatype::FloatingPoint {
size: 1,
byte_order: DatatypeByteOrder::LittleEndian,
bit_offset: 0,
bit_precision: 4,
exponent_location: 1,
exponent_size: 2,
mantissa_location: 0,
mantissa_size: 1,
exponent_bias: 1,
};
let raw = [0x01, 0x05, 0xF5, 0x06, 0x0E, 0x07, 0x0F];
let got = read_as_f64(&raw, &fp4).unwrap();
assert_eq!(
&got[..5],
&[0.5, 3.0, 3.0, f64::INFINITY, f64::NEG_INFINITY]
);
assert_eq!(got[5].to_bits(), 0x7FFF_FFFF_FFFF_FFFF);
assert_eq!(got[6].to_bits(), 0xFFFF_FFFF_FFFF_FFFF);
let got = read_as_f32(&raw, &fp4).unwrap();
assert_eq!(&got[..3], &[0.5, 3.0, 3.0]);
assert_eq!(got[5].to_bits(), 0x7FFF_FFFF);
assert_eq!(got[6].to_bits(), 0xFFFF_FFFF);
}
#[test]
fn full_width_signed_unchanged() {
// Regression: full-width 32-bit signed must be unaffected.
+6 -1
View File
@@ -37,7 +37,12 @@ with clawhdf5.File("data.h5", "r") as f:
`dtype.metadata['enum']`), complex, `S<n>` fixed strings, `object` for
variable-length strings (`bytes` values) and sequences (array values),
`V<n>` opaque, array types, and compounds as structured dtypes.
Other types raise `TypeError`.
Non-IEEE floats (HDF5 2.x's bfloat16, FP8, FP6 and FP4) read as h5py
3.16 reads them: as the narrowest IEEE float that holds them (`float32`
for bfloat16, `float16` for the 1-byte formats) in the file's byte order,
with libhdf5's values (`tests/test_small_floats.py`); inside a compound or
array type they raise `TypeError`, and writing them raises
`NotImplementedError`. Other types raise `TypeError`.
- Keys are h5py's: integers, slices with a positive step, `...`, one
increasing list of integers, compound field names. Each maps onto a
hyperslab selection. `None` and negative steps are refused
+117 -4
View File
@@ -8,10 +8,16 @@
//! returns become the array's buffer as they are: the `Vec<u8>` is handed to
//! numpy without a copy and viewed as the dtype.
//!
//! Anything this mapping cannot describe exactly — non-IEEE floats, integers
//! with padding bits, VAX byte order, references, bitfields, time, and
//! variable-length members inside compounds or arrays — is a `TypeError`,
//! never a best-effort guess.
//! The one exception is a dataset or attribute of a non-IEEE float
//! (bfloat16, FP8 E4M3/E5M2, FP6, FP4, ...): like h5py, it reads as the
//! narrowest of `float16`/`float32`/`float64` that holds every value of the
//! format exactly, in the file's byte order, and the values are converted
//! (as libhdf5 converts them, NaN bits included).
//!
//! Anything this mapping cannot describe exactly — non-IEEE floats inside
//! compounds or arrays, integers with padding bits, VAX byte order,
//! references, bitfields, time, and variable-length members inside compounds
//! or arrays — is a `TypeError`, never a best-effort guess.
use std::collections::HashMap;
@@ -35,6 +41,13 @@ pub(crate) enum Layout {
VlString { utf8: bool },
/// Variable-length sequence of a fixed-size base type.
VlSequence,
/// A non-IEEE float (`source`), converted to the IEEE float of `size`
/// bytes numpy reports (see [`widened_float`]).
Float {
source: Datatype,
size: usize,
big_endian: bool,
},
}
/// Everything needed to turn a dataset's or attribute's bytes into numpy.
@@ -139,6 +152,74 @@ fn float_format(dt: &Datatype) -> PyResult<String> {
Ok(format!("{}f{size}", byte_order_char(byte_order, *size)?))
}
/// Whether `dt` is an IEEE 754 binary16/32/64 float numpy reads as it is.
pub(crate) fn is_ieee_float(dt: &Datatype) -> bool {
float_format(dt).is_ok()
}
/// The IEEE float h5py reads a non-IEEE float as (`TypeFloatID.py_dtype` in
/// h5py 3.16): the first of `float16`, `float32`, `float64` at least as wide
/// whose mantissa holds the format's and whose normal exponent range covers
/// it — bfloat16 as `float32`, FP8/FP6/FP4 as `float16` — in the file's byte
/// order. Returns `(size in bytes, big endian)`, or `None` when no IEEE type
/// holds it (or the byte order is VAX).
fn widened_float(dt: &Datatype) -> Option<(usize, bool)> {
let Datatype::FloatingPoint {
size,
byte_order,
exponent_size,
mantissa_size,
exponent_bias,
..
} = dt
else {
return None;
};
let big_endian = match byte_order {
DatatypeByteOrder::LittleEndian => false,
DatatypeByteOrder::BigEndian => true,
DatatypeByteOrder::Vax => return None,
};
if *exponent_size >= 32 {
return None;
}
let max_exp = (1i64 << exponent_size) - i64::from(*exponent_bias) - 1;
let min_exp = 1 - i64::from(*exponent_bias);
// numpy's finfo: (itemsize, nmant, maxexp, minexp).
[(2, 10, 16, -14), (4, 23, 128, -126), (8, 52, 1024, -1022)]
.into_iter()
.find(|&(bytes, nmant, maxexp, minexp)| {
bytes >= *size && *mantissa_size <= nmant && max_exp <= maxexp && min_exp >= minexp
})
.map(|(bytes, ..)| (bytes as usize, big_endian))
}
/// Encode `values` as IEEE floats of `size` bytes. Every value is exact in
/// the target (`widened_float` chose it so); a NaN keeps its sign and gets
/// every mantissa bit set, as libhdf5 converts NaNs.
fn encode_floats(values: &[f64], size: usize, big_endian: bool) -> Vec<u8> {
let mut out = Vec::with_capacity(values.len() * size);
for &v in values {
let bits: u64 = if v.is_nan() {
let sign = u64::from(v.is_sign_negative()) << (size * 8 - 1);
sign | ((1u64 << (size * 8 - 1)) - 1)
} else {
match size {
2 => u64::from(clawhdf5_format::float16::f32_to_f16_bits(v as f32)),
4 => u64::from((v as f32).to_bits()),
_ => v.to_bits(),
}
};
let bytes = bits.to_le_bytes();
if big_endian {
out.extend(bytes[..size].iter().rev());
} else {
out.extend_from_slice(&bytes[..size]);
}
}
out
}
/// `r`/`i` compounds of two identical IEEE floats are complex numbers in h5py.
fn complex_format(
size: u32,
@@ -366,6 +447,25 @@ impl Converter {
vl_unit: 0,
})
}
Datatype::FloatingPoint { .. } if !is_ieee_float(dt) => {
let Some((size, big_endian)) = widened_float(dt) else {
// The error names the layout.
return Err(float_format(dt).expect_err("not IEEE"));
};
let order = if big_endian { '>' } else { '<' };
let dtype = np_dtype(py, format!("{order}f{size}"))?;
Ok(Self {
view: dtype.clone().unbind(),
dtype: dtype.unbind(),
layout: Layout::Float {
source: dt.clone(),
size,
big_endian,
},
elem_size: dt.type_size() as usize,
vl_unit: 0,
})
}
_ => {
let dtype = fixed_dtype(py, dt)?;
Ok(Self {
@@ -416,6 +516,19 @@ impl Converter {
(Elements::Bytes(bytes), Layout::Fixed) => {
bytes_as_array(py, bytes, self.view.bind(py), shape)
}
(
Elements::Bytes(bytes),
Layout::Float {
source,
size,
big_endian,
},
) => {
let values = clawhdf5_format::data_read::read_as_f64(&bytes, source)
.map_err(|e| unsupported(e.to_string()))?;
let converted = encode_floats(&values, *size, *big_endian);
bytes_as_array(py, converted, self.view.bind(py), shape)
}
(Elements::Bytes(bytes), Layout::Subarray(dims)) => {
let mut full = shape.to_vec();
full.extend_from_slice(dims);
+6 -1
View File
@@ -48,7 +48,12 @@ fn not_implemented(what: impl std::fmt::Display) -> PyErr {
pub(crate) fn category(dt: &Datatype) -> PyResult<&'static str> {
match dt {
Datatype::FixedPoint { .. } => Ok("int"),
Datatype::FloatingPoint { .. } => Ok("float"),
Datatype::FloatingPoint { .. } if crate::convert::is_ieee_float(dt) => Ok("float"),
// Read as a wider IEEE float; writing would need the reverse
// conversion (rounding into bfloat16, FP8, ...).
Datatype::FloatingPoint { .. } => Err(not_implemented(
"writing non-IEEE floats (bfloat16, FP8, FP6, FP4, ...)",
)),
Datatype::Enumeration {
base_type, members, ..
} => {
@@ -0,0 +1,92 @@
"""Non-IEEE floats (bfloat16, FP8 E4M3/E5M2, FP6 E2M3/E3M2, FP4 E2M1) read
as h5py 3.16 reads them: as the narrowest IEEE float that holds every value
(bfloat16 as float32, the 1-byte formats as float16), in the file's byte
order, with the values libhdf5 converts them to.
The fixture was written by libhdf5 2.2.0; the JSON next to it holds what
libhdf5 2.2.0 itself returns for every element (see gen_mx_floats.py)."""
import json
import os
import shutil
import numpy as np
import pytest
import clawhdf5
FIXTURES = os.path.join(os.path.dirname(__file__), "..", "..", "clawhdf5", "tests", "fixtures")
FILE = os.path.join(FIXTURES, "mx_floats_hdf5_2_2.h5")
REFERENCE = json.load(open(os.path.join(FIXTURES, "mx_floats_hdf5_2_2.json")))["objects"]
# What h5py 3.16 reports for each (checked 2026-09-28).
DTYPES = {
"bf16le": "<f4",
"bf16be": ">f4",
"f8e4m3": "<f2",
"f8e5m2": "<f2",
"f6e2m3": "<f2",
"f6e2m3_pad": "<f2",
"f6e3m2": "<f2",
"f6e3m2_pad": "<f2",
"f4e2m1": "<f2",
"f4e2m1_pad": "<f2",
}
def _expected(name, dtype):
"""The libhdf5 2.2.0 values as `dtype`, NaN bits included (sign kept,
every mantissa bit set)."""
out = []
for text in REFERENCE[name]["f64"]:
if text.startswith("nan:"):
negative = int(text[4:], 16) >> 63
bits = dtype.itemsize * 8
word = (negative << (bits - 1)) | ((1 << (bits - 1)) - 1)
out.append(np.frombuffer(word.to_bytes(dtype.itemsize, "little"), dtype.newbyteorder("<"))[0])
else:
out.append(float(text))
return np.array(out, dtype=dtype.newbyteorder("<")).astype(dtype)
def test_values_match_libhdf5_2_2():
assert set(DTYPES) == set(REFERENCE)
with clawhdf5.File(FILE, "r") as f:
for name, dtype in DTYPES.items():
ds = f[name]
assert ds.dtype == np.dtype(dtype), name
assert ds.dtype.str == dtype, name
want = _expected(name, np.dtype(dtype))
got = ds[()]
assert got.dtype.str == dtype
assert got.tobytes() == want.tobytes(), name
# Selections convert the same way.
assert ds[3:9].tobytes() == want[3:9].tobytes(), name
assert ds[[0, 5, 7]].tobytes() == want[[0, 5, 7]].tobytes(), name
assert ds[5].tobytes() == want[5].tobytes(), name
if REFERENCE[name]["attribute"]:
attr = f.attrs[name]
assert attr.dtype.str == dtype
assert attr.tobytes() == want.tobytes(), name
def test_matches_h5py(h5py):
with clawhdf5.File(FILE, "r") as ours, h5py.File(FILE, "r") as theirs:
for name in DTYPES:
assert ours[name].dtype == theirs[name].dtype, name
assert ours[name][()].tobytes() == theirs[name][()].tobytes(), name
if name in theirs.attrs:
assert ours.attrs[name].dtype == theirs.attrs[name].dtype
assert ours.attrs[name].tobytes() == theirs.attrs[name].tobytes()
def test_writing_is_refused(tmp_path):
path = tmp_path / "copy.h5"
shutil.copyfile(FILE, path)
with clawhdf5.File(str(path), "r+") as f:
with pytest.raises(NotImplementedError, match="non-IEEE"):
f["bf16le"][0] = 1.0
with pytest.raises(NotImplementedError, match="non-IEEE"):
f["f4e2m1"][:] = np.zeros(16)
with open(path, "rb") as a, open(FILE, "rb") as b:
assert a.read() == b.read()
+10 -1
View File
@@ -109,7 +109,16 @@ virtual datasets. Known differences from
h5dump:
- Floats print at their own precision (a `float32` 0.1 prints as `0.1`),
which for some values is more digits than h5dump's `%g`.
which for some values is more digits than h5dump's `%g`; non-finite
values print as `Inf`, `-Inf` and `NaN` (h5dump: `inf`, `-inf`, `nan`,
`-nan`).
- HDF5 2.x's predefined small floats print under h5dump 2.x's names
(`H5T_FLOAT_BFLOAT16LE`, `H5T_FLOAT_F8E4M3`, ..., `H5T_FLOAT_F4E2M1`;
`ls -v` as h5ls 2.x does: `FP4 E2M1 4-bit float`).
`tests/mx_floats_dump.rs` checks this, and the values, against the output
of h5dump 2.2.0 stored with the fixture. h5dump 1.14 describes these types
instead (`8-bit floating-point 4-bit precision`); other non-IEEE floats
print as an `H5T_FLOAT { ... }` block.
- A compound nested in a compound prints inline (`{ 1, 2.5 }`) where
h5dump prints it as an indented block, one member per line; only the
outer compound is a block.
+78
View File
@@ -59,8 +59,81 @@ pub fn is_ieee(dt: &Datatype) -> bool {
) == std
}
/// A float datatype predefined by libhdf5 2.x besides the IEEE ones.
struct SmallFloat {
/// h5dump's name (`H5T_FLOAT_F8E4M3`).
ddl: &'static str,
/// h5ls's description (`FP8 E4M3 8-bit float`).
long: &'static str,
/// The short name `ls` lists (`float8-e4m3`).
short: &'static str,
}
/// The libhdf5 2.x predefined float `dt` is equal to (as `H5Tequal` sees it:
/// the same size, byte order, precision, offset, fields and bias), if any.
/// The 1-byte types are predefined little-endian only.
fn small_float(dt: &Datatype) -> Option<SmallFloat> {
let Datatype::FloatingPoint {
size,
byte_order,
bit_offset,
bit_precision,
exponent_location,
exponent_size,
mantissa_location,
mantissa_size,
exponent_bias,
} = dt
else {
return None;
};
if *bit_offset != 0 || *mantissa_location != 0 {
return None;
}
let f = |ddl, long, short| Some(SmallFloat { ddl, long, short });
let fields = (
*size,
*bit_precision,
*exponent_location,
*exponent_size,
*mantissa_size,
*exponent_bias,
);
match (fields, byte_order) {
((2, 16, 7, 8, 7, 127), DatatypeByteOrder::LittleEndian) => f(
"H5T_FLOAT_BFLOAT16LE",
"bfloat16 16-bit little-endian float",
"bfloat16",
),
((2, 16, 7, 8, 7, 127), DatatypeByteOrder::BigEndian) => f(
"H5T_FLOAT_BFLOAT16BE",
"bfloat16 16-bit big-endian float",
"bfloat16-be",
),
((1, 8, 3, 4, 3, 7), DatatypeByteOrder::LittleEndian) => {
f("H5T_FLOAT_F8E4M3", "FP8 E4M3 8-bit float", "float8-e4m3")
}
((1, 8, 2, 5, 2, 15), DatatypeByteOrder::LittleEndian) => {
f("H5T_FLOAT_F8E5M2", "FP8 E5M2 8-bit float", "float8-e5m2")
}
((1, 6, 3, 2, 3, 1), DatatypeByteOrder::LittleEndian) => {
f("H5T_FLOAT_F6E2M3", "FP6 E2M3 6-bit float", "float6-e2m3")
}
((1, 6, 2, 3, 2, 3), DatatypeByteOrder::LittleEndian) => {
f("H5T_FLOAT_F6E3M2", "FP6 E3M2 6-bit float", "float6-e3m2")
}
((1, 4, 1, 2, 1, 1), DatatypeByteOrder::LittleEndian) => {
f("H5T_FLOAT_F4E2M1", "FP4 E2M1 4-bit float", "float4-e2m1")
}
_ => None,
}
}
/// Short name used by `ls`: `int32`, `float64-be`, `string[3]`, ...
pub fn short(dt: &Datatype) -> String {
if let Some(f) = small_float(dt) {
return f.short.into();
}
match dt {
Datatype::FixedPoint {
size,
@@ -136,6 +209,9 @@ fn cset_word(c: &CharacterSet) -> &'static str {
/// h5ls -v style description.
pub fn long(dt: &Datatype) -> String {
if let Some(f) = small_float(dt) {
return f.long.into();
}
match dt {
Datatype::FixedPoint {
size,
@@ -277,6 +353,8 @@ fn atomic_ddl(dt: &Datatype) -> Option<String> {
u64::from(*size) * 8,
order_suffix(byte_order)
)),
// As h5dump 2.x names them (checked against h5dump 2.2.0).
Datatype::FloatingPoint { .. } => small_float(dt).map(|f| f.ddl.to_string()),
Datatype::BitField {
size, byte_order, ..
} => Some(format!(
@@ -0,0 +1,111 @@
//! `h5rs dump` and `h5rs ls` of the small floats libhdf5 2.x predefines
//! (bfloat16, FP8 E4M3/E5M2, FP6 E2M3/E3M2, FP4 E2M1), against the output
//! of h5dump 2.2.0 and h5ls 2.2.0 stored next to the fixture (the Debian
//! h5dump CI installs, 1.14.x, predates these types). See
//! `crates/clawhdf5/tests/fixtures/gen_mx_floats.py`.
use std::path::PathBuf;
use std::process::Command;
fn fixture(name: &str) -> PathBuf {
PathBuf::from(env!("CARGO_MANIFEST_DIR"))
.join("../clawhdf5/tests/fixtures")
.join(name)
}
fn h5rs(args: &[&str]) -> String {
let out = Command::new(env!("CARGO_BIN_EXE_h5rs"))
.args(args)
.output()
.unwrap();
assert!(out.status.success(), "h5rs {args:?}: {out:?}");
String::from_utf8(out.stdout).unwrap()
}
/// Split a dump into the lines outside `DATA { ... }` blocks and the values
/// inside them, per block.
fn split(ddl: &str) -> (Vec<&str>, Vec<Vec<String>>) {
let (mut frame, mut data) = (Vec::new(), Vec::new());
let mut values: Option<Vec<String>> = None;
for line in ddl.lines() {
match &mut values {
None if line.trim() == "DATA {" => values = Some(Vec::new()),
None => frame.push(line),
Some(v) if line.trim() == "}" => {
data.push(std::mem::take(v));
values = None;
}
Some(v) => {
let (_, body) = line.split_once("):").expect("an indexed data line");
v.extend(
body.split(',')
.map(str::trim)
.filter(|s| !s.is_empty())
.map(String::from),
);
}
}
}
(frame, data)
}
#[test]
fn dump_matches_h5dump_2_2() {
let path = fixture("mx_floats_hdf5_2_2.h5");
let ours = h5rs(&["dump", path.to_str().unwrap()]);
let reference = std::fs::read_to_string(fixture("mx_floats_hdf5_2_2.ddl")).unwrap();
let (our_frame, our_data) = split(&ours);
let (ref_frame, ref_data) = split(&reference);
// Everything but the values is h5dump's byte for byte: the datatypes
// print as H5T_FLOAT_BFLOAT16LE, H5T_FLOAT_F4E2M1, ...
assert_eq!(our_frame, ref_frame);
assert_eq!(our_data.len(), 17);
assert_eq!(our_data.len(), ref_data.len());
// Values: h5dump prints `%g` (6 significant digits) and inf/-inf/nan/
// -nan; h5rs prints the shortest string that round-trips and Inf/-Inf/NaN.
for (block, (ours, theirs)) in our_data.iter().zip(&ref_data).enumerate() {
assert_eq!(ours.len(), theirs.len(), "block {block}");
for (i, (o, r)) in ours.iter().zip(theirs).enumerate() {
let o: f64 = o.parse().unwrap();
let r: f64 = r.parse().unwrap();
let same = if r.is_nan() || r.is_infinite() {
o.is_nan() == r.is_nan() && (r.is_nan() || o == r)
} else {
// Both round to the same f32 (a bfloat16 subnormal prints
// as its shortest f32 string, `9.1835e-41`, fewer digits
// than h5dump's), or agree to h5dump's 6 digits.
o as f32 == r as f32 || (o - r).abs() <= 5e-6 * r.abs()
};
assert!(same, "block {block}[{i}]: h5rs {o}, h5dump {r}");
}
}
}
#[test]
fn ls_names_small_floats_like_h5ls_2_2() {
let path = fixture("mx_floats_hdf5_2_2.h5");
// h5ls 2.2.0 -v, `Type:` line of each dataset.
for (name, long, short) in [
("bf16le", "bfloat16 16-bit little-endian float", "bfloat16"),
("bf16be", "bfloat16 16-bit big-endian float", "bfloat16-be"),
("f8e4m3", "FP8 E4M3 8-bit float", "float8-e4m3"),
("f8e5m2", "FP8 E5M2 8-bit float", "float8-e5m2"),
("f6e2m3", "FP6 E2M3 6-bit float", "float6-e2m3"),
("f6e3m2", "FP6 E3M2 6-bit float", "float6-e3m2"),
("f4e2m1", "FP4 E2M1 4-bit float", "float4-e2m1"),
] {
let target = format!("{}/{name}", path.display());
let verbose = h5rs(&["ls", "-v", &target]);
assert!(
verbose.contains(&format!(" Type: {long}\n")),
"{name}:\n{verbose}"
);
let listing = h5rs(&["ls", path.to_str().unwrap()]);
assert!(
listing
.lines()
.any(|l| l.starts_with(&format!("{name} ")) && l.ends_with(&format!(" {short}"))),
"{name}:\n{listing}"
);
}
}
+218
View File
@@ -0,0 +1,218 @@
"""Generate mx_floats_hdf5_2_2.h5 and mx_floats_hdf5_2_2.json.
The file holds one dataset and one root attribute per small-float datatype
predefined by libhdf5 2.x:
H5T_FLOAT_F4E2M1 all 16 bit patterns
H5T_FLOAT_F6E2M3 all 64 bit patterns
H5T_FLOAT_F6E3M2 all 64 bit patterns
H5T_FLOAT_F8E4M3 all 256 bit patterns
H5T_FLOAT_F8E5M2 all 256 bit patterns
H5T_FLOAT_BFLOAT16LE a representative set (zeros, subnormals, normals,
H5T_FLOAT_BFLOAT16BE max, +-inf, quiet/signalling/negative NaN)
plus `<type>_pad` datasets for the 4- and 6-bit types whose bytes also have
every unused high bit set (libhdf5 ignores bits outside the precision).
The raw bit patterns are written with the file type as the memory type (no
conversion). The JSON is the reference: for every element, the bits of the
f64 and f32 values that libhdf5 itself returns from H5Dread/H5Aread into
H5T_NATIVE_DOUBLE/H5T_NATIVE_FLOAT, and each conversion exception libhdf5
raised through H5Pset_type_conv_cb (the callback returns UNHANDLED, so the
values are libhdf5's defaults).
It talks to libhdf5 2.2.0 through ctypes only (no h5py: h5py 3.16 bundles
libhdf5 2.0.0, which predates these types). libhdf5 was built from source:
git clone --depth 1 --branch 2.2.0 https://github.com/HDFGroup/hdf5 \
~/.cache/hdf5-2.2.0-src # commit 49df1b4e5ca4108d56e80f06321a749da1535268
cmake -S ~/.cache/hdf5-2.2.0-src -B ~/.cache/hdf5-2.2.0-build \
-DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=$HOME/.cache/hdf5-2.2.0 \
-DBUILD_SHARED_LIBS=ON -DBUILD_STATIC_LIBS=OFF -DHDF5_ENABLE_ZLIB_SUPPORT=ON \
-DHDF5_BUILD_FORTRAN=OFF -DHDF5_BUILD_JAVA=OFF -DHDF5_BUILD_CPP_LIB=OFF \
-DBUILD_TESTING=OFF -DHDF5_BUILD_EXAMPLES=OFF -DHDF5_BUILD_TOOLS=ON \
-DHDF5_BUILD_HL_LIB=OFF
cmake --build ~/.cache/hdf5-2.2.0-build -j8
cmake --install ~/.cache/hdf5-2.2.0-build
Re-run only if the fixture ever needs regenerating (generated 2026-09-28):
python gen_mx_floats.py ~/.cache/hdf5-2.2.0/lib/libhdf5.so mx_floats_hdf5_2_2
(h5dump from the same build writes mx_floats_hdf5_2_2.ddl:
`h5dump mx_floats_hdf5_2_2.h5 > mx_floats_hdf5_2_2.ddl`.)
"""
import ctypes
import json
import struct
import sys
lib = ctypes.CDLL(sys.argv[1])
out = sys.argv[2]
hid = ctypes.c_int64
herr = ctypes.c_int
assert lib.H5open() >= 0
majnum, minnum, relnum = ctypes.c_uint(), ctypes.c_uint(), ctypes.c_uint()
lib.H5get_libversion(ctypes.byref(majnum), ctypes.byref(minnum), ctypes.byref(relnum))
version = f"{majnum.value}.{minnum.value}.{relnum.value}"
assert version == "2.2.0", version
def g(name):
return hid.in_dll(lib, name + "_g").value
for fn, res, args in [
("H5Fcreate", hid, [ctypes.c_char_p, ctypes.c_uint, hid, hid]),
("H5Fclose", herr, [hid]),
("H5Fopen", hid, [ctypes.c_char_p, ctypes.c_uint, hid]),
("H5Screate_simple", hid, [ctypes.c_int, ctypes.POINTER(ctypes.c_uint64), ctypes.c_void_p]),
("H5Sclose", herr, [hid]),
("H5Dcreate2", hid, [hid, ctypes.c_char_p, hid, hid, hid, hid, hid]),
("H5Dopen2", hid, [hid, ctypes.c_char_p, hid]),
("H5Dwrite", herr, [hid, hid, hid, hid, hid, ctypes.c_void_p]),
("H5Dread", herr, [hid, hid, hid, hid, hid, ctypes.c_void_p]),
("H5Dclose", herr, [hid]),
("H5Acreate2", hid, [hid, ctypes.c_char_p, hid, hid, hid, hid]),
("H5Aopen", hid, [hid, ctypes.c_char_p, hid]),
("H5Awrite", herr, [hid, hid, ctypes.c_void_p]),
("H5Aread", herr, [hid, hid, ctypes.c_void_p]),
("H5Aclose", herr, [hid]),
("H5Pcreate", hid, [hid]),
("H5Pclose", herr, [hid]),
("H5Pset_type_conv_cb", herr, [hid, ctypes.c_void_p, ctypes.c_void_p]),
("H5Tget_fields", herr, [hid] + [ctypes.POINTER(ctypes.c_size_t)] * 5),
("H5Tget_ebias", ctypes.c_size_t, [hid]),
("H5Tget_precision", ctypes.c_size_t, [hid]),
("H5Tget_offset", ctypes.c_int, [hid]),
("H5Tget_size", ctypes.c_size_t, [hid]),
("H5Tget_order", ctypes.c_int, [hid]),
]:
getattr(lib, fn).restype = res
getattr(lib, fn).argtypes = args
H5F_ACC_TRUNC, H5F_ACC_RDONLY = 0x0002, 0x0000
H5P_DEFAULT, H5S_ALL = 0, 0
NATIVE_DOUBLE = g("H5T_NATIVE_DOUBLE")
NATIVE_FLOAT = g("H5T_NATIVE_FLOAT")
BF16 = [0x0000, 0x8000, 0x3F80, 0xBF80, 0x3FC0, 0xC010, 0x4049, 0x3F81,
0x0001, 0x8001, 0x007F, 0x0080, 0x7F7F, 0xFF7F, 0x7F80, 0xFF80,
0x7FC0, 0x7F81, 0xFFC1, 0x7FFF]
# name -> (libhdf5 global, raw element bytes)
TYPES = {
"f4e2m1": ("H5T_FLOAT_F4E2M1", [bytes([b]) for b in range(16)]),
"f4e2m1_pad": ("H5T_FLOAT_F4E2M1", [bytes([0xF0 | b]) for b in range(16)]),
"f6e2m3": ("H5T_FLOAT_F6E2M3", [bytes([b]) for b in range(64)]),
"f6e2m3_pad": ("H5T_FLOAT_F6E2M3", [bytes([0xC0 | b]) for b in range(64)]),
"f6e3m2": ("H5T_FLOAT_F6E3M2", [bytes([b]) for b in range(64)]),
"f6e3m2_pad": ("H5T_FLOAT_F6E3M2", [bytes([0xC0 | b]) for b in range(64)]),
"f8e4m3": ("H5T_FLOAT_F8E4M3", [bytes([b]) for b in range(256)]),
"f8e5m2": ("H5T_FLOAT_F8E5M2", [bytes([b]) for b in range(256)]),
"bf16le": ("H5T_FLOAT_BFLOAT16LE", [struct.pack("<H", v) for v in BF16]),
"bf16be": ("H5T_FLOAT_BFLOAT16BE", [struct.pack(">H", v) for v in BF16]),
}
EXCEPT_NAMES = ["RANGE_HI", "RANGE_LOW", "PRECISION", "TRUNCATE", "PINF", "NINF", "NAN"]
fid = lib.H5Fcreate((out + ".h5").encode(), H5F_ACC_TRUNC, H5P_DEFAULT, H5P_DEFAULT)
assert fid >= 0
for name, (tname, raw) in TYPES.items():
t = g(tname)
buf = ctypes.create_string_buffer(b"".join(raw), len(raw) * len(raw[0]))
sid = lib.H5Screate_simple(1, (ctypes.c_uint64 * 1)(len(raw)), None)
did = lib.H5Dcreate2(fid, name.encode(), t, sid, H5P_DEFAULT, H5P_DEFAULT, H5P_DEFAULT)
assert did >= 0 and lib.H5Dwrite(did, t, H5S_ALL, H5S_ALL, H5P_DEFAULT, buf) >= 0
lib.H5Dclose(did)
if not name.endswith("_pad"):
aid = lib.H5Acreate2(fid, name.encode(), t, sid, H5P_DEFAULT, H5P_DEFAULT)
assert aid >= 0 and lib.H5Awrite(aid, t, buf) >= 0
lib.H5Aclose(aid)
lib.H5Sclose(sid)
assert lib.H5Fclose(fid) >= 0
# --- reference: what libhdf5 converts each element to --------------------
CB = ctypes.CFUNCTYPE(ctypes.c_int, ctypes.c_int, hid, hid, ctypes.c_void_p, ctypes.c_void_p, ctypes.c_void_p)
events = []
@CB
def on_except(kind, src_id, dst_id, src_buf, dst_buf, user):
events.append((EXCEPT_NAMES[kind], ctypes.string_at(src_buf, 1 if lib.H5Tget_size(src_id) == 1 else 2).hex()))
return 0 # H5T_CONV_UNHANDLED: let libhdf5 apply its default
def f64_text(b):
"""The f64 as Python's round-tripping repr ("0.125", "inf", "-inf"), or
"nan:<bits>" for a NaN, whose sign and payload are kept exactly."""
(bits,) = struct.unpack("<Q", b)
(v,) = struct.unpack("<d", b)
return "nan:%016x" % bits if v != v else repr(v)
def read_all(obj, is_dataset, mem, width, n):
buf = ctypes.create_string_buffer(n * width)
events.clear()
if is_dataset:
assert lib.H5Dread(obj, mem, H5S_ALL, H5S_ALL, dxpl, buf) >= 0
else:
# H5Aread takes no transfer property list, so no exception callback.
assert lib.H5Aread(obj, mem, buf) >= 0
return [buf.raw[i * width:(i + 1) * width] for i in range(n)], sorted(events)
fid = lib.H5Fopen((out + ".h5").encode(), H5F_ACC_RDONLY, H5P_DEFAULT)
dxpl = lib.H5Pcreate(g("H5P_CLS_DATASET_XFER_ID"))
assert lib.H5Pset_type_conv_cb(dxpl, ctypes.cast(on_except, ctypes.c_void_p), None) >= 0
ref = {
"libhdf5": version,
"generator": "gen_mx_floats.py",
"note": "f64: what H5Dread into H5T_NATIVE_DOUBLE returns per element; "
"H5Aread (attribute of the same name) and H5T_NATIVE_FLOAT return the same values "
"(f32 NaN bits in f32_nan); exceptions: H5Pset_type_conv_cb events by source bytes",
"objects": {},
}
for name, (tname, raw) in TYPES.items():
t = g(tname)
sizes = [ctypes.c_size_t() for _ in range(5)]
lib.H5Tget_fields(t, *[ctypes.byref(s) for s in sizes])
n = len(raw)
did = lib.H5Dopen2(fid, name.encode(), H5P_DEFAULT)
d64, exc64 = read_all(did, True, NATIVE_DOUBLE, 8, n)
d32, exc32 = read_all(did, True, NATIVE_FLOAT, 4, n)
lib.H5Dclose(did)
assert exc32 == exc64, name
for a, b in zip(d64, d32):
(x,) = struct.unpack("<d", a)
(y,) = struct.unpack("<f", b)
assert (x != x and y != y) or x == y, (name, x, y)
f32_nan = sorted({"%08x" % struct.unpack("<I", b)[0] for b in d32 if struct.unpack("<f", b)[0] != struct.unpack("<f", b)[0]})
if not name.endswith("_pad"):
aid = lib.H5Aopen(fid, name.encode(), H5P_DEFAULT)
assert read_all(aid, False, NATIVE_DOUBLE, 8, n)[0] == d64, name
assert read_all(aid, False, NATIVE_FLOAT, 4, n)[0] == d32, name
lib.H5Aclose(aid)
exceptions = {}
for kind, src in exc64:
exceptions.setdefault(kind, []).append(src)
ref["objects"][name] = {
"type": tname,
"size": lib.H5Tget_size(t),
"precision": lib.H5Tget_precision(t),
"offset": lib.H5Tget_offset(t),
"order": ["LE", "BE"][lib.H5Tget_order(t)],
"fields": [s.value for s in sizes], # spos, epos, esize, mpos, msize
"ebias": lib.H5Tget_ebias(t),
"attribute": not name.endswith("_pad"),
"raw": "".join(r.hex() for r in raw),
"f64": [f64_text(b) for b in d64],
"f32_nan": f32_nan,
"exceptions": exceptions,
}
lib.H5Pclose(dxpl)
lib.H5Fclose(fid)
with open(out + ".json", "w") as fh:
json.dump(ref, fh, separators=(",", ":"))
fh.write("\n")
+290
View File
@@ -0,0 +1,290 @@
HDF5 "mx_floats_hdf5_2_2.h5" {
GROUP "/" {
ATTRIBUTE "bf16be" {
DATATYPE H5T_FLOAT_BFLOAT16BE
DATASPACE SIMPLE { ( 20 ) / ( 20 ) }
DATA {
(0): 0, -0, 1, -1, 1.5, -2.25, 3.14062, 1.00781, 9.18355e-41,
(9): -9.18355e-41, 1.16631e-38, 1.17549e-38, 3.38953e+38, -3.38953e+38,
(14): inf, -inf, nan, nan, -nan, nan
}
}
ATTRIBUTE "bf16le" {
DATATYPE H5T_FLOAT_BFLOAT16LE
DATASPACE SIMPLE { ( 20 ) / ( 20 ) }
DATA {
(0): 0, -0, 1, -1, 1.5, -2.25, 3.14062, 1.00781, 9.18355e-41,
(9): -9.18355e-41, 1.16631e-38, 1.17549e-38, 3.38953e+38, -3.38953e+38,
(14): inf, -inf, nan, nan, -nan, nan
}
}
ATTRIBUTE "f4e2m1" {
DATATYPE H5T_FLOAT_F4E2M1
DATASPACE SIMPLE { ( 16 ) / ( 16 ) }
DATA {
(0): 0, 0.5, 1, 1.5, 2, 3, inf, nan, -0, -0.5, -1, -1.5, -2, -3, -inf,
(15): -nan
}
}
ATTRIBUTE "f6e2m3" {
DATATYPE H5T_FLOAT_F6E2M3
DATASPACE SIMPLE { ( 64 ) / ( 64 ) }
DATA {
(0): 0, 0.125, 0.25, 0.375, 0.5, 0.625, 0.75, 0.875, 1, 1.125, 1.25,
(11): 1.375, 1.5, 1.625, 1.75, 1.875, 2, 2.25, 2.5, 2.75, 3, 3.25, 3.5,
(23): 3.75, inf, nan, nan, nan, nan, nan, nan, nan, -0, -0.125, -0.25,
(35): -0.375, -0.5, -0.625, -0.75, -0.875, -1, -1.125, -1.25, -1.375,
(44): -1.5, -1.625, -1.75, -1.875, -2, -2.25, -2.5, -2.75, -3, -3.25,
(54): -3.5, -3.75, -inf, -nan, -nan, -nan, -nan, -nan, -nan, -nan
}
}
ATTRIBUTE "f6e3m2" {
DATATYPE H5T_FLOAT_F6E3M2
DATASPACE SIMPLE { ( 64 ) / ( 64 ) }
DATA {
(0): 0, 0.0625, 0.125, 0.1875, 0.25, 0.3125, 0.375, 0.4375, 0.5, 0.625,
(10): 0.75, 0.875, 1, 1.25, 1.5, 1.75, 2, 2.5, 3, 3.5, 4, 5, 6, 7, 8,
(25): 10, 12, 14, inf, nan, nan, nan, -0, -0.0625, -0.125, -0.1875,
(36): -0.25, -0.3125, -0.375, -0.4375, -0.5, -0.625, -0.75, -0.875, -1,
(45): -1.25, -1.5, -1.75, -2, -2.5, -3, -3.5, -4, -5, -6, -7, -8, -10,
(58): -12, -14, -inf, -nan, -nan, -nan
}
}
ATTRIBUTE "f8e4m3" {
DATATYPE H5T_FLOAT_F8E4M3
DATASPACE SIMPLE { ( 256 ) / ( 256 ) }
DATA {
(0): 0, 0.00195312, 0.00390625, 0.00585938, 0.0078125, 0.00976562,
(6): 0.0117188, 0.0136719, 0.015625, 0.0175781, 0.0195312, 0.0214844,
(12): 0.0234375, 0.0253906, 0.0273438, 0.0292969, 0.03125, 0.0351562,
(18): 0.0390625, 0.0429688, 0.046875, 0.0507812, 0.0546875, 0.0585938,
(24): 0.0625, 0.0703125, 0.078125, 0.0859375, 0.09375, 0.101562,
(30): 0.109375, 0.117188, 0.125, 0.140625, 0.15625, 0.171875, 0.1875,
(37): 0.203125, 0.21875, 0.234375, 0.25, 0.28125, 0.3125, 0.34375,
(44): 0.375, 0.40625, 0.4375, 0.46875, 0.5, 0.5625, 0.625, 0.6875,
(52): 0.75, 0.8125, 0.875, 0.9375, 1, 1.125, 1.25, 1.375, 1.5, 1.625,
(62): 1.75, 1.875, 2, 2.25, 2.5, 2.75, 3, 3.25, 3.5, 3.75, 4, 4.5, 5,
(75): 5.5, 6, 6.5, 7, 7.5, 8, 9, 10, 11, 12, 13, 14, 15, 16, 18, 20,
(91): 22, 24, 26, 28, 30, 32, 36, 40, 44, 48, 52, 56, 60, 64, 72, 80,
(107): 88, 96, 104, 112, 120, 128, 144, 160, 176, 192, 208, 224, 240,
(120): inf, nan, nan, nan, nan, nan, nan, nan, -0, -0.00195312,
(130): -0.00390625, -0.00585938, -0.0078125, -0.00976562, -0.0117188,
(135): -0.0136719, -0.015625, -0.0175781, -0.0195312, -0.0214844,
(140): -0.0234375, -0.0253906, -0.0273438, -0.0292969, -0.03125,
(145): -0.0351562, -0.0390625, -0.0429688, -0.046875, -0.0507812,
(150): -0.0546875, -0.0585938, -0.0625, -0.0703125, -0.078125,
(155): -0.0859375, -0.09375, -0.101562, -0.109375, -0.117188, -0.125,
(161): -0.140625, -0.15625, -0.171875, -0.1875, -0.203125, -0.21875,
(167): -0.234375, -0.25, -0.28125, -0.3125, -0.34375, -0.375, -0.40625,
(174): -0.4375, -0.46875, -0.5, -0.5625, -0.625, -0.6875, -0.75,
(181): -0.8125, -0.875, -0.9375, -1, -1.125, -1.25, -1.375, -1.5,
(189): -1.625, -1.75, -1.875, -2, -2.25, -2.5, -2.75, -3, -3.25, -3.5,
(199): -3.75, -4, -4.5, -5, -5.5, -6, -6.5, -7, -7.5, -8, -9, -10, -11,
(212): -12, -13, -14, -15, -16, -18, -20, -22, -24, -26, -28, -30, -32,
(225): -36, -40, -44, -48, -52, -56, -60, -64, -72, -80, -88, -96,
(237): -104, -112, -120, -128, -144, -160, -176, -192, -208, -224,
(247): -240, -inf, -nan, -nan, -nan, -nan, -nan, -nan, -nan
}
}
ATTRIBUTE "f8e5m2" {
DATATYPE H5T_FLOAT_F8E5M2
DATASPACE SIMPLE { ( 256 ) / ( 256 ) }
DATA {
(0): 0, 1.52588e-05, 3.05176e-05, 4.57764e-05, 6.10352e-05,
(5): 7.62939e-05, 9.15527e-05, 0.000106812, 0.00012207, 0.000152588,
(10): 0.000183105, 0.000213623, 0.000244141, 0.000305176, 0.000366211,
(15): 0.000427246, 0.000488281, 0.000610352, 0.000732422, 0.000854492,
(20): 0.000976562, 0.0012207, 0.00146484, 0.00170898, 0.00195312,
(25): 0.00244141, 0.00292969, 0.00341797, 0.00390625, 0.00488281,
(30): 0.00585938, 0.00683594, 0.0078125, 0.00976562, 0.0117188,
(35): 0.0136719, 0.015625, 0.0195312, 0.0234375, 0.0273438, 0.03125,
(41): 0.0390625, 0.046875, 0.0546875, 0.0625, 0.078125, 0.09375,
(47): 0.109375, 0.125, 0.15625, 0.1875, 0.21875, 0.25, 0.3125, 0.375,
(55): 0.4375, 0.5, 0.625, 0.75, 0.875, 1, 1.25, 1.5, 1.75, 2, 2.5, 3,
(67): 3.5, 4, 5, 6, 7, 8, 10, 12, 14, 16, 20, 24, 28, 32, 40, 48, 56,
(84): 64, 80, 96, 112, 128, 160, 192, 224, 256, 320, 384, 448, 512,
(97): 640, 768, 896, 1024, 1280, 1536, 1792, 2048, 2560, 3072, 3584,
(108): 4096, 5120, 6144, 7168, 8192, 10240, 12288, 14336, 16384, 20480,
(118): 24576, 28672, 32768, 40960, 49152, 57344, inf, nan, nan, nan,
(128): -0, -1.52588e-05, -3.05176e-05, -4.57764e-05, -6.10352e-05,
(133): -7.62939e-05, -9.15527e-05, -0.000106812, -0.00012207,
(137): -0.000152588, -0.000183105, -0.000213623, -0.000244141,
(141): -0.000305176, -0.000366211, -0.000427246, -0.000488281,
(145): -0.000610352, -0.000732422, -0.000854492, -0.000976562,
(149): -0.0012207, -0.00146484, -0.00170898, -0.00195312, -0.00244141,
(154): -0.00292969, -0.00341797, -0.00390625, -0.00488281, -0.00585938,
(159): -0.00683594, -0.0078125, -0.00976562, -0.0117188, -0.0136719,
(164): -0.015625, -0.0195312, -0.0234375, -0.0273438, -0.03125,
(169): -0.0390625, -0.046875, -0.0546875, -0.0625, -0.078125, -0.09375,
(175): -0.109375, -0.125, -0.15625, -0.1875, -0.21875, -0.25, -0.3125,
(182): -0.375, -0.4375, -0.5, -0.625, -0.75, -0.875, -1, -1.25, -1.5,
(191): -1.75, -2, -2.5, -3, -3.5, -4, -5, -6, -7, -8, -10, -12, -14,
(204): -16, -20, -24, -28, -32, -40, -48, -56, -64, -80, -96, -112,
(216): -128, -160, -192, -224, -256, -320, -384, -448, -512, -640,
(226): -768, -896, -1024, -1280, -1536, -1792, -2048, -2560, -3072,
(235): -3584, -4096, -5120, -6144, -7168, -8192, -10240, -12288,
(243): -14336, -16384, -20480, -24576, -28672, -32768, -40960, -49152,
(251): -57344, -inf, -nan, -nan, -nan
}
}
DATASET "bf16be" {
DATATYPE H5T_FLOAT_BFLOAT16BE
DATASPACE SIMPLE { ( 20 ) / ( 20 ) }
DATA {
(0): 0, -0, 1, -1, 1.5, -2.25, 3.14062, 1.00781, 9.18355e-41,
(9): -9.18355e-41, 1.16631e-38, 1.17549e-38, 3.38953e+38, -3.38953e+38,
(14): inf, -inf, nan, nan, -nan, nan
}
}
DATASET "bf16le" {
DATATYPE H5T_FLOAT_BFLOAT16LE
DATASPACE SIMPLE { ( 20 ) / ( 20 ) }
DATA {
(0): 0, -0, 1, -1, 1.5, -2.25, 3.14062, 1.00781, 9.18355e-41,
(9): -9.18355e-41, 1.16631e-38, 1.17549e-38, 3.38953e+38, -3.38953e+38,
(14): inf, -inf, nan, nan, -nan, nan
}
}
DATASET "f4e2m1" {
DATATYPE H5T_FLOAT_F4E2M1
DATASPACE SIMPLE { ( 16 ) / ( 16 ) }
DATA {
(0): 0, 0.5, 1, 1.5, 2, 3, inf, nan, -0, -0.5, -1, -1.5, -2, -3, -inf,
(15): -nan
}
}
DATASET "f4e2m1_pad" {
DATATYPE H5T_FLOAT_F4E2M1
DATASPACE SIMPLE { ( 16 ) / ( 16 ) }
DATA {
(0): 0, 0.5, 1, 1.5, 2, 3, inf, nan, -0, -0.5, -1, -1.5, -2, -3, -inf,
(15): -nan
}
}
DATASET "f6e2m3" {
DATATYPE H5T_FLOAT_F6E2M3
DATASPACE SIMPLE { ( 64 ) / ( 64 ) }
DATA {
(0): 0, 0.125, 0.25, 0.375, 0.5, 0.625, 0.75, 0.875, 1, 1.125, 1.25,
(11): 1.375, 1.5, 1.625, 1.75, 1.875, 2, 2.25, 2.5, 2.75, 3, 3.25, 3.5,
(23): 3.75, inf, nan, nan, nan, nan, nan, nan, nan, -0, -0.125, -0.25,
(35): -0.375, -0.5, -0.625, -0.75, -0.875, -1, -1.125, -1.25, -1.375,
(44): -1.5, -1.625, -1.75, -1.875, -2, -2.25, -2.5, -2.75, -3, -3.25,
(54): -3.5, -3.75, -inf, -nan, -nan, -nan, -nan, -nan, -nan, -nan
}
}
DATASET "f6e2m3_pad" {
DATATYPE H5T_FLOAT_F6E2M3
DATASPACE SIMPLE { ( 64 ) / ( 64 ) }
DATA {
(0): 0, 0.125, 0.25, 0.375, 0.5, 0.625, 0.75, 0.875, 1, 1.125, 1.25,
(11): 1.375, 1.5, 1.625, 1.75, 1.875, 2, 2.25, 2.5, 2.75, 3, 3.25, 3.5,
(23): 3.75, inf, nan, nan, nan, nan, nan, nan, nan, -0, -0.125, -0.25,
(35): -0.375, -0.5, -0.625, -0.75, -0.875, -1, -1.125, -1.25, -1.375,
(44): -1.5, -1.625, -1.75, -1.875, -2, -2.25, -2.5, -2.75, -3, -3.25,
(54): -3.5, -3.75, -inf, -nan, -nan, -nan, -nan, -nan, -nan, -nan
}
}
DATASET "f6e3m2" {
DATATYPE H5T_FLOAT_F6E3M2
DATASPACE SIMPLE { ( 64 ) / ( 64 ) }
DATA {
(0): 0, 0.0625, 0.125, 0.1875, 0.25, 0.3125, 0.375, 0.4375, 0.5, 0.625,
(10): 0.75, 0.875, 1, 1.25, 1.5, 1.75, 2, 2.5, 3, 3.5, 4, 5, 6, 7, 8,
(25): 10, 12, 14, inf, nan, nan, nan, -0, -0.0625, -0.125, -0.1875,
(36): -0.25, -0.3125, -0.375, -0.4375, -0.5, -0.625, -0.75, -0.875, -1,
(45): -1.25, -1.5, -1.75, -2, -2.5, -3, -3.5, -4, -5, -6, -7, -8, -10,
(58): -12, -14, -inf, -nan, -nan, -nan
}
}
DATASET "f6e3m2_pad" {
DATATYPE H5T_FLOAT_F6E3M2
DATASPACE SIMPLE { ( 64 ) / ( 64 ) }
DATA {
(0): 0, 0.0625, 0.125, 0.1875, 0.25, 0.3125, 0.375, 0.4375, 0.5, 0.625,
(10): 0.75, 0.875, 1, 1.25, 1.5, 1.75, 2, 2.5, 3, 3.5, 4, 5, 6, 7, 8,
(25): 10, 12, 14, inf, nan, nan, nan, -0, -0.0625, -0.125, -0.1875,
(36): -0.25, -0.3125, -0.375, -0.4375, -0.5, -0.625, -0.75, -0.875, -1,
(45): -1.25, -1.5, -1.75, -2, -2.5, -3, -3.5, -4, -5, -6, -7, -8, -10,
(58): -12, -14, -inf, -nan, -nan, -nan
}
}
DATASET "f8e4m3" {
DATATYPE H5T_FLOAT_F8E4M3
DATASPACE SIMPLE { ( 256 ) / ( 256 ) }
DATA {
(0): 0, 0.00195312, 0.00390625, 0.00585938, 0.0078125, 0.00976562,
(6): 0.0117188, 0.0136719, 0.015625, 0.0175781, 0.0195312, 0.0214844,
(12): 0.0234375, 0.0253906, 0.0273438, 0.0292969, 0.03125, 0.0351562,
(18): 0.0390625, 0.0429688, 0.046875, 0.0507812, 0.0546875, 0.0585938,
(24): 0.0625, 0.0703125, 0.078125, 0.0859375, 0.09375, 0.101562,
(30): 0.109375, 0.117188, 0.125, 0.140625, 0.15625, 0.171875, 0.1875,
(37): 0.203125, 0.21875, 0.234375, 0.25, 0.28125, 0.3125, 0.34375,
(44): 0.375, 0.40625, 0.4375, 0.46875, 0.5, 0.5625, 0.625, 0.6875,
(52): 0.75, 0.8125, 0.875, 0.9375, 1, 1.125, 1.25, 1.375, 1.5, 1.625,
(62): 1.75, 1.875, 2, 2.25, 2.5, 2.75, 3, 3.25, 3.5, 3.75, 4, 4.5, 5,
(75): 5.5, 6, 6.5, 7, 7.5, 8, 9, 10, 11, 12, 13, 14, 15, 16, 18, 20,
(91): 22, 24, 26, 28, 30, 32, 36, 40, 44, 48, 52, 56, 60, 64, 72, 80,
(107): 88, 96, 104, 112, 120, 128, 144, 160, 176, 192, 208, 224, 240,
(120): inf, nan, nan, nan, nan, nan, nan, nan, -0, -0.00195312,
(130): -0.00390625, -0.00585938, -0.0078125, -0.00976562, -0.0117188,
(135): -0.0136719, -0.015625, -0.0175781, -0.0195312, -0.0214844,
(140): -0.0234375, -0.0253906, -0.0273438, -0.0292969, -0.03125,
(145): -0.0351562, -0.0390625, -0.0429688, -0.046875, -0.0507812,
(150): -0.0546875, -0.0585938, -0.0625, -0.0703125, -0.078125,
(155): -0.0859375, -0.09375, -0.101562, -0.109375, -0.117188, -0.125,
(161): -0.140625, -0.15625, -0.171875, -0.1875, -0.203125, -0.21875,
(167): -0.234375, -0.25, -0.28125, -0.3125, -0.34375, -0.375, -0.40625,
(174): -0.4375, -0.46875, -0.5, -0.5625, -0.625, -0.6875, -0.75,
(181): -0.8125, -0.875, -0.9375, -1, -1.125, -1.25, -1.375, -1.5,
(189): -1.625, -1.75, -1.875, -2, -2.25, -2.5, -2.75, -3, -3.25, -3.5,
(199): -3.75, -4, -4.5, -5, -5.5, -6, -6.5, -7, -7.5, -8, -9, -10, -11,
(212): -12, -13, -14, -15, -16, -18, -20, -22, -24, -26, -28, -30, -32,
(225): -36, -40, -44, -48, -52, -56, -60, -64, -72, -80, -88, -96,
(237): -104, -112, -120, -128, -144, -160, -176, -192, -208, -224,
(247): -240, -inf, -nan, -nan, -nan, -nan, -nan, -nan, -nan
}
}
DATASET "f8e5m2" {
DATATYPE H5T_FLOAT_F8E5M2
DATASPACE SIMPLE { ( 256 ) / ( 256 ) }
DATA {
(0): 0, 1.52588e-05, 3.05176e-05, 4.57764e-05, 6.10352e-05,
(5): 7.62939e-05, 9.15527e-05, 0.000106812, 0.00012207, 0.000152588,
(10): 0.000183105, 0.000213623, 0.000244141, 0.000305176, 0.000366211,
(15): 0.000427246, 0.000488281, 0.000610352, 0.000732422, 0.000854492,
(20): 0.000976562, 0.0012207, 0.00146484, 0.00170898, 0.00195312,
(25): 0.00244141, 0.00292969, 0.00341797, 0.00390625, 0.00488281,
(30): 0.00585938, 0.00683594, 0.0078125, 0.00976562, 0.0117188,
(35): 0.0136719, 0.015625, 0.0195312, 0.0234375, 0.0273438, 0.03125,
(41): 0.0390625, 0.046875, 0.0546875, 0.0625, 0.078125, 0.09375,
(47): 0.109375, 0.125, 0.15625, 0.1875, 0.21875, 0.25, 0.3125, 0.375,
(55): 0.4375, 0.5, 0.625, 0.75, 0.875, 1, 1.25, 1.5, 1.75, 2, 2.5, 3,
(67): 3.5, 4, 5, 6, 7, 8, 10, 12, 14, 16, 20, 24, 28, 32, 40, 48, 56,
(84): 64, 80, 96, 112, 128, 160, 192, 224, 256, 320, 384, 448, 512,
(97): 640, 768, 896, 1024, 1280, 1536, 1792, 2048, 2560, 3072, 3584,
(108): 4096, 5120, 6144, 7168, 8192, 10240, 12288, 14336, 16384, 20480,
(118): 24576, 28672, 32768, 40960, 49152, 57344, inf, nan, nan, nan,
(128): -0, -1.52588e-05, -3.05176e-05, -4.57764e-05, -6.10352e-05,
(133): -7.62939e-05, -9.15527e-05, -0.000106812, -0.00012207,
(137): -0.000152588, -0.000183105, -0.000213623, -0.000244141,
(141): -0.000305176, -0.000366211, -0.000427246, -0.000488281,
(145): -0.000610352, -0.000732422, -0.000854492, -0.000976562,
(149): -0.0012207, -0.00146484, -0.00170898, -0.00195312, -0.00244141,
(154): -0.00292969, -0.00341797, -0.00390625, -0.00488281, -0.00585938,
(159): -0.00683594, -0.0078125, -0.00976562, -0.0117188, -0.0136719,
(164): -0.015625, -0.0195312, -0.0234375, -0.0273438, -0.03125,
(169): -0.0390625, -0.046875, -0.0546875, -0.0625, -0.078125, -0.09375,
(175): -0.109375, -0.125, -0.15625, -0.1875, -0.21875, -0.25, -0.3125,
(182): -0.375, -0.4375, -0.5, -0.625, -0.75, -0.875, -1, -1.25, -1.5,
(191): -1.75, -2, -2.5, -3, -3.5, -4, -5, -6, -7, -8, -10, -12, -14,
(204): -16, -20, -24, -28, -32, -40, -48, -56, -64, -80, -96, -112,
(216): -128, -160, -192, -224, -256, -320, -384, -448, -512, -640,
(226): -768, -896, -1024, -1280, -1536, -1792, -2048, -2560, -3072,
(235): -3584, -4096, -5120, -6144, -7168, -8192, -10240, -12288,
(243): -14336, -16384, -20480, -24576, -28672, -32768, -40960, -49152,
(251): -57344, -inf, -nan, -nan, -nan
}
}
}
}
Binary file not shown.
File diff suppressed because one or more lines are too long
+159
View File
@@ -0,0 +1,159 @@
//! Small floats predefined by libhdf5 2.x (FP4 E2M1, FP6 E2M3/E3M2, FP8
//! E4M3/E5M2, bfloat16 LE/BE), decoded bit for bit as libhdf5 2.2.0 decodes
//! them.
//!
//! `fixtures/mx_floats_hdf5_2_2.h5` and the reference
//! `fixtures/mx_floats_hdf5_2_2.json` were written by libhdf5 2.2.0 itself
//! (`fixtures/gen_mx_floats.py`): every bit pattern of the 4-, 6- and 8-bit
//! types, a representative bfloat16 set, and what `H5Dread`/`H5Aread` into
//! `H5T_NATIVE_DOUBLE`/`H5T_NATIVE_FLOAT` return for each. libhdf5 treats an
//! all-ones exponent as infinity/NaN in every one of these formats, including
//! the OCP MX formats that have no infinity (FP4, FP6) or only one NaN (FP8
//! E4M3); clawhdf5 matches libhdf5, NaN sign and payload included.
use clawhdf5::{AttrValue, File};
const FIXTURE: &str = concat!(
env!("CARGO_MANIFEST_DIR"),
"/tests/fixtures/mx_floats_hdf5_2_2.h5"
);
const REFERENCE: &str = include_str!("fixtures/mx_floats_hdf5_2_2.json");
struct Expected {
name: String,
attribute: bool,
/// Bits of the `f64` libhdf5 returns for each element.
f64_bits: Vec<u64>,
/// Bits of the `f32` NaNs libhdf5 returns (every non-NaN value is the
/// `f64` value, exact in `f32`).
f32_nan: Vec<u32>,
}
fn reference() -> Vec<Expected> {
let json: serde_json::Value = serde_json::from_str(REFERENCE).unwrap();
assert_eq!(json["libhdf5"], "2.2.0");
json["objects"]
.as_object()
.unwrap()
.iter()
.map(|(name, obj)| Expected {
name: name.clone(),
attribute: obj["attribute"].as_bool().unwrap(),
f64_bits: obj["f64"]
.as_array()
.unwrap()
.iter()
.map(|v| {
let text = v.as_str().unwrap();
match text.strip_prefix("nan:") {
Some(bits) => u64::from_str_radix(bits, 16).unwrap(),
None => text.parse::<f64>().unwrap().to_bits(),
}
})
.collect(),
f32_nan: obj["f32_nan"]
.as_array()
.unwrap()
.iter()
.map(|v| u32::from_str_radix(v.as_str().unwrap(), 16).unwrap())
.collect(),
})
.collect()
}
fn check_f64(what: &str, got: &[f64], want: &[u64]) {
assert_eq!(got.len(), want.len(), "{what}: element count");
for (i, (g, w)) in got.iter().zip(want).enumerate() {
assert_eq!(
g.to_bits(),
*w,
"{what}[{i}]: clawhdf5 {g:?} ({:#018x}), libhdf5 {:?} ({w:#018x})",
g.to_bits(),
f64::from_bits(*w)
);
}
}
fn check_f32(what: &str, got: &[f32], want: &Expected) {
assert_eq!(got.len(), want.f64_bits.len(), "{what}: element count");
for (i, (g, w)) in got.iter().zip(&want.f64_bits).enumerate() {
let w = f64::from_bits(*w);
if w.is_nan() {
// libhdf5 keeps the sign and sets every mantissa bit.
let expected = if w.is_sign_negative() {
0xFFFF_FFFF
} else {
0x7FFF_FFFF
};
assert!(want.f32_nan.contains(&expected));
assert_eq!(g.to_bits(), expected, "{what}[{i}]: NaN bits");
} else {
assert_eq!(g.to_bits(), (w as f32).to_bits(), "{what}[{i}]: {g} vs {w}");
}
}
}
#[test]
fn small_floats_decode_as_libhdf5_2_2_does() {
let file = File::open(FIXTURE).unwrap();
let root_attrs = file.root().attrs().unwrap();
let expected = reference();
assert_eq!(expected.len(), 10);
for want in &expected {
let name = &want.name;
let ds = file.dataset(name).unwrap();
check_f64(name, &ds.read_f64().unwrap(), &want.f64_bits);
check_f32(name, &ds.read_f32().unwrap(), want);
if want.attribute {
match &root_attrs[name.as_str()] {
AttrValue::F64Array(values) => {
check_f64(&format!("attribute {name}"), values, &want.f64_bits)
}
other => panic!("attribute {name}: {other:?}"),
}
}
}
}
#[test]
fn small_float_datatypes_keep_their_fields() {
use clawhdf5_format::datatype::Datatype;
let json: serde_json::Value = serde_json::from_str(REFERENCE).unwrap();
let file = File::open(FIXTURE).unwrap();
for (name, obj) in json["objects"].as_object().unwrap() {
let dt = file.dataset(name).unwrap().raw_datatype().unwrap();
let Datatype::FloatingPoint {
size,
bit_offset,
bit_precision,
exponent_location,
exponent_size,
mantissa_location,
mantissa_size,
exponent_bias,
..
} = dt
else {
panic!("{name}: {dt:?}");
};
let field = |i: usize| obj["fields"][i].as_u64().unwrap();
assert_eq!(u64::from(size), obj["size"].as_u64().unwrap(), "{name}");
assert_eq!(u64::from(bit_offset), obj["offset"].as_u64().unwrap());
assert_eq!(
u64::from(bit_precision),
obj["precision"].as_u64().unwrap(),
"{name}"
);
// Fields: sign position, exponent position and size, mantissa
// position and size.
assert_eq!(
u64::from(exponent_location) + u64::from(exponent_size),
field(0)
);
assert_eq!(u64::from(exponent_location), field(1), "{name}");
assert_eq!(u64::from(exponent_size), field(2), "{name}");
assert_eq!(u64::from(mantissa_location), field(3), "{name}");
assert_eq!(u64::from(mantissa_size), field(4), "{name}");
assert_eq!(u64::from(exponent_bias), obj["ebias"].as_u64().unwrap());
}
}