feat: check HDF5 2.x small floats against libhdf5 2.2.0
Fixture written by libhdf5 2.2.0 (built from tag 2.2.0) through ctypes: every bit pattern of FP4 E2M1, FP6 E2M3/E3M2, FP8 E4M3/E5M2 and a bfloat16 LE/BE set, as datasets and attributes, with what H5Dread/H5Aread return into double and float and the conversion exceptions libhdf5 raises. clawhdf5 already decoded every value as libhdf5 does, including an all-ones exponent as inf/NaN in the OCP formats that have none (documented as a deliberate match in known-issues). - data_read: NaNs of non-native float layouts get libhdf5's bits (sign kept, every mantissa bit set) in f64 and f32. - h5rs dump/ls name these types as h5dump/h5ls 2.x do (H5T_FLOAT_F4E2M1, "FP4 E2M1 4-bit float", float4-e2m1 ...), checked against h5dump 2.2.0's output of the fixture. - Python bindings read them as h5py 3.16 does (float32 for bfloat16, float16 for the 1-byte formats, file byte order, same bytes as h5py); writing them in 'r+' is refused. Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
@@ -37,7 +37,12 @@ with clawhdf5.File("data.h5", "r") as f:
|
||||
`dtype.metadata['enum']`), complex, `S<n>` fixed strings, `object` for
|
||||
variable-length strings (`bytes` values) and sequences (array values),
|
||||
`V<n>` opaque, array types, and compounds as structured dtypes.
|
||||
Other types raise `TypeError`.
|
||||
Non-IEEE floats (HDF5 2.x's bfloat16, FP8, FP6 and FP4) read as h5py
|
||||
3.16 reads them: as the narrowest IEEE float that holds them (`float32`
|
||||
for bfloat16, `float16` for the 1-byte formats) in the file's byte order,
|
||||
with libhdf5's values (`tests/test_small_floats.py`); inside a compound or
|
||||
array type they raise `TypeError`, and writing them raises
|
||||
`NotImplementedError`. Other types raise `TypeError`.
|
||||
- Keys are h5py's: integers, slices with a positive step, `...`, one
|
||||
increasing list of integers, compound field names. Each maps onto a
|
||||
hyperslab selection. `None` and negative steps are refused
|
||||
|
||||
@@ -8,10 +8,16 @@
|
||||
//! returns become the array's buffer as they are: the `Vec<u8>` is handed to
|
||||
//! numpy without a copy and viewed as the dtype.
|
||||
//!
|
||||
//! Anything this mapping cannot describe exactly — non-IEEE floats, integers
|
||||
//! with padding bits, VAX byte order, references, bitfields, time, and
|
||||
//! variable-length members inside compounds or arrays — is a `TypeError`,
|
||||
//! never a best-effort guess.
|
||||
//! The one exception is a dataset or attribute of a non-IEEE float
|
||||
//! (bfloat16, FP8 E4M3/E5M2, FP6, FP4, ...): like h5py, it reads as the
|
||||
//! narrowest of `float16`/`float32`/`float64` that holds every value of the
|
||||
//! format exactly, in the file's byte order, and the values are converted
|
||||
//! (as libhdf5 converts them, NaN bits included).
|
||||
//!
|
||||
//! Anything this mapping cannot describe exactly — non-IEEE floats inside
|
||||
//! compounds or arrays, integers with padding bits, VAX byte order,
|
||||
//! references, bitfields, time, and variable-length members inside compounds
|
||||
//! or arrays — is a `TypeError`, never a best-effort guess.
|
||||
|
||||
use std::collections::HashMap;
|
||||
|
||||
@@ -35,6 +41,13 @@ pub(crate) enum Layout {
|
||||
VlString { utf8: bool },
|
||||
/// Variable-length sequence of a fixed-size base type.
|
||||
VlSequence,
|
||||
/// A non-IEEE float (`source`), converted to the IEEE float of `size`
|
||||
/// bytes numpy reports (see [`widened_float`]).
|
||||
Float {
|
||||
source: Datatype,
|
||||
size: usize,
|
||||
big_endian: bool,
|
||||
},
|
||||
}
|
||||
|
||||
/// Everything needed to turn a dataset's or attribute's bytes into numpy.
|
||||
@@ -139,6 +152,74 @@ fn float_format(dt: &Datatype) -> PyResult<String> {
|
||||
Ok(format!("{}f{size}", byte_order_char(byte_order, *size)?))
|
||||
}
|
||||
|
||||
/// Whether `dt` is an IEEE 754 binary16/32/64 float numpy reads as it is.
|
||||
pub(crate) fn is_ieee_float(dt: &Datatype) -> bool {
|
||||
float_format(dt).is_ok()
|
||||
}
|
||||
|
||||
/// The IEEE float h5py reads a non-IEEE float as (`TypeFloatID.py_dtype` in
|
||||
/// h5py 3.16): the first of `float16`, `float32`, `float64` at least as wide
|
||||
/// whose mantissa holds the format's and whose normal exponent range covers
|
||||
/// it — bfloat16 as `float32`, FP8/FP6/FP4 as `float16` — in the file's byte
|
||||
/// order. Returns `(size in bytes, big endian)`, or `None` when no IEEE type
|
||||
/// holds it (or the byte order is VAX).
|
||||
fn widened_float(dt: &Datatype) -> Option<(usize, bool)> {
|
||||
let Datatype::FloatingPoint {
|
||||
size,
|
||||
byte_order,
|
||||
exponent_size,
|
||||
mantissa_size,
|
||||
exponent_bias,
|
||||
..
|
||||
} = dt
|
||||
else {
|
||||
return None;
|
||||
};
|
||||
let big_endian = match byte_order {
|
||||
DatatypeByteOrder::LittleEndian => false,
|
||||
DatatypeByteOrder::BigEndian => true,
|
||||
DatatypeByteOrder::Vax => return None,
|
||||
};
|
||||
if *exponent_size >= 32 {
|
||||
return None;
|
||||
}
|
||||
let max_exp = (1i64 << exponent_size) - i64::from(*exponent_bias) - 1;
|
||||
let min_exp = 1 - i64::from(*exponent_bias);
|
||||
// numpy's finfo: (itemsize, nmant, maxexp, minexp).
|
||||
[(2, 10, 16, -14), (4, 23, 128, -126), (8, 52, 1024, -1022)]
|
||||
.into_iter()
|
||||
.find(|&(bytes, nmant, maxexp, minexp)| {
|
||||
bytes >= *size && *mantissa_size <= nmant && max_exp <= maxexp && min_exp >= minexp
|
||||
})
|
||||
.map(|(bytes, ..)| (bytes as usize, big_endian))
|
||||
}
|
||||
|
||||
/// Encode `values` as IEEE floats of `size` bytes. Every value is exact in
|
||||
/// the target (`widened_float` chose it so); a NaN keeps its sign and gets
|
||||
/// every mantissa bit set, as libhdf5 converts NaNs.
|
||||
fn encode_floats(values: &[f64], size: usize, big_endian: bool) -> Vec<u8> {
|
||||
let mut out = Vec::with_capacity(values.len() * size);
|
||||
for &v in values {
|
||||
let bits: u64 = if v.is_nan() {
|
||||
let sign = u64::from(v.is_sign_negative()) << (size * 8 - 1);
|
||||
sign | ((1u64 << (size * 8 - 1)) - 1)
|
||||
} else {
|
||||
match size {
|
||||
2 => u64::from(clawhdf5_format::float16::f32_to_f16_bits(v as f32)),
|
||||
4 => u64::from((v as f32).to_bits()),
|
||||
_ => v.to_bits(),
|
||||
}
|
||||
};
|
||||
let bytes = bits.to_le_bytes();
|
||||
if big_endian {
|
||||
out.extend(bytes[..size].iter().rev());
|
||||
} else {
|
||||
out.extend_from_slice(&bytes[..size]);
|
||||
}
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
/// `r`/`i` compounds of two identical IEEE floats are complex numbers in h5py.
|
||||
fn complex_format(
|
||||
size: u32,
|
||||
@@ -366,6 +447,25 @@ impl Converter {
|
||||
vl_unit: 0,
|
||||
})
|
||||
}
|
||||
Datatype::FloatingPoint { .. } if !is_ieee_float(dt) => {
|
||||
let Some((size, big_endian)) = widened_float(dt) else {
|
||||
// The error names the layout.
|
||||
return Err(float_format(dt).expect_err("not IEEE"));
|
||||
};
|
||||
let order = if big_endian { '>' } else { '<' };
|
||||
let dtype = np_dtype(py, format!("{order}f{size}"))?;
|
||||
Ok(Self {
|
||||
view: dtype.clone().unbind(),
|
||||
dtype: dtype.unbind(),
|
||||
layout: Layout::Float {
|
||||
source: dt.clone(),
|
||||
size,
|
||||
big_endian,
|
||||
},
|
||||
elem_size: dt.type_size() as usize,
|
||||
vl_unit: 0,
|
||||
})
|
||||
}
|
||||
_ => {
|
||||
let dtype = fixed_dtype(py, dt)?;
|
||||
Ok(Self {
|
||||
@@ -416,6 +516,19 @@ impl Converter {
|
||||
(Elements::Bytes(bytes), Layout::Fixed) => {
|
||||
bytes_as_array(py, bytes, self.view.bind(py), shape)
|
||||
}
|
||||
(
|
||||
Elements::Bytes(bytes),
|
||||
Layout::Float {
|
||||
source,
|
||||
size,
|
||||
big_endian,
|
||||
},
|
||||
) => {
|
||||
let values = clawhdf5_format::data_read::read_as_f64(&bytes, source)
|
||||
.map_err(|e| unsupported(e.to_string()))?;
|
||||
let converted = encode_floats(&values, *size, *big_endian);
|
||||
bytes_as_array(py, converted, self.view.bind(py), shape)
|
||||
}
|
||||
(Elements::Bytes(bytes), Layout::Subarray(dims)) => {
|
||||
let mut full = shape.to_vec();
|
||||
full.extend_from_slice(dims);
|
||||
|
||||
@@ -48,7 +48,12 @@ fn not_implemented(what: impl std::fmt::Display) -> PyErr {
|
||||
pub(crate) fn category(dt: &Datatype) -> PyResult<&'static str> {
|
||||
match dt {
|
||||
Datatype::FixedPoint { .. } => Ok("int"),
|
||||
Datatype::FloatingPoint { .. } => Ok("float"),
|
||||
Datatype::FloatingPoint { .. } if crate::convert::is_ieee_float(dt) => Ok("float"),
|
||||
// Read as a wider IEEE float; writing would need the reverse
|
||||
// conversion (rounding into bfloat16, FP8, ...).
|
||||
Datatype::FloatingPoint { .. } => Err(not_implemented(
|
||||
"writing non-IEEE floats (bfloat16, FP8, FP6, FP4, ...)",
|
||||
)),
|
||||
Datatype::Enumeration {
|
||||
base_type, members, ..
|
||||
} => {
|
||||
|
||||
@@ -0,0 +1,92 @@
|
||||
"""Non-IEEE floats (bfloat16, FP8 E4M3/E5M2, FP6 E2M3/E3M2, FP4 E2M1) read
|
||||
as h5py 3.16 reads them: as the narrowest IEEE float that holds every value
|
||||
(bfloat16 as float32, the 1-byte formats as float16), in the file's byte
|
||||
order, with the values libhdf5 converts them to.
|
||||
|
||||
The fixture was written by libhdf5 2.2.0; the JSON next to it holds what
|
||||
libhdf5 2.2.0 itself returns for every element (see gen_mx_floats.py)."""
|
||||
|
||||
import json
|
||||
import os
|
||||
import shutil
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
|
||||
import clawhdf5
|
||||
|
||||
FIXTURES = os.path.join(os.path.dirname(__file__), "..", "..", "clawhdf5", "tests", "fixtures")
|
||||
FILE = os.path.join(FIXTURES, "mx_floats_hdf5_2_2.h5")
|
||||
REFERENCE = json.load(open(os.path.join(FIXTURES, "mx_floats_hdf5_2_2.json")))["objects"]
|
||||
|
||||
# What h5py 3.16 reports for each (checked 2026-09-28).
|
||||
DTYPES = {
|
||||
"bf16le": "<f4",
|
||||
"bf16be": ">f4",
|
||||
"f8e4m3": "<f2",
|
||||
"f8e5m2": "<f2",
|
||||
"f6e2m3": "<f2",
|
||||
"f6e2m3_pad": "<f2",
|
||||
"f6e3m2": "<f2",
|
||||
"f6e3m2_pad": "<f2",
|
||||
"f4e2m1": "<f2",
|
||||
"f4e2m1_pad": "<f2",
|
||||
}
|
||||
|
||||
|
||||
def _expected(name, dtype):
|
||||
"""The libhdf5 2.2.0 values as `dtype`, NaN bits included (sign kept,
|
||||
every mantissa bit set)."""
|
||||
out = []
|
||||
for text in REFERENCE[name]["f64"]:
|
||||
if text.startswith("nan:"):
|
||||
negative = int(text[4:], 16) >> 63
|
||||
bits = dtype.itemsize * 8
|
||||
word = (negative << (bits - 1)) | ((1 << (bits - 1)) - 1)
|
||||
out.append(np.frombuffer(word.to_bytes(dtype.itemsize, "little"), dtype.newbyteorder("<"))[0])
|
||||
else:
|
||||
out.append(float(text))
|
||||
return np.array(out, dtype=dtype.newbyteorder("<")).astype(dtype)
|
||||
|
||||
|
||||
def test_values_match_libhdf5_2_2():
|
||||
assert set(DTYPES) == set(REFERENCE)
|
||||
with clawhdf5.File(FILE, "r") as f:
|
||||
for name, dtype in DTYPES.items():
|
||||
ds = f[name]
|
||||
assert ds.dtype == np.dtype(dtype), name
|
||||
assert ds.dtype.str == dtype, name
|
||||
want = _expected(name, np.dtype(dtype))
|
||||
got = ds[()]
|
||||
assert got.dtype.str == dtype
|
||||
assert got.tobytes() == want.tobytes(), name
|
||||
# Selections convert the same way.
|
||||
assert ds[3:9].tobytes() == want[3:9].tobytes(), name
|
||||
assert ds[[0, 5, 7]].tobytes() == want[[0, 5, 7]].tobytes(), name
|
||||
assert ds[5].tobytes() == want[5].tobytes(), name
|
||||
if REFERENCE[name]["attribute"]:
|
||||
attr = f.attrs[name]
|
||||
assert attr.dtype.str == dtype
|
||||
assert attr.tobytes() == want.tobytes(), name
|
||||
|
||||
|
||||
def test_matches_h5py(h5py):
|
||||
with clawhdf5.File(FILE, "r") as ours, h5py.File(FILE, "r") as theirs:
|
||||
for name in DTYPES:
|
||||
assert ours[name].dtype == theirs[name].dtype, name
|
||||
assert ours[name][()].tobytes() == theirs[name][()].tobytes(), name
|
||||
if name in theirs.attrs:
|
||||
assert ours.attrs[name].dtype == theirs.attrs[name].dtype
|
||||
assert ours.attrs[name].tobytes() == theirs.attrs[name].tobytes()
|
||||
|
||||
|
||||
def test_writing_is_refused(tmp_path):
|
||||
path = tmp_path / "copy.h5"
|
||||
shutil.copyfile(FILE, path)
|
||||
with clawhdf5.File(str(path), "r+") as f:
|
||||
with pytest.raises(NotImplementedError, match="non-IEEE"):
|
||||
f["bf16le"][0] = 1.0
|
||||
with pytest.raises(NotImplementedError, match="non-IEEE"):
|
||||
f["f4e2m1"][:] = np.zeros(16)
|
||||
with open(path, "rb") as a, open(FILE, "rb") as b:
|
||||
assert a.read() == b.read()
|
||||
Reference in New Issue
Block a user