- Datatype::Complex serializes class 11 version 5 byte-identically to
libhdf5 2.2.0; containers holding it are written as version 5.
- DatasetBuilder::with_complex_f32/f64_data (h5py's {r, i} compound,
default) and with_native_complex_f32/f64_data (class 11, opt-in);
make_(native_)complex_f32/f64_type for attributes.
- Dataset::read_complex_f64/f32 read either form.
- Python create_dataset accepts complex64/complex128 (compound form).
- Parsing unchanged: class 11 still surfaces as {r, i}.
- Tests vs h5py 3.16 / libhdf5 2.0.0 and h5dump 2.2.0; docs.
Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
1164 lines
43 KiB
Rust
1164 lines
43 KiB
Rust
//! h5py round-trip tests for the file writer.
|
|
//!
|
|
//! These tests write HDF5 files with our writer and verify h5py can read them
|
|
//! (and vice versa). They require python3 + h5py to be installed.
|
|
|
|
use clawhdf5_format::file_writer::{AttrValue, CompoundTypeBuilder, EnumTypeBuilder, FileWriter};
|
|
/// The Python interpreter to drive interop checks with.
|
|
///
|
|
/// `CLAWHDF5_PYTHON` lets these run against a virtualenv holding h5py, which
|
|
/// on a PEP 668 "externally managed" system is the only place it can be
|
|
/// installed. Without it the suite silently skips, and a silent skip here is
|
|
/// how a datatype bug once reached a release.
|
|
fn python() -> String {
|
|
std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string())
|
|
}
|
|
|
|
fn h5py_available() -> bool {
|
|
std::process::Command::new(python())
|
|
.args(["-c", "import h5py"])
|
|
.output()
|
|
.map(|o| o.status.success())
|
|
.unwrap_or(false)
|
|
}
|
|
|
|
fn h5py_read(_path: &std::path::Path, script: &str) -> String {
|
|
if !h5py_available() {
|
|
panic!("h5py not installed — skipping interop test");
|
|
}
|
|
let o = std::process::Command::new(python())
|
|
.args(["-c", script])
|
|
.output()
|
|
.expect("python interpreter");
|
|
if !o.status.success() {
|
|
panic!("h5py: {}", String::from_utf8_lossy(&o.stderr));
|
|
}
|
|
String::from_utf8(o.stdout).unwrap().trim().to_string()
|
|
}
|
|
|
|
// ---- h5py round-trip: basic datasets ----
|
|
|
|
#[test]
|
|
#[ignore = "requires Python h5py module"]
|
|
fn h5py_reads_our_f64_dataset() {
|
|
let mut fw = FileWriter::new();
|
|
fw.create_dataset("data")
|
|
.with_f64_data(&[1.0, 2.0, 3.0])
|
|
.with_shape(&[3]);
|
|
let bytes = fw.finish().unwrap();
|
|
let path = std::env::temp_dir().join("clawhdf5_test_f64.h5");
|
|
std::fs::write(&path, &bytes).unwrap();
|
|
let script = format!(
|
|
"import h5py, json; f=h5py.File('{}','r'); print(json.dumps(f['data'][:].tolist()))",
|
|
path.display()
|
|
);
|
|
let stdout = h5py_read(&path, &script);
|
|
let values: Vec<f64> = serde_json::from_str(&stdout).unwrap();
|
|
assert_eq!(values, vec![1.0, 2.0, 3.0]);
|
|
}
|
|
|
|
#[test]
|
|
#[ignore = "requires Python h5py module"]
|
|
fn h5py_reads_our_i32_dataset() {
|
|
let mut fw = FileWriter::new();
|
|
fw.create_dataset("ints").with_i32_data(&[10, 20, 30]);
|
|
let bytes = fw.finish().unwrap();
|
|
let path = std::env::temp_dir().join("clawhdf5_test_i32.h5");
|
|
std::fs::write(&path, &bytes).unwrap();
|
|
let script = format!(
|
|
"import h5py, json; f=h5py.File('{}','r'); print(json.dumps(f['ints'][:].tolist()))",
|
|
path.display()
|
|
);
|
|
let stdout = h5py_read(&path, &script);
|
|
let values: Vec<i32> = serde_json::from_str(&stdout).unwrap();
|
|
assert_eq!(values, vec![10, 20, 30]);
|
|
}
|
|
|
|
#[test]
|
|
#[ignore = "requires Python h5py module"]
|
|
fn h5py_reads_dataset_with_attrs() {
|
|
let mut fw = FileWriter::new();
|
|
fw.create_dataset("data")
|
|
.with_f64_data(&[1.0, 2.0])
|
|
.set_attr("scale", AttrValue::F64(0.5));
|
|
let bytes = fw.finish().unwrap();
|
|
let path = std::env::temp_dir().join("clawhdf5_test_attrs.h5");
|
|
std::fs::write(&path, &bytes).unwrap();
|
|
let script = format!(
|
|
"import h5py, json; f=h5py.File('{}','r'); d=f['data']; print(json.dumps({{'data': d[:].tolist(), 'scale': float(d.attrs['scale'])}}))",
|
|
path.display()
|
|
);
|
|
let stdout = h5py_read(&path, &script);
|
|
let v: serde_json::Value = serde_json::from_str(&stdout).unwrap();
|
|
assert_eq!(v["data"], serde_json::json!([1.0, 2.0]));
|
|
assert_eq!(v["scale"], serde_json::json!(0.5));
|
|
}
|
|
|
|
#[test]
|
|
#[ignore = "requires Python h5py module"]
|
|
fn h5py_reads_group_with_dataset() {
|
|
let mut fw = FileWriter::new();
|
|
let mut gb = fw.create_group("grp");
|
|
gb.create_dataset("vals").with_f64_data(&[10.0, 20.0]);
|
|
fw.add_group(gb.finish());
|
|
let bytes = fw.finish().unwrap();
|
|
let path = std::env::temp_dir().join("clawhdf5_test_grp.h5");
|
|
std::fs::write(&path, &bytes).unwrap();
|
|
let script = format!(
|
|
"import h5py, json; f=h5py.File('{}','r'); print(json.dumps(f['grp/vals'][:].tolist()))",
|
|
path.display()
|
|
);
|
|
let stdout = h5py_read(&path, &script);
|
|
let values: Vec<f64> = serde_json::from_str(&stdout).unwrap();
|
|
assert_eq!(values, vec![10.0, 20.0]);
|
|
}
|
|
|
|
#[test]
|
|
#[ignore = "requires Python h5py module"]
|
|
fn h5py_reads_root_attrs() {
|
|
let mut fw = FileWriter::new();
|
|
fw.set_root_attr("version", AttrValue::I64(42));
|
|
fw.create_dataset("dummy").with_f64_data(&[0.0]);
|
|
let bytes = fw.finish().unwrap();
|
|
let path = std::env::temp_dir().join("clawhdf5_test_root_attrs.h5");
|
|
std::fs::write(&path, &bytes).unwrap();
|
|
let script = format!(
|
|
"import h5py, json; f=h5py.File('{}','r'); print(int(f.attrs['version']))",
|
|
path.display()
|
|
);
|
|
let stdout = h5py_read(&path, &script);
|
|
assert_eq!(stdout, "42");
|
|
}
|
|
|
|
#[test]
|
|
#[ignore = "requires Python h5py module"]
|
|
fn h5py_reads_multiple_datasets() {
|
|
let mut fw = FileWriter::new();
|
|
fw.create_dataset("a").with_f64_data(&[1.0]);
|
|
fw.create_dataset("b").with_f64_data(&[2.0]);
|
|
fw.create_dataset("c").with_f64_data(&[3.0]);
|
|
let bytes = fw.finish().unwrap();
|
|
let path = std::env::temp_dir().join("clawhdf5_test_multi.h5");
|
|
std::fs::write(&path, &bytes).unwrap();
|
|
let script = format!(
|
|
"import h5py, json; f=h5py.File('{}','r'); print(json.dumps({{k: f[k][:].tolist() for k in ['a','b','c']}}))",
|
|
path.display()
|
|
);
|
|
let stdout = h5py_read(&path, &script);
|
|
let v: serde_json::Value = serde_json::from_str(&stdout).unwrap();
|
|
assert_eq!(v["a"], serde_json::json!([1.0]));
|
|
assert_eq!(v["b"], serde_json::json!([2.0]));
|
|
assert_eq!(v["c"], serde_json::json!([3.0]));
|
|
}
|
|
|
|
// ---- Compound / Enum / Array h5py round-trips ----
|
|
|
|
#[test]
|
|
#[ignore = "requires Python h5py module"]
|
|
fn h5py_reads_our_compound_dataset() {
|
|
let ct = CompoundTypeBuilder::new()
|
|
.f64_field("x")
|
|
.f64_field("y")
|
|
.i32_field("id")
|
|
.build();
|
|
let mut raw = Vec::new();
|
|
raw.extend_from_slice(&1.5f64.to_le_bytes());
|
|
raw.extend_from_slice(&2.5f64.to_le_bytes());
|
|
raw.extend_from_slice(&10i32.to_le_bytes());
|
|
raw.extend_from_slice(&3.5f64.to_le_bytes());
|
|
raw.extend_from_slice(&4.5f64.to_le_bytes());
|
|
raw.extend_from_slice(&20i32.to_le_bytes());
|
|
|
|
let mut fw = FileWriter::new();
|
|
fw.create_dataset("particles")
|
|
.with_compound_data(ct, raw, 2);
|
|
let bytes = fw.finish().unwrap();
|
|
let path = std::env::temp_dir().join("clawhdf5_compound.h5");
|
|
std::fs::write(&path, &bytes).unwrap();
|
|
let script = format!(
|
|
"import h5py,json; f=h5py.File('{}','r'); d=f['particles']; print(json.dumps({{'x':d['x'].tolist(),'y':d['y'].tolist(),'id':d['id'].tolist()}}))",
|
|
path.display()
|
|
);
|
|
let stdout = h5py_read(&path, &script);
|
|
let v: serde_json::Value = serde_json::from_str(&stdout).unwrap();
|
|
assert_eq!(v["x"], serde_json::json!([1.5, 3.5]));
|
|
assert_eq!(v["y"], serde_json::json!([2.5, 4.5]));
|
|
assert_eq!(v["id"], serde_json::json!([10, 20]));
|
|
}
|
|
|
|
#[test]
|
|
#[ignore = "requires Python h5py module"]
|
|
fn h5py_reads_our_enum_dataset() {
|
|
let et = EnumTypeBuilder::i32_based()
|
|
.value("RED", 0)
|
|
.value("GREEN", 1)
|
|
.value("BLUE", 2)
|
|
.build();
|
|
|
|
let mut fw = FileWriter::new();
|
|
fw.create_dataset("colors")
|
|
.with_enum_i32_data(et, &[1, 0, 2]);
|
|
let bytes = fw.finish().unwrap();
|
|
let path = std::env::temp_dir().join("clawhdf5_enum.h5");
|
|
std::fs::write(&path, &bytes).unwrap();
|
|
let script = format!(
|
|
"import h5py,json; f=h5py.File('{}','r'); d=f['colors']; print(json.dumps(d[:].tolist()))",
|
|
path.display()
|
|
);
|
|
let stdout = h5py_read(&path, &script);
|
|
let values: Vec<i32> = serde_json::from_str(&stdout).unwrap();
|
|
assert_eq!(values, vec![1, 0, 2]);
|
|
}
|
|
|
|
#[test]
|
|
#[ignore = "requires Python h5py module"]
|
|
fn h5py_reads_our_array_dataset() {
|
|
let mut raw = Vec::new();
|
|
for v in &[1.0f64, 2.0, 3.0, 4.0, 5.0, 6.0] {
|
|
raw.extend_from_slice(&v.to_le_bytes());
|
|
}
|
|
|
|
let mut fw = FileWriter::new();
|
|
fw.create_dataset("vectors").with_array_data(
|
|
clawhdf5_format::type_builders::make_f64_type(),
|
|
&[3],
|
|
raw,
|
|
2,
|
|
);
|
|
let bytes = fw.finish().unwrap();
|
|
let path = std::env::temp_dir().join("clawhdf5_array.h5");
|
|
std::fs::write(&path, &bytes).unwrap();
|
|
let script = format!(
|
|
"import h5py,json; f=h5py.File('{}','r'); d=f['vectors']; print(json.dumps({{'shape':list(d.shape),'dtype':str(d.dtype),'values':d[:].tolist()}}))",
|
|
path.display()
|
|
);
|
|
let stdout = h5py_read(&path, &script);
|
|
let v: serde_json::Value = serde_json::from_str(&stdout).unwrap();
|
|
assert_eq!(v["shape"], serde_json::json!([2]));
|
|
assert_eq!(
|
|
v["values"],
|
|
serde_json::json!([[1.0, 2.0, 3.0], [4.0, 5.0, 6.0]])
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
#[ignore = "requires Python h5py module"]
|
|
fn read_h5py_generated_compound() {
|
|
check_h5py_generated_compound("latest", ", libver='latest'");
|
|
}
|
|
|
|
/// Same file written with h5py's default format bounds. HDF5 2.0 raised the
|
|
/// default low bound to 1.8, so "default" files exercise different on-disk
|
|
/// structures than both `libver='latest'` and pre-2.0 defaults.
|
|
#[test]
|
|
#[ignore = "requires Python h5py module"]
|
|
fn read_h5py_generated_compound_default_libver() {
|
|
check_h5py_generated_compound("default", "");
|
|
}
|
|
|
|
fn check_h5py_generated_compound(tag: &str, libver_kw: &str) {
|
|
let path = std::env::temp_dir().join(format!("clawhdf5_h5py_compound_{tag}.h5"));
|
|
let gen_script = format!(
|
|
r#"
|
|
import h5py, numpy as np
|
|
dt = np.dtype([('x', 'f8'), ('y', 'f8'), ('id', 'i4')])
|
|
data = np.array([(1.0, 2.0, 10), (3.0, 4.0, 20)], dtype=dt)
|
|
f = h5py.File('{}', 'w'{})
|
|
f.create_dataset('particles', data=data)
|
|
f.close()
|
|
"#,
|
|
path.display(),
|
|
libver_kw
|
|
);
|
|
h5py_read(&path, &gen_script);
|
|
|
|
let bytes = std::fs::read(&path).unwrap();
|
|
let sig = clawhdf5_format::signature::find_signature(&bytes).unwrap();
|
|
let sb = clawhdf5_format::superblock::Superblock::parse(&bytes, sig).unwrap();
|
|
let addr = clawhdf5_format::group_v2::resolve_path_any(&bytes, &sb, "particles").unwrap();
|
|
let hdr = clawhdf5_format::object_header::ObjectHeader::parse(
|
|
&bytes,
|
|
addr as usize,
|
|
sb.offset_size,
|
|
sb.length_size,
|
|
)
|
|
.unwrap();
|
|
let dt_data = &hdr
|
|
.messages
|
|
.iter()
|
|
.find(|m| m.msg_type == clawhdf5_format::message_type::MessageType::Datatype)
|
|
.unwrap()
|
|
.data;
|
|
let ds_data = &hdr
|
|
.messages
|
|
.iter()
|
|
.find(|m| m.msg_type == clawhdf5_format::message_type::MessageType::Dataspace)
|
|
.unwrap()
|
|
.data;
|
|
let dl_data = &hdr
|
|
.messages
|
|
.iter()
|
|
.find(|m| m.msg_type == clawhdf5_format::message_type::MessageType::DataLayout)
|
|
.unwrap()
|
|
.data;
|
|
let (dt, _) = clawhdf5_format::datatype::Datatype::parse(dt_data).unwrap();
|
|
let ds = clawhdf5_format::dataspace::Dataspace::parse(ds_data, sb.length_size).unwrap();
|
|
let dl =
|
|
clawhdf5_format::data_layout::DataLayout::parse(dl_data, sb.offset_size, sb.length_size)
|
|
.unwrap();
|
|
let raw = clawhdf5_format::data_read::read_raw_data(&bytes, &dl, &ds, &dt).unwrap();
|
|
let fields = clawhdf5_format::data_read::read_compound_fields(&raw, &dt).unwrap();
|
|
assert_eq!(fields.len(), 3);
|
|
assert_eq!(fields[0].name, "x");
|
|
let x_vals =
|
|
clawhdf5_format::data_read::read_as_f64(&fields[0].raw_data, &fields[0].datatype).unwrap();
|
|
assert_eq!(x_vals, vec![1.0, 3.0]);
|
|
}
|
|
|
|
#[test]
|
|
#[ignore = "requires Python h5py module"]
|
|
fn read_h5py_generated_native_complex() {
|
|
// HDF5 2.0 native complex (datatype class 11, version 5), written through
|
|
// h5py's low-level API. Skips when the linked HDF5 predates 2.0.
|
|
let path = std::env::temp_dir().join("clawhdf5_h5py_native_complex.h5");
|
|
let gen_script = format!(
|
|
r#"
|
|
import h5py, numpy as np
|
|
from h5py import h5t, h5s, h5d, h5f, h5p
|
|
if not getattr(h5py.get_config(), 'has_native_complex', False):
|
|
print('SKIP')
|
|
else:
|
|
fapl = h5p.create(h5p.FILE_ACCESS)
|
|
fapl.set_libver_bounds(h5f.LIBVER_LATEST, h5f.LIBVER_LATEST)
|
|
fid = h5f.create(b'{}', h5f.ACC_TRUNC, fapl=fapl)
|
|
t = h5t.COMPLEX_IEEE_F64LE
|
|
d = h5d.create(fid, b'z', t, h5s.create_simple((2,)))
|
|
d.write(h5s.ALL, h5s.ALL, np.array([1+2j, 3+4j], dtype=np.complex128), mtype=t)
|
|
fid.close()
|
|
"#,
|
|
path.display()
|
|
);
|
|
if h5py_read(&path, &gen_script) == "SKIP" {
|
|
eprintln!("HDF5 < 2.0: no native complex support, skipping");
|
|
return;
|
|
}
|
|
|
|
let bytes = std::fs::read(&path).unwrap();
|
|
let sig = clawhdf5_format::signature::find_signature(&bytes).unwrap();
|
|
let sb = clawhdf5_format::superblock::Superblock::parse(&bytes, sig).unwrap();
|
|
let addr = clawhdf5_format::group_v2::resolve_path_any(&bytes, &sb, "z").unwrap();
|
|
let hdr = clawhdf5_format::object_header::ObjectHeader::parse(
|
|
&bytes,
|
|
addr as usize,
|
|
sb.offset_size,
|
|
sb.length_size,
|
|
)
|
|
.unwrap();
|
|
let msg = |t: clawhdf5_format::message_type::MessageType| {
|
|
&hdr.messages.iter().find(|m| m.msg_type == t).unwrap().data
|
|
};
|
|
let (dt, _) = clawhdf5_format::datatype::Datatype::parse(msg(
|
|
clawhdf5_format::message_type::MessageType::Datatype,
|
|
))
|
|
.unwrap();
|
|
let ds = clawhdf5_format::dataspace::Dataspace::parse(
|
|
msg(clawhdf5_format::message_type::MessageType::Dataspace),
|
|
sb.length_size,
|
|
)
|
|
.unwrap();
|
|
let dl = clawhdf5_format::data_layout::DataLayout::parse(
|
|
msg(clawhdf5_format::message_type::MessageType::DataLayout),
|
|
sb.offset_size,
|
|
sb.length_size,
|
|
)
|
|
.unwrap();
|
|
let raw = clawhdf5_format::data_read::read_raw_data(&bytes, &dl, &ds, &dt).unwrap();
|
|
let fields = clawhdf5_format::data_read::read_compound_fields(&raw, &dt).unwrap();
|
|
assert_eq!(fields.len(), 2);
|
|
let re =
|
|
clawhdf5_format::data_read::read_as_f64(&fields[0].raw_data, &fields[0].datatype).unwrap();
|
|
let im =
|
|
clawhdf5_format::data_read::read_as_f64(&fields[1].raw_data, &fields[1].datatype).unwrap();
|
|
assert_eq!((fields[0].name.as_str(), re), ("r", vec![1.0, 3.0]));
|
|
assert_eq!((fields[1].name.as_str(), im), ("i", vec![2.0, 4.0]));
|
|
}
|
|
|
|
/// Complex data both ways we write it: h5py's compound `{r, i}` (the
|
|
/// default, readable everywhere) and HDF5 2.0's native complex type (class
|
|
/// 11, opt-in). h5py on libhdf5 2.0+ must read the native datasets and
|
|
/// attributes as numpy `complex64`/`complex128`, and a compound holding a
|
|
/// native complex member (written as datatype version 5, as libhdf5 does).
|
|
/// Skips the native checks when h5py's libhdf5 predates 2.0; also runs
|
|
/// h5dump 2.x when `CLAWHDF5_H5DUMP2` names one.
|
|
#[test]
|
|
#[ignore = "requires Python h5py module"]
|
|
fn h5py_reads_our_complex_datasets_and_attributes() {
|
|
use clawhdf5_format::type_builders::{
|
|
make_complex_f64_type, make_native_complex_f32_type, make_native_complex_f64_type,
|
|
};
|
|
let path = std::env::temp_dir().join("clawhdf5_test_complex.h5");
|
|
let native_ok = h5py_read(
|
|
&path,
|
|
"import h5py; print(int(getattr(h5py.get_config(), 'has_native_complex', False)))",
|
|
) == "1";
|
|
|
|
let z64 = [[1.5f32, -2.0], [0.0, 3.25], [-7.0, 1.0e-3]];
|
|
let z128 = [[1.0f64, 2.0], [-3.5, 4.0e300], [0.25, -0.0]];
|
|
let big: Vec<[f64; 2]> = (0..600).map(|k| [k as f64, -(k as f64) / 4.0]).collect();
|
|
let c128 = |re: f64, im: f64| [re.to_le_bytes(), im.to_le_bytes()].concat();
|
|
|
|
let mut fw = FileWriter::new();
|
|
fw.create_dataset("compound128")
|
|
.with_complex_f64_data(&z128)
|
|
.set_attr(
|
|
"c",
|
|
AttrValue::Raw {
|
|
datatype: make_complex_f64_type(),
|
|
shape: vec![],
|
|
data: c128(1.0, -1.0),
|
|
},
|
|
);
|
|
if native_ok {
|
|
fw.create_dataset("native64")
|
|
.with_native_complex_f32_data(&z64)
|
|
.set_attr(
|
|
"c",
|
|
AttrValue::Raw {
|
|
datatype: make_native_complex_f64_type(),
|
|
shape: vec![2],
|
|
data: [c128(0.5, -1.5), c128(2.0, 3.0)].concat(),
|
|
},
|
|
);
|
|
fw.create_dataset("native128")
|
|
.with_native_complex_f64_data(&z128);
|
|
fw.create_dataset("native_chunked")
|
|
.with_native_complex_f64_data(&big)
|
|
.with_shape(&[20, 30])
|
|
.with_chunks(&[7, 16])
|
|
.with_deflate(4);
|
|
// A compound with a native complex member, and an array of them.
|
|
let rec = CompoundTypeBuilder::new()
|
|
.field("z", make_native_complex_f64_type())
|
|
.i64_field("k")
|
|
.build();
|
|
let raw = [
|
|
c128(1.0, 2.0),
|
|
7i64.to_le_bytes().to_vec(),
|
|
c128(-1.0, 0.5),
|
|
(-8i64).to_le_bytes().to_vec(),
|
|
]
|
|
.concat();
|
|
fw.create_dataset("records").with_compound_data(rec, raw, 2);
|
|
let arr: Vec<u8> = [1.0f32, 2.0, 3.0, 4.0, 5.0, 6.0, 7.0, 8.0]
|
|
.iter()
|
|
.flat_map(|v| v.to_le_bytes())
|
|
.collect();
|
|
fw.create_dataset("arrays")
|
|
.with_array_data(make_native_complex_f32_type(), &[2], arr, 2);
|
|
fw.set_root_attr(
|
|
"zroot",
|
|
AttrValue::Raw {
|
|
datatype: make_native_complex_f32_type(),
|
|
shape: vec![],
|
|
data: [4.0f32.to_le_bytes(), (-4.0f32).to_le_bytes()].concat(),
|
|
},
|
|
);
|
|
}
|
|
std::fs::write(&path, fw.finish().unwrap()).unwrap();
|
|
|
|
let script = format!(
|
|
r#"
|
|
import h5py, json, numpy as np
|
|
f = h5py.File('{}', 'r')
|
|
def z(a): return [[float(np.real(v)), float(np.imag(v))] for v in np.asarray(a).ravel()]
|
|
out = {{}}
|
|
d = f['compound128']
|
|
out['compound128'] = [str(d.dtype), z(d[()]), str(d.attrs['c'].dtype), z(d.attrs['c'])]
|
|
if {native}:
|
|
for n in ['native64', 'native128', 'native_chunked']:
|
|
d = f[n]
|
|
out[n] = [str(d.dtype), list(d.shape), z(d[()]), d.id.get_type().get_class()]
|
|
a = f['native64'].attrs['c']
|
|
out['attr'] = [str(a.dtype), z(a)]
|
|
r = f['records'][()]
|
|
out['records'] = [str(r.dtype['z']), z(r['z']), r['k'].tolist()]
|
|
a = f['arrays'][()]
|
|
out['arrays'] = [str(a.dtype), list(a.shape), z(a)]
|
|
a = f.attrs['zroot']
|
|
out['zroot'] = [str(a.dtype), z(a)]
|
|
out['CLASS'] = h5py.h5t.COMPLEX
|
|
print(json.dumps(out))
|
|
"#,
|
|
path.display(),
|
|
native = if native_ok { "True" } else { "False" },
|
|
);
|
|
let v: serde_json::Value = serde_json::from_str(&h5py_read(&path, &script)).unwrap();
|
|
let pairs = |p: &[[f64; 2]]| serde_json::json!(p);
|
|
assert_eq!(v["compound128"][0], "complex128");
|
|
assert_eq!(v["compound128"][1], pairs(&z128));
|
|
assert_eq!(v["compound128"][2], "complex128");
|
|
assert_eq!(v["compound128"][3], pairs(&[[1.0, -1.0]]));
|
|
if !native_ok {
|
|
eprintln!("HDF5 < 2.0: native complex not checked");
|
|
return;
|
|
}
|
|
let z64_wide: Vec<[f64; 2]> = z64.iter().map(|p| [p[0].into(), p[1].into()]).collect();
|
|
let class_complex = v["CLASS"].clone();
|
|
for (name, dtype, shape, values) in [
|
|
("native64", "complex64", vec![3], z64_wide.clone()),
|
|
("native128", "complex128", vec![3], z128.to_vec()),
|
|
("native_chunked", "complex128", vec![20, 30], big.clone()),
|
|
] {
|
|
assert_eq!(v[name][0], dtype, "{name}");
|
|
assert_eq!(v[name][1], serde_json::json!(shape), "{name}");
|
|
assert_eq!(v[name][2], pairs(&values), "{name}");
|
|
// Stored as class 11, not converted from a compound.
|
|
assert_eq!(v[name][3], class_complex, "{name}");
|
|
}
|
|
assert_eq!(v["attr"][0], "complex128");
|
|
assert_eq!(v["attr"][1], pairs(&[[0.5, -1.5], [2.0, 3.0]]));
|
|
assert_eq!(v["records"][0], "complex128");
|
|
assert_eq!(v["records"][1], pairs(&[[1.0, 2.0], [-1.0, 0.5]]));
|
|
assert_eq!(v["records"][2], serde_json::json!([7, -8]));
|
|
assert_eq!(v["arrays"][0], "complex64");
|
|
assert_eq!(v["arrays"][1], serde_json::json!([2, 2]));
|
|
assert_eq!(
|
|
v["arrays"][2],
|
|
pairs(&[[1.0, 2.0], [3.0, 4.0], [5.0, 6.0], [7.0, 8.0]])
|
|
);
|
|
assert_eq!(v["zroot"][0], "complex64");
|
|
assert_eq!(v["zroot"][1], pairs(&[[4.0, -4.0]]));
|
|
|
|
// h5dump from libhdf5 2.x, when available (Debian's 1.14 cannot read
|
|
// class 11 at all).
|
|
if let Ok(h5dump) = std::env::var("CLAWHDF5_H5DUMP2") {
|
|
let o = std::process::Command::new(&h5dump)
|
|
.arg(&path)
|
|
.output()
|
|
.expect("run h5dump");
|
|
let text = String::from_utf8_lossy(&o.stdout);
|
|
assert!(
|
|
o.status.success(),
|
|
"h5dump: {}",
|
|
String::from_utf8_lossy(&o.stderr)
|
|
);
|
|
assert!(text.contains("H5T_COMPLEX"), "{text}");
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
#[ignore = "requires Python h5py module"]
|
|
fn read_h5py_generated_enum() {
|
|
check_h5py_generated_enum("latest", ", libver='latest'");
|
|
}
|
|
|
|
/// Same file written with h5py's default format bounds. HDF5 2.0 raised the
|
|
/// default low bound to 1.8, so "default" files exercise different on-disk
|
|
/// structures than both `libver='latest'` and pre-2.0 defaults.
|
|
#[test]
|
|
#[ignore = "requires Python h5py module"]
|
|
fn read_h5py_generated_enum_default_libver() {
|
|
check_h5py_generated_enum("default", "");
|
|
}
|
|
|
|
fn check_h5py_generated_enum(tag: &str, libver_kw: &str) {
|
|
let path = std::env::temp_dir().join(format!("clawhdf5_h5py_enum_{tag}.h5"));
|
|
let gen_script = format!(
|
|
r#"
|
|
import h5py, numpy as np
|
|
dt = h5py.enum_dtype({{"RED": 0, "GREEN": 1, "BLUE": 2}}, basetype=np.int32)
|
|
data = np.array([1, 0, 2, 1], dtype=np.int32)
|
|
f = h5py.File('{}', 'w'{})
|
|
f.create_dataset('colors', data=data, dtype=dt)
|
|
f.close()
|
|
"#,
|
|
path.display(),
|
|
libver_kw
|
|
);
|
|
h5py_read(&path, &gen_script);
|
|
|
|
let bytes = std::fs::read(&path).unwrap();
|
|
let sig = clawhdf5_format::signature::find_signature(&bytes).unwrap();
|
|
let sb = clawhdf5_format::superblock::Superblock::parse(&bytes, sig).unwrap();
|
|
let addr = clawhdf5_format::group_v2::resolve_path_any(&bytes, &sb, "colors").unwrap();
|
|
let hdr = clawhdf5_format::object_header::ObjectHeader::parse(
|
|
&bytes,
|
|
addr as usize,
|
|
sb.offset_size,
|
|
sb.length_size,
|
|
)
|
|
.unwrap();
|
|
let dt_data = &hdr
|
|
.messages
|
|
.iter()
|
|
.find(|m| m.msg_type == clawhdf5_format::message_type::MessageType::Datatype)
|
|
.unwrap()
|
|
.data;
|
|
let ds_data = &hdr
|
|
.messages
|
|
.iter()
|
|
.find(|m| m.msg_type == clawhdf5_format::message_type::MessageType::Dataspace)
|
|
.unwrap()
|
|
.data;
|
|
let dl_data = &hdr
|
|
.messages
|
|
.iter()
|
|
.find(|m| m.msg_type == clawhdf5_format::message_type::MessageType::DataLayout)
|
|
.unwrap()
|
|
.data;
|
|
let (dt, _) = clawhdf5_format::datatype::Datatype::parse(dt_data).unwrap();
|
|
let ds = clawhdf5_format::dataspace::Dataspace::parse(ds_data, sb.length_size).unwrap();
|
|
let dl =
|
|
clawhdf5_format::data_layout::DataLayout::parse(dl_data, sb.offset_size, sb.length_size)
|
|
.unwrap();
|
|
let raw = clawhdf5_format::data_read::read_raw_data(&bytes, &dl, &ds, &dt).unwrap();
|
|
let names = clawhdf5_format::data_read::read_enum_names(&raw, &dt).unwrap();
|
|
assert_eq!(names, vec!["GREEN", "RED", "BLUE", "GREEN"]);
|
|
}
|
|
|
|
// ---- h5py chunked round-trip tests ----
|
|
|
|
#[test]
|
|
#[ignore = "requires Python h5py module"]
|
|
fn h5py_reads_chunked_no_compression() {
|
|
let mut fw = FileWriter::new();
|
|
let data: Vec<f64> = (0..100).map(|i| i as f64).collect();
|
|
fw.create_dataset("data")
|
|
.with_f64_data(&data)
|
|
.with_shape(&[100])
|
|
.with_chunks(&[20]);
|
|
let bytes = fw.finish().unwrap();
|
|
let path = std::env::temp_dir().join("clawhdf5_chunked_nocomp.h5");
|
|
std::fs::write(&path, &bytes).unwrap();
|
|
let script = format!(
|
|
"import h5py,json; f=h5py.File('{}','r'); d=f['data']; print(json.dumps({{'values':d[:].tolist(),'chunks':list(d.chunks)}}))",
|
|
path.display()
|
|
);
|
|
let stdout = h5py_read(&path, &script);
|
|
let v: serde_json::Value = serde_json::from_str(&stdout).unwrap();
|
|
let values: Vec<f64> = serde_json::from_value(v["values"].clone()).unwrap();
|
|
assert_eq!(values, data);
|
|
assert_eq!(v["chunks"], serde_json::json!([20]));
|
|
}
|
|
|
|
#[test]
|
|
#[ignore = "requires Python h5py module"]
|
|
fn h5py_reads_chunked_deflate() {
|
|
let mut fw = FileWriter::new();
|
|
let data: Vec<f64> = (0..100).map(|i| i as f64).collect();
|
|
fw.create_dataset("data")
|
|
.with_f64_data(&data)
|
|
.with_shape(&[100])
|
|
.with_chunks(&[20])
|
|
.with_deflate(6);
|
|
let bytes = fw.finish().unwrap();
|
|
let path = std::env::temp_dir().join("clawhdf5_chunked_deflate.h5");
|
|
std::fs::write(&path, &bytes).unwrap();
|
|
let script = format!(
|
|
"import h5py,json; f=h5py.File('{}','r'); d=f['data']; print(json.dumps({{'values':d[:].tolist(),'chunks':list(d.chunks),'compression':d.compression}}))",
|
|
path.display()
|
|
);
|
|
let stdout = h5py_read(&path, &script);
|
|
let v: serde_json::Value = serde_json::from_str(&stdout).unwrap();
|
|
let values: Vec<f64> = serde_json::from_value(v["values"].clone()).unwrap();
|
|
assert_eq!(values, data);
|
|
assert_eq!(v["compression"], serde_json::json!("gzip"));
|
|
}
|
|
|
|
#[test]
|
|
#[ignore = "requires Python h5py module"]
|
|
fn h5py_reads_chunked_shuffle_deflate() {
|
|
let mut fw = FileWriter::new();
|
|
let data: Vec<f64> = (0..100).map(|i| i as f64).collect();
|
|
fw.create_dataset("data")
|
|
.with_f64_data(&data)
|
|
.with_shape(&[100])
|
|
.with_chunks(&[50])
|
|
.with_shuffle()
|
|
.with_deflate(6);
|
|
let bytes = fw.finish().unwrap();
|
|
let path = std::env::temp_dir().join("clawhdf5_chunked_shuffle_deflate.h5");
|
|
std::fs::write(&path, &bytes).unwrap();
|
|
let script = format!(
|
|
"import h5py,json; f=h5py.File('{}','r'); d=f['data']; print(json.dumps({{'values':d[:].tolist(),'shuffle':bool(d.shuffle),'compression':d.compression}}))",
|
|
path.display()
|
|
);
|
|
let stdout = h5py_read(&path, &script);
|
|
let v: serde_json::Value = serde_json::from_str(&stdout).unwrap();
|
|
let values: Vec<f64> = serde_json::from_value(v["values"].clone()).unwrap();
|
|
assert_eq!(values, data);
|
|
assert_eq!(v["shuffle"], serde_json::json!(true));
|
|
assert_eq!(v["compression"], serde_json::json!("gzip"));
|
|
}
|
|
|
|
#[test]
|
|
#[ignore = "requires Python h5py module"]
|
|
fn h5py_reads_chunked_fletcher32() {
|
|
let mut fw = FileWriter::new();
|
|
let data: Vec<f64> = (0..100).map(|i| i as f64).collect();
|
|
fw.create_dataset("data")
|
|
.with_f64_data(&data)
|
|
.with_shape(&[100])
|
|
.with_chunks(&[100])
|
|
.with_fletcher32();
|
|
let bytes = fw.finish().unwrap();
|
|
let path = std::env::temp_dir().join("clawhdf5_chunked_fletcher32.h5");
|
|
std::fs::write(&path, &bytes).unwrap();
|
|
let script = format!(
|
|
"import h5py,json; f=h5py.File('{}','r'); d=f['data']; print(json.dumps({{'values':d[:].tolist(),'fletcher32':bool(d.fletcher32)}}))",
|
|
path.display()
|
|
);
|
|
let stdout = h5py_read(&path, &script);
|
|
let v: serde_json::Value = serde_json::from_str(&stdout).unwrap();
|
|
let values: Vec<f64> = serde_json::from_value(v["values"].clone()).unwrap();
|
|
assert_eq!(values, data);
|
|
assert_eq!(v["fletcher32"], serde_json::json!(true));
|
|
}
|
|
|
|
#[test]
|
|
#[ignore = "requires Python h5py module"]
|
|
fn h5py_reads_chunked_2d() {
|
|
let mut fw = FileWriter::new();
|
|
let data: Vec<f64> = (0..24).map(|i| i as f64).collect();
|
|
fw.create_dataset("data")
|
|
.with_f64_data(&data)
|
|
.with_shape(&[4, 6])
|
|
.with_chunks(&[2, 3]);
|
|
let bytes = fw.finish().unwrap();
|
|
let path = std::env::temp_dir().join("clawhdf5_chunked_2d.h5");
|
|
std::fs::write(&path, &bytes).unwrap();
|
|
let script = format!(
|
|
"import h5py,json; f=h5py.File('{}','r'); d=f['data']; print(json.dumps({{'shape':list(d.shape),'chunks':list(d.chunks),'values':d[:].flatten().tolist()}}))",
|
|
path.display()
|
|
);
|
|
let stdout = h5py_read(&path, &script);
|
|
let v: serde_json::Value = serde_json::from_str(&stdout).unwrap();
|
|
let values: Vec<f64> = serde_json::from_value(v["values"].clone()).unwrap();
|
|
assert_eq!(values, data);
|
|
assert_eq!(v["shape"], serde_json::json!([4, 6]));
|
|
assert_eq!(v["chunks"], serde_json::json!([2, 3]));
|
|
}
|
|
|
|
#[test]
|
|
#[ignore = "requires Python h5py module"]
|
|
fn h5py_reads_2d_data() {
|
|
let mut fw = FileWriter::new();
|
|
fw.create_dataset("matrix")
|
|
.with_f64_data(&[1.0, 2.0, 3.0, 4.0, 5.0, 6.0])
|
|
.with_shape(&[2, 3]);
|
|
let bytes = fw.finish().unwrap();
|
|
let path = std::env::temp_dir().join("clawhdf5_test_2d.h5");
|
|
std::fs::write(&path, &bytes).unwrap();
|
|
let script = format!(
|
|
"import h5py, json; f=h5py.File('{}','r'); d=f['matrix']; print(json.dumps({{'shape': list(d.shape), 'data': d[:].flatten().tolist()}}))",
|
|
path.display()
|
|
);
|
|
let stdout = h5py_read(&path, &script);
|
|
let v: serde_json::Value = serde_json::from_str(&stdout).unwrap();
|
|
assert_eq!(v["shape"], serde_json::json!([2, 3]));
|
|
assert_eq!(v["data"], serde_json::json!([1.0, 2.0, 3.0, 4.0, 5.0, 6.0]));
|
|
}
|
|
|
|
// ---- Extensible Array / resizable dataset h5py tests ----
|
|
|
|
#[test]
|
|
#[ignore = "requires Python h5py module"]
|
|
fn h5py_reads_resizable_dataset() {
|
|
let mut fw = FileWriter::new();
|
|
let data: Vec<f64> = (0..50).map(|i| i as f64).collect();
|
|
fw.create_dataset("data")
|
|
.with_f64_data(&data)
|
|
.with_shape(&[50])
|
|
.with_chunks(&[10])
|
|
.with_maxshape(&[u64::MAX]);
|
|
let bytes = fw.finish().unwrap();
|
|
let path = std::env::temp_dir().join("clawhdf5_ea_resizable.h5");
|
|
std::fs::write(&path, &bytes).unwrap();
|
|
let script = format!(
|
|
"import h5py,json; f=h5py.File('{}','r'); d=f['data']; print(json.dumps({{'values':d[:].tolist(),'chunks':list(d.chunks),'maxshape':list(None if x is None else x for x in d.maxshape)}}))",
|
|
path.display()
|
|
);
|
|
let stdout = h5py_read(&path, &script);
|
|
let v: serde_json::Value = serde_json::from_str(&stdout).unwrap();
|
|
let values: Vec<f64> = serde_json::from_value(v["values"].clone()).unwrap();
|
|
assert_eq!(values, data);
|
|
assert_eq!(v["chunks"], serde_json::json!([10]));
|
|
}
|
|
|
|
#[test]
|
|
#[ignore = "requires Python h5py module"]
|
|
fn read_h5py_generated_ea_file() {
|
|
let path = std::env::temp_dir().join("clawhdf5_h5py_ea.h5");
|
|
let gen_script = format!(
|
|
"import h5py,numpy as np; f=h5py.File('{}','w'); d=f.create_dataset('data',data=np.arange(30,dtype='float64'),chunks=(10,),maxshape=(None,)); f.close()",
|
|
path.display()
|
|
);
|
|
h5py_read(&path, &gen_script);
|
|
|
|
let bytes = std::fs::read(&path).unwrap();
|
|
let sig = clawhdf5_format::signature::find_signature(&bytes).unwrap();
|
|
let sb = clawhdf5_format::superblock::Superblock::parse(&bytes, sig).unwrap();
|
|
let addr = clawhdf5_format::group_v2::resolve_path_any(&bytes, &sb, "data").unwrap();
|
|
let hdr = clawhdf5_format::object_header::ObjectHeader::parse(
|
|
&bytes,
|
|
addr as usize,
|
|
sb.offset_size,
|
|
sb.length_size,
|
|
)
|
|
.unwrap();
|
|
let dt_data = &hdr
|
|
.messages
|
|
.iter()
|
|
.find(|m| m.msg_type == clawhdf5_format::message_type::MessageType::Datatype)
|
|
.unwrap()
|
|
.data;
|
|
let ds_data = &hdr
|
|
.messages
|
|
.iter()
|
|
.find(|m| m.msg_type == clawhdf5_format::message_type::MessageType::Dataspace)
|
|
.unwrap()
|
|
.data;
|
|
let dl_data = &hdr
|
|
.messages
|
|
.iter()
|
|
.find(|m| m.msg_type == clawhdf5_format::message_type::MessageType::DataLayout)
|
|
.unwrap()
|
|
.data;
|
|
let (dt, _) = clawhdf5_format::datatype::Datatype::parse(dt_data).unwrap();
|
|
let ds = clawhdf5_format::dataspace::Dataspace::parse(ds_data, sb.length_size).unwrap();
|
|
let dl =
|
|
clawhdf5_format::data_layout::DataLayout::parse(dl_data, sb.offset_size, sb.length_size)
|
|
.unwrap();
|
|
let raw = match &dl {
|
|
clawhdf5_format::data_layout::DataLayout::Chunked { .. } => {
|
|
let pipeline = hdr
|
|
.messages
|
|
.iter()
|
|
.find(|m| m.msg_type == clawhdf5_format::message_type::MessageType::FilterPipeline)
|
|
.map(|m| clawhdf5_format::filter_pipeline::FilterPipeline::parse(&m.data).unwrap());
|
|
clawhdf5_format::chunked_read::read_chunked_data(
|
|
&bytes,
|
|
&dl,
|
|
&ds,
|
|
&dt,
|
|
pipeline.as_ref(),
|
|
sb.offset_size,
|
|
sb.length_size,
|
|
)
|
|
.unwrap()
|
|
}
|
|
_ => clawhdf5_format::data_read::read_raw_data(&bytes, &dl, &ds, &dt).unwrap(),
|
|
};
|
|
let result = clawhdf5_format::data_read::read_as_f64(&raw, &dt).unwrap();
|
|
let expected: Vec<f64> = (0..30).map(|i| i as f64).collect();
|
|
assert_eq!(result, expected);
|
|
}
|
|
|
|
#[test]
|
|
#[ignore = "requires Python h5py module"]
|
|
fn h5py_append_and_verify() {
|
|
let mut fw = FileWriter::new();
|
|
let initial: Vec<f64> = (0..10).map(|i| i as f64).collect();
|
|
fw.create_dataset("data")
|
|
.with_f64_data(&initial)
|
|
.with_shape(&[10])
|
|
.with_chunks(&[10])
|
|
.with_maxshape(&[u64::MAX]);
|
|
let bytes = fw.finish().unwrap();
|
|
let path = std::env::temp_dir().join("clawhdf5_ea_append.h5");
|
|
std::fs::write(&path, &bytes).unwrap();
|
|
|
|
let script = format!(
|
|
r#"
|
|
import h5py, json, numpy as np
|
|
f = h5py.File('{}', 'a')
|
|
d = f['data']
|
|
for batch in range(3):
|
|
old_size = d.shape[0]
|
|
new_data = np.arange(old_size, old_size + 10, dtype='float64')
|
|
d.resize(old_size + 10, axis=0)
|
|
d[old_size:] = new_data
|
|
f.close()
|
|
f = h5py.File('{}', 'r')
|
|
d = f['data']
|
|
result = d[:].tolist()
|
|
shape = list(d.shape)
|
|
f.close()
|
|
print(json.dumps({{'values': result, 'shape': shape}}))
|
|
"#,
|
|
path.display(),
|
|
path.display()
|
|
);
|
|
let stdout = h5py_read(&path, &script);
|
|
let v: serde_json::Value = serde_json::from_str(&stdout).unwrap();
|
|
let values: Vec<f64> = serde_json::from_value(v["values"].clone()).unwrap();
|
|
let expected: Vec<f64> = (0..40).map(|i| i as f64).collect();
|
|
assert_eq!(values, expected);
|
|
assert_eq!(v["shape"], serde_json::json!([40]));
|
|
}
|
|
|
|
#[test]
|
|
#[ignore = "requires Python h5py module"]
|
|
fn h5py_reads_resizable_single_chunk() {
|
|
let mut fw = FileWriter::new();
|
|
let data: Vec<f64> = (0..5).map(|i| i as f64).collect();
|
|
fw.create_dataset("data")
|
|
.with_f64_data(&data)
|
|
.with_shape(&[5])
|
|
.with_chunks(&[10])
|
|
.with_maxshape(&[u64::MAX]);
|
|
let bytes = fw.finish().unwrap();
|
|
let path = std::env::temp_dir().join("clawhdf5_ea_single.h5");
|
|
std::fs::write(&path, &bytes).unwrap();
|
|
let script = format!(
|
|
"import h5py,json; f=h5py.File('{}','r'); d=f['data']; print(json.dumps(d[:].tolist()))",
|
|
path.display()
|
|
);
|
|
let stdout = h5py_read(&path, &script);
|
|
let values: Vec<f64> = serde_json::from_str(&stdout).unwrap();
|
|
assert_eq!(values, data);
|
|
}
|
|
|
|
// ---- Dense attribute h5py round-trip ----
|
|
|
|
#[test]
|
|
#[ignore = "requires Python h5py module"]
|
|
fn h5py_reads_dense_attrs() {
|
|
let mut fw = FileWriter::new();
|
|
let ds = fw.create_dataset("data");
|
|
ds.with_f64_data(&[1.0, 2.0, 3.0]);
|
|
for i in 0..20 {
|
|
ds.set_attr(&format!("attr_{i:03}"), AttrValue::F64(i as f64 * 1.5));
|
|
}
|
|
let bytes = fw.finish().unwrap();
|
|
let path = std::env::temp_dir().join("clawhdf5_dense_attrs.h5");
|
|
std::fs::write(&path, &bytes).unwrap();
|
|
|
|
let script = format!(
|
|
r#"
|
|
import h5py, json
|
|
f = h5py.File('{}', 'r')
|
|
d = f['data']
|
|
attrs = {{k: float(v) for k, v in d.attrs.items()}}
|
|
data = d[:].tolist()
|
|
f.close()
|
|
print(json.dumps({{'data': data, 'num_attrs': len(attrs), 'attr_000': attrs.get('attr_000'), 'attr_010': attrs.get('attr_010'), 'attr_019': attrs.get('attr_019')}}))
|
|
"#,
|
|
path.display()
|
|
);
|
|
let stdout = h5py_read(&path, &script);
|
|
let v: serde_json::Value = serde_json::from_str(&stdout).unwrap();
|
|
assert_eq!(v["data"], serde_json::json!([1.0, 2.0, 3.0]));
|
|
assert_eq!(v["num_attrs"], serde_json::json!(20));
|
|
assert_eq!(v["attr_000"], serde_json::json!(0.0));
|
|
assert_eq!(v["attr_010"], serde_json::json!(15.0));
|
|
assert_eq!(v["attr_019"], serde_json::json!(28.5));
|
|
}
|
|
|
|
#[test]
|
|
#[ignore = "requires Python h5py module"]
|
|
fn h5py_reads_50_dense_attrs() {
|
|
let mut fw = FileWriter::new();
|
|
let ds = fw.create_dataset("data");
|
|
ds.with_f64_data(&[42.0]);
|
|
for i in 0..50 {
|
|
ds.set_attr(&format!("attr_{i:03}"), AttrValue::F64(i as f64 * 1.5));
|
|
}
|
|
let bytes = fw.finish().unwrap();
|
|
let path = std::env::temp_dir().join("clawhdf5_dense_50_attrs.h5");
|
|
std::fs::write(&path, &bytes).unwrap();
|
|
|
|
let script = format!(
|
|
r#"
|
|
import h5py, json
|
|
f = h5py.File('{}', 'r')
|
|
d = f['data']
|
|
attrs = {{k: float(v) for k, v in d.attrs.items()}}
|
|
f.close()
|
|
print(json.dumps({{'num_attrs': len(attrs), 'first': attrs.get('attr_000'), 'last': attrs.get('attr_049')}}))
|
|
"#,
|
|
path.display()
|
|
);
|
|
let stdout = h5py_read(&path, &script);
|
|
let v: serde_json::Value = serde_json::from_str(&stdout).unwrap();
|
|
assert_eq!(v["num_attrs"], serde_json::json!(50));
|
|
assert_eq!(v["first"], serde_json::json!(0.0));
|
|
assert_eq!(v["last"], serde_json::json!(73.5));
|
|
}
|
|
|
|
// ---- SHINES provenance h5py round-trips ----
|
|
|
|
#[test]
|
|
#[ignore = "requires Python h5py module"]
|
|
fn h5py_reads_provenance_attrs() {
|
|
let mut fw = FileWriter::new();
|
|
let ds = fw.create_dataset("sensor");
|
|
ds.with_f64_data(&[1.0, 2.0, 3.0, 4.0]).with_provenance(
|
|
"clawhdf5/test",
|
|
"2026-02-19T12:00:00Z",
|
|
Some("bench_42"),
|
|
);
|
|
let bytes = fw.finish().unwrap();
|
|
let path = std::env::temp_dir().join("clawhdf5_provenance.h5");
|
|
std::fs::write(&path, &bytes).unwrap();
|
|
|
|
let script = format!(
|
|
r#"
|
|
import h5py, json, hashlib, struct
|
|
f = h5py.File('{}', 'r')
|
|
d = f['sensor']
|
|
data = d[:].tolist()
|
|
sha = d.attrs['_provenance_sha256']
|
|
if isinstance(sha, bytes):
|
|
sha = sha.decode('utf-8')
|
|
sha = sha.rstrip('\x00')
|
|
creator = d.attrs['_provenance_creator']
|
|
if isinstance(creator, bytes):
|
|
creator = creator.decode('utf-8')
|
|
creator = creator.rstrip('\x00')
|
|
ts = d.attrs['_provenance_timestamp']
|
|
if isinstance(ts, bytes):
|
|
ts = ts.decode('utf-8')
|
|
ts = ts.rstrip('\x00')
|
|
source = d.attrs['_provenance_source']
|
|
if isinstance(source, bytes):
|
|
source = source.decode('utf-8')
|
|
source = source.rstrip('\x00')
|
|
# Verify SHA-256 matches the raw little-endian f64 bytes
|
|
raw = struct.pack('<4d', *data)
|
|
expected = hashlib.sha256(raw).hexdigest()
|
|
f.close()
|
|
print(json.dumps({{'data': data, 'sha256': sha, 'expected_sha256': expected, 'creator': creator, 'timestamp': ts, 'source': source}}))
|
|
"#,
|
|
path.display()
|
|
);
|
|
let stdout = h5py_read(&path, &script);
|
|
let v: serde_json::Value = serde_json::from_str(&stdout).unwrap();
|
|
assert_eq!(v["data"], serde_json::json!([1.0, 2.0, 3.0, 4.0]));
|
|
assert_eq!(v["sha256"], v["expected_sha256"]);
|
|
assert_eq!(v["creator"], serde_json::json!("clawhdf5/test"));
|
|
assert_eq!(v["timestamp"], serde_json::json!("2026-02-19T12:00:00Z"));
|
|
assert_eq!(v["source"], serde_json::json!("bench_42"));
|
|
}
|
|
|
|
#[test]
|
|
#[ignore = "requires Python h5py module"]
|
|
fn h5py_reads_provenance_no_source() {
|
|
let mut fw = FileWriter::new();
|
|
let ds = fw.create_dataset("data");
|
|
ds.with_i32_data(&[10, 20, 30])
|
|
.with_provenance("test-writer", "2026-01-01T00:00:00Z", None);
|
|
let bytes = fw.finish().unwrap();
|
|
let path = std::env::temp_dir().join("clawhdf5_provenance_nosrc.h5");
|
|
std::fs::write(&path, &bytes).unwrap();
|
|
|
|
let script = format!(
|
|
r#"
|
|
import h5py, json, hashlib, struct
|
|
f = h5py.File('{}', 'r')
|
|
d = f['data']
|
|
attr_names = sorted(d.attrs.keys())
|
|
sha = d.attrs['_provenance_sha256']
|
|
if isinstance(sha, bytes):
|
|
sha = sha.decode('utf-8')
|
|
sha = sha.rstrip('\x00')
|
|
raw = struct.pack('<3i', *d[:].tolist())
|
|
expected = hashlib.sha256(raw).hexdigest()
|
|
has_source = '_provenance_source' in d.attrs
|
|
f.close()
|
|
print(json.dumps({{'attrs': attr_names, 'sha_ok': sha == expected, 'has_source': has_source}}))
|
|
"#,
|
|
path.display()
|
|
);
|
|
let stdout = h5py_read(&path, &script);
|
|
let v: serde_json::Value = serde_json::from_str(&stdout).unwrap();
|
|
assert_eq!(v["sha_ok"], serde_json::json!(true));
|
|
assert_eq!(v["has_source"], serde_json::json!(false));
|
|
}
|
|
|
|
#[test]
|
|
fn provenance_verify_written_file() {
|
|
let mut fw = FileWriter::new();
|
|
let ds = fw.create_dataset("values");
|
|
ds.with_f64_data(&[100.0, 200.0, 300.0]).with_provenance(
|
|
"integrity-test",
|
|
"2026-02-19T00:00:00Z",
|
|
None,
|
|
);
|
|
let bytes = fw.finish().unwrap();
|
|
|
|
// Use our verification API to check integrity
|
|
let sig = clawhdf5_format::signature::find_signature(&bytes).unwrap();
|
|
let sb = clawhdf5_format::superblock::Superblock::parse(&bytes, sig).unwrap();
|
|
let addr = clawhdf5_format::group_v2::resolve_path_any(&bytes, &sb, "values").unwrap();
|
|
let hdr = clawhdf5_format::object_header::ObjectHeader::parse(
|
|
&bytes,
|
|
addr as usize,
|
|
sb.offset_size,
|
|
sb.length_size,
|
|
)
|
|
.unwrap();
|
|
let result =
|
|
clawhdf5_format::provenance::verify_dataset(&bytes, &hdr, sb.offset_size, sb.length_size)
|
|
.unwrap();
|
|
assert_eq!(result, clawhdf5_format::provenance::VerifyResult::Ok);
|
|
}
|
|
|
|
// ---- hdf5plugin interop: registered third-party compression filters ----
|
|
|
|
/// Write `data` (f64, 1-D, chunked) with `configure` applied, then read it
|
|
/// back with h5py + hdf5plugin (libhdf5's registered filter plugins) and
|
|
/// return the values it decodes.
|
|
#[cfg(any(feature = "lz4", feature = "zstd"))]
|
|
fn hdf5plugin_roundtrip(
|
|
tag: &str,
|
|
data: &[f64],
|
|
configure: impl FnOnce(&mut clawhdf5_format::type_builders::DatasetBuilder),
|
|
) -> Vec<f64> {
|
|
let mut fw = FileWriter::new();
|
|
let ds = fw.create_dataset("data");
|
|
ds.with_f64_data(data)
|
|
.with_shape(&[data.len() as u64])
|
|
.with_chunks(&[250]);
|
|
configure(ds);
|
|
let bytes = fw.finish().unwrap();
|
|
let path = std::env::temp_dir().join(format!("clawhdf5_hdf5plugin_{tag}.h5"));
|
|
std::fs::write(&path, &bytes).unwrap();
|
|
let script = format!(
|
|
"import h5py,hdf5plugin,json; f=h5py.File('{}','r'); print(json.dumps(f['data'][:].tolist()))",
|
|
path.display()
|
|
);
|
|
let stdout = h5py_read(&path, &script);
|
|
serde_json::from_str(&stdout).unwrap()
|
|
}
|
|
|
|
/// libhdf5's LZ4 plugin must decode what we write (it could not while we
|
|
/// wrote a private 4-byte-LE-size framing).
|
|
#[cfg(feature = "lz4")]
|
|
#[test]
|
|
#[ignore = "requires Python h5py + hdf5plugin"]
|
|
fn hdf5plugin_reads_our_lz4() {
|
|
let data: Vec<f64> = (0..1000).map(|i| (i % 37) as f64 * 0.5).collect();
|
|
let got = hdf5plugin_roundtrip("lz4", &data, |ds| {
|
|
ds.with_lz4();
|
|
});
|
|
assert_eq!(got, data);
|
|
let got = hdf5plugin_roundtrip("lz4_noshuffle", &data, |ds| {
|
|
ds.with_lz4().without_shuffle();
|
|
});
|
|
assert_eq!(got, data);
|
|
}
|
|
|
|
/// libhdf5's Zstandard plugin must decode what we write (it could not while
|
|
/// our frames lacked the content size).
|
|
#[cfg(feature = "zstd")]
|
|
#[test]
|
|
#[ignore = "requires Python h5py + hdf5plugin"]
|
|
fn hdf5plugin_reads_our_zstd() {
|
|
let data: Vec<f64> = (0..1000).map(|i| (i % 37) as f64 * 0.5).collect();
|
|
let got = hdf5plugin_roundtrip("zstd", &data, |ds| {
|
|
ds.with_zstd(3);
|
|
});
|
|
assert_eq!(got, data);
|
|
}
|