Write files HDF5 1.8 can read: FileWriter/FileBuilder::libver_bounds
New `LibVer` (V18, V110, V112, V114, V200, Latest) and
`libver_bounds(low, high)` on the format crate's `FileWriter` and the
facade's `FileBuilder`, as libhdf5's H5Pset_libver_bounds / h5py's
libver=(low, high). The default stays (V110, Latest), byte for byte what
was written before.
With a low bound of 1.8: superblock version 2, layout message version 3
(contiguous, compact, chunked) and a version-1 B-tree chunk index for
every chunked dataset, resizable ones included -- what libhdf5 2.x writes
under libver=('v108', 'latest'). The new chunk B-tree writer
(btree_v1_write.rs) replays H5B_insert with the H5Dbtree.c callbacks for
row-major insertion (split ratios 0.1/0.5/0.9, right keys moved as
H5D__btree_cmp3 moves them, root kept in place): its trees equal
libhdf5's node for node for 1-D/2-D/3-D, 2- and 3-level, filtered and
unfiltered datasets (libhdf5 writing without a chunk cache).
The high bound refuses, with FormatError::LibverBound before anything is
written, what needs a newer format: virtual datasets and the paged
file-space strategy (1.10), the 1.12 reference types (datatype v4),
native complex (datatype v5, HDF5 2.0), and a low bound above the high.
Tests: tools/tests/libver_v18.rs writes every writer feature under
(V18, V18), and HDF5 1.8.23's h5dump (scripts/build-hdf5-1.8.sh; skipped
when absent) dumps it exactly as h5dump 1.14 does and returns our bytes
for every numeric dataset; h5py, clawhdf5 and h5rs check --data agree;
then FileEditor grows/appends/annotates it and h5py appends, and every
reader checks again. read_harness gains --v18 and --chunk N.
Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
@@ -0,0 +1,736 @@
|
||||
//! Files written with `FileBuilder::libver_bounds(LibVer::V18, LibVer::V18)`
|
||||
//! must be readable by HDF5 1.8. Every writer feature is written under that
|
||||
//! bound and read back by HDF5 1.8.23's h5dump (values dumped in binary and
|
||||
//! compared), by h5py (libhdf5 2.x), by h5dump 1.14, by clawhdf5 and by
|
||||
//! `h5rs check --data`; then `FileEditor` grows, appends to and annotates
|
||||
//! the file (splitting version-1 B-tree nodes) and every reader checks it
|
||||
//! again. The version-1 B-trees we write are compared node by node with
|
||||
//! the ones libhdf5 writes for the same data under h5py's
|
||||
//! `libver=('v108', 'latest')`.
|
||||
//!
|
||||
//! HDF5 1.8 is found through `CLAWHDF5_H5DUMP18` (the path to its h5dump)
|
||||
//! or at `~/.cache/hdf5-1.8.23/bin/h5dump`, where
|
||||
//! `scripts/build-hdf5-1.8.sh` builds it; without it the 1.8 checks are
|
||||
//! skipped (CI has no HDF5 1.8), even with `CLAWHDF5_REQUIRE_INTEROP=1`.
|
||||
//! h5py/numpy (`CLAWHDF5_PYTHON`) and h5dump are needed otherwise; they
|
||||
//! skip when missing unless `CLAWHDF5_REQUIRE_INTEROP=1`.
|
||||
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::process::{Command, Output};
|
||||
|
||||
use clawhdf5::{AttrValue, File, FileBuilder, FileEditor, LibVer, Selection};
|
||||
use clawhdf5_format::datatype::{CharacterSet, Datatype, StringPadding};
|
||||
use clawhdf5_format::file_writer::{CompoundTypeBuilder, EnumTypeBuilder};
|
||||
use clawhdf5_format::type_builders::make_i32_type;
|
||||
|
||||
fn python() -> String {
|
||||
std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string())
|
||||
}
|
||||
|
||||
fn interop_required() -> bool {
|
||||
std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1")
|
||||
}
|
||||
|
||||
fn available(cmd: &str, args: &[&str]) -> bool {
|
||||
Command::new(cmd)
|
||||
.args(args)
|
||||
.output()
|
||||
.map(|o| o.status.success())
|
||||
.unwrap_or(false)
|
||||
}
|
||||
|
||||
fn tools_ok() -> bool {
|
||||
let ok =
|
||||
available(&python(), &["-c", "import h5py, numpy"]) && available("h5dump", &["--version"]);
|
||||
if !ok {
|
||||
assert!(
|
||||
!interop_required(),
|
||||
"CLAWHDF5_REQUIRE_INTEROP=1 but h5py/numpy or h5dump is not available"
|
||||
);
|
||||
eprintln!("SKIP: h5py/numpy or h5dump not available");
|
||||
}
|
||||
ok
|
||||
}
|
||||
|
||||
/// HDF5 1.8's h5dump, when there is one.
|
||||
fn h5dump18() -> Option<PathBuf> {
|
||||
let p = match std::env::var_os("CLAWHDF5_H5DUMP18") {
|
||||
Some(p) => PathBuf::from(p),
|
||||
None => PathBuf::from(std::env::var_os("HOME")?).join(".cache/hdf5-1.8.23/bin/h5dump"),
|
||||
};
|
||||
let o = Command::new(&p).arg("--version").output().ok()?;
|
||||
let v = String::from_utf8_lossy(&o.stdout).to_string();
|
||||
if !v.contains("1.8.") {
|
||||
eprintln!("SKIP 1.8 checks: {} is not HDF5 1.8 ({v})", p.display());
|
||||
return None;
|
||||
}
|
||||
Some(p)
|
||||
}
|
||||
|
||||
fn py(script: &str) -> String {
|
||||
let o = Command::new(python())
|
||||
.args(["-c", script])
|
||||
.output()
|
||||
.expect("run python");
|
||||
assert!(
|
||||
o.status.success(),
|
||||
"python failed:\n{script}\nSTDOUT: {}\nSTDERR: {}",
|
||||
String::from_utf8_lossy(&o.stdout),
|
||||
String::from_utf8_lossy(&o.stderr)
|
||||
);
|
||||
String::from_utf8_lossy(&o.stdout).trim().to_string()
|
||||
}
|
||||
|
||||
fn text(o: &Output) -> String {
|
||||
format!(
|
||||
"{}{}",
|
||||
String::from_utf8_lossy(&o.stdout),
|
||||
String::from_utf8_lossy(&o.stderr)
|
||||
)
|
||||
}
|
||||
|
||||
fn tmpdir() -> tempfile::TempDir {
|
||||
tempfile::TempDir::new_in(env!("CARGO_TARGET_TMPDIR")).unwrap()
|
||||
}
|
||||
|
||||
/// The hyperslab of `count` elements from `start`.
|
||||
fn block(start: &[u64], count: &[u64]) -> Selection {
|
||||
Selection::Hyperslab {
|
||||
start: start.to_vec(),
|
||||
stride: vec![1; start.len()],
|
||||
count: count.to_vec(),
|
||||
block: vec![1; start.len()],
|
||||
}
|
||||
}
|
||||
|
||||
fn le<T: Copy, const N: usize>(v: &[T], f: impl Fn(T) -> [u8; N]) -> Vec<u8> {
|
||||
v.iter().flat_map(|&x| f(x)).collect()
|
||||
}
|
||||
|
||||
/// The datasets of the test file and the bytes each holds (little-endian,
|
||||
/// row-major, as h5py's `tobytes()` and h5dump's `-b LE` give them).
|
||||
struct Expect {
|
||||
datasets: Vec<(String, Vec<u8>)>,
|
||||
}
|
||||
|
||||
impl Expect {
|
||||
fn set(&mut self, name: &str, bytes: Vec<u8>) {
|
||||
match self.datasets.iter_mut().find(|(n, _)| n == name) {
|
||||
Some(e) => e.1 = bytes,
|
||||
None => self.datasets.push((name.to_string(), bytes)),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const MANY: usize = 100_000;
|
||||
|
||||
/// Write every feature under the 1.8 bound.
|
||||
fn write_file(path: &Path) -> Expect {
|
||||
let mut e = Expect {
|
||||
datasets: Vec::new(),
|
||||
};
|
||||
let mut b = FileBuilder::new();
|
||||
b.libver_bounds(LibVer::V18, LibVer::V18);
|
||||
|
||||
// Contiguous, with dense attributes (more than 8).
|
||||
let v: Vec<f64> = (0..1000).map(|i| i as f64 * 0.5).collect();
|
||||
let d = b.create_dataset("contig").with_f64_data(&v);
|
||||
for i in 0..12 {
|
||||
d.set_attr(&format!("a{i:02}"), AttrValue::I64(i));
|
||||
}
|
||||
e.set("/contig", le(&v, f64::to_le_bytes));
|
||||
b.create_dataset("empty").with_f64_data(&[]);
|
||||
e.set("/empty", vec![]);
|
||||
b.create_dataset("scalar")
|
||||
.with_f64_data(&[2.5])
|
||||
.with_shape(&[]);
|
||||
e.set("/scalar", 2.5f64.to_le_bytes().to_vec());
|
||||
let v: Vec<i32> = (0..16).map(|i| i * 3 - 7).collect();
|
||||
b.create_dataset("compact").with_i32_data(&v).compact();
|
||||
e.set("/compact", le(&v, i32::to_le_bytes));
|
||||
let v: Vec<f32> = (0..20).map(|i| i as f32 / 3.0).collect();
|
||||
b.create_dataset("f32").with_f32_data(&v);
|
||||
e.set("/f32", le(&v, f32::to_le_bytes));
|
||||
let v: Vec<f32> = vec![0.5, -2.0, 1024.0, 0.0];
|
||||
b.create_dataset("f16").with_f16_data(&v);
|
||||
e.set(
|
||||
"/f16",
|
||||
le(&v, |x| {
|
||||
clawhdf5_format::float16::f32_to_f16_bits(x).to_le_bytes()
|
||||
}),
|
||||
);
|
||||
|
||||
// Chunked with every built-in filter HDF5 1.8 has, and edge chunks.
|
||||
let v: Vec<f32> = (0..60 * 70).map(|i| (i % 97) as f32 * 1.25).collect();
|
||||
b.create_dataset("chunked")
|
||||
.with_f32_data(&v)
|
||||
.with_shape(&[60, 70])
|
||||
.with_chunks(&[16, 16])
|
||||
.with_deflate(6)
|
||||
.with_shuffle()
|
||||
.with_fletcher32();
|
||||
e.set("/chunked", le(&v, f32::to_le_bytes));
|
||||
// What the 1.10 indexes would be: Extensible Array, version-2 B-tree,
|
||||
// Fixed Array, single chunk. All become version-1 B-trees.
|
||||
let v: Vec<i32> = (0..25).collect();
|
||||
b.create_dataset("resizable")
|
||||
.with_i32_data(&v)
|
||||
.with_maxshape(&[u64::MAX])
|
||||
.with_chunks(&[4]);
|
||||
e.set("/resizable", le(&v, i32::to_le_bytes));
|
||||
let v: Vec<i64> = (0..30).map(|i| i * 1_000_000_007).collect();
|
||||
b.create_dataset("resizable2")
|
||||
.with_i64_data(&v)
|
||||
.with_shape(&[5, 6])
|
||||
.with_maxshape(&[u64::MAX, u64::MAX])
|
||||
.with_chunks(&[2, 4]);
|
||||
e.set("/resizable2", le(&v, i64::to_le_bytes));
|
||||
let v: Vec<i32> = (0..10).map(|i| -i).collect();
|
||||
b.create_dataset("fixedmax")
|
||||
.with_i32_data(&v)
|
||||
.with_maxshape(&[100])
|
||||
.with_chunks(&[3])
|
||||
.with_deflate(1);
|
||||
e.set("/fixedmax", le(&v, i32::to_le_bytes));
|
||||
let v: Vec<i32> = (0..8).map(|i| i * i).collect();
|
||||
b.create_dataset("single")
|
||||
.with_i32_data(&v)
|
||||
.with_chunks(&[8]);
|
||||
e.set("/single", le(&v, i32::to_le_bytes));
|
||||
// Enough chunks for a three-level tree, and a 2-D two-level one.
|
||||
let v: Vec<i32> = (0..MANY as i32).map(|i| i ^ 0x5a5a).collect();
|
||||
b.create_dataset("many")
|
||||
.with_i32_data(&v)
|
||||
.with_maxshape(&[u64::MAX])
|
||||
.with_chunks(&[1]);
|
||||
e.set("/many", le(&v, i32::to_le_bytes));
|
||||
let v: Vec<f64> = (0..100 * 100).map(|i| i as f64).collect();
|
||||
b.create_dataset("grid")
|
||||
.with_f64_data(&v)
|
||||
.with_shape(&[100, 100])
|
||||
.with_chunks(&[1, 1]);
|
||||
e.set("/grid", le(&v, f64::to_le_bytes));
|
||||
b.create_dataset("empty_chunked")
|
||||
.with_f64_data(&[])
|
||||
.with_maxshape(&[u64::MAX])
|
||||
.with_chunks(&[10]);
|
||||
e.set("/empty_chunked", vec![]);
|
||||
let v: Vec<i32> = (0..12).collect();
|
||||
b.create_dataset("filled")
|
||||
.with_i32_data(&v)
|
||||
.with_maxshape(&[u64::MAX])
|
||||
.with_chunks(&[5])
|
||||
.with_fill_value(&(-9i32).to_le_bytes());
|
||||
e.set("/filled", le(&v, i32::to_le_bytes));
|
||||
|
||||
// Datatypes.
|
||||
let raw = b"abcdehello\0\0\0\0\0".to_vec();
|
||||
b.create_dataset("strings").with_compound_data(
|
||||
Datatype::String {
|
||||
size: 5,
|
||||
padding: StringPadding::NullPad,
|
||||
charset: CharacterSet::Ascii,
|
||||
},
|
||||
raw.clone(),
|
||||
3,
|
||||
);
|
||||
e.set("/strings", raw);
|
||||
let ct = CompoundTypeBuilder::new()
|
||||
.i32_field("a")
|
||||
.f64_field("b")
|
||||
.build();
|
||||
let mut raw = Vec::new();
|
||||
for i in 0..5i32 {
|
||||
raw.extend_from_slice(&i.to_le_bytes());
|
||||
raw.extend_from_slice(&(f64::from(i) * 1.5).to_le_bytes());
|
||||
}
|
||||
b.create_dataset("compound")
|
||||
.with_compound_data(ct, raw.clone(), 5);
|
||||
e.set("/compound", raw);
|
||||
let et = EnumTypeBuilder::i32_based()
|
||||
.value("RED", 0)
|
||||
.value("GREEN", 1)
|
||||
.value("BLUE", 7)
|
||||
.build();
|
||||
let v = [0, 7, 1, 1, 0];
|
||||
b.create_dataset("enum").with_enum_i32_data(et, &v);
|
||||
e.set("/enum", le(&v, i32::to_le_bytes));
|
||||
let v: Vec<i32> = (0..24).collect();
|
||||
b.create_dataset("array").with_array_data(
|
||||
make_i32_type(),
|
||||
&[2, 3],
|
||||
le(&v, i32::to_le_bytes),
|
||||
4,
|
||||
);
|
||||
e.set("/array", le(&v, i32::to_le_bytes));
|
||||
let v: Vec<i64> = vec![-1, 0, i64::MAX];
|
||||
b.create_dataset("i64").with_i64_data(&v);
|
||||
e.set("/i64", le(&v, i64::to_le_bytes));
|
||||
|
||||
// Groups: compact with links of every kind, dense (links and
|
||||
// attributes), and tracking creation order.
|
||||
let mut g = b.create_group("g_compact");
|
||||
g.create_dataset("x").with_i32_data(&[1, 2, 3]);
|
||||
e.set("/g_compact/x", le(&[1i32, 2, 3], i32::to_le_bytes));
|
||||
g.add_soft_link("soft", "/contig");
|
||||
g.add_external_link("ext", "other.h5", "/data");
|
||||
g.set_attr("title", AttrValue::String("compact group".into()));
|
||||
b.add_group(g.finish());
|
||||
b.add_hard_link("hard", "/g_compact/x");
|
||||
let mut g = b.create_group("g_dense");
|
||||
for i in 0..20 {
|
||||
let v = [i, i + 1];
|
||||
g.create_dataset(&format!("d{i:02}")).with_i32_data(&v);
|
||||
e.set(&format!("/g_dense/d{i:02}"), le(&v, i32::to_le_bytes));
|
||||
}
|
||||
for i in 0..12 {
|
||||
g.set_attr(&format!("attr{i:02}"), AttrValue::F64(f64::from(i) / 4.0));
|
||||
}
|
||||
b.add_group(g.finish());
|
||||
let mut g = b.create_group("g_order");
|
||||
g.track_order(true);
|
||||
for i in 0..10 {
|
||||
let n = format!("z{}", 9 - i);
|
||||
g.create_dataset(&n).with_i32_data(&[i]);
|
||||
e.set(&format!("/g_order/{n}"), i.to_le_bytes().to_vec());
|
||||
}
|
||||
for i in 0..10 {
|
||||
g.set_attr(&format!("b{}", 9 - i), AttrValue::I64(i));
|
||||
}
|
||||
b.add_group(g.finish());
|
||||
b.create_dataset("deep/er/path").with_i32_data(&[42]);
|
||||
e.set("/deep/er/path", 42i32.to_le_bytes().to_vec());
|
||||
|
||||
b.set_attr("version", AttrValue::F64(1.8));
|
||||
b.set_attr("ints", AttrValue::I64Array(vec![1, -2, 3]));
|
||||
b.set_attr("text", AttrValue::String("readable by 1.8".into()));
|
||||
b.set_attr(
|
||||
"texts",
|
||||
AttrValue::StringArray(vec!["a".into(), "bb".into(), "ccc".into()]),
|
||||
);
|
||||
b.set_attr("big", AttrValue::U64(u64::MAX));
|
||||
b.write(path).unwrap();
|
||||
e
|
||||
}
|
||||
|
||||
/// Structural facts of a file: superblock and layout message versions.
|
||||
fn check_versions(path: &Path) {
|
||||
use clawhdf5_format::message_type::MessageType;
|
||||
use clawhdf5_format::object_header::ObjectHeader;
|
||||
let bytes = std::fs::read(path).unwrap();
|
||||
assert_eq!(bytes[8], 2, "superblock version");
|
||||
let f = File::open(path).unwrap();
|
||||
let sb = f.superblock().clone();
|
||||
for name in [
|
||||
"contig",
|
||||
"compact",
|
||||
"chunked",
|
||||
"resizable",
|
||||
"resizable2",
|
||||
"many",
|
||||
] {
|
||||
let addr = clawhdf5_format::group_v2::resolve_path_any(&bytes, &sb, name).unwrap();
|
||||
let oh = ObjectHeader::parse(&bytes, addr as usize, 8, 8).unwrap();
|
||||
let layout = oh
|
||||
.messages
|
||||
.iter()
|
||||
.find(|m| m.msg_type == MessageType::DataLayout)
|
||||
.unwrap();
|
||||
assert_eq!(layout.data[0], 3, "layout version of {name}");
|
||||
}
|
||||
}
|
||||
|
||||
/// clawhdf5 reads every dataset's bytes back.
|
||||
fn check_ours(path: &Path, e: &Expect) {
|
||||
let f = File::open(path).unwrap();
|
||||
for (name, want) in &e.datasets {
|
||||
let got = f
|
||||
.dataset(name)
|
||||
.unwrap()
|
||||
.read_selection(&Selection::All)
|
||||
.unwrap();
|
||||
assert!(got == *want, "our read of {name} differs");
|
||||
}
|
||||
let attrs = f.dataset("contig").unwrap().attrs().unwrap();
|
||||
assert!((12..=13).contains(&attrs.len()), "{attrs:?}");
|
||||
}
|
||||
|
||||
/// h5py (libhdf5 2.x) reads every dataset's bytes back.
|
||||
fn check_h5py(path: &Path, e: &Expect, dir: &Path) {
|
||||
let mut script = format!(
|
||||
"import h5py, numpy as np\nf = h5py.File({p:?}, 'r')\n",
|
||||
p = path.to_str().unwrap()
|
||||
);
|
||||
for (i, (name, want)) in e.datasets.iter().enumerate() {
|
||||
let exp = dir.join(format!("expect{i}.bin"));
|
||||
std::fs::write(&exp, want).unwrap();
|
||||
script.push_str(&format!(
|
||||
"a = np.ascontiguousarray(f[{name:?}][()]).tobytes()\n\
|
||||
assert a == open({x:?}, 'rb').read(), {name:?}\n",
|
||||
x = exp.to_str().unwrap()
|
||||
));
|
||||
}
|
||||
script.push_str(
|
||||
"assert f.attrs['text'] in (b'readable by 1.8', 'readable by 1.8')\n\
|
||||
assert list(f.attrs['ints']) == [1, -2, 3]\n\
|
||||
assert len(f['g_dense'].attrs) == 12 and len(f['g_dense']) == 20\n\
|
||||
assert list(f['g_order']) == ['z9', 'z8', 'z7', 'z6', 'z5', 'z4', 'z3', 'z2', 'z1', 'z0']\n\
|
||||
assert f['g_compact/soft'].shape == (1000,)\n\
|
||||
assert f['resizable'].maxshape == (None,)\n\
|
||||
assert f['filled'].fillvalue == -9\n\
|
||||
print('ok')\n",
|
||||
);
|
||||
assert_eq!(py(&script), "ok");
|
||||
}
|
||||
|
||||
/// h5dump (1.14) and `h5rs check --data` accept the file.
|
||||
fn check_tools(path: &Path) {
|
||||
let p = path.to_str().unwrap();
|
||||
let o = Command::new(env!("CARGO_BIN_EXE_h5rs"))
|
||||
.args(["check", "--data", "-q", p])
|
||||
.output()
|
||||
.unwrap();
|
||||
assert!(o.status.success(), "h5rs check --data {p}:\n{}", text(&o));
|
||||
let o = Command::new("h5dump")
|
||||
.args(["-o", "/dev/null", p])
|
||||
.output()
|
||||
.unwrap();
|
||||
assert!(
|
||||
o.status.success() && o.stderr.is_empty(),
|
||||
"h5dump {p}:\n{}",
|
||||
text(&o)
|
||||
);
|
||||
}
|
||||
|
||||
/// HDF5 1.8's h5dump reads the whole file exactly as h5dump 1.14 does, and
|
||||
/// dumps each numeric dataset's values as the bytes we wrote.
|
||||
fn check_18(h5dump: &Path, path: &Path, e: &Expect, dir: &Path) {
|
||||
let p = path.to_str().unwrap();
|
||||
let o = Command::new(h5dump).arg(p).output().unwrap();
|
||||
assert!(
|
||||
o.status.success() && o.stderr.is_empty(),
|
||||
"h5dump 1.8 {p}:\n{}",
|
||||
text(&o)
|
||||
);
|
||||
let out18 = String::from_utf8_lossy(&o.stdout).to_string();
|
||||
assert!(out18.contains("EXTERNAL_LINK \"ext\""), "{out18}");
|
||||
let o = Command::new("h5dump").arg(p).output().unwrap();
|
||||
assert!(o.status.success(), "h5dump {p}:\n{}", text(&o));
|
||||
// HDF5 1.8 has no name for IEEE half floats.
|
||||
let out = String::from_utf8_lossy(&o.stdout).replace(
|
||||
"H5T_IEEE_F16LE",
|
||||
"16-bit little-endian floating-point 16-bit precision",
|
||||
);
|
||||
if let Some((n, (a, b))) = out18
|
||||
.lines()
|
||||
.zip(out.lines())
|
||||
.enumerate()
|
||||
.find(|(_, (a, b))| a != b)
|
||||
{
|
||||
panic!("h5dump 1.8 and h5dump differ at line {}:\n{a}\n{b}", n + 1);
|
||||
}
|
||||
assert_eq!(out18.lines().count(), out.lines().count());
|
||||
// `-b` writes nothing for compounds, enums, arrays, strings and half
|
||||
// floats (HDF5 1.8
|
||||
// has no native half float), for h5py's files too: those are covered
|
||||
// by the text above.
|
||||
let skip = ["/f16", "/strings", "/compound", "/enum", "/array"];
|
||||
for (i, (name, want)) in e.datasets.iter().enumerate() {
|
||||
if want.is_empty() || skip.contains(&name.as_str()) {
|
||||
continue;
|
||||
}
|
||||
let bin = dir.join(format!("dump18_{i}.bin"));
|
||||
let o = Command::new(h5dump)
|
||||
.args(["-d", name, "-b", "LE", "-o"])
|
||||
.arg(&bin)
|
||||
.arg(p)
|
||||
.output()
|
||||
.unwrap();
|
||||
assert!(o.status.success(), "h5dump 1.8 -d {name}:\n{}", text(&o));
|
||||
let got = std::fs::read(&bin).unwrap();
|
||||
assert!(got == *want, "HDF5 1.8 read of {name} differs");
|
||||
}
|
||||
}
|
||||
|
||||
fn check_all(path: &Path, e: &Expect, dir: &Path, h5dump: Option<&Path>) {
|
||||
check_ours(path, e);
|
||||
check_h5py(path, e, dir);
|
||||
check_tools(path);
|
||||
if let Some(h) = h5dump {
|
||||
check_18(h, path, e, dir);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn every_feature_reads_in_hdf5_1_8() {
|
||||
if !tools_ok() {
|
||||
return;
|
||||
}
|
||||
let h18 = h5dump18();
|
||||
let dir = tmpdir();
|
||||
let path = dir.path().join("v18.h5");
|
||||
let mut e = write_file(&path);
|
||||
check_versions(&path);
|
||||
check_all(&path, &e, dir.path(), h18.as_deref());
|
||||
|
||||
// FileEditor: grow the unlimited datasets (splitting B-tree nodes),
|
||||
// overwrite values in filtered and unfiltered chunks, set attributes
|
||||
// (compact and dense).
|
||||
let mut ed = FileEditor::open(&path).unwrap();
|
||||
let mut res: Vec<i32> = (0..25).collect();
|
||||
for round in 0..300usize {
|
||||
let n = res.len() as u64;
|
||||
let add = 1 + (round % 9) as u64;
|
||||
ed.resize("resizable", &[n + add]).unwrap();
|
||||
let vals: Vec<i32> = (0..add).map(|k| (n + k) as i32 * 7 - 3).collect();
|
||||
ed.write_values("resizable", &block(&[n], &[add]), &vals)
|
||||
.unwrap();
|
||||
res.extend(&vals);
|
||||
}
|
||||
e.set("/resizable", le(&res, i32::to_le_bytes));
|
||||
let mut many: Vec<i32> = (0..MANY as i32).map(|i| i ^ 0x5a5a).collect();
|
||||
ed.resize("many", &[MANY as u64 + 500]).unwrap();
|
||||
let vals: Vec<i32> = (0..500).collect();
|
||||
ed.write_values("many", &block(&[MANY as u64], &[500]), &vals)
|
||||
.unwrap();
|
||||
many.extend(&vals);
|
||||
many[12_345] = -1;
|
||||
ed.write_values("many", &block(&[12_345], &[1]), &[-1i32])
|
||||
.unwrap();
|
||||
e.set("/many", le(&many, i32::to_le_bytes));
|
||||
let mut chunked: Vec<f32> = (0..60 * 70).map(|i| (i % 97) as f32 * 1.25).collect();
|
||||
let row: Vec<f32> = (0..70).map(|i| -(i as f32)).collect();
|
||||
ed.write_values("chunked", &block(&[31, 0], &[1, 70]), &row)
|
||||
.unwrap();
|
||||
chunked[31 * 70..32 * 70].copy_from_slice(&row);
|
||||
e.set("/chunked", le(&chunked, f32::to_le_bytes));
|
||||
ed.resize("resizable2", &[9, 6]).unwrap();
|
||||
let mut r2: Vec<i64> = (0..30).map(|i| i * 1_000_000_007).collect();
|
||||
r2.extend(std::iter::repeat_n(0, 24));
|
||||
e.set("/resizable2", le(&r2, i64::to_le_bytes));
|
||||
ed.set_attr("contig", "added", &AttrValue::F64(3.25))
|
||||
.unwrap();
|
||||
ed.set_attr("g_dense", "attr03", &AttrValue::F64(-1.0))
|
||||
.unwrap();
|
||||
ed.set_attr("/", "text", &AttrValue::String("edited".into()))
|
||||
.unwrap();
|
||||
drop(ed);
|
||||
|
||||
check_ours(&path, &e);
|
||||
let f = File::open(&path).unwrap();
|
||||
assert!(matches!(
|
||||
f.dataset("contig").unwrap().attr("added").unwrap(),
|
||||
Some(AttrValue::F64(v)) if v == 3.25
|
||||
));
|
||||
drop(f);
|
||||
// Everything but the root's "text" attribute is as check_h5py expects.
|
||||
py(&format!(
|
||||
"import h5py\nf = h5py.File({p:?}, 'r')\n\
|
||||
assert f.attrs['text'] in (b'edited', 'edited')\n\
|
||||
assert f['contig'].attrs['added'] == 3.25\n\
|
||||
assert f['g_dense'].attrs['attr03'] == -1.0\n",
|
||||
p = path.to_str().unwrap()
|
||||
));
|
||||
let o = Command::new(env!("CARGO_BIN_EXE_h5rs"))
|
||||
.args(["check", "--data", "-q", path.to_str().unwrap()])
|
||||
.output()
|
||||
.unwrap();
|
||||
assert!(o.status.success(), "h5rs check after edits:\n{}", text(&o));
|
||||
if let Some(h) = h18.as_deref() {
|
||||
check_18(h, &path, &e, dir.path());
|
||||
}
|
||||
|
||||
// h5py (libhdf5 2.x) goes on appending to the 1.8-format file, and 1.8
|
||||
// still reads it.
|
||||
py(&format!(
|
||||
"import h5py, numpy as np\n\
|
||||
with h5py.File({p:?}, 'r+') as f:\n\
|
||||
\x20 d = f['resizable']\n\
|
||||
\x20 n = d.shape[0]\n\
|
||||
\x20 d.resize((n + 100,))\n\
|
||||
\x20 d[n:] = np.arange(100, dtype='<i4') + 5000\n",
|
||||
p = path.to_str().unwrap()
|
||||
));
|
||||
res.extend((0..100).map(|k| 5000 + k));
|
||||
e.set("/resizable", le(&res, i32::to_le_bytes));
|
||||
check_ours(&path, &e);
|
||||
if let Some(h) = h18.as_deref() {
|
||||
check_18(h, &path, &e, dir.path());
|
||||
}
|
||||
}
|
||||
|
||||
// ---- The trees against libhdf5's ----
|
||||
|
||||
/// One node of a version-1 chunk B-tree: level, and per key its stored
|
||||
/// size and offsets; children below.
|
||||
#[derive(Debug, PartialEq)]
|
||||
struct TreeNode {
|
||||
level: u8,
|
||||
keys: Vec<(u32, Vec<u64>)>,
|
||||
children: Vec<TreeNode>,
|
||||
}
|
||||
|
||||
fn read_tree(bytes: &[u8], addr: u64, ndims: usize, sizes: bool) -> TreeNode {
|
||||
let a = addr as usize;
|
||||
assert_eq!(&bytes[a..a + 4], b"TREE");
|
||||
let level = bytes[a + 5];
|
||||
let n = u16::from_le_bytes([bytes[a + 6], bytes[a + 7]]) as usize;
|
||||
let mut p = a + 24;
|
||||
let mut keys = Vec::new();
|
||||
let mut kids = Vec::new();
|
||||
for i in 0..=n {
|
||||
let size = u32::from_le_bytes(bytes[p..p + 4].try_into().unwrap());
|
||||
let offs = (0..ndims)
|
||||
.map(|d| u64::from_le_bytes(bytes[p + 8 + 8 * d..p + 16 + 8 * d].try_into().unwrap()))
|
||||
.collect();
|
||||
keys.push((if sizes { size } else { 0 }, offs));
|
||||
p += 8 + 8 * ndims;
|
||||
if i < n {
|
||||
kids.push(u64::from_le_bytes(bytes[p..p + 8].try_into().unwrap()));
|
||||
p += 8;
|
||||
}
|
||||
}
|
||||
let children = if level > 0 {
|
||||
kids.iter()
|
||||
.map(|&c| read_tree(bytes, c, ndims, sizes))
|
||||
.collect()
|
||||
} else {
|
||||
Vec::new()
|
||||
};
|
||||
TreeNode {
|
||||
level,
|
||||
keys,
|
||||
children,
|
||||
}
|
||||
}
|
||||
|
||||
/// Where two trees first differ (libhdf5's first).
|
||||
fn first_difference(a: &TreeNode, b: &TreeNode, at: &str) -> Option<String> {
|
||||
if a.level != b.level || a.keys.len() != b.keys.len() {
|
||||
return Some(format!(
|
||||
"{at}: level {} with {} keys vs level {} with {} keys",
|
||||
a.level,
|
||||
a.keys.len(),
|
||||
b.level,
|
||||
b.keys.len()
|
||||
));
|
||||
}
|
||||
if let Some(i) = (0..a.keys.len()).find(|&i| a.keys[i] != b.keys[i]) {
|
||||
return Some(format!("{at} key {i}: {:?} vs {:?}", a.keys[i], b.keys[i]));
|
||||
}
|
||||
a.children
|
||||
.iter()
|
||||
.zip(&b.children)
|
||||
.enumerate()
|
||||
.find_map(|(i, (x, y))| first_difference(x, y, &format!("{at}/{i}")))
|
||||
}
|
||||
|
||||
/// The chunk B-tree of dataset `name`: (tree, ndims) from its layout.
|
||||
fn tree_of(path: &Path, name: &str, sizes: bool) -> TreeNode {
|
||||
use clawhdf5_format::message_type::MessageType;
|
||||
use clawhdf5_format::object_header::ObjectHeader;
|
||||
let bytes = std::fs::read(path).unwrap();
|
||||
let f = File::open(path).unwrap();
|
||||
let sb = f.superblock().clone();
|
||||
let addr = clawhdf5_format::group_v2::resolve_path_any(&bytes, &sb, name).unwrap();
|
||||
let oh = ObjectHeader::parse(&bytes, addr as usize, 8, 8).unwrap();
|
||||
let l = &oh
|
||||
.messages
|
||||
.iter()
|
||||
.find(|m| m.msg_type == MessageType::DataLayout)
|
||||
.unwrap()
|
||||
.data;
|
||||
assert_eq!((l[0], l[1]), (3, 2), "{name}: layout v3, chunked");
|
||||
let ndims = l[2] as usize;
|
||||
let root = u64::from_le_bytes(l[3..11].try_into().unwrap());
|
||||
read_tree(&bytes, root, ndims, sizes)
|
||||
}
|
||||
|
||||
/// Our version-1 chunk B-trees are libhdf5's, node for node (levels, child
|
||||
/// counts, every key's offsets, and its chunk size where the chunks are
|
||||
/// the same bytes), for 1-D, 2-D and 3-D datasets with two- and three-level
|
||||
/// trees, filtered or not.
|
||||
///
|
||||
/// libhdf5 inserts each chunk into the tree when it leaves its chunk cache.
|
||||
/// A whole-dataset write with no cache (`rdcc_nbytes=0`, or chunks larger
|
||||
/// than the cache) inserts them in row-major order, as we build the tree;
|
||||
/// with the default cache small chunks of a 1-D dataset still arrive in
|
||||
/// order, but those of a multi-dimensional one arrive in the order the
|
||||
/// cache's hash evicts them, which gives the same keys in differently
|
||||
/// filled nodes. Both trees index the same chunks; we do not model the
|
||||
/// cache.
|
||||
#[test]
|
||||
fn chunk_btrees_match_libhdf5() {
|
||||
if !tools_ok() {
|
||||
return;
|
||||
}
|
||||
let dir = tmpdir();
|
||||
let theirs = dir.path().join("libhdf5.h5");
|
||||
py(&format!(
|
||||
"import h5py, numpy as np\n\
|
||||
with h5py.File({p:?}, 'w', libver=('v108', 'latest'), rdcc_nbytes=0) as f:\n\
|
||||
\x20 f.create_dataset('d1000', data=np.arange(10000.0), chunks=(10,))\n\
|
||||
\x20 f.create_dataset('d999', data=np.arange(9990.0), chunks=(10,))\n\
|
||||
\x20 f.create_dataset('big', data=np.arange(100000, dtype='<i4'), chunks=(1,), maxshape=(None,))\n\
|
||||
\x20 f.create_dataset('grid', data=np.arange(10000.0).reshape(100, 100), chunks=(1, 1))\n\
|
||||
\x20 f.create_dataset('cube', data=np.arange(27000, dtype='<i2').reshape(30, 30, 30), chunks=(2, 3, 5))\n\
|
||||
\x20 f.create_dataset('gz', data=np.arange(20000, dtype='<i8') % 13, chunks=(7,), compression='gzip')\n",
|
||||
p = theirs.to_str().unwrap()
|
||||
));
|
||||
let ours = dir.path().join("ours.h5");
|
||||
let mut b = FileBuilder::new();
|
||||
b.libver_bounds(LibVer::V18, LibVer::V18);
|
||||
let v: Vec<f64> = (0..10000).map(f64::from).collect();
|
||||
b.create_dataset("d1000")
|
||||
.with_f64_data(&v)
|
||||
.with_chunks(&[10]);
|
||||
b.create_dataset("d999")
|
||||
.with_f64_data(&v[..9990])
|
||||
.with_chunks(&[10]);
|
||||
let v: Vec<i32> = (0..100_000).collect();
|
||||
b.create_dataset("big")
|
||||
.with_i32_data(&v)
|
||||
.with_chunks(&[1])
|
||||
.with_maxshape(&[u64::MAX]);
|
||||
let v: Vec<f64> = (0..10000).map(f64::from).collect();
|
||||
b.create_dataset("grid")
|
||||
.with_f64_data(&v)
|
||||
.with_shape(&[100, 100])
|
||||
.with_chunks(&[1, 1]);
|
||||
let raw: Vec<u8> = (0..27000i16).flat_map(|x| x.to_le_bytes()).collect();
|
||||
b.create_dataset("cube")
|
||||
.with_compound_data(
|
||||
clawhdf5_format::datatype::Datatype::FixedPoint {
|
||||
size: 2,
|
||||
byte_order: clawhdf5_format::datatype::DatatypeByteOrder::LittleEndian,
|
||||
signed: true,
|
||||
bit_offset: 0,
|
||||
bit_precision: 16,
|
||||
},
|
||||
raw,
|
||||
27000,
|
||||
)
|
||||
.with_shape(&[30, 30, 30])
|
||||
.with_chunks(&[2, 3, 5]);
|
||||
let v: Vec<i64> = (0..20000).map(|i| i % 13).collect();
|
||||
b.create_dataset("gz")
|
||||
.with_i64_data(&v)
|
||||
.with_chunks(&[7])
|
||||
.with_deflate(4);
|
||||
b.write(&ours).unwrap();
|
||||
for (name, sizes) in [
|
||||
("d1000", true),
|
||||
("d999", true),
|
||||
("big", true),
|
||||
("grid", true),
|
||||
("cube", true),
|
||||
// Compressed sizes differ between zlib-rs and zlib.
|
||||
("gz", false),
|
||||
] {
|
||||
let a = tree_of(&theirs, name, sizes);
|
||||
let b = tree_of(&ours, name, sizes);
|
||||
if let Some(d) = first_difference(&a, &b, "root") {
|
||||
panic!("{name}: our chunk B-tree differs from libhdf5's: {d}");
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user