From 310448bfcb1c7ee153ce1c3e82e7aeb0594b4f1b Mon Sep 17 00:00:00 2001 From: osobh Date: Sat, 26 Sep 2026 00:29:28 -0500 Subject: [PATCH] feat(tools): h5rs, pure-Rust HDF5 tools (ls, dump, stat, diff, check) New workspace crate clawhdf5-tools with one binary, h5rs, built only on the clawhdf5 facade and clawhdf5-format (no libhdf5, no C): - ls [-r] [-v] FILE[/path]: h5ls's listing (same text in its first two columns) plus the datatype; -v adds address, link count, layout and chunk index, chunk size, storage, filters, datatype and attributes. - dump [--json] [-A] [-p] [-d PATH] FILE: h5dump DDL (byte-identical to h5dump 1.14.6 on the test files) or hdf5-json. - stat FILE: h5stat's object/link/rank/layout/filter/attribute counts, raw data and total size. - diff [-r] [-q] [-d D] [-p R] A B [OBJ1 [OBJ2]]: structural and value differences, exit 0/1/2 like h5diff. - check [--data] FILE: walks every object, parses every message, verifies the checksums of every v2+ structure (including the fractal heap blocks the library never checks), checks chunk indexes against their datasets and raw data for out-of-file or overlapping extents; every problem with its address. Values over --max-bytes are reported, not read; dense-storage heaps are verified before objects are read from them; panics are caught (exit 3). Tests compare with h5ls, h5stat, h5dump and h5diff and with h5py's values, and flip the checksum of every checksummed structure in a v1.14-format file. Co-Authored-By: Claude Opus 5.5 (1M context) --- Cargo.toml | 1 + crates/clawhdf5-tools/Cargo.toml | 23 + crates/clawhdf5-tools/README.md | 231 +++++ crates/clawhdf5-tools/src/check.rs | 825 +++++++++++++++++ crates/clawhdf5-tools/src/cli.rs | 45 + crates/clawhdf5-tools/src/diff.rs | 703 +++++++++++++++ crates/clawhdf5-tools/src/dtype.rs | 529 +++++++++++ crates/clawhdf5-tools/src/dump.rs | 944 ++++++++++++++++++++ crates/clawhdf5-tools/src/h5.rs | 718 +++++++++++++++ crates/clawhdf5-tools/src/heap_blocks.rs | 308 +++++++ crates/clawhdf5-tools/src/info.rs | 245 +++++ crates/clawhdf5-tools/src/lib.rs | 63 ++ crates/clawhdf5-tools/src/ls.rs | 385 ++++++++ crates/clawhdf5-tools/src/main.rs | 58 ++ crates/clawhdf5-tools/src/stat.rs | 305 +++++++ crates/clawhdf5-tools/src/value.rs | 483 ++++++++++ crates/clawhdf5-tools/tests/gen_files.py | 115 +++ crates/clawhdf5-tools/tests/h5rs_interop.rs | 569 ++++++++++++ 18 files changed, 6550 insertions(+) create mode 100644 crates/clawhdf5-tools/Cargo.toml create mode 100644 crates/clawhdf5-tools/README.md create mode 100644 crates/clawhdf5-tools/src/check.rs create mode 100644 crates/clawhdf5-tools/src/cli.rs create mode 100644 crates/clawhdf5-tools/src/diff.rs create mode 100644 crates/clawhdf5-tools/src/dtype.rs create mode 100644 crates/clawhdf5-tools/src/dump.rs create mode 100644 crates/clawhdf5-tools/src/h5.rs create mode 100644 crates/clawhdf5-tools/src/heap_blocks.rs create mode 100644 crates/clawhdf5-tools/src/info.rs create mode 100644 crates/clawhdf5-tools/src/lib.rs create mode 100644 crates/clawhdf5-tools/src/ls.rs create mode 100644 crates/clawhdf5-tools/src/main.rs create mode 100644 crates/clawhdf5-tools/src/stat.rs create mode 100644 crates/clawhdf5-tools/src/value.rs create mode 100644 crates/clawhdf5-tools/tests/gen_files.py create mode 100644 crates/clawhdf5-tools/tests/h5rs_interop.rs diff --git a/Cargo.toml b/Cargo.toml index 2e1f6fe..b86d8b6 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -16,6 +16,7 @@ members = [ "crates/clawhdf5-cli", "crates/clawhdf5-napi", "crates/clawhdf5-bench", + "crates/clawhdf5-tools", "crates/libaec-sys", ] resolver = "2" diff --git a/crates/clawhdf5-tools/Cargo.toml b/crates/clawhdf5-tools/Cargo.toml new file mode 100644 index 0000000..d54f39d --- /dev/null +++ b/crates/clawhdf5-tools/Cargo.toml @@ -0,0 +1,23 @@ +[package] +name = "clawhdf5-tools" +version = "2.7.0" +edition = "2024" +rust-version.workspace = true +license = "MIT" +description = "h5rs: pure-Rust HDF5 command-line tools (ls, dump, stat, diff, check) without libhdf5" +repository = "https://git.redclaw.dev/quantumclaw/clawhdf5" +keywords = ["hdf5", "h5dump", "h5ls", "cli", "science"] +categories = ["command-line-utilities", "science"] +readme = "README.md" + +[[bin]] +name = "h5rs" +path = "src/main.rs" + +[dependencies] +clawhdf5 = { path = "../clawhdf5", version = "2.7.0" } +clawhdf5-format = { path = "../clawhdf5-format", version = "2.7.0" } +serde_json = "1" + +[dev-dependencies] +tempfile = { workspace = true } diff --git a/crates/clawhdf5-tools/README.md b/crates/clawhdf5-tools/README.md new file mode 100644 index 0000000..342756b --- /dev/null +++ b/crates/clawhdf5-tools/README.md @@ -0,0 +1,231 @@ +# clawhdf5-tools: `h5rs` + +HDF5 command-line tools in pure Rust, built only on the `clawhdf5` facade and +`clawhdf5-format`. No libhdf5 and no C code, so the binary also builds as a +fully static executable: `cargo build --release -p clawhdf5-tools --target +x86_64-unknown-linux-musl` gives a static-pie `h5rs` of 1.8 MB with no +shared-library dependencies (built and run on tank, 2026-09-26, Rust 1.98). + +| Command | Modelled on | What it does | +|---------|-------------|--------------| +| `h5rs ls` | `h5ls` | list objects: name, kind, shape, datatype; `-v` adds layout, chunking, storage, filters, attributes | +| `h5rs dump` | `h5dump` | the file's structure and values as DDL text, or as JSON (hdf5-json layout) | +| `h5rs stat` | `h5stat` | object, link, rank, layout, filter and attribute counts; raw-data and file size | +| `h5rs diff` | `h5diff` | structural and value differences between two files or objects | +| `h5rs check` | `h5check` | structural validator: every object and header message, the checksums of version 2+ structures, chunk index consistency | + +```bash +cargo install --path crates/clawhdf5-tools # or: cargo build --release -p clawhdf5-tools +h5rs --help +h5rs --help +``` + +Every command takes `--max-bytes N` where it reads values (default 1 GiB): a +dataset whose dataspace claims more than that is reported instead of read, so +a corrupt size cannot exhaust memory. + +## `h5rs ls` + +```console +$ h5rs ls -r data.h5 # an extract of the output +/ Group +/external External Link {other.h5//x} +/grp Group +/grp/ext2 Dataset {6/Inf, 10/Inf} float32 +/grp/gz Dataset {1000} int32 +/hard2 Group, same as /grp/sub +/named_t Type +/soft Soft Link {/contig} +``` + +The first two columns are h5ls's (`h5ls -r` prints the same text); h5rs adds +the datatype. `FILE/path` lists a group's members or one dataset, as h5ls +does. `-v` prints, per object: + +```console +$ h5rs ls -v data.h5/grp/gz +gz Dataset {1000/1000} + Address: 1120 + Links: 1 + Layout: chunked (fixed array index) + Chunks: {100} 400 bytes + Storage: 4000 logical bytes, 1237 allocated bytes, 323.36% utilization + Filter-0: shuffle-2 OPT {4} + Filter-1: deflate-1 OPT {4} + Filter-2: fletcher32-3 + Type: 32-bit little-endian integer +``` + +## `h5rs dump` + +```console +$ h5rs dump data.h5 # DDL, like h5dump +$ h5rs dump -A data.h5 # no dataset values (attributes still shown), like h5dump -A +$ h5rs dump -d /grp/gz data.h5 # one dataset +$ h5rs dump -p data.h5 # also STORAGE_LAYOUT and FILTERS blocks +$ h5rs dump --json data.h5 # hdf5-json +``` + +The DDL output is h5dump's: on the test files of `tests/gen_files.py` +(compact, contiguous and chunked datasets with every chunk index; v1 and v2 +groups; integers of both byte orders, floats, compound with an array member, +enum, fixed and variable-length strings; soft, external and hard links; a +named datatype; compact and dense attributes) `h5rs dump` and `h5rs dump -A` +print the same bytes as h5dump 1.14.6 — `dump_matches_h5dump` in +`tests/h5rs_interop.rs` checks this. Not covered by that test: references, +opaque, bitfield, variable-length sequences and virtual datasets. Floats print at their own precision (a `float32` 0.1 prints as +`0.1`), which for some values is more digits than h5dump's `%g`. The `-p` +block is h5rs's own (it names the chunk index), not h5dump's. + +### JSON schema + +`--json` follows the HDF Group's [hdf5-json](https://github.com/HDFGroup/hdf5-json) +layout: + +```json +{ + "apiVersion": "1.1.1", + "root": "g-0000000000000060", + "groups": { "": { "alias": ["/grp"], "attributes": [...], "links": [...] } }, + "datasets": { "": { "alias": [...], "attributes": [...], "shape": {...}, + "type": {...}, "creationProperties": {...}, "value": ... } }, + "datatypes": { "": { "alias": [...], "attributes": [...], "type": {...} } } +} +``` + +- **ids** are `g-`/`d-`/`t-` plus the object header address in 16 hex digits + (hdf5-json uses UUIDs; these are stable for a given file). `alias` lists + every path that reaches the object. +- **links**: `{"class": "H5L_TYPE_HARD", "title", "collection", "id"}`, + `{"class": "H5L_TYPE_SOFT", "title", "h5path"}`, + `{"class": "H5L_TYPE_EXTERNAL", "title", "file", "h5path"}`, + `{"class": "H5L_TYPE_USER_DEFINED", "title", "linkClass"}`. +- **shape**: `{"class": "H5S_NULL"}`, `{"class": "H5S_SCALAR"}` or + `{"class": "H5S_SIMPLE", "dims": [...], "maxdims": [...]}` with + `"H5S_UNLIMITED"` for an unlimited dimension. +- **type**: `{"class": "H5T_INTEGER" | "H5T_FLOAT" | "H5T_BITFIELD", "base": + "H5T_STD_I32LE" ...}`, `{"class": "H5T_STRING", "charSet", "strPad", + "length": n | "H5T_VARIABLE"}`, `{"class": "H5T_COMPOUND", "fields": [{"name", + "type"}]}`, `{"class": "H5T_ARRAY", "base", "dims"}`, `{"class": "H5T_ENUM", + "base", "mapping": {"NAME": value}}`, `{"class": "H5T_VLEN", "base"}`, + `{"class": "H5T_OPAQUE", "size", "tag"}`, `{"class": "H5T_REFERENCE", "base": + "H5T_STD_REF_OBJ" | "H5T_STD_REF_DSETREG" | "H5T_STD_REF"}`. +- **value**: nested lists in the dataset's shape (a scalar is the bare value, + a null dataspace `null`). A compound element is a list of its members, an + enum element its integer value, a string a JSON string, opaque/bitfield data + a `0x...` hex string, an object reference the referenced object's path, + NaN/infinities the strings `"NaN"`, `"Infinity"`, `"-Infinity"`, and an + integer beyond 64 bits a decimal string. +- **creationProperties**: `layout` (`{"class": "H5D_CHUNKED", "dims": [...]}` + etc.) and `filters` (`[{"id", "name", "class", "parameters"}]`). + +A value that cannot be read is replaced by `"value_error": ""` and +the command exits 1. + +## `h5rs stat` + +Prints h5stat's report sections with the same labels for the facts it +computes — object and link counts, max links to an object, max objects in a +group, dataset ranks, layout counts, filter counts, attribute counts, total +raw data size and total file size (all equal to h5stat's on the test files; +`stat_matches_h5stat` checks them). It does not break metadata space down by +structure as h5stat does; it reports metadata and free space as one figure. + +## `h5rs diff` + +```console +$ h5rs diff a.h5 b.h5 # whole files +$ h5rs diff a.h5 b.h5 /grp # one object and everything below it +$ h5rs diff a.h5 b.h5 /x /y # different paths in each +$ h5rs diff -r a.h5 b.h5 /d # list every differing element +$ h5rs diff -d 0.001 a.h5 b.h5 # |a - b| > 0.001 is a difference +$ h5rs diff -p 0.01 a.h5 b.h5 # |a - b| / |a| > 1% is a difference +``` + +Exit status: 0 no differences, 1 differences, 2 error — the same as h5diff's +on the cases `diff_exit_codes_match_h5diff` runs. Compared: which objects +exist, their kinds, datatypes and shapes, attribute sets and values, dataset +values, and soft/external link targets. Two differences from h5diff, on +purpose: objects that cannot be compared (different shapes or datatype +classes) count as a difference (h5diff warns and exits 0), and two NaNs are +equal. + +## `h5rs check` + +```console +$ h5rs check data.h5 +checked data.h5: superblock v3, 33 objects (4 groups, 28 datasets, 1 named datatypes), 139 header messages, 34 chunks +checksums verified: superblock 1, v2 object headers 33, v2 B-trees 3, fractal heaps 3 (+3 blocks), chunk indexes 5 +no problems found + +$ h5rs check damaged.h5 # one bit flipped in /grp/ext1's chunk index header +problem: 0x79c /grp/ext1: chunk index (extensible array): checksum mismatch: expected 0xcfe2391f, computed 0x2dd59bed +checked damaged.h5: superblock v3, 33 objects (4 groups, 28 datasets, 1 named datatypes), 139 header messages, 27 chunks +checksums verified: superblock 1, v2 object headers 33, v2 B-trees 3, fractal heaps 3 (+3 blocks), chunk indexes 4 +1 problem found +``` + +Each problem line is `problem:
: `; addresses +are HDF5 addresses (relative to the superblock, as h5dump and h5ls print +them). + +It walks every object reachable from the root group (and the superblock +extension) and checks: + +- the superblock (and its checksum, version 2+) and that the file is not + shorter than the superblock's end-of-file address; +- every object header, with the checksum of version 2 headers and of their + continuation chunks, and every header message parsed by type (dataspace, + datatype, fill value, layout, filter pipeline, attributes, link info, + links, group info, symbol table), including shared messages; +- groups: the symbol table (v1 B-tree, local heap, symbol nodes), or the + links, and for dense storage the fractal heap — header, and every direct + and indirect block with its checksum, back-pointer and heap offset — and + the v2 B-tree name and creation-order indexes (every node's checksum, and + the record count against the header's); +- dense attribute storage the same way; +- datasets: the layout against the dataspace and datatype (compact and + contiguous sizes, chunk rank), and for chunked datasets the whole chunk + index (v1 B-tree, single chunk, implicit, fixed array, extensible array, + v2 B-tree, with the checksums of the last three): every chunk's offset must + be a multiple of the chunk size and inside the extent, appear once, and + have a plausible size; +- that all raw data (contiguous blocks and chunks) lies inside the file and + no two pieces overlap. + +`--data` also reads every dataset, decoding every chunk through its filters +(which catches corrupt compressed data and Fletcher-32 mismatches). Data the +tool cannot decode (a filter it does not implement, such as szip, or a +dataset over `--max-bytes`) is a `note:`, not a problem. Every problem is +printed with the address of the structure involved; the exit status is 0 +when there are none, 1 when there are, 2 for a usage error or a missing +file. + +libhdf5's h5check understands only the HDF5 1.8 file format; `h5rs check` +also covers the structures HDF5 1.10+ writes (fixed/extensible array and v2 +B-tree chunk indexes, and version 3 superblocks). + +What it does not check: free-space manager and shared-message (SOHM) table +checksums, global heap collections other than those a value read touches, +and objects reachable only by external links. + +## Robustness + +A panic is a bug: `h5rs` catches it, prints `internal error`, and exits 3 +(`check` records it against the object and carries on). `scripts/h5rs-fuzz.sh` +runs every subcommand over every file of a corpus (by default the HDF Group's +CVE reproducers, fetched by `conformance/fetch-corpus.sh`), optionally with +byte-flipped copies (`MUTATE=N`), under a timeout and a memory limit, with +overflow checks on, and fails on any panic, crash or hang. +`scripts/h5rs-check-ok-files.sh` runs `check --data` over the conformance +files that both clawhdf5 and h5py read in full, which must all pass. + +## Tests + +```bash +CLAWHDF5_PYTHON=.venv/bin/python CLAWHDF5_REQUIRE_INTEROP=1 cargo test -p clawhdf5-tools +``` + +The interop tests write their files with h5py and compare with h5ls, h5stat, +h5dump and h5diff; each skips when what it needs is missing unless +`CLAWHDF5_REQUIRE_INTEROP=1`. diff --git a/crates/clawhdf5-tools/src/check.rs b/crates/clawhdf5-tools/src/check.rs new file mode 100644 index 0000000..635b234 --- /dev/null +++ b/crates/clawhdf5-tools/src/check.rs @@ -0,0 +1,825 @@ +//! `h5rs check`: a structural validator. +//! +//! It walks every object reachable from the root group (and the superblock +//! extension), parses every header message, verifies the checksum of every +//! checksummed (version 2+) structure it meets, checks each chunked +//! dataset's chunk index against the dataset's shape, and checks that raw +//! data lies inside the file without overlapping other raw data. Every +//! problem is reported with the address of the structure involved. + +use std::collections::HashSet; +use std::panic::{self, AssertUnwindSafe}; + +use clawhdf5_format::attribute_info::AttributeInfoMessage; +use clawhdf5_format::btree_v2::{BTreeV2Header, collect_btree_v2_records}; +use clawhdf5_format::data_layout::DataLayout; +use clawhdf5_format::dataspace::{Dataspace, DataspaceType}; +use clawhdf5_format::datatype::Datatype; +use clawhdf5_format::group_info::GroupInfoMessage; +use clawhdf5_format::link_info::LinkInfoMessage; +use clawhdf5_format::message_type::MessageType; +use clawhdf5_format::object_header::ObjectHeader; +use clawhdf5_format::symbol_table::SymbolTableMessage; + +use crate::cli::{Args, Out}; +use crate::h5::{Error, ErrorKind, H5, Kind}; +use crate::info::{self, DsInfo}; + +pub const USAGE: &str = "\ +usage: h5rs check [--data] [-q] [--max-bytes N] FILE + +Validate FILE's structure: walk every object from the root group, parse +every header message, verify the checksums of version 2+ structures +(superblock, object headers and continuation chunks, v2 B-tree nodes, +fractal heap headers and blocks, extensible/fixed array chunk indexes), +check each chunked dataset's chunk index against its shape (offsets aligned +to the chunk size and inside the extent, no duplicates, sizes plausible), +and check that all raw data lies inside the file without overlaps. Every +problem is printed with the address of the structure involved. + + --data also read every dataset, decoding every chunk through its + filters (catches corrupt compressed data and Fletcher-32 + mismatches) + -q, --quiet print only the problems, not the summary + --max-bytes N largest dataset read by --data (default 1 GiB) + +Exit status: 0 no problems, 1 problems found, 2 usage error or file not +found, 3 internal error."; + +const MAX_CHUNKS_CHECKED: usize = 10_000_000; + +#[derive(Default)] +struct Counts { + objects: u64, + groups: u64, + datasets: u64, + datatypes: u64, + messages: u64, + chunks: u64, + datasets_read: u64, + sb_checksum: u64, + ohdr_v2: u64, + btree_v2: u64, + heaps: u64, + heap_block_checksums: u64, + chunk_index_checksummed: u64, +} + +struct Problem { + addr: u64, + path: String, + msg: String, +} + +struct Checker<'a> { + h5: &'a H5, + read_data: bool, + eof: u64, + problems: Vec, + /// Things not checked, which are not problems with the file. + notes: Vec, + counts: Counts, + /// Raw data extents: (start, end, owner path). + extents: Vec<(u64, u64, String)>, + heaps_seen: HashSet, + btrees_seen: HashSet, + panicked: bool, +} + +pub fn run(args: &mut Args, out: &mut Out) -> std::io::Result { + let mut read_data = false; + let mut quiet = false; + let mut max_bytes = None; + let mut file = None; + while let Some(a) = args.next() { + match a.as_str() { + "--data" => read_data = true, + "-q" | "--quiet" => quiet = true, + "--max-bytes" => match args.number() { + Some(n) => max_bytes = Some(n), + None => return args.usage_error(out, "--max-bytes needs a number", USAGE), + }, + "-h" | "--help" => { + writeln!(out.o, "{USAGE}")?; + return Ok(0); + } + s if s.starts_with('-') && s.len() > 1 => { + return args.usage_error(out, &format!("unknown option {a}"), USAGE); + } + _ if file.is_none() => file = Some(a), + _ => return args.usage_error(out, &format!("unexpected argument {a}"), USAGE), + } + } + let Some(file) = file else { + return args.usage_error(out, "missing FILE", USAGE); + }; + let path = std::path::Path::new(&file); + if !path.is_file() { + writeln!(out.e, "h5rs check: {file}: no such file")?; + return Ok(2); + } + let mut h5 = match H5::open(path) { + Ok(h) => h, + Err(_) => return unopenable(path, out), + }; + if let Some(m) = max_bytes { + h5.max_bytes = m; + } + let mut c = Checker { + h5: &h5, + read_data, + eof: h5.data().len() as u64, + problems: Vec::new(), + notes: Vec::new(), + counts: Counts::default(), + extents: Vec::new(), + heaps_seen: HashSet::new(), + btrees_seen: HashSet::new(), + panicked: false, + }; + c.superblock(); + c.objects(); + c.overlaps(); + for p in &c.problems { + writeln!(out.o, "problem: {:#x} {}: {}", p.addr, p.path, p.msg)?; + } + if !quiet { + for n in &c.notes { + writeln!(out.o, "note: {:#x} {}: {}", n.addr, n.path, n.msg)?; + } + } + if !quiet { + c.summary(&file, out)?; + } + Ok(if c.panicked { + 3 + } else if c.problems.is_empty() { + 0 + } else { + 1 + }) +} + +/// The file could not be opened at all: say why, as precisely as possible. +fn unopenable(path: &std::path::Path, out: &mut Out) -> std::io::Result { + let data = match std::fs::read(path) { + Ok(d) => d, + Err(e) => { + writeln!(out.e, "h5rs check: {}: {e}", path.display())?; + return Ok(2); + } + }; + let msg = match clawhdf5_format::signature::find_signature(&data) { + Err(_) => ( + 0, + "no HDF5 signature at offset 0 or any power of two from 512".to_string(), + ), + Ok(off) => match clawhdf5_format::superblock::Superblock::parse(&data[off..], 0) { + Ok(_) => (off as u64, "file cannot be opened".to_string()), + Err(e) => (off as u64, format!("superblock: {e}")), + }, + }; + writeln!(out.o, "problem: {:#x} /: {}", msg.0, msg.1)?; + writeln!( + out.o, + "checked {}: 1 problem found (file cannot be opened)", + path.display() + )?; + Ok(1) +} + +fn msg_name(t: MessageType) -> String { + match t { + MessageType::Unknown(n) => format!("message type {n:#x}"), + t => format!("{t:?} message"), + } +} + +impl Checker<'_> { + fn problem(&mut self, addr: u64, path: &str, msg: impl Into) { + self.problems.push(Problem { + addr, + path: path.to_string(), + msg: msg.into(), + }); + } + + fn err(&mut self, default_addr: u64, path: &str, e: &Error) { + // Reading attributes or links fails on a damaged heap or B-tree that + // the structure checks have already reported at the same address. + if let Some(a) = e.addr + && self.problems.iter().any(|p| p.addr == a && p.path == path) + { + return; + } + self.problem(e.addr.unwrap_or(default_addr), path, e.msg.clone()); + } + + /// Run `f`; a panic inside it becomes a problem instead of aborting the + /// whole check. + fn guarded(&mut self, addr: u64, path: &str, f: impl FnOnce(&mut Self)) { + let r = panic::catch_unwind(AssertUnwindSafe(|| f(self))); + if r.is_err() { + self.panicked = true; + self.problem( + addr, + path, + "internal error while checking this object (see stderr)", + ); + } + } + + fn superblock(&mut self) { + let sb = self.h5.sb().clone(); + if sb.version >= 2 { + // Superblock::parse verified it, or the file would not be open. + self.counts.sb_checksum += 1; + } + // libhdf5 stores the end-of-file address counting the user block + // (h5py's `userblock_size` files show it), unlike every other + // address, which is relative to the superblock. + let file_len = self.eof.saturating_add(self.h5.file.user_block_size()); + if sb.eof_address > file_len { + self.problem( + 0, + "/", + format!( + "file is truncated: the superblock's end-of-file address is {:#x} but the file \ + is {file_len:#x} bytes long", + sb.eof_address + ), + ); + } + if sb.root_group_address >= self.eof { + self.problem( + 0, + "/", + format!( + "root group address {:#x} is past the end of the file", + sb.root_group_address + ), + ); + } + let undef = if sb.offset_size >= 8 { + u64::MAX + } else { + (1u64 << (8 * u32::from(sb.offset_size))) - 1 + }; + if let Some(ext) = sb.superblock_extension_address.filter(|&a| a != undef) { + self.guarded(ext, "", |c| match c.h5.header(ext) { + Ok(h) => { + if h.version >= 2 { + c.counts.ohdr_v2 += 1; + } + c.messages(ext, "", &h); + } + Err(e) => c.err(ext, "", &e), + }); + } + } + + fn objects(&mut self) { + let h5 = self.h5; + // Collect first, then check: the walk borrows the file, the checks + // borrow `self` mutably. + let (items, walk) = h5.walk_collect(); + if let Err(e) = walk { + self.err(h5.root(), "/", &e); + } + for it in items { + let path = it.path; + // Soft, external and user-defined links have no header here. + let (Some(addr), None, Some(header)) = (it.addr, it.first_path, it.header) else { + continue; + }; + if addr >= self.eof { + self.problem( + addr, + &path, + "object header address is past the end of the file", + ); + continue; + } + let h = match header { + Ok(h) => h, + Err(e) => { + self.err(addr, &path, &e); + continue; + } + }; + self.counts.objects += 1; + self.guarded(addr, &path, |c| c.object(addr, &path, &h)); + } + } + + fn object(&mut self, addr: u64, path: &str, h: &ObjectHeader) { + if h.version >= 2 { + self.counts.ohdr_v2 += 1; + } + self.messages(addr, path, h); + match self.h5.attributes(h) { + Ok((_, errs)) => { + for e in errs { + self.problem(addr, path, format!("attribute: {e}")); + } + } + Err(e) => self.err(addr, path, &e), + } + let is_root = addr == self.h5.root(); + match Kind::of(h) { + Kind::Group => { + self.counts.groups += 1; + self.group(addr, path, h); + } + Kind::Dataset => { + self.counts.datasets += 1; + self.dataset(addr, path, h); + } + Kind::Datatype => self.counts.datatypes += 1, + Kind::Unknown if is_root => self.counts.groups += 1, + Kind::Unknown => { + self.problem( + addr, + path, + "object header describes no group, dataset or named datatype", + ); + } + } + } + + /// Parse every message of a header by type. + fn messages(&mut self, addr: u64, path: &str, h: &ObjectHeader) { + let (os, ls) = (self.h5.os(), self.h5.ls()); + for m in &h.messages { + self.counts.messages += 1; + let data = + match clawhdf5_format::shared_message::message_data(self.h5.data(), m, os, ls) { + Ok(d) => d.into_owned(), + Err(e) => { + self.problem(addr, path, format!("shared {}: {e}", msg_name(m.msg_type))); + continue; + } + }; + let r: Result<(), String> = match m.msg_type { + MessageType::Dataspace => Dataspace::parse(&data, ls) + .map(drop) + .map_err(|e| e.to_string()), + MessageType::Datatype => { + Datatype::parse(&data).map(drop).map_err(|e| e.to_string()) + } + MessageType::FillValue | MessageType::FillValueOld => { + let mut mm = m.clone(); + mm.data = data.clone(); + clawhdf5_format::fill_value::parse_fill_value(&mm) + .map(drop) + .map_err(|e| e.to_string()) + } + MessageType::DataLayout => DataLayout::parse(&data, os, ls) + .map(drop) + .map_err(|e| e.to_string()), + MessageType::FilterPipeline => { + clawhdf5_format::filter_pipeline::FilterPipeline::parse(&data) + .map(drop) + .map_err(|e| e.to_string()) + } + MessageType::Attribute => { + clawhdf5_format::attribute::AttributeMessage::parse_in_file( + &data, + self.h5.data(), + os, + ls, + ) + .map(drop) + .map_err(|e| e.to_string()) + } + MessageType::AttributeInfo => match AttributeInfoMessage::parse(&data, os) { + Ok(ai) => { + self.dense_storage( + addr, + path, + "attribute", + ai.fractal_heap_address, + [ai.btree_name_index_address, ai.btree_creation_order_address], + ); + Ok(()) + } + Err(e) => Err(e.to_string()), + }, + MessageType::LinkInfo => LinkInfoMessage::parse(&data, os) + .map(drop) + .map_err(|e| e.to_string()), + MessageType::Link => clawhdf5_format::link_message::LinkMessage::parse(&data, os) + .map(drop) + .or_else(|e| match e { + clawhdf5_format::error::FormatError::InvalidLinkType(t) if t >= 65 => { + Ok(()) + } + e => Err(e.to_string()), + }), + MessageType::GroupInfo => GroupInfoMessage::parse(&data) + .map(drop) + .map_err(|e| e.to_string()), + MessageType::SymbolTable => SymbolTableMessage::parse(&data, os) + .map(drop) + .map_err(|e| e.to_string()), + _ => Ok(()), + }; + if let Err(e) = r { + self.problem(addr, path, format!("{}: {e}", msg_name(m.msg_type))); + } + } + } + + /// Dense (fractal heap + v2 B-tree) link or attribute storage. + fn dense_storage( + &mut self, + addr: u64, + path: &str, + what: &str, + heap: Option, + btrees: [Option; 2], + ) { + if let Some(fh) = heap + && self.heaps_seen.insert(fh) + { + self.counts.heaps += 1; + let r = crate::heap_blocks::verify(self.h5, fh); + self.counts.heap_block_checksums += r.checksums as u64; + for e in r.problems { + self.err(fh, path, &e.context(&format!("dense {what} storage"))); + } + } + for bt in btrees.into_iter().flatten() { + if !self.btrees_seen.insert(bt) { + continue; + } + let (os, ls) = (self.h5.os(), self.h5.ls()); + let Ok(off) = usize::try_from(bt) else { + self.problem(bt, path, format!("{what} index address out of range")); + continue; + }; + match BTreeV2Header::parse(self.h5.data(), off, os, ls) { + Ok(hdr) => { + self.counts.btree_v2 += 1; + match collect_btree_v2_records(self.h5.data(), &hdr, os, ls) { + Ok(recs) => { + if recs.len() as u64 != hdr.total_records { + self.problem( + bt, + path, + format!( + "{what} index: v2 B-tree header counts {} records, the tree holds {}", + hdr.total_records, + recs.len() + ), + ); + } + } + Err(e) => self.problem(bt, path, format!("{what} index (v2 B-tree): {e}")), + } + } + Err(e) => self.problem(bt, path, format!("{what} index (v2 B-tree header): {e}")), + } + } + let _ = addr; + } + + fn group(&mut self, addr: u64, path: &str, h: &ObjectHeader) { + if let Some(m) = h + .messages + .iter() + .find(|m| m.msg_type == MessageType::LinkInfo) + && let Ok(li) = LinkInfoMessage::parse(&m.data, self.h5.os()) + { + self.dense_storage( + addr, + path, + "link", + li.fractal_heap_address, + [li.btree_name_index_address, li.btree_creation_order_address], + ); + } + if let Err(e) = self.h5.links(h) { + self.err(addr, path, &e.context("cannot list the group")); + } + if let Some(m) = h + .messages + .iter() + .find(|m| m.msg_type == MessageType::SymbolTable) + && let Ok(st) = SymbolTableMessage::parse(&m.data, self.h5.os()) + { + for (a, what) in [ + (st.btree_address, "B-tree"), + (st.local_heap_address, "local heap"), + ] { + if a >= self.eof { + self.problem( + addr, + path, + format!("symbol table {what} address {a:#x} is past the end of the file"), + ); + } + } + } + } + + fn dataset(&mut self, addr: u64, path: &str, h: &ObjectHeader) { + let info = DsInfo::read(self.h5, path, h); + for e in [ + info.dt.as_ref().err(), + info.ds.as_ref().err(), + info.layout.as_ref().err(), + info.filters.as_ref().err(), + ] + .into_iter() + .flatten() + { + let e = e.clone(); + self.err(addr, path, &e); + } + let (Ok(dt), Ok(ds), Ok(layout)) = (&info.dt, &info.ds, &info.layout) else { + return; + }; + let esize = u64::from(dt.type_size()); + if esize == 0 { + self.problem(addr, path, "datatype has size 0"); + return; + } + let need = match crate::h5::byte_len(ds, dt) { + Ok(n) => n, + Err(e) => { + self.err(addr, path, &e); + return; + } + }; + if let Some(max) = &ds.max_dimensions { + for (i, (&d, &m)) in ds.dimensions.iter().zip(max).enumerate() { + if m != u64::MAX && d > m { + self.problem( + addr, + path, + format!("dimension {i} is {d}, above its maximum {m}"), + ); + } + } + } + match layout { + DataLayout::Compact { data } => { + if (data.len() as u64) < need { + self.problem( + addr, + path, + format!( + "compact data holds {} bytes; the dataset needs {need}", + data.len() + ), + ); + } + } + DataLayout::Contiguous { address, size } => { + if let Some(a) = address { + if !info.external && *size < need { + self.problem( + *a, + path, + format!("contiguous storage is {size} bytes; the dataset needs {need}"), + ); + } + self.extent(*a, *size, path); + } + } + DataLayout::Chunked { .. } => self.chunked(addr, path, &info, dt, ds, layout), + DataLayout::Virtual { .. } => { + let mut l = layout.clone(); + if let Err(e) = l.resolve_vds_mappings(self.h5.data(), self.h5.ls()) { + self.problem(addr, path, format!("virtual dataset mappings: {e}")); + } + } + } + if self.read_data { + match self.h5.read_dataset(path, dt, ds) { + Ok(_) => self.counts.datasets_read += 1, + // Valid data this tool cannot decode is not a problem with + // the file. + Err(e) if e.kind != ErrorKind::Corrupt => self.notes.push(Problem { + addr, + path: path.to_string(), + msg: format!("data not read: {}", e.msg), + }), + Err(e) => self.problem(addr, path, format!("reading the data: {}", e.msg)), + } + } + } + + fn chunked( + &mut self, + addr: u64, + path: &str, + info: &DsInfo, + dt: &Datatype, + ds: &Dataspace, + layout: &DataLayout, + ) { + let DataLayout::Chunked { + chunk_dimensions, + btree_address, + chunk_index_type, + .. + } = layout + else { + return; + }; + let rank = match ds.space_type { + DataspaceType::Simple => ds.dimensions.len(), + _ => 0, + }; + if chunk_dimensions.len() != rank + 1 { + self.problem( + addr, + path, + format!( + "chunked layout has {} chunk dimensions for a rank-{rank} dataset (expected {})", + chunk_dimensions.len(), + rank + 1 + ), + ); + return; + } + let cdims = &chunk_dimensions[..rank]; + if cdims.contains(&0) { + self.problem(addr, path, "a chunk dimension is 0"); + return; + } + let Some(index_addr) = *btree_address else { + return; // no chunk allocated yet + }; + if index_addr >= self.eof { + self.problem( + addr, + path, + format!("chunk index address {index_addr:#x} is past the end of the file"), + ); + return; + } + let chunks = match info::chunks(self.h5, layout, ds, dt) { + Ok(c) => c, + Err(e) => { + self.err(index_addr, path, &e); + return; + } + }; + if matches!(chunk_index_type, Some(3..=5)) { + self.counts.chunk_index_checksummed += 1; + } + let filtered = matches!(&info.filters, Ok(Some(p)) if !p.filters.is_empty()); + let chunk_bytes = cdims.iter().try_fold(u64::from(dt.type_size()), |a, &d| { + a.checked_mul(u64::from(d)) + }); + let max = ds.max_dimensions.clone(); + let mut seen: HashSet> = HashSet::with_capacity(chunks.len().min(1 << 20)); + let mut reported = 0usize; + for (n, c) in chunks.iter().enumerate() { + if n >= MAX_CHUNKS_CHECKED { + self.problem(index_addr, path, "too many chunks; stopped checking them"); + break; + } + self.counts.chunks += 1; + let mut bad = Vec::new(); + if c.offsets.len() < rank { + bad.push(format!( + "has {} coordinates for a rank-{rank} dataset", + c.offsets.len() + )); + } else { + for (i, (&o, &cd)) in c.offsets.iter().zip(cdims).enumerate() { + if o % u64::from(cd) != 0 { + bad.push(format!( + "offset {o} in dimension {i} is not a multiple of the chunk size {cd}" + )); + } + let limit = match max.as_ref().and_then(|m| m.get(i)) { + Some(&u64::MAX) | None => ds.dimensions[i], + Some(&m) => m.max(ds.dimensions[i]), + }; + if o >= limit && !(limit == 0 && o == 0) { + bad.push(format!( + "offset {o} in dimension {i} is outside the extent {limit}" + )); + } + } + if !seen.insert(c.offsets[..rank].to_vec()) { + bad.push("appears twice in the chunk index".into()); + } + } + if c.chunk_size == 0 { + bad.push("has size 0".into()); + } else if !filtered + && let Some(cb) = chunk_bytes + && u64::from(c.chunk_size) != cb + { + bad.push(format!( + "is {} bytes; an unfiltered chunk is {cb}", + c.chunk_size + )); + } + if !bad.is_empty() { + reported += 1; + if reported <= 50 { + self.problem( + c.address, + path, + format!("chunk at {:?} {}", c.offsets, bad.join("; ")), + ); + } + } + self.extent(c.address, u64::from(c.chunk_size), path); + } + if reported > 50 { + self.problem( + index_addr, + path, + format!("{} more bad chunks not listed", reported - 50), + ); + } + } + + /// Record raw data at `[start, start + len)`, checking it is in the file. + fn extent(&mut self, start: u64, len: u64, path: &str) { + if len == 0 { + return; + } + match start.checked_add(len) { + Some(end) if end <= self.eof => self.extents.push((start, end, path.to_string())), + _ => self.problem( + start, + path, + format!("raw data ({len} bytes) extends past the end of the file"), + ), + } + } + + fn overlaps(&mut self) { + let mut ext = std::mem::take(&mut self.extents); + ext.sort_unstable_by_key(|e| (e.0, e.1)); + let mut reported = 0usize; + let mut far: Option<(u64, String)> = None; + let mut msgs = Vec::new(); + for (s, e, p) in &ext { + if let Some((end, owner)) = &far + && s < end + { + reported += 1; + if reported <= 50 { + msgs.push(( + *s, + p.clone(), + format!("raw data at {s:#x} overlaps raw data of {owner}"), + )); + } + } + if far.as_ref().is_none_or(|(end, _)| e > end) { + far = Some((*e, p.clone())); + } + } + for (a, p, m) in msgs { + self.problem(a, &p, m); + } + if reported > 50 { + self.problem( + 0, + "/", + format!("{} more raw data overlaps not listed", reported - 50), + ); + } + } + + fn summary(&self, file: &str, out: &mut Out) -> std::io::Result<()> { + let c = &self.counts; + let sb = self.h5.sb(); + writeln!( + out.o, + "checked {file}: superblock v{}, {} objects ({} groups, {} datasets, {} named datatypes), \ + {} header messages, {} chunks", + sb.version, c.objects, c.groups, c.datasets, c.datatypes, c.messages, c.chunks + )?; + writeln!( + out.o, + "checksums verified: superblock {}, v2 object headers {}, v2 B-trees {}, \ + fractal heaps {} (+{} blocks), chunk indexes {}", + c.sb_checksum, + c.ohdr_v2, + c.btree_v2, + c.heaps, + c.heap_block_checksums, + c.chunk_index_checksummed + )?; + if self.read_data { + writeln!(out.o, "datasets read: {}", c.datasets_read)?; + } + match self.problems.len() { + 0 => writeln!(out.o, "no problems found"), + 1 => writeln!(out.o, "1 problem found"), + n => writeln!(out.o, "{n} problems found"), + } + } +} diff --git a/crates/clawhdf5-tools/src/cli.rs b/crates/clawhdf5-tools/src/cli.rs new file mode 100644 index 0000000..e4ba795 --- /dev/null +++ b/crates/clawhdf5-tools/src/cli.rs @@ -0,0 +1,45 @@ +//! Argument handling and output plumbing shared by the subcommands. + +use std::io::Write; + +/// Where a subcommand writes: `o` for results, `e` for diagnostics. +pub struct Out<'a> { + pub o: &'a mut dyn Write, + pub e: &'a mut dyn Write, +} + +/// The arguments after the subcommand name. +pub struct Args { + cmd: &'static str, + rest: std::vec::IntoIter, +} + +impl Args { + pub fn new(cmd: &'static str, rest: Vec) -> Self { + Self { + cmd, + rest: rest.into_iter(), + } + } + + #[allow(clippy::should_implement_trait)] + pub fn next(&mut self) -> Option { + self.rest.next() + } + + /// The value after an option such as `--max-bytes`. + pub fn value(&mut self) -> Option { + self.rest.next() + } + + /// A numeric option value. + pub fn number(&mut self) -> Option { + self.rest.next().and_then(|s| s.parse().ok()) + } + + /// Report a usage problem; exit status 2. + pub fn usage_error(&self, out: &mut Out, msg: &str, usage: &str) -> std::io::Result { + writeln!(out.e, "h5rs {}: {msg}\n\n{usage}", self.cmd)?; + Ok(2) + } +} diff --git a/crates/clawhdf5-tools/src/diff.rs b/crates/clawhdf5-tools/src/diff.rs new file mode 100644 index 0000000..ba8d324 --- /dev/null +++ b/crates/clawhdf5-tools/src/diff.rs @@ -0,0 +1,703 @@ +//! `h5rs diff`: compare two files (or two objects) like h5diff. + +use std::cell::OnceCell; +use std::collections::{BTreeMap, HashMap}; + +use clawhdf5_format::attribute::AttributeMessage; +use clawhdf5_format::dataspace::{Dataspace, DataspaceType}; +use clawhdf5_format::datatype::Datatype; +use clawhdf5_format::object_header::ObjectHeader; + +use crate::cli::{Args, Out}; +use crate::h5::{H5, Kind, LinkKind}; +use crate::value::{self, Decoder, Value}; + +pub const USAGE: &str = "\ +usage: h5rs diff [options] FILE1 FILE2 [OBJ1 [OBJ2]] + +Compare FILE1 and FILE2 (or OBJ1 in FILE1 with OBJ2 in FILE2, and everything +below them): the objects present, their kinds, datatypes, shapes, attribute +sets and values, and soft/external link targets. + + -r, --report list every differing element (position, values, difference) + -q, --quiet print nothing; only the exit status + -d, --delta D numbers differ only when |a - b| > D + -p, --relative R numbers differ only when |a - b| / |a| > R + -c, --count N list at most N differing elements per object with -r + --max-bytes N largest dataset read (default 1 GiB); a larger one is an error + +Two NaNs compare equal. Unlike h5diff, objects that cannot be compared +(different kinds, datatype classes or shapes) count as a difference. + +Exit status: 0 no differences, 1 differences found, 2 error (a file or +object could not be opened or read)."; + +#[derive(Clone, Copy)] +enum Tol { + Exact, + Delta(f64), + Relative(f64), +} + +struct Opts { + report: bool, + quiet: bool, + tol: Tol, + count: usize, +} + +/// What a relative path names in one file. +#[derive(Clone)] +enum Entry { + Obj(u64, Kind), + Soft(String), + External(String, String), + UserDefined(u8), + Broken(String), +} + +struct Side<'a> { + h5: &'a H5, + label: String, + base: String, + /// Relative path -> what it is. + entries: BTreeMap, + /// Address -> relative path, for comparing references. + rel_of: OnceCell>, +} + +impl Side<'_> { + fn full(&self, rel: &str) -> String { + if rel.is_empty() { + if self.base.is_empty() { + "/".into() + } else { + self.base.clone() + } + } else { + format!("{}{rel}", self.base) + } + } + + fn rel_paths(&self) -> &HashMap { + self.rel_of.get_or_init(|| { + let mut m = HashMap::new(); + let _ = self.h5.walk(|it| { + if let (Some(a), None) = (it.addr, it.first_path) { + m.entry(a).or_insert_with(|| it.path.to_string()); + } + }); + m + }) + } +} + +struct Diff { + opts: Opts, + diffs: u64, + /// Differences already reported under an object heading. + per_object: u64, + errors: u64, +} + +pub fn run(args: &mut Args, out: &mut Out) -> std::io::Result { + let mut opts = Opts { + report: false, + quiet: false, + tol: Tol::Exact, + count: usize::MAX, + }; + let mut max_bytes = None; + let mut pos = Vec::new(); + while let Some(a) = args.next() { + match a.as_str() { + "-r" | "--report" => opts.report = true, + "-q" | "--quiet" => opts.quiet = true, + "-d" | "--delta" => match args.number::() { + Some(d) if d >= 0.0 => opts.tol = Tol::Delta(d), + _ => return args.usage_error(out, "--delta needs a number >= 0", USAGE), + }, + "-p" | "--relative" => match args.number::() { + Some(r) if r >= 0.0 => opts.tol = Tol::Relative(r), + _ => return args.usage_error(out, "--relative needs a number >= 0", USAGE), + }, + "-c" | "--count" => match args.number() { + Some(n) => opts.count = n, + None => return args.usage_error(out, "--count needs a number", USAGE), + }, + "--max-bytes" => match args.number() { + Some(n) => max_bytes = Some(n), + None => return args.usage_error(out, "--max-bytes needs a number", USAGE), + }, + "-h" | "--help" => { + writeln!(out.o, "{USAGE}")?; + return Ok(0); + } + s if s.starts_with('-') && s.len() > 1 => { + return args.usage_error(out, &format!("unknown option {a}"), USAGE); + } + _ => pos.push(a), + } + } + if pos.len() < 2 || pos.len() > 4 { + return args.usage_error(out, "expected FILE1 FILE2 [OBJ1 [OBJ2]]", USAGE); + } + let mut files = Vec::new(); + for f in &pos[..2] { + match H5::open(std::path::Path::new(f)) { + Ok(mut h) => { + if let Some(m) = max_bytes { + h.max_bytes = m; + } + files.push(h); + } + Err(e) => { + writeln!(out.e, "h5rs diff: {e}")?; + return Ok(2); + } + } + } + let obj1 = pos.get(2).cloned().unwrap_or_else(|| "/".into()); + let obj2 = pos.get(3).cloned().unwrap_or_else(|| obj1.clone()); + let mut sides = Vec::new(); + for (h5, (obj, f)) in files.iter().zip([(&obj1, &pos[0]), (&obj2, &pos[1])]) { + let addr = match h5.resolve(obj) { + Ok(a) => a, + Err(_) => { + writeln!( + out.e, + "h5rs diff: object <{obj}> could not be found in <{f}>" + )?; + return Ok(2); + } + }; + let base = obj.trim_end_matches('/').to_string(); + let base = if base.is_empty() || base.starts_with('/') { + base + } else { + format!("/{base}") + }; + let mut entries = BTreeMap::new(); + let walk = h5.walk_from(addr, "", &mut |it| { + let e = match (it.link.map(|l| &l.kind), it.header) { + (Some(LinkKind::Soft(t)), _) => Entry::Soft(t.clone()), + (Some(LinkKind::External { file, path }), _) => { + Entry::External(file.clone(), path.clone()) + } + (Some(LinkKind::UserDefined(t)), _) => Entry::UserDefined(*t), + (_, Some(Err(e))) => Entry::Broken(e.to_string()), + (_, Some(Ok(h))) => Entry::Obj(it.addr.unwrap_or(0), Kind::of(h)), + // A second hard link to an object already compared. + (_, None) => return, + }; + entries.insert(it.path.to_string(), e); + }); + if let Err(e) = walk { + writeln!(out.e, "h5rs diff: {f}: {e}")?; + return Ok(2); + } + sides.push(Side { + h5, + label: f.clone(), + base, + entries, + rel_of: OnceCell::new(), + }); + } + let (a, b) = (&sides[0], &sides[1]); + let mut d = Diff { + opts, + diffs: 0, + per_object: 0, + errors: 0, + }; + let mut names: Vec<&String> = a.entries.keys().chain(b.entries.keys()).collect(); + names.sort(); + names.dedup(); + for rel in names { + match (a.entries.get(rel), b.entries.get(rel)) { + (Some(_), None) => { + d.diffs += 1; + d.say( + out, + &format!("<{}> exists only in <{}>", a.full(rel), a.label), + )?; + } + (None, Some(_)) => { + d.diffs += 1; + d.say( + out, + &format!("<{}> exists only in <{}>", b.full(rel), b.label), + )?; + } + (Some(ea), Some(eb)) => d.entry(out, a, b, rel, ea, eb)?, + (None, None) => {} + } + } + // Per-object counts were printed with each object; add a total when + // they do not already tell the whole story. + if !d.opts.quiet && d.diffs > 0 && d.diffs != d.per_object { + writeln!(out.o, "{} difference(s) found in total", d.diffs)?; + } + Ok(if d.errors > 0 { + 2 + } else if d.diffs > 0 { + 1 + } else { + 0 + }) +} + +fn kind_word(k: Kind) -> &'static str { + match k { + Kind::Group => "group", + Kind::Dataset => "dataset", + Kind::Datatype => "datatype", + Kind::Unknown => "object", + } +} + +impl Diff { + fn say(&self, out: &mut Out, msg: &str) -> std::io::Result<()> { + if self.opts.quiet { + Ok(()) + } else { + writeln!(out.o, "{msg}") + } + } + + fn error(&mut self, out: &mut Out, msg: &str) -> std::io::Result<()> { + self.errors += 1; + writeln!(out.e, "h5rs diff: {msg}") + } + + fn entry( + &mut self, + out: &mut Out, + a: &Side, + b: &Side, + rel: &str, + ea: &Entry, + eb: &Entry, + ) -> std::io::Result<()> { + let (pa, pb) = (a.full(rel), b.full(rel)); + match (ea, eb) { + (Entry::Broken(e), _) => self.error(out, &format!("<{pa}> in <{}>: {e}", a.label)), + (_, Entry::Broken(e)) => self.error(out, &format!("<{pb}> in <{}>: {e}", b.label)), + (Entry::Soft(x), Entry::Soft(y)) => { + if x != y { + self.diffs += 1; + self.say(out, &format!("soft link: <{pa}> -> {x} and <{pb}> -> {y}"))?; + } + Ok(()) + } + (Entry::External(f1, p1), Entry::External(f2, p2)) => { + if (f1, p1) != (f2, p2) { + self.diffs += 1; + self.say( + out, + &format!("external link: <{pa}> -> {f1}:{p1} and <{pb}> -> {f2}:{p2}"), + )?; + } + Ok(()) + } + (Entry::UserDefined(x), Entry::UserDefined(y)) => { + if x != y { + self.diffs += 1; + self.say(out, &format!("user-defined link: <{pa}> and <{pb}> differ"))?; + } + Ok(()) + } + (Entry::Obj(aa, ka), Entry::Obj(ab, kb)) if ka == kb => { + let (ha, hb) = match (a.h5.header(*aa), b.h5.header(*ab)) { + (Ok(x), Ok(y)) => (x, y), + (Err(e), _) | (_, Err(e)) => return self.error(out, &format!("<{pa}>: {e}")), + }; + let before = self.diffs; + let mut rows = Vec::new(); + match ka { + Kind::Dataset => self.dataset(out, a, b, &pa, &pb, &ha, &hb, &mut rows)?, + Kind::Datatype => match (a.h5.datatype(&ha), b.h5.datatype(&hb)) { + (Ok(x), Ok(y)) => { + if x != y { + self.diffs += 1; + rows.push("datatypes differ".to_string()); + } + } + (Err(e), _) | (_, Err(e)) => self.error(out, &format!("<{pa}>: {e}"))?, + }, + _ => {} + } + self.attributes(out, a, b, &pa, &ha, &hb, &mut rows)?; + let n = self.diffs - before; + if n > 0 && !self.opts.quiet { + writeln!(out.o, "{}: <{pa}> and <{pb}>", kind_word(*ka))?; + for r in &rows { + writeln!(out.o, "{r}")?; + } + writeln!(out.o, "{n} difference(s) found")?; + self.per_object += n; + } + Ok(()) + } + _ => { + self.diffs += 1; + self.say( + out, + &format!( + "Not comparable: <{pa}> is a {} and <{pb}> is a {}", + entry_word(ea), + entry_word(eb) + ), + ) + } + } + } + + #[allow(clippy::too_many_arguments)] + fn dataset( + &mut self, + out: &mut Out, + a: &Side, + b: &Side, + pa: &str, + pb: &str, + ha: &ObjectHeader, + hb: &ObjectHeader, + rows: &mut Vec, + ) -> std::io::Result<()> { + let got = (|| -> crate::h5::Result<_> { + Ok(( + a.h5.datatype(ha)?, + a.h5.resolved_dataspace(pa, ha)?, + b.h5.datatype(hb)?, + b.h5.resolved_dataspace(pb, hb)?, + )) + })(); + let (dta, dsa, dtb, dsb) = match got { + Ok(x) => x, + Err(e) => return self.error(out, &format!("<{pa}>: {e}")), + }; + if let Some(why) = not_comparable(&dta, &dsa, &dtb, &dsb) { + self.diffs += 1; + rows.push(format!("Not comparable: {why}")); + return Ok(()); + } + let raw = + a.h5.read_dataset(pa, &dta, &dsa) + .and_then(|x| Ok((x, b.h5.read_dataset(pb, &dtb, &dsb)?))); + let (ra, rb) = match raw { + Ok(x) => x, + Err(e) => return self.error(out, &format!("<{pa}>: {e}")), + }; + self.values(out, a, b, pa, (&dta, &ra), (&dtb, &rb), &dsa, rows) + } + + #[allow(clippy::too_many_arguments)] + fn values( + &mut self, + out: &mut Out, + a: &Side, + b: &Side, + what: &str, + (dta, ra): (&Datatype, &[u8]), + (dtb, rb): (&Datatype, &[u8]), + ds: &Dataspace, + rows: &mut Vec, + ) -> std::io::Result<()> { + let n = crate::h5::num_elements(ds).unwrap_or(0) as usize; + let (da, db) = (Decoder::new(a.h5), Decoder::new(b.h5)); + let dims: Vec = match ds.space_type { + DataspaceType::Simple => ds.dimensions.clone(), + _ => vec![1], + }; + let mut found = 0u64; + let mut header = false; + for i in 0..n { + let va = da.element(dta, ra, i); + let vb = db.element(dtb, rb, i); + if let (Value::Error(e), _) | (_, Value::Error(e)) = (&va, &vb) { + return self.error(out, &format!("<{what}> element {i}: {e}")); + } + if self.equal(a, b, &va, &vb) { + continue; + } + found += 1; + if self.opts.report && (found as usize) <= self.opts.count { + if !header { + header = true; + rows.push(format!( + "{:<24}{:<24}{:<24}{}", + "position", "value 1", "value 2", "difference" + )); + rows.push("-".repeat(80)); + } + let pos = format!("[ {} ]", index(i as u64, &dims)); + let ta = value::text(&va, &|_| None); + let tb = value::text(&vb, &|_| None); + let dif = match (number(&va), number(&vb)) { + (Some(x), Some(y)) => value::fmt_float((x - y).abs(), 64), + _ => String::new(), + }; + rows.push(format!("{pos:<24}{ta:<24}{tb:<24}{dif}")); + } + } + self.diffs += found; + Ok(()) + } + + #[allow(clippy::too_many_arguments)] + fn attributes( + &mut self, + out: &mut Out, + a: &Side, + b: &Side, + path: &str, + ha: &ObjectHeader, + hb: &ObjectHeader, + rows: &mut Vec, + ) -> std::io::Result<()> { + let (la, lb) = match (a.h5.attributes(ha), b.h5.attributes(hb)) { + (Ok(x), Ok(y)) => (x, y), + (Err(e), _) | (_, Err(e)) => return self.error(out, &format!("<{path}>: {e}")), + }; + for e in la.1.iter().chain(lb.1.iter()) { + self.error(out, &format!("<{path}>: attribute: {e}"))?; + } + let ma: BTreeMap<&str, &AttributeMessage> = + la.0.iter().map(|x| (x.name.as_str(), x)).collect(); + let mb: BTreeMap<&str, &AttributeMessage> = + lb.0.iter().map(|x| (x.name.as_str(), x)).collect(); + let mut names: Vec<&str> = ma.keys().chain(mb.keys()).copied().collect(); + names.sort_unstable(); + names.dedup(); + for n in names { + match (ma.get(n), mb.get(n)) { + (Some(x), Some(y)) => { + if let Some(why) = + not_comparable(&x.datatype, &x.dataspace, &y.datatype, &y.dataspace) + { + self.diffs += 1; + rows.push(format!("attribute \"{n}\": not comparable: {why}")); + continue; + } + for (m, h5) in [(x, a.h5), (y, b.h5)] { + let need = + crate::h5::byte_len(&m.dataspace, &m.datatype).unwrap_or(u64::MAX); + if need > h5.max_bytes || (m.raw_data.len() as u64) < need { + return self.error( + out, + &format!( + "<{path}> attribute \"{n}\": value is truncated or too large" + ), + ); + } + } + let before = rows.len(); + let what = format!("{path}\" attribute \"{n}"); + self.values( + out, + a, + b, + &what, + (&x.datatype, &x.raw_data), + (&y.datatype, &y.raw_data), + &x.dataspace, + rows, + )?; + if rows.len() > before { + rows.insert(before, format!("attribute \"{n}\":")); + } + } + (Some(_), None) => { + self.diffs += 1; + rows.push(format!("attribute \"{n}\" exists only in <{}>", a.label)); + } + (None, Some(_)) => { + self.diffs += 1; + rows.push(format!("attribute \"{n}\" exists only in <{}>", b.label)); + } + (None, None) => {} + } + } + Ok(()) + } + + fn equal(&self, a: &Side, b: &Side, x: &Value, y: &Value) -> bool { + if let (Some(p), Some(q)) = (number(x), number(y)) { + let exact_ints = matches!((x, y), (Value::Int(_), Value::Int(_))) + || matches!((x, y), (Value::Enum(..), Value::Enum(..))); + if exact_ints && matches!(self.opts.tol, Tol::Exact) { + return int_of(x) == int_of(y); + } + return close(p, q, self.opts.tol); + } + match (x, y) { + (Value::Str(p), Value::Str(q)) => p == q, + (Value::Bytes(p), Value::Bytes(q)) | (Value::OtherRef(p), Value::OtherRef(q)) => p == q, + (Value::Compound(p), Value::Compound(q)) => { + p.len() == q.len() + && p.iter() + .zip(q) + .all(|((_, u), (_, v))| self.equal(a, b, u, v)) + } + (Value::Array(p), Value::Array(q)) | (Value::Seq(p), Value::Seq(q)) => { + p.len() == q.len() && p.iter().zip(q).all(|(u, v)| self.equal(a, b, u, v)) + } + (Value::Ref(None), Value::Ref(None)) => true, + (Value::Ref(Some(p)), Value::Ref(Some(q))) => { + // Addresses mean nothing across files: compare the paths the + // references lead to. + let (pp, qq) = (a.rel_paths().get(p), b.rel_paths().get(q)); + pp.is_some() && pp == qq + } + _ => false, + } + } +} + +fn int_of(v: &Value) -> Option { + match v { + Value::Int(i) | Value::Enum(_, i) => Some(*i), + _ => None, + } +} + +fn number(v: &Value) -> Option { + match v { + Value::Int(i) | Value::Enum(_, i) => Some(*i as f64), + Value::Float(f, _) => Some(*f), + _ => None, + } +} + +fn close(a: f64, b: f64, tol: Tol) -> bool { + if a.is_nan() || b.is_nan() { + return a.is_nan() && b.is_nan(); + } + if a == b { + return true; + } + let d = (a - b).abs(); + match tol { + Tol::Exact => false, + Tol::Delta(t) => d <= t, + Tol::Relative(r) => a != 0.0 && d / a.abs() <= r, + } +} + +fn entry_word(e: &Entry) -> &'static str { + match e { + Entry::Obj(_, k) => kind_word(*k), + Entry::Soft(_) => "soft link", + Entry::External(..) => "external link", + Entry::UserDefined(_) => "user-defined link", + Entry::Broken(_) => "unreadable object", + } +} + +/// Why two datasets/attributes cannot be compared element by element. +fn not_comparable( + dta: &Datatype, + dsa: &Dataspace, + dtb: &Datatype, + dsb: &Dataspace, +) -> Option { + let rank = |d: &Dataspace| match d.space_type { + DataspaceType::Simple => Some(d.dimensions.clone()), + DataspaceType::Scalar => Some(Vec::new()), + DataspaceType::Null => None, + }; + let (sa, sb) = (rank(dsa), rank(dsb)); + if sa != sb { + let show = |s: &Option>| match s { + None => "null".to_string(), + Some(d) if d.is_empty() => "scalar".to_string(), + Some(d) => format!("{d:?}"), + }; + return Some(format!("shapes differ: {} and {}", show(&sa), show(&sb))); + } + if !types_comparable(dta, dtb) { + return Some(format!( + "datatypes differ: {} and {}", + crate::dtype::short(dta), + crate::dtype::short(dtb) + )); + } + None +} + +fn types_comparable(a: &Datatype, b: &Datatype) -> bool { + use crate::dtype::class; + let numeric = |t: &Datatype| matches!(class(t), "integer" | "float"); + if numeric(a) && numeric(b) { + return class(a) == class(b); + } + if class(a) != class(b) { + return false; + } + match (a, b) { + (Datatype::Compound { members: ma, .. }, Datatype::Compound { members: mb, .. }) => { + ma.len() == mb.len() + && ma + .iter() + .zip(mb) + .all(|(x, y)| x.name == y.name && types_comparable(&x.datatype, &y.datatype)) + } + ( + Datatype::Array { + base_type: x, + dimensions: dx, + }, + Datatype::Array { + base_type: y, + dimensions: dy, + }, + ) => dx == dy && types_comparable(x, y), + ( + Datatype::VariableLength { base_type: x, .. }, + Datatype::VariableLength { base_type: y, .. }, + ) => class(a) == "string" || types_comparable(x, y), + ( + Datatype::Enumeration { base_type: x, .. }, + Datatype::Enumeration { base_type: y, .. }, + ) => types_comparable(x, y), + _ => true, + } +} + +fn index(mut i: u64, dims: &[u64]) -> String { + let mut idx = vec![0u64; dims.len()]; + for (k, &d) in dims.iter().enumerate().rev() { + let d = d.max(1); + idx[k] = i % d; + i /= d; + } + idx.iter() + .map(|x| x.to_string()) + .collect::>() + .join(" ") +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn tolerances() { + assert!(close(1.0, 1.0, Tol::Exact)); + assert!(!close(1.0, 1.001, Tol::Exact)); + assert!(close(1.0, 1.001, Tol::Delta(0.01))); + assert!(!close(1.0, 1.1, Tol::Delta(0.01))); + assert!(close(100.0, 101.0, Tol::Relative(0.02))); + assert!(!close(0.0, 1e-9, Tol::Relative(0.5))); + assert!(close(f64::NAN, f64::NAN, Tol::Exact)); + assert!(!close(f64::NAN, 1.0, Tol::Delta(1e9))); + } + + #[test] + fn positions() { + assert_eq!(index(5, &[3, 4]), "1 1"); + assert_eq!(index(0, &[1]), "0"); + } +} diff --git a/crates/clawhdf5-tools/src/dtype.rs b/crates/clawhdf5-tools/src/dtype.rs new file mode 100644 index 0000000..c42c4eb --- /dev/null +++ b/crates/clawhdf5-tools/src/dtype.rs @@ -0,0 +1,529 @@ +//! Names for datatypes: a short one for listings, h5ls's long form, h5dump's +//! DDL and the HDF Group's hdf5-json type objects. + +use clawhdf5_format::datatype::{ + CharacterSet, Datatype, DatatypeByteOrder, ReferenceType, StringPadding, +}; +use serde_json::{Value as J, json}; + +fn be(o: &DatatypeByteOrder) -> bool { + matches!(o, DatatypeByteOrder::BigEndian) +} + +fn order_suffix(o: &DatatypeByteOrder) -> &'static str { + match o { + DatatypeByteOrder::LittleEndian => "LE", + DatatypeByteOrder::BigEndian => "BE", + DatatypeByteOrder::Vax => "VAX", + } +} + +fn order_word(o: &DatatypeByteOrder) -> &'static str { + match o { + DatatypeByteOrder::LittleEndian => "little-endian", + DatatypeByteOrder::BigEndian => "big-endian", + DatatypeByteOrder::Vax => "VAX-order", + } +} + +/// True when a float is laid out exactly as IEEE 754 binary16/32/64. +pub fn is_ieee(dt: &Datatype) -> bool { + let Datatype::FloatingPoint { + size, + bit_offset, + bit_precision, + exponent_location, + exponent_size, + mantissa_location, + mantissa_size, + exponent_bias, + .. + } = dt + else { + return false; + }; + let std = match size { + 2 => (16, 10, 5, 10, 15), + 4 => (32, 23, 8, 23, 127), + 8 => (64, 52, 11, 52, 1023), + _ => return false, + }; + *bit_offset == 0 + && *mantissa_location == 0 + && ( + *bit_precision, + *exponent_location, + *exponent_size, + *mantissa_size, + *exponent_bias, + ) == std +} + +/// Short name used by `ls`: `int32`, `float64-be`, `string[3]`, ... +pub fn short(dt: &Datatype) -> String { + match dt { + Datatype::FixedPoint { + size, + signed, + byte_order, + .. + } => format!( + "{}int{}{}", + if *signed { "" } else { "u" }, + u64::from(*size) * 8, + if be(byte_order) { "-be" } else { "" } + ), + Datatype::FloatingPoint { + size, byte_order, .. + } => format!( + "float{}{}", + u64::from(*size) * 8, + if be(byte_order) { "-be" } else { "" } + ), + Datatype::Time { size, .. } => format!("time{}", u64::from(*size) * 8), + Datatype::String { size, .. } => format!("string[{size}]"), + Datatype::BitField { size, .. } => format!("bitfield{}", u64::from(*size) * 8), + Datatype::Opaque { size, .. } => format!("opaque[{size}]"), + Datatype::Compound { members, .. } => format!( + "compound{{{}}}", + members + .iter() + .map(|m| format!("{}: {}", m.name, short(&m.datatype))) + .collect::>() + .join(", ") + ), + Datatype::Reference { ref_type, .. } => match ref_type { + ReferenceType::Object | ReferenceType::Object2 => "object-reference".into(), + ReferenceType::DatasetRegion | ReferenceType::DatasetRegion2 => { + "region-reference".into() + } + ReferenceType::Attribute => "attribute-reference".into(), + }, + Datatype::Enumeration { base_type, .. } => format!("enum<{}>", short(base_type)), + Datatype::VariableLength { + is_string: true, .. + } => "vlen-string".into(), + Datatype::VariableLength { base_type, .. } => format!("vlen<{}>", short(base_type)), + Datatype::Array { + base_type, + dimensions, + } => format!( + "array[{}]<{}>", + dimensions + .iter() + .map(|d| d.to_string()) + .collect::>() + .join(","), + short(base_type) + ), + } +} + +fn pad_word(p: &StringPadding) -> &'static str { + match p { + StringPadding::NullTerminate => "null-terminated", + StringPadding::NullPad => "null-padded", + StringPadding::SpacePad => "space-padded", + } +} + +fn cset_word(c: &CharacterSet) -> &'static str { + match c { + CharacterSet::Ascii => "ASCII", + CharacterSet::Utf8 => "UTF-8", + } +} + +/// h5ls -v style description. +pub fn long(dt: &Datatype) -> String { + match dt { + Datatype::FixedPoint { + size, + signed, + byte_order, + .. + } => format!( + "{}-bit {} {}integer", + u64::from(*size) * 8, + order_word(byte_order), + if *signed { "" } else { "unsigned " } + ), + Datatype::FloatingPoint { + size, byte_order, .. + } => { + if is_ieee(dt) { + format!( + "IEEE {}-bit {} float", + u64::from(*size) * 8, + order_word(byte_order) + ) + } else { + format!( + "{}-bit {} non-IEEE float", + u64::from(*size) * 8, + order_word(byte_order) + ) + } + } + Datatype::Time { size, .. } => format!("{}-bit time", u64::from(*size) * 8), + Datatype::String { + size, + padding, + charset, + } => format!( + "{size}-byte {} {} string", + pad_word(padding), + cset_word(charset) + ), + Datatype::BitField { + size, byte_order, .. + } => format!( + "{}-bit {} bitfield", + u64::from(*size) * 8, + order_word(byte_order) + ), + Datatype::Opaque { size, tag } => { + let tag = String::from_utf8_lossy(tag); + let tag = tag.trim_end_matches('\0'); + if tag.is_empty() { + format!("{size}-byte opaque type") + } else { + format!("{size}-byte opaque type (tag \"{tag}\")") + } + } + Datatype::Compound { size, members } => { + let mut s = String::from("struct {"); + for m in members { + s.push_str(&format!( + "\n \"{}\" +{} {}", + m.name, + m.byte_offset, + long(&m.datatype) + )); + } + s.push_str(&format!("\n }} {size} bytes")); + s + } + Datatype::Reference { ref_type, .. } => match ref_type { + ReferenceType::Object | ReferenceType::Object2 => "object reference".into(), + ReferenceType::DatasetRegion | ReferenceType::DatasetRegion2 => { + "dataset region reference".into() + } + ReferenceType::Attribute => "attribute reference".into(), + }, + Datatype::Enumeration { + base_type, members, .. + } => { + let vals: Vec = members + .iter() + .map(|m| format!("{} = {}", m.name, enum_value_text(base_type, &m.value))) + .collect(); + format!("enum {} {{{}}}", long(base_type), vals.join(", ")) + } + Datatype::VariableLength { + is_string: true, + padding, + charset, + .. + } => format!( + "variable-length {} {} string", + padding.as_ref().map(pad_word).unwrap_or("null-terminated"), + charset.as_ref().map(cset_word).unwrap_or("ASCII") + ), + Datatype::VariableLength { base_type, .. } => { + format!("variable length of {}", long(base_type)) + } + Datatype::Array { + base_type, + dimensions, + } => format!( + "[{}] {}", + dimensions + .iter() + .map(|d| d.to_string()) + .collect::>() + .join(","), + long(base_type) + ), + } +} + +/// The integer an enum member's value bytes hold, as text. +pub fn enum_value_text(base: &Datatype, bytes: &[u8]) -> String { + match crate::value::decode_int(base, bytes) { + Some(v) => v.to_string(), + None => crate::value::hex(bytes), + } +} + +/// h5dump's predefined name for an atomic type, if it has one. +fn atomic_ddl(dt: &Datatype) -> Option { + match dt { + Datatype::FixedPoint { + size, + signed, + byte_order, + .. + } => Some(format!( + "H5T_STD_{}{}{}", + if *signed { "I" } else { "U" }, + u64::from(*size) * 8, + order_suffix(byte_order) + )), + Datatype::FloatingPoint { + size, byte_order, .. + } if is_ieee(dt) => Some(format!( + "H5T_IEEE_F{}{}", + u64::from(*size) * 8, + order_suffix(byte_order) + )), + Datatype::BitField { + size, byte_order, .. + } => Some(format!( + "H5T_STD_B{}{}", + u64::from(*size) * 8, + order_suffix(byte_order) + )), + Datatype::Time { .. } => Some("H5T_TIME".into()), + _ => None, + } +} + +/// h5dump DDL for a datatype; `ind` is the indentation of the line the type +/// starts on (continuation lines are indented relative to it). +pub fn ddl(dt: &Datatype, ind: usize) -> String { + let pad = " ".repeat(ind + 3); + let end = " ".repeat(ind); + if let Some(s) = atomic_ddl(dt) { + return s; + } + match dt { + Datatype::FloatingPoint { + size, + byte_order, + bit_offset, + bit_precision, + exponent_location, + exponent_size, + mantissa_location, + mantissa_size, + exponent_bias, + } => format!( + "H5T_FLOAT {{\n{pad}SIZE {size};\n{pad}ORDER H5T_ORDER_{};\n{pad}OFFSET {bit_offset};\n\ + {pad}PRECISION {bit_precision};\n{pad}EXPONENT {exponent_location} {exponent_size} \ + BIAS {exponent_bias};\n{pad}MANTISSA {mantissa_location} {mantissa_size};\n{end}}}", + order_suffix(byte_order) + ), + Datatype::String { + size, + padding, + charset, + } => string_ddl(&size.to_string(), padding, charset, &pad, &end), + Datatype::VariableLength { + is_string: true, + padding, + charset, + .. + } => string_ddl( + "H5T_VARIABLE", + padding.as_ref().unwrap_or(&StringPadding::NullTerminate), + charset.as_ref().unwrap_or(&CharacterSet::Ascii), + &pad, + &end, + ), + Datatype::VariableLength { base_type, .. } => { + format!("H5T_VLEN {{ {} }}", ddl(base_type, ind)) + } + Datatype::Opaque { size, tag } => { + let tag = String::from_utf8_lossy(tag); + format!( + "H5T_OPAQUE {{\n{pad}OPAQUE_SIZE {size};\n{pad}OPAQUE_TAG \"{}\";\n{end}}}", + tag.trim_end_matches('\0') + ) + } + Datatype::Compound { members, .. } => { + let mut s = String::from("H5T_COMPOUND {\n"); + for m in members { + s.push_str(&format!( + "{pad}{} \"{}\";\n", + ddl(&m.datatype, ind + 3), + m.name + )); + } + s.push_str(&end); + s.push('}'); + s + } + Datatype::Reference { ref_type, .. } => match ref_type { + ReferenceType::Object => "H5T_REFERENCE { H5T_STD_REF_OBJECT }".into(), + ReferenceType::DatasetRegion => "H5T_REFERENCE { H5T_STD_REF_DSETREG }".into(), + _ => "H5T_REFERENCE { H5T_STD_REF }".into(), + }, + Datatype::Enumeration { + base_type, members, .. + } => { + let mut s = format!("H5T_ENUM {{\n{pad}{};\n", ddl(base_type, ind + 3)); + for m in members { + let name = format!("\"{}\"", m.name); + s.push_str(&format!( + "{pad}{name:<18} {};\n", + enum_value_text(base_type, &m.value) + )); + } + s.push_str(&end); + s.push('}'); + s + } + Datatype::Array { + base_type, + dimensions, + } => format!( + "H5T_ARRAY {{ {} {} }}", + dimensions + .iter() + .map(|d| format!("[{d}]")) + .collect::(), + ddl(base_type, ind) + ), + // Atomic types were handled above. + _ => short(dt), + } +} + +fn string_ddl( + size: &str, + padding: &StringPadding, + charset: &CharacterSet, + pad: &str, + end: &str, +) -> String { + let p = match padding { + StringPadding::NullTerminate => "H5T_STR_NULLTERM", + StringPadding::NullPad => "H5T_STR_NULLPAD", + StringPadding::SpacePad => "H5T_STR_SPACEPAD", + }; + let c = match charset { + CharacterSet::Ascii => "H5T_CSET_ASCII", + CharacterSet::Utf8 => "H5T_CSET_UTF8", + }; + format!( + "H5T_STRING {{\n{pad}STRSIZE {size};\n{pad}STRPAD {p};\n{pad}CSET {c};\n{pad}CTYPE H5T_C_S1;\n{end}}}" + ) +} + +/// hdf5-json type object. +pub fn json(dt: &Datatype) -> J { + match dt { + Datatype::FixedPoint { .. } => { + json!({"class": "H5T_INTEGER", "base": atomic_ddl(dt)}) + } + Datatype::FloatingPoint { .. } if is_ieee(dt) => { + json!({"class": "H5T_FLOAT", "base": atomic_ddl(dt)}) + } + Datatype::FloatingPoint { + size, byte_order, .. + } => json!({ + "class": "H5T_FLOAT", + "size": size, + "order": format!("H5T_ORDER_{}", order_suffix(byte_order)), + }), + Datatype::BitField { .. } => json!({"class": "H5T_BITFIELD", "base": atomic_ddl(dt)}), + Datatype::Time { size, .. } => json!({"class": "H5T_TIME", "size": size}), + Datatype::String { + size, + padding, + charset, + } => json!({ + "class": "H5T_STRING", + "charSet": cset_json(charset), + "strPad": pad_json(padding), + "length": size, + }), + Datatype::VariableLength { + is_string: true, + padding, + charset, + .. + } => json!({ + "class": "H5T_STRING", + "charSet": cset_json(charset.as_ref().unwrap_or(&CharacterSet::Ascii)), + "strPad": pad_json(padding.as_ref().unwrap_or(&StringPadding::NullTerminate)), + "length": "H5T_VARIABLE", + }), + Datatype::VariableLength { base_type, .. } => { + json!({"class": "H5T_VLEN", "base": json(base_type)}) + } + Datatype::Opaque { size, tag } => json!({ + "class": "H5T_OPAQUE", + "size": size, + "tag": String::from_utf8_lossy(tag).trim_end_matches('\0'), + }), + Datatype::Compound { members, .. } => json!({ + "class": "H5T_COMPOUND", + "fields": members + .iter() + .map(|m| json!({"name": m.name, "type": json(&m.datatype)})) + .collect::>(), + }), + Datatype::Reference { ref_type, .. } => json!({ + "class": "H5T_REFERENCE", + "base": match ref_type { + ReferenceType::Object => "H5T_STD_REF_OBJ", + ReferenceType::DatasetRegion => "H5T_STD_REF_DSETREG", + _ => "H5T_STD_REF", + }, + }), + Datatype::Enumeration { + base_type, members, .. + } => { + let mut mapping = serde_json::Map::new(); + for m in members { + let v = crate::value::decode_int(base_type, &m.value) + .and_then(|v| i64::try_from(v).ok()) + .map(J::from) + .unwrap_or_else(|| J::from(crate::value::hex(&m.value))); + mapping.insert(m.name.clone(), v); + } + json!({"class": "H5T_ENUM", "base": json(base_type), "mapping": mapping}) + } + Datatype::Array { + base_type, + dimensions, + } => json!({"class": "H5T_ARRAY", "base": json(base_type), "dims": dimensions}), + } +} + +fn cset_json(c: &CharacterSet) -> &'static str { + match c { + CharacterSet::Ascii => "H5T_CSET_ASCII", + CharacterSet::Utf8 => "H5T_CSET_UTF8", + } +} + +fn pad_json(p: &StringPadding) -> &'static str { + match p { + StringPadding::NullTerminate => "H5T_STR_NULLTERM", + StringPadding::NullPad => "H5T_STR_NULLPAD", + StringPadding::SpacePad => "H5T_STR_SPACEPAD", + } +} + +/// Class name used to decide whether two datatypes can be compared. +pub fn class(dt: &Datatype) -> &'static str { + match dt { + Datatype::FixedPoint { .. } => "integer", + Datatype::FloatingPoint { .. } => "float", + Datatype::Time { .. } => "time", + Datatype::String { .. } => "string", + Datatype::BitField { .. } => "bitfield", + Datatype::Opaque { .. } => "opaque", + Datatype::Compound { .. } => "compound", + Datatype::Reference { .. } => "reference", + Datatype::Enumeration { .. } => "enum", + Datatype::VariableLength { + is_string: true, .. + } => "string", + Datatype::VariableLength { .. } => "vlen", + Datatype::Array { .. } => "array", + } +} diff --git a/crates/clawhdf5-tools/src/dump.rs b/crates/clawhdf5-tools/src/dump.rs new file mode 100644 index 0000000..e743d36 --- /dev/null +++ b/crates/clawhdf5-tools/src/dump.rs @@ -0,0 +1,944 @@ +//! `h5rs dump`: the whole file (or one dataset) as h5dump-style DDL text or +//! as hdf5-json. + +use std::cell::OnceCell; +use std::collections::HashMap; + +use clawhdf5_format::attribute::AttributeMessage; +use clawhdf5_format::data_layout::DataLayout; +use clawhdf5_format::dataspace::{Dataspace, DataspaceType}; +use clawhdf5_format::datatype::{Datatype, StringPadding}; +use clawhdf5_format::object_header::ObjectHeader; +use serde_json::{Map, Value as J, json}; + +use crate::cli::{Args, Out}; +use crate::h5::{Error, H5, Kind, Link, LinkKind}; +use crate::info::{self, DsInfo}; +use crate::value::{self, Decoder, Value}; + +pub const USAGE: &str = "\ +usage: h5rs dump [--json] [-A] [-p] [-d PATH] [--max-bytes N] FILE + +Print FILE's groups, datasets, named datatypes, links and attributes, with +their values, as h5dump-style DDL text (default) or as JSON. + + --json hdf5-json layout (see the crate README for the schema) + -A, --header no dataset values (attribute values are still printed, + as with h5dump -A) + -p, --properties also print each dataset's storage layout and filters + -d, --dataset P dump only the dataset at path P + --max-bytes N largest dataset or attribute decoded (default 1 GiB); + a larger one is reported instead of read + +Exit status: 0 dumped, 1 something could not be read (reported on stderr +and marked in the output), 2 error."; + +struct Opts { + json: bool, + header_only: bool, + props: bool, +} + +struct Dump<'a> { + h5: &'a H5, + opts: Opts, + problems: usize, + paths: OnceCell>, +} + +pub fn run(args: &mut Args, out: &mut Out) -> std::io::Result { + let mut opts = Opts { + json: false, + header_only: false, + props: false, + }; + let mut dataset = None; + let mut max_bytes = None; + let mut file = None; + while let Some(a) = args.next() { + match a.as_str() { + "--json" | "-j" => opts.json = true, + "-A" | "--header" => opts.header_only = true, + "-p" | "--properties" => opts.props = true, + "-d" | "--dataset" => match args.value() { + Some(p) => dataset = Some(p), + None => return args.usage_error(out, "-d needs a path", USAGE), + }, + "--max-bytes" => match args.number() { + Some(n) => max_bytes = Some(n), + None => return args.usage_error(out, "--max-bytes needs a number", USAGE), + }, + "-h" | "--help" => { + writeln!(out.o, "{USAGE}")?; + return Ok(0); + } + s if s.starts_with('-') && s.len() > 1 => { + return args.usage_error(out, &format!("unknown option {a}"), USAGE); + } + _ if file.is_none() => file = Some(a), + _ => return args.usage_error(out, &format!("unexpected argument {a}"), USAGE), + } + } + let Some(file) = file else { + return args.usage_error(out, "missing FILE", USAGE); + }; + let mut h5 = match H5::open(std::path::Path::new(&file)) { + Ok(h) => h, + Err(e) => { + writeln!(out.e, "h5rs dump: {e}")?; + return Ok(2); + } + }; + if let Some(m) = max_bytes { + h5.max_bytes = m; + } + let mut d = Dump { + h5: &h5, + opts, + problems: 0, + paths: OnceCell::new(), + }; + let fname = std::path::Path::new(&file) + .file_name() + .map(|s| s.to_string_lossy().into_owned()) + .unwrap_or(file.clone()); + let code = if d.opts.json { + d.json(out, dataset.as_deref())? + } else { + d.ddl(out, &fname, dataset.as_deref())? + }; + if code != 0 { + return Ok(code); + } + Ok(if d.problems > 0 { 1 } else { 0 }) +} + +/// A full path for `name` inside the group at `base`. +fn join(base: &str, name: &str) -> String { + if base == "/" { + format!("/{name}") + } else { + format!("{base}/{name}") + } +} + +fn quote(s: &str) -> String { + s.replace('\\', "\\\\").replace('"', "\\\"") +} + +impl Dump<'_> { + /// Object address -> first path, for printing references. + fn paths(&self) -> &HashMap { + self.paths.get_or_init(|| { + let mut m = HashMap::new(); + let _ = self.h5.walk(|it| { + if let (Some(a), None) = (it.addr, it.first_path) { + m.entry(a).or_insert_with(|| it.path.to_string()); + } + }); + m + }) + } + + fn problem(&mut self, out: &mut Out, what: &str, e: &Error) -> std::io::Result<()> { + self.problems += 1; + writeln!(out.e, "h5rs dump: {what}: {e}") + } + + // ------------------------------------------------------------------ + // DDL + // ------------------------------------------------------------------ + + fn ddl(&mut self, out: &mut Out, fname: &str, only: Option<&str>) -> std::io::Result { + writeln!(out.o, "HDF5 \"{}\" {{", quote(fname))?; + if let Some(p) = only { + let h = match self.h5.resolve(p).and_then(|a| self.h5.header(a)) { + Ok(h) => h, + Err(e) => { + writeln!(out.o, "}}")?; + writeln!(out.e, "h5rs dump: {p}: {e}")?; + return Ok(2); + } + }; + if Kind::of(&h) != Kind::Dataset { + writeln!(out.o, "}}")?; + writeln!(out.e, "h5rs dump: {p}: not a dataset")?; + return Ok(2); + } + let full = if p.starts_with('/') { + p.to_string() + } else { + format!("/{p}") + }; + self.ddl_dataset(out, &full, &full, &h, 0)?; + } else { + let root = self.h5.root(); + let mut seen = HashMap::new(); + match self.h5.header(root) { + Ok(h) => self.ddl_group(out, "/", "/", root, &h, 0, &mut seen)?, + Err(e) => { + writeln!(out.o, "GROUP \"/\" {{\n}}")?; + self.problem(out, "/", &e)?; + } + } + } + writeln!(out.o, "}}")?; + Ok(0) + } + + #[allow(clippy::too_many_arguments)] + fn ddl_group( + &mut self, + out: &mut Out, + name: &str, + path: &str, + addr: u64, + h: &ObjectHeader, + ind: usize, + seen: &mut HashMap, + ) -> std::io::Result<()> { + let pad = " ".repeat(ind); + seen.insert(addr, path.to_string()); + writeln!(out.o, "{pad}GROUP \"{}\" {{", quote(name))?; + self.ddl_attributes(out, path, h, ind + 3)?; + let links = match self.h5.links(h) { + Ok(l) => l, + Err(e) => { + self.problem(out, path, &e)?; + Vec::new() + } + }; + // The DDL recursion follows the group nesting; bound it (the walk used + // by the other commands is iterative). + if seen.len() > crate::h5::MAX_OBJECTS || ind > 3 * MAX_DDL_DEPTH { + let e = Error::new("too many objects, or groups nested too deeply; stopped"); + self.problem(out, path, &e)?; + return writeln!(out.o, "{pad}}}"); + } + for l in &links { + self.ddl_link(out, path, l, ind + 3, seen)?; + } + writeln!(out.o, "{pad}}}") + } + + fn ddl_link( + &mut self, + out: &mut Out, + base: &str, + l: &Link, + ind: usize, + seen: &mut HashMap, + ) -> std::io::Result<()> { + let pad = " ".repeat(ind); + let inner = " ".repeat(ind + 3); + let name = quote(&l.name); + let path = join(base, &l.name); + match &l.kind { + LinkKind::Soft(t) => writeln!( + out.o, + "{pad}SOFTLINK \"{name}\" {{\n{inner}LINKTARGET \"{}\"\n{pad}}}", + quote(t) + ), + LinkKind::External { file, path: p } => writeln!( + out.o, + "{pad}EXTERNAL_LINK \"{name}\" {{\n{inner}TARGETFILE \"{}\"\n{inner}TARGETPATH \"{}\"\n{pad}}}", + quote(file), + quote(p) + ), + LinkKind::UserDefined(t) => writeln!( + out.o, + "{pad}USERDEFINED_LINK \"{name}\" {{\n{inner}LINKCLASS {t}\n{pad}}}" + ), + LinkKind::Hard(a) => { + let h = match self.h5.header(*a) { + Ok(h) => h, + Err(e) => { + self.problem(out, &path, &e)?; + return writeln!(out.o, "{pad}UNKNOWN_OBJECT \"{name}\" {{\n{pad}}}"); + } + }; + let kind = Kind::of(&h); + if let Some(first) = seen.get(a) { + let word = match kind { + Kind::Group => "GROUP", + Kind::Dataset => "DATASET", + Kind::Datatype => "DATATYPE", + Kind::Unknown => "OBJECT", + }; + return writeln!( + out.o, + "{pad}{word} \"{name}\" {{\n{inner}HARDLINK \"{}\"\n{pad}}}", + quote(first) + ); + } + match kind { + Kind::Group => self.ddl_group(out, &l.name, &path, *a, &h, ind, seen), + Kind::Dataset => { + seen.insert(*a, path.clone()); + self.ddl_dataset(out, &l.name, &path, &h, ind) + } + Kind::Datatype => { + seen.insert(*a, path.clone()); + match self.h5.datatype(&h) { + Ok(dt) => writeln!( + out.o, + "{pad}DATATYPE \"{name}\" {};", + crate::dtype::ddl(&dt, ind) + ), + Err(e) => { + self.problem(out, &path, &e)?; + writeln!(out.o, "{pad}DATATYPE \"{name}\" ?;") + } + } + } + Kind::Unknown => { + seen.insert(*a, path.clone()); + writeln!(out.o, "{pad}UNKNOWN_OBJECT \"{name}\" {{\n{pad}}}") + } + } + } + } + } + + fn ddl_dataset( + &mut self, + out: &mut Out, + name: &str, + path: &str, + h: &ObjectHeader, + ind: usize, + ) -> std::io::Result<()> { + let pad = " ".repeat(ind); + let inner = " ".repeat(ind + 3); + writeln!(out.o, "{pad}DATASET \"{}\" {{", quote(name))?; + let info = DsInfo::read(self.h5, path, h); + match &info.dt { + Ok(dt) => writeln!(out.o, "{inner}DATATYPE {}", crate::dtype::ddl(dt, ind + 3))?, + Err(e) => self.problem(out, path, e)?, + } + match &info.ds { + Ok(ds) => writeln!(out.o, "{inner}DATASPACE {}", info::dataspace_ddl(ds))?, + Err(e) => self.problem(out, path, e)?, + } + if self.opts.props { + self.ddl_properties(out, path, &info, ind + 3)?; + } + if !self.opts.header_only + && let (Ok(dt), Ok(ds)) = (&info.dt, &info.ds) + { + match self.h5.read_dataset(path, dt, ds) { + Ok(raw) => self.ddl_data(out, dt, ds, &raw, ind + 3)?, + Err(e) => { + writeln!(out.o, "{inner}DATA {{\n{inner}\n{inner}}}")?; + self.problem(out, path, &e)?; + } + } + } + self.ddl_attributes(out, path, h, ind + 3)?; + writeln!(out.o, "{pad}}}") + } + + fn ddl_properties( + &mut self, + out: &mut Out, + path: &str, + info: &DsInfo, + ind: usize, + ) -> std::io::Result<()> { + let pad = " ".repeat(ind); + let inner = " ".repeat(ind + 3); + let Ok(layout) = &info.layout else { + if let Err(e) = &info.layout { + self.problem(out, path, e)?; + } + return Ok(()); + }; + writeln!(out.o, "{pad}STORAGE_LAYOUT {{")?; + match layout { + DataLayout::Chunked { + chunk_dimensions, .. + } => { + let rank = chunk_dimensions.len().saturating_sub(1); + writeln!( + out.o, + "{inner}CHUNKED ( {} )", + chunk_dimensions[..rank] + .iter() + .map(|d| d.to_string()) + .collect::>() + .join(", ") + )?; + writeln!( + out.o, + "{inner}INDEX {}", + info::chunk_index_name(layout).to_uppercase() + )?; + } + DataLayout::Contiguous { address, size } => { + writeln!(out.o, "{inner}CONTIGUOUS")?; + match address { + Some(a) => writeln!(out.o, "{inner}OFFSET {a}")?, + None => writeln!(out.o, "{inner}NOT ALLOCATED")?, + } + writeln!(out.o, "{inner}SIZE {size}")?; + } + DataLayout::Compact { data } => { + writeln!(out.o, "{inner}COMPACT")?; + writeln!(out.o, "{inner}SIZE {}", data.len())?; + } + DataLayout::Virtual { .. } => writeln!(out.o, "{inner}VIRTUAL")?, + } + if let Ok(a) = info::allocated_bytes(self.h5, info) + && matches!(layout, DataLayout::Chunked { .. }) + { + writeln!(out.o, "{inner}SIZE {a}")?; + } + writeln!(out.o, "{pad}}}")?; + writeln!(out.o, "{pad}FILTERS {{")?; + match &info.filters { + Ok(Some(p)) if !p.filters.is_empty() => { + for f in &p.filters { + writeln!( + out.o, + "{inner}{} {{ ID {}; PARAMS {{ {} }} }}", + info::filter_name(f).to_uppercase(), + f.filter_id, + f.client_data + .iter() + .map(|c| c.to_string()) + .collect::>() + .join(" ") + )?; + } + } + Ok(_) => writeln!(out.o, "{inner}NONE")?, + Err(e) => self.problem(out, path, e)?, + } + writeln!(out.o, "{pad}}}") + } + + fn ddl_attributes( + &mut self, + out: &mut Out, + path: &str, + h: &ObjectHeader, + ind: usize, + ) -> std::io::Result<()> { + let (attrs, errs) = match self.h5.attributes(h) { + Ok(x) => x, + Err(e) => return self.problem(out, path, &e), + }; + for e in errs { + self.problem(out, path, &Error::new(format!("attribute: {e}")))?; + } + for a in &attrs { + self.ddl_attribute(out, path, a, ind)?; + } + Ok(()) + } + + fn ddl_attribute( + &mut self, + out: &mut Out, + path: &str, + a: &AttributeMessage, + ind: usize, + ) -> std::io::Result<()> { + let pad = " ".repeat(ind); + let inner = " ".repeat(ind + 3); + writeln!(out.o, "{pad}ATTRIBUTE \"{}\" {{", quote(&a.name))?; + writeln!( + out.o, + "{inner}DATATYPE {}", + crate::dtype::ddl(&a.datatype, ind + 3) + )?; + writeln!( + out.o, + "{inner}DATASPACE {}", + info::dataspace_ddl(&a.dataspace) + )?; + { + match self.checked_attr(a) { + Ok(()) => self.ddl_data(out, &a.datatype, &a.dataspace, &a.raw_data, ind + 3)?, + Err(e) => { + writeln!(out.o, "{inner}DATA {{\n{inner}\n{inner}}}")?; + self.problem(out, &format!("{path} attribute {}", a.name), &e)?; + } + } + } + writeln!(out.o, "{pad}}}") + } + + /// An attribute's value holds as many bytes as its dataspace says. + fn checked_attr(&self, a: &AttributeMessage) -> crate::h5::Result<()> { + let need = crate::h5::byte_len(&a.dataspace, &a.datatype)?; + if need > self.h5.max_bytes { + return Err(Error::new(format!( + "attribute is {need} bytes, over the {} byte limit (--max-bytes)", + self.h5.max_bytes + ))); + } + if (a.raw_data.len() as u64) < need { + return Err(Error::new(format!( + "attribute holds {} bytes, its dataspace needs {need}", + a.raw_data.len() + ))); + } + Ok(()) + } + + /// An h5dump DATA block: elements row by row, each line starting with + /// the index of its first element, wrapped where h5dump wraps (78 columns). + fn ddl_data( + &mut self, + out: &mut Out, + dt: &Datatype, + ds: &Dataspace, + raw: &[u8], + ind: usize, + ) -> std::io::Result<()> { + let pad = " ".repeat(ind); + writeln!(out.o, "{pad}DATA {{")?; + let dims: Vec = match ds.space_type { + DataspaceType::Null => { + return writeln!(out.o, "{pad}}}"); + } + DataspaceType::Scalar => vec![1], + DataspaceType::Simple => ds.dimensions.clone(), + }; + let n = crate::h5::num_elements(ds).unwrap_or(0) as usize; + let dec = Decoder::new(self.h5); + let last = dims.last().copied().unwrap_or(1).max(1) as usize; + let paths = |a: u64| self.paths().get(&a).cloned(); + let mut line = String::new(); + let mut errors = 0usize; + for i in 0..n { + let v = dec.element(dt, raw, i); + if matches!(v, Value::Error(_)) { + errors += 1; + } + let comma = if i + 1 < n { "," } else { "" }; + if let Value::Compound(members) = &v { + // h5dump prints each compound element as a block, one + // member per line. + if !line.is_empty() { + writeln!(out.o, "{line}")?; + line.clear(); + } + writeln!(out.o, "{pad}({}): {{", index_text(i as u64, &dims))?; + for (k, (_, m)) in members.iter().enumerate() { + let sep = if k + 1 < members.len() { "," } else { "" }; + writeln!(out.o, "{pad} {}{sep}", value::text(m, &paths))?; + } + writeln!(out.o, "{pad} }}{comma}")?; + continue; + } + let t = match dt { + // h5dump shows a null-padded string's padding. + Datatype::String { + padding: StringPadding::NullPad, + size, + .. + } if !matches!(v, Value::Error(_)) => { + let sz = *size as usize; + let b = raw.get(i * sz..(i + 1) * sz).unwrap_or_default(); + format!("{}{comma}", value::quote_bytes(b)) + } + _ => format!("{}{comma}", value::text(&v, &paths)), + }; + let row_start = i % last == 0; + if row_start || line.len() + 1 + t.len() > 77 { + if !line.is_empty() { + writeln!(out.o, "{line}")?; + } + line = format!("{pad}({}): {t}", index_text(i as u64, &dims)); + } else { + line.push(' '); + line.push_str(&t); + } + } + if !line.is_empty() { + writeln!(out.o, "{line}")?; + } + if errors > 0 { + self.problems += 1; + writeln!(out.e, "h5rs dump: {errors} element(s) could not be decoded")?; + } + writeln!(out.o, "{pad}}}") + } + + // ------------------------------------------------------------------ + // JSON + // ------------------------------------------------------------------ + + fn json(&mut self, out: &mut Out, only: Option<&str>) -> std::io::Result { + let mut groups = Map::new(); + let mut datasets = Map::new(); + let mut datatypes = Map::new(); + let mut doc = Map::new(); + doc.insert("apiVersion".into(), json!("1.1.1")); + if let Some(p) = only { + let full = if p.starts_with('/') { + p.to_string() + } else { + format!("/{p}") + }; + let r = self + .h5 + .resolve(&full) + .and_then(|a| Ok((a, self.h5.header(a)?))); + match r { + Ok((a, h)) if Kind::of(&h) == Kind::Dataset => { + let obj = self.json_dataset(out, &full, &h)?; + datasets.insert(obj_id(Kind::Dataset, a), obj); + } + Ok(_) => { + writeln!(out.e, "h5rs dump: {p}: not a dataset")?; + return Ok(2); + } + Err(e) => { + writeln!(out.e, "h5rs dump: {p}: {e}")?; + return Ok(2); + } + } + } else { + doc.insert("root".into(), json!(obj_id(Kind::Group, self.h5.root()))); + let (items, walk) = self.h5.walk_collect(); + if let Err(e) = walk { + self.problem(out, "/", &e)?; + } + // Aliases: every path an object is reachable by. + let mut aliases: HashMap> = HashMap::new(); + for it in &items { + if let Some(a) = it.addr { + aliases.entry(a).or_default().push(it.path.clone()); + } + } + for it in items { + let path = it.path; + let (Some(addr), None, Some(header)) = (it.addr, it.first_path, it.header) else { + continue; + }; + let h = match header { + Ok(h) => h, + Err(e) => { + self.problem(out, &path, &e)?; + continue; + } + }; + let kind = Kind::of(&h); + let alias = aliases.remove(&addr).unwrap_or_default(); + let mut obj = match kind { + Kind::Dataset => self.json_dataset(out, &path, &h)?, + Kind::Datatype => { + let mut m = Map::new(); + match self.h5.datatype(&h) { + Ok(dt) => { + m.insert("type".into(), crate::dtype::json(&dt)); + } + Err(e) => self.problem(out, &path, &e)?, + } + m.insert("attributes".into(), self.json_attributes(out, &path, &h)?); + J::Object(m) + } + _ => { + let mut m = Map::new(); + m.insert("attributes".into(), self.json_attributes(out, &path, &h)?); + let links = match self.h5.links(&h) { + Ok(l) => l, + Err(e) => { + if kind == Kind::Group { + self.problem(out, &path, &e)?; + } + Vec::new() + } + }; + let mut jl = Vec::new(); + for l in &links { + jl.push(self.json_link(l)); + } + m.insert("links".into(), J::Array(jl)); + J::Object(m) + } + }; + if let J::Object(m) = &mut obj { + m.insert("alias".into(), json!(alias)); + } + let id = obj_id(kind, addr); + match kind { + Kind::Dataset => datasets.insert(id, obj), + Kind::Datatype => datatypes.insert(id, obj), + _ => groups.insert(id, obj), + }; + } + doc.insert("groups".into(), J::Object(groups)); + } + doc.insert("datasets".into(), J::Object(datasets)); + if only.is_none() { + doc.insert("datatypes".into(), J::Object(datatypes)); + } + let text = serde_json::to_string_pretty(&J::Object(doc)).map_err(std::io::Error::other)?; + writeln!(out.o, "{text}")?; + Ok(0) + } + + fn json_link(&self, l: &Link) -> J { + match &l.kind { + LinkKind::Hard(a) => { + let kind = self + .h5 + .header(*a) + .map(|h| Kind::of(&h)) + .unwrap_or(Kind::Unknown); + let coll = match kind { + Kind::Dataset => "datasets", + Kind::Datatype => "datatypes", + _ => "groups", + }; + json!({"class": "H5L_TYPE_HARD", "title": l.name, "collection": coll, + "id": obj_id(kind, *a)}) + } + LinkKind::Soft(t) => json!({"class": "H5L_TYPE_SOFT", "title": l.name, "h5path": t}), + LinkKind::External { file, path } => json!({"class": "H5L_TYPE_EXTERNAL", + "title": l.name, "file": file, "h5path": path}), + LinkKind::UserDefined(t) => json!({"class": "H5L_TYPE_USER_DEFINED", + "title": l.name, "linkClass": t}), + } + } + + fn json_dataset(&mut self, out: &mut Out, path: &str, h: &ObjectHeader) -> std::io::Result { + let info = DsInfo::read(self.h5, path, h); + let mut m = Map::new(); + match &info.dt { + Ok(dt) => { + m.insert("type".into(), crate::dtype::json(dt)); + } + Err(e) => self.problem(out, path, e)?, + } + match &info.ds { + Ok(ds) => { + m.insert("shape".into(), shape_json(ds)); + } + Err(e) => self.problem(out, path, e)?, + } + let mut cp = Map::new(); + if let Ok(l) = &info.layout { + let mut lj = Map::new(); + lj.insert( + "class".into(), + json!(match l { + DataLayout::Compact { .. } => "H5D_COMPACT", + DataLayout::Contiguous { .. } => "H5D_CONTIGUOUS", + DataLayout::Chunked { .. } => "H5D_CHUNKED", + DataLayout::Virtual { .. } => "H5D_VIRTUAL", + }), + ); + if let DataLayout::Chunked { + chunk_dimensions, .. + } = l + { + let rank = chunk_dimensions.len().saturating_sub(1); + lj.insert("dims".into(), json!(chunk_dimensions[..rank])); + } + cp.insert("layout".into(), J::Object(lj)); + } else if let Err(e) = &info.layout { + self.problem(out, path, e)?; + } + if let Ok(Some(p)) = &info.filters { + let fl: Vec = p + .filters + .iter() + .map(|f| { + json!({"id": f.filter_id, "name": info::filter_name(f), + "class": filter_class(f.filter_id), "parameters": f.client_data}) + }) + .collect(); + cp.insert("filters".into(), J::Array(fl)); + } + m.insert("creationProperties".into(), J::Object(cp)); + if !self.opts.header_only + && let (Ok(dt), Ok(ds)) = (&info.dt, &info.ds) + { + // JSON values are built in memory, some 32+ bytes per element: + // hold them to the same budget as the raw data. + let n = crate::h5::num_elements(ds).unwrap_or(u64::MAX); + let read = if n.saturating_mul(JSON_BYTES_PER_ELEMENT) > self.h5.max_bytes { + Err(Error::new(format!( + "{n} elements are too many to hold as JSON within --max-bytes {}", + self.h5.max_bytes + )) + .with_kind(crate::h5::ErrorKind::Limit)) + } else { + self.h5.read_dataset(path, dt, ds) + }; + match read { + Ok(raw) => { + let v = self.json_values(out, dt, ds, &raw)?; + m.insert("value".into(), v); + } + Err(e) => { + m.insert("value_error".into(), json!(e.to_string())); + self.problem(out, path, &e)?; + } + } + } + m.insert("attributes".into(), self.json_attributes(out, path, h)?); + Ok(J::Object(m)) + } + + fn json_attributes( + &mut self, + out: &mut Out, + path: &str, + h: &ObjectHeader, + ) -> std::io::Result { + let (attrs, errs) = match self.h5.attributes(h) { + Ok(x) => x, + Err(e) => { + self.problem(out, path, &e)?; + return Ok(J::Array(Vec::new())); + } + }; + for e in errs { + self.problem(out, path, &Error::new(format!("attribute: {e}")))?; + } + let mut v = Vec::new(); + for a in &attrs { + let mut m = Map::new(); + m.insert("name".into(), json!(a.name)); + m.insert("shape".into(), shape_json(&a.dataspace)); + m.insert("type".into(), crate::dtype::json(&a.datatype)); + { + match self.checked_attr(a) { + Ok(()) => { + let val = self.json_values(out, &a.datatype, &a.dataspace, &a.raw_data)?; + m.insert("value".into(), val); + } + Err(e) => { + m.insert("value_error".into(), json!(e.to_string())); + self.problem(out, &format!("{path} attribute {}", a.name), &e)?; + } + } + } + v.push(J::Object(m)); + } + Ok(J::Array(v)) + } + + fn json_values( + &mut self, + out: &mut Out, + dt: &Datatype, + ds: &Dataspace, + raw: &[u8], + ) -> std::io::Result { + let n = crate::h5::num_elements(ds).unwrap_or(0) as usize; + let dec = Decoder::new(self.h5); + let paths = |a: u64| self.paths().get(&a).cloned(); + let mut flat = Vec::with_capacity(n); + let mut errors = 0usize; + for i in 0..n { + let v = dec.element(dt, raw, i); + if matches!(v, Value::Error(_)) { + errors += 1; + } + flat.push(value::to_json(&v, &paths)); + } + if errors > 0 { + self.problems += 1; + writeln!(out.e, "h5rs dump: {errors} element(s) could not be decoded")?; + } + Ok(match ds.space_type { + DataspaceType::Null => J::Null, + DataspaceType::Scalar => flat.into_iter().next().unwrap_or(J::Null), + DataspaceType::Simple => nest(flat, &ds.dimensions), + }) + } +} + +/// Deepest group nesting the DDL output follows. +const MAX_DDL_DEPTH: usize = 256; + +/// Memory budgeted per element for a dataset's value held as JSON. +const JSON_BYTES_PER_ELEMENT: u64 = 64; + +/// `(i,j,k)` index of flat element `i` of an array with `dims`. +fn index_text(mut i: u64, dims: &[u64]) -> String { + let mut idx = vec![0u64; dims.len()]; + for (k, &d) in dims.iter().enumerate().rev() { + let d = d.max(1); + idx[k] = i % d; + i /= d; + } + idx.iter() + .map(|x| x.to_string()) + .collect::>() + .join(",") +} + +/// Turn a flat row-major list into nested lists of shape `dims`. +fn nest(flat: Vec, dims: &[u64]) -> J { + if dims.len() <= 1 { + return J::Array(flat); + } + let inner: usize = dims[1..].iter().map(|&d| d as usize).product(); + if inner == 0 { + return J::Array((0..dims[0]).map(|_| nest(Vec::new(), &dims[1..])).collect()); + } + let mut it = flat.into_iter(); + let mut outer = Vec::with_capacity(dims[0] as usize); + for _ in 0..dims[0] { + let part: Vec = it.by_ref().take(inner).collect(); + outer.push(nest(part, &dims[1..])); + } + J::Array(outer) +} + +fn shape_json(ds: &Dataspace) -> J { + match ds.space_type { + DataspaceType::Null => json!({"class": "H5S_NULL"}), + DataspaceType::Scalar => json!({"class": "H5S_SCALAR"}), + DataspaceType::Simple => { + let mut m = Map::new(); + m.insert("class".into(), json!("H5S_SIMPLE")); + m.insert("dims".into(), json!(ds.dimensions)); + if let Some(max) = &ds.max_dimensions { + let mx: Vec = max + .iter() + .map(|&d| { + if d == u64::MAX { + json!("H5S_UNLIMITED") + } else { + json!(d) + } + }) + .collect(); + m.insert("maxdims".into(), J::Array(mx)); + } + J::Object(m) + } + } +} + +fn filter_class(id: u16) -> &'static str { + match id { + 1 => "H5Z_FILTER_DEFLATE", + 2 => "H5Z_FILTER_SHUFFLE", + 3 => "H5Z_FILTER_FLETCHER32", + 4 => "H5Z_FILTER_SZIP", + 5 => "H5Z_FILTER_NBIT", + 6 => "H5Z_FILTER_SCALEOFFSET", + _ => "H5Z_FILTER_USER", + } +} + +/// Deterministic object id: kind prefix + header address (hdf5-json uses +/// UUIDs; these are stable for a given file instead). +pub fn obj_id(kind: Kind, addr: u64) -> String { + let p = match kind { + Kind::Dataset => 'd', + Kind::Datatype => 't', + _ => 'g', + }; + format!("{p}-{addr:016x}") +} diff --git a/crates/clawhdf5-tools/src/h5.rs b/crates/clawhdf5-tools/src/h5.rs new file mode 100644 index 0000000..e2e7931 --- /dev/null +++ b/crates/clawhdf5-tools/src/h5.rs @@ -0,0 +1,718 @@ +//! The file model the tools share: an open file, its objects, their links +//! and the messages that describe a dataset, all read through +//! `clawhdf5-format` (and the `clawhdf5` facade for dataset values). +//! +//! Nothing here trusts the file: every size is checked before it is used, +//! and every failure is an [`Error`] carrying the address it happened at. + +use std::cell::RefCell; +use std::collections::HashMap; +use std::path::{Path, PathBuf}; +use std::rc::Rc; + +use clawhdf5::File; +use clawhdf5_format::attribute::{AttributeMessage, extract_attributes_tolerant}; +use clawhdf5_format::attribute_info::AttributeInfoMessage; +use clawhdf5_format::btree_v2::{BTreeV2Header, collect_btree_v2_records}; +use clawhdf5_format::data_layout::DataLayout; +use clawhdf5_format::dataspace::{Dataspace, DataspaceType}; +use clawhdf5_format::datatype::Datatype; +use clawhdf5_format::error::FormatError; +use clawhdf5_format::filter_pipeline::FilterPipeline; +use clawhdf5_format::fractal_heap::FractalHeapHeader; +use clawhdf5_format::global_heap::GlobalHeapCollection; +use clawhdf5_format::group_v1; +use clawhdf5_format::link_info::LinkInfoMessage; +use clawhdf5_format::link_message::{LinkMessage, LinkTarget}; +use clawhdf5_format::message_type::MessageType; +use clawhdf5_format::object_header::ObjectHeader; +use clawhdf5_format::superblock::Superblock; +use clawhdf5_format::symbol_table::SymbolTableMessage; + +/// Largest dataset or attribute (in bytes) the tools will decode by default. +/// A corrupt dataspace can claim far more elements than the file holds; the +/// limit keeps such a file from exhausting memory. `--max-bytes` changes it. +pub const DEFAULT_MAX_BYTES: u64 = 1 << 30; + +/// Upper bound on the objects one traversal visits, so a crafted file with +/// millions of links cannot make a walk run for ever. +pub const MAX_OBJECTS: usize = 1_000_000; + +/// A tool error: what went wrong and, when known, the file address (relative +/// to the superblock, as every HDF5 address is) of the structure involved. +#[derive(Debug, Clone)] +pub struct Error { + pub addr: Option, + pub msg: String, + pub kind: ErrorKind, +} + +/// Whether an error says something is wrong with the file. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum ErrorKind { + /// The file is damaged or not valid HDF5 (as far as clawhdf5 knows). + Corrupt, + /// Valid HDF5 that clawhdf5 cannot decode (a filter it does not + /// implement, external raw data files, ...). + Unsupported, + /// Refused by a limit of the tool (`--max-bytes`). + Limit, +} + +impl Error { + pub fn new(msg: impl Into) -> Self { + Self { + addr: None, + msg: msg.into(), + kind: ErrorKind::Corrupt, + } + } + + pub fn at(addr: u64, msg: impl Into) -> Self { + Self { + addr: Some(addr), + msg: msg.into(), + kind: ErrorKind::Corrupt, + } + } + + pub fn with_kind(mut self, kind: ErrorKind) -> Self { + self.kind = kind; + self + } + + /// The same error with `prefix: ` in front of its message. + pub fn context(mut self, prefix: &str) -> Self { + self.msg = format!("{prefix}: {}", self.msg); + self + } +} + +fn format_error_kind(e: &FormatError) -> ErrorKind { + match e { + FormatError::UnsupportedFilter(_) + | FormatError::ExternalDataFilesUnsupported + | FormatError::ExternalLinkUnsupported { .. } => ErrorKind::Unsupported, + _ => ErrorKind::Corrupt, + } +} + +impl std::fmt::Display for Error { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self.addr { + Some(a) => write!(f, "{} (at address {a:#x})", self.msg), + None => write!(f, "{}", self.msg), + } + } +} + +impl From for Error { + fn from(e: FormatError) -> Self { + Error::new(e.to_string()).with_kind(format_error_kind(&e)) + } +} + +impl From for Error { + fn from(e: clawhdf5::Error) -> Self { + let kind = match &e { + clawhdf5::Error::Format(f) => format_error_kind(f), + _ => ErrorKind::Corrupt, + }; + Error::new(e.to_string()).with_kind(kind) + } +} + +pub type Result = std::result::Result; + +/// Attach an address to a format error. +pub fn fe_at(addr: u64) -> impl Fn(FormatError) -> Error { + move |e| Error::at(addr, e.to_string()) +} + +/// What an object header describes. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Kind { + Group, + Dataset, + Datatype, + Unknown, +} + +impl Kind { + pub fn of(h: &ObjectHeader) -> Kind { + let has = |t: MessageType| h.messages.iter().any(|m| m.msg_type == t); + if has(MessageType::DataLayout) { + Kind::Dataset + } else if has(MessageType::SymbolTable) + || has(MessageType::LinkInfo) + || has(MessageType::Link) + || has(MessageType::GroupInfo) + { + Kind::Group + } else if has(MessageType::Datatype) { + Kind::Datatype + } else { + Kind::Unknown + } + } + + pub fn name(self) -> &'static str { + match self { + Kind::Group => "Group", + Kind::Dataset => "Dataset", + Kind::Datatype => "Type", + Kind::Unknown => "Unknown", + } + } +} + +/// Where a link points. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum LinkKind { + Hard(u64), + Soft(String), + External { + file: String, + path: String, + }, + /// A user-defined link class (type 65-255): its target means something + /// only to the application that registered the class. + UserDefined(u8), +} + +#[derive(Debug, Clone)] +pub struct Link { + pub name: String, + pub kind: LinkKind, +} + +/// An open file. +pub struct H5 { + pub path: PathBuf, + pub file: File, + pub max_bytes: u64, + heaps: RefCell, String>>>, + /// Fractal heaps whose blocks were verified: `None` = sound. + verified_heaps: RefCell>>, +} + +impl H5 { + pub fn open(path: &Path) -> Result
{ + if !path.exists() { + return Err(Error::new(format!("{}: no such file", path.display()))); + } + let file = File::open(path).map_err(|e| { + Error::new(format!( + "{}: not an HDF5 file this tool can open: {e}", + path.display() + )) + })?; + Ok(H5 { + path: path.to_path_buf(), + file, + max_bytes: DEFAULT_MAX_BYTES, + heaps: RefCell::new(HashMap::new()), + verified_heaps: RefCell::new(HashMap::new()), + }) + } + + /// The file's bytes from the superblock on: what every address indexes. + pub fn data(&self) -> &[u8] { + self.file.as_bytes() + } + + pub fn sb(&self) -> &Superblock { + self.file.superblock() + } + + pub fn os(&self) -> u8 { + self.sb().offset_size + } + + pub fn ls(&self) -> u8 { + self.sb().length_size + } + + pub fn root(&self) -> u64 { + self.sb().root_group_address + } + + pub fn header(&self, addr: u64) -> Result { + let off = usize::try_from(addr).map_err(|_| Error::at(addr, "address out of range"))?; + ObjectHeader::parse(self.data(), off, self.os(), self.ls()) + .map_err(|e| Error::at(addr, format!("object header: {e}"))) + } + + /// The payload of the first message of type `t`, resolving a shared + /// message to the message it points at. + pub fn payload(&self, h: &ObjectHeader, t: MessageType) -> Result>> { + match h.messages.iter().find(|m| m.msg_type == t) { + None => Ok(None), + Some(m) => { + clawhdf5_format::shared_message::message_data(self.data(), m, self.os(), self.ls()) + .map(|c| Some(c.into_owned())) + .map_err(|e| Error::new(format!("{t:?} message: {e}"))) + } + } + } + + pub fn datatype(&self, h: &ObjectHeader) -> Result { + let b = self + .payload(h, MessageType::Datatype)? + .ok_or_else(|| Error::new("no datatype message"))?; + Datatype::parse(&b) + .map(|(d, _)| d) + .map_err(|e| Error::new(format!("datatype message: {e}"))) + } + + pub fn dataspace(&self, h: &ObjectHeader) -> Result { + let b = self + .payload(h, MessageType::Dataspace)? + .ok_or_else(|| Error::new("no dataspace message"))?; + Dataspace::parse(&b, self.ls()).map_err(|e| Error::new(format!("dataspace message: {e}"))) + } + + pub fn layout(&self, h: &ObjectHeader) -> Result { + let m = h + .messages + .iter() + .find(|m| m.msg_type == MessageType::DataLayout) + .ok_or_else(|| Error::new("no layout message"))?; + DataLayout::parse(&m.data, self.os(), self.ls()) + .map_err(|e| Error::new(format!("layout message: {e}"))) + } + + pub fn filters(&self, h: &ObjectHeader) -> Result> { + match self.payload(h, MessageType::FilterPipeline)? { + None => Ok(None), + Some(b) => FilterPipeline::parse(&b) + .map(Some) + .map_err(|e| Error::new(format!("filter pipeline message: {e}"))), + } + } + + /// Every attribute that can be read, plus one error per attribute that + /// cannot. + pub fn attributes(&self, h: &ObjectHeader) -> Result<(Vec, Vec)> { + if let Some(m) = h + .messages + .iter() + .find(|m| m.msg_type == MessageType::AttributeInfo) + && let Ok(ai) = AttributeInfoMessage::parse(&m.data, self.os()) + && let Some(fh) = ai.fractal_heap_address + { + self.verified_heap(fh) + .map_err(|e| e.context("dense attribute storage"))?; + } + let (mut attrs, errs) = extract_attributes_tolerant(self.data(), h, self.os(), self.ls()) + .map_err(|e| Error::new(format!("attributes: {e}")))?; + attrs.sort_by(|a, b| a.name.cmp(&b.name)); + Ok((attrs, errs.iter().map(|e| e.to_string()).collect())) + } + + /// Every link of the group whose header is `h`, sorted by name. An + /// object that is not a group has none. + pub fn links(&self, h: &ObjectHeader) -> Result> { + let os = self.os(); + let ls = self.ls(); + let data = self.data(); + let mut out = Vec::new(); + if let Some(m) = h + .messages + .iter() + .find(|m| m.msg_type == MessageType::SymbolTable) + { + let stm = SymbolTableMessage::parse(&m.data, os) + .map_err(|e| Error::new(format!("symbol table message: {e}")))?; + let entries = group_v1::resolve_v1_group_entries(data, &stm, os, ls) + .map_err(|e| Error::at(stm.btree_address, format!("symbol table: {e}")))?; + let has_soft = entries.iter().any(group_v1::is_v1_soft_link); + for e in entries { + if !group_v1::is_v1_soft_link(&e) { + out.push(Link { + name: e.name, + kind: LinkKind::Hard(e.object_header_address), + }); + } + } + if has_soft { + let soft = group_v1::v1_soft_links(data, &stm, os, ls) + .map_err(|e| Error::at(stm.btree_address, format!("soft links: {e}")))?; + for (name, target) in soft { + out.push(Link { + name, + kind: LinkKind::Soft(target), + }); + } + } + } else { + let li = match h + .messages + .iter() + .find(|m| m.msg_type == MessageType::LinkInfo) + { + Some(m) => Some( + LinkInfoMessage::parse(&m.data, os) + .map_err(|e| Error::new(format!("link info message: {e}")))?, + ), + None => None, + }; + for m in h + .messages + .iter() + .filter(|m| m.msg_type == MessageType::Link) + { + out.push(link_from_message(&m.data, os)?); + } + if let Some(li) = li + && let Some(fh) = li.fractal_heap_address + { + for bytes in self.dense_heap_objects(fh, li.btree_name_index_address, 5)? { + out.push(link_from_message(&bytes, os)?); + } + } + } + out.sort_by(|a, b| a.name.cmp(&b.name)); + Ok(out) + } + + /// The objects a dense-storage B-tree (`btree`, of record type + /// `name_type`: 5 for links, 8 for attributes) indexes in the fractal + /// heap at `heap`. + pub fn dense_heap_objects( + &self, + heap: u64, + btree: Option, + name_type: u8, + ) -> Result>> { + self.verified_heap(heap)?; + let data = self.data(); + let os = self.os(); + let ls = self.ls(); + let fh = FractalHeapHeader::parse(data, to_usize(heap)?, os, ls) + .map_err(|e| Error::at(heap, format!("fractal heap header: {e}")))?; + let bt = btree.ok_or_else(|| Error::at(heap, "dense storage without a name index"))?; + let hdr = BTreeV2Header::parse(data, to_usize(bt)?, os, ls) + .map_err(|e| Error::at(bt, format!("v2 B-tree header: {e}")))?; + let recs = collect_btree_v2_records(data, &hdr, os, ls) + .map_err(|e| Error::at(bt, format!("v2 B-tree: {e}")))?; + // Name-index records: hash(4) + heap ID; creation-order ones: order(8) + heap ID. + let skip = if hdr.tree_type == name_type { 4 } else { 8 }; + let idlen = usize::from(fh.heap_id_length); + let mut out = Vec::with_capacity(recs.len()); + for r in &recs { + let id = r + .data + .get(skip..skip + idlen) + .ok_or_else(|| Error::at(bt, "v2 B-tree record shorter than a heap ID"))?; + let obj = fh + .read_managed_object(data, id, os) + .map_err(|e| Error::at(heap, format!("fractal heap object: {e}")))?; + out.push(obj); + } + Ok(out) + } + + /// Verify every block of the fractal heap at `addr` (once per heap): + /// the library reads heap objects without checking the blocks' + /// checksums, which libhdf5 does, so a damaged block would otherwise be + /// read as if it were sound. + pub fn verified_heap(&self, addr: u64) -> Result<()> { + if let Some(r) = self.verified_heaps.borrow().get(&addr) { + return r.clone().map_or(Ok(()), Err); + } + let report = crate::heap_blocks::verify(self, addr); + let n = report.problems.len(); + let r = report.problems.into_iter().next().map(|mut e| { + if n > 1 { + e.msg = format!("{} (and {} more problems in this heap)", e.msg, n - 1); + } + e + }); + self.verified_heaps.borrow_mut().insert(addr, r.clone()); + r.map_or(Ok(()), Err) + } + + /// The global heap object `idx` of the collection at `addr` (cached per + /// collection). + pub fn heap_object(&self, addr: u64, idx: u32) -> Result> { + let coll = { + let mut cache = self.heaps.borrow_mut(); + cache + .entry(addr) + .or_insert_with(|| match usize::try_from(addr) { + Ok(a) => GlobalHeapCollection::parse(self.data(), a, self.ls()) + .map(Rc::new) + .map_err(|e| e.to_string()), + Err(_) => Err("address out of range".into()), + }) + .clone() + .map_err(|e| Error::at(addr, format!("global heap: {e}")))? + }; + let idx16 = u16::try_from(idx) + .map_err(|_| Error::at(addr, format!("global heap object index {idx} out of range")))?; + coll.get_object(idx16) + .map(|o| o.data.clone()) + .ok_or_else(|| Error::at(addr, format!("global heap has no object {idx}"))) + } + + /// The dataspace of the dataset at `path` with a virtual dataset's + /// extent resolved from its sources (as libhdf5 reports it) instead of + /// the stored one. + pub fn resolved_dataspace(&self, path: &str, h: &ObjectHeader) -> Result { + let mut ds = self.dataspace(h)?; + if matches!(self.layout(h), Ok(DataLayout::Virtual { .. })) { + let d = self.file.dataset(path)?; + ds.dimensions = d.shape()?; + if ds.space_type == DataspaceType::Simple + && let Some(m) = &ds.max_dimensions + && m.len() != ds.dimensions.len() + { + ds.max_dimensions = None; + } + } + Ok(ds) + } + + /// Read a dataset's values (raw, in file byte order) through the + /// `clawhdf5` facade, refusing one larger than `max_bytes`. `ds` must be + /// the [resolved](Self::resolved_dataspace) dataspace. + pub fn read_dataset(&self, path: &str, dt: &Datatype, ds: &Dataspace) -> Result> { + let need = byte_len(ds, dt)?; + if need > self.max_bytes { + return Err(Error::new(format!( + "dataset is {need} bytes, over the {} byte limit (--max-bytes)", + self.max_bytes + )) + .with_kind(ErrorKind::Limit)); + } + let d = self.file.dataset(path)?; + let raw = d.read_selection(&clawhdf5::Selection::All)?; + if raw.len() as u64 != need { + return Err(Error::new(format!( + "read {} bytes, expected {need}", + raw.len() + ))); + } + Ok(raw) + } + + /// Walk every object reachable by hard links from the root, depth first + /// in name order, calling `visit` once per link (the root is visited + /// first with an empty link name). Each object is described once; later + /// hard links to it are reported with `first_path` set. + pub fn walk(&self, mut visit: impl FnMut(&WalkItem<'_>)) -> Result<()> { + self.walk_from(self.root(), "/", &mut visit) + } + + pub fn walk_from( + &self, + start: u64, + start_path: &str, + visit: &mut dyn FnMut(&WalkItem<'_>), + ) -> Result<()> { + let mut seen: HashMap = HashMap::new(); + let mut stack: Vec<(u64, String, Option, usize)> = + vec![(start, start_path.to_string(), None, 0)]; + let mut count = 0usize; + while let Some((addr, path, link, depth)) = stack.pop() { + count += 1; + if count > MAX_OBJECTS { + return Err(Error::new(format!( + "more than {MAX_OBJECTS} links; stopped walking" + ))); + } + // Soft, external and user-defined links have no address. + if let Some(Link { kind, .. }) = &link + && !matches!(kind, LinkKind::Hard(_)) + { + visit(&WalkItem { + path: &path, + link: link.as_ref(), + addr: None, + first_path: None, + header: None, + depth, + }); + continue; + } + if let Some(first) = seen.get(&addr) { + visit(&WalkItem { + path: &path, + link: link.as_ref(), + addr: Some(addr), + first_path: Some(first), + header: None, + depth, + }); + continue; + } + seen.insert(addr, path.clone()); + let header = self.header(addr); + visit(&WalkItem { + path: &path, + link: link.as_ref(), + addr: Some(addr), + first_path: None, + header: Some(&header), + depth, + }); + let Ok(h) = &header else { continue }; + if Kind::of(h) != Kind::Group { + continue; + } + // Link errors are the visitor's to report (it sees the header). + let Ok(links) = self.links(h) else { continue }; + let base = if path == "/" { "" } else { path.as_str() }; + for l in links.into_iter().rev() { + let child = format!("{base}/{}", l.name); + let a = match l.kind { + LinkKind::Hard(a) => a, + _ => 0, + }; + stack.push((a, child, Some(l), depth + 1)); + } + } + Ok(()) + } + + /// Resolve an absolute object path to its header address, following + /// soft links like the library does. + pub fn resolve(&self, path: &str) -> Result { + let p = path.trim_matches('/'); + if p.is_empty() { + return Ok(self.root()); + } + clawhdf5_format::group_v2::resolve_path_any(self.data(), self.sb(), p) + .map_err(|e| Error::new(format!("{path}: {e}"))) + } +} + +/// An owned copy of a [`WalkItem`], from [`H5::walk_collect`]. +pub struct Visited { + pub path: String, + pub addr: Option, + pub first_path: Option, + pub header: Option>, +} + +impl H5 { + /// Every step of [`H5::walk`], collected, and the walk's own error. + pub fn walk_collect(&self) -> (Vec, Result<()>) { + let mut items = Vec::new(); + let r = self.walk(|it| { + items.push(Visited { + path: it.path.to_string(), + addr: it.addr, + first_path: it.first_path.map(str::to_string), + header: it.header.cloned(), + }); + }); + (items, r) + } +} + +/// One step of [`H5::walk`]. +pub struct WalkItem<'a> { + pub path: &'a str, + /// The link that led here (`None` for the start object). + pub link: Option<&'a Link>, + /// Header address (`None` for a soft/external/user-defined link). + pub addr: Option, + /// Set when this object was already visited under another path. + pub first_path: Option<&'a str>, + /// The parsed header, for an object seen for the first time. + pub header: Option<&'a Result>, + pub depth: usize, +} + +fn link_from_message(data: &[u8], os: u8) -> Result { + match LinkMessage::parse(data, os) { + Ok(l) => Ok(Link { + name: l.name, + kind: match l.link_target { + LinkTarget::Hard { + object_header_address, + } => LinkKind::Hard(object_header_address), + LinkTarget::Soft { target_path } => LinkKind::Soft(target_path), + LinkTarget::External { + filename, + object_path, + } => LinkKind::External { + file: filename, + path: object_path, + }, + }, + }), + Err(FormatError::InvalidLinkType(t)) if t >= 65 => Ok(Link { + name: user_defined_link_name(data).unwrap_or_else(|| "?".into()), + kind: LinkKind::UserDefined(t), + }), + Err(e) => Err(Error::new(format!("link message: {e}"))), + } +} + +/// The name of a user-defined link, which `LinkMessage::parse` refuses. +fn user_defined_link_name(d: &[u8]) -> Option { + // version(1) flags(1) [type(1)] [corder(8)] [cset(1)] len(1|2|4|8) name + let flags = *d.get(1)?; + let mut p = 2usize; + if flags & 0x08 != 0 { + p += 1; + } + if flags & 0x04 != 0 { + p += 8; + } + if flags & 0x10 != 0 { + p += 1; + } + let w = 1usize << (flags & 0x03); + let mut n = 0usize; + for i in 0..w { + n |= usize::from(*d.get(p + i)?) << (8 * i); + } + p += w; + let name = d.get(p..p.checked_add(n)?)?; + Some(String::from_utf8_lossy(name).into_owned()) +} + +pub fn to_usize(a: u64) -> Result { + usize::try_from(a).map_err(|_| Error::at(a, "address out of range")) +} + +/// Number of elements a dataspace holds (0 for a null dataspace). +pub fn num_elements(ds: &Dataspace) -> Result { + match ds.space_type { + DataspaceType::Null => Ok(0), + DataspaceType::Scalar => Ok(1), + DataspaceType::Simple => ds + .dimensions + .iter() + .try_fold(1u64, |a, &d| a.checked_mul(d)) + .ok_or_else(|| Error::new("dataspace element count overflows")), + } +} + +/// Bytes needed for all of a dataspace's elements of type `dt`. +pub fn byte_len(ds: &Dataspace, dt: &Datatype) -> Result { + num_elements(ds)? + .checked_mul(u64::from(dt.type_size())) + .ok_or_else(|| Error::new("dataset byte size overflows")) +} + +/// Split a `FILE[/object/path]` argument the way h5ls does: the longest +/// prefix that is an existing file is the file. +pub fn split_file_arg(arg: &str) -> (String, Option) { + if Path::new(arg).is_file() { + return (arg.to_string(), None); + } + let mut idx: Vec = arg.match_indices('/').map(|(i, _)| i).collect(); + idx.reverse(); + for i in idx { + let (f, rest) = arg.split_at(i); + if !f.is_empty() && Path::new(f).is_file() { + return (f.to_string(), Some(rest.to_string())); + } + } + (arg.to_string(), None) +} diff --git a/crates/clawhdf5-tools/src/heap_blocks.rs b/crates/clawhdf5-tools/src/heap_blocks.rs new file mode 100644 index 0000000..a1c40cd --- /dev/null +++ b/crates/clawhdf5-tools/src/heap_blocks.rs @@ -0,0 +1,308 @@ +//! Walk every block of a fractal heap and verify it: signatures, the +//! back-pointer to the heap header, each block's heap offset, and the +//! checksums (always present on indirect blocks; on direct blocks when the +//! heap header's flag says so). The library reads only the blocks an object +//! lives in and does not verify block checksums, so `check` does it here. + +use std::collections::HashSet; + +use clawhdf5_format::checksum::jenkins_lookup3; +use clawhdf5_format::fractal_heap::FractalHeapHeader; + +use crate::h5::{Error, H5}; + +/// Heap header flag bit 1: direct blocks carry a checksum. +const FLAG_CHECKSUM_DBLOCKS: u8 = 0x02; +const MAX_DEPTH: u32 = 16; +const MAX_BLOCKS: usize = 1 << 20; + +#[derive(Default, Debug)] +pub struct HeapReport { + pub direct_blocks: usize, + pub indirect_blocks: usize, + /// Blocks whose checksum was verified. + pub checksums: usize, + pub problems: Vec, +} + +struct Walk<'a> { + data: &'a [u8], + heap: u64, + fh: FractalHeapHeader, + checksum_dblocks: bool, + boff_bytes: usize, + os: usize, + ls: usize, + seen: HashSet, + r: HeapReport, +} + +fn le(b: &[u8]) -> u64 { + b.iter() + .take(8) + .enumerate() + .fold(0u64, |a, (i, &x)| a | (u64::from(x) << (8 * i))) +} + +fn undefined(v: u64, os: usize) -> bool { + if os >= 8 { + v == u64::MAX + } else { + v == (1u64 << (8 * os)) - 1 + } +} + +fn log2(v: u64) -> u32 { + 63u32.saturating_sub(v.max(1).leading_zeros()) +} + +/// Verify the fractal heap whose header is at `heap`. The header itself is +/// parsed (and its checksum verified) by the library; an error there is +/// returned as the only problem. +pub fn verify(h5: &H5, heap: u64) -> HeapReport { + let data = h5.data(); + let Ok(off) = usize::try_from(heap) else { + return HeapReport { + problems: vec![Error::at(heap, "fractal heap address out of range")], + ..Default::default() + }; + }; + let fh = match FractalHeapHeader::parse(data, off, h5.os(), h5.ls()) { + Ok(f) => f, + Err(e) => { + return HeapReport { + problems: vec![Error::at(heap, format!("fractal heap header: {e}"))], + ..Default::default() + }; + } + }; + // Flags: signature(4) version(1) heap ID length(2) filter length(2) flags(1). + let flags = data.get(off + 9).copied().unwrap_or(0); + let mut w = Walk { + data, + heap, + checksum_dblocks: flags & FLAG_CHECKSUM_DBLOCKS != 0 && fh.filter_pipeline.is_none(), + boff_bytes: usize::from(fh.max_heap_size).div_ceil(8), + os: usize::from(h5.os()), + ls: usize::from(h5.ls()), + fh, + seen: HashSet::new(), + r: HeapReport::default(), + }; + if w.fh.table_width == 0 || w.fh.starting_block_size == 0 || !w.fh.table_width.is_power_of_two() + { + w.problem(heap, "fractal heap header: invalid doubling table geometry"); + return w.r; + } + let root = w.fh.root_block_address; + if !undefined(root, w.os) { + if w.fh.current_rows_in_root_indirect_block == 0 { + let size = w.fh.starting_block_size; + w.direct(root, size, 0); + } else { + let rows = w.fh.current_rows_in_root_indirect_block; + w.indirect(root, rows, 0, 0); + } + } + w.r +} + +impl Walk<'_> { + fn problem(&mut self, addr: u64, msg: impl Into) { + self.r.problems.push(Error::at(addr, msg)); + } + + fn row_size(&self, row: usize) -> Option { + let s = self.fh.starting_block_size; + if row <= 1 { + Some(s) + } else { + let sh = u32::try_from(row - 1).ok()?; + s.checked_mul(1u64.checked_shl(sh)?) + } + } + + fn max_direct_rows(&self) -> usize { + let ratio = (self.fh.max_direct_block_size / self.fh.starting_block_size).max(1); + log2(ratio) as usize + 2 + } + + fn rows_for_size(&self, size: u64) -> u16 { + let first = log2(self.fh.starting_block_size) + log2(u64::from(self.fh.table_width)); + (log2(size).saturating_sub(first) + 1) as u16 + } + + /// Common block prefix: signature, version, heap header address and + /// block offset. Returns the position after it, or `None` after + /// recording a problem. + fn prefix(&mut self, addr: u64, sig: &[u8; 4], what: &str, heap_offset: u64) -> Option { + if !self.seen.insert(addr) { + self.problem( + addr, + format!("fractal heap {what} block reached twice (cycle)"), + ); + return None; + } + if self.seen.len() > MAX_BLOCKS { + self.problem(self.heap, "fractal heap has too many blocks; stopped"); + return None; + } + let Ok(start) = usize::try_from(addr) else { + self.problem( + addr, + format!("fractal heap {what} block address out of range"), + ); + return None; + }; + let hdr_len = 5 + self.os + self.boff_bytes; + let Some(b) = start + .checked_add(hdr_len) + .and_then(|e| self.data.get(start..e)) + else { + self.problem( + addr, + format!("fractal heap {what} block lies past the end of the file"), + ); + return None; + }; + if &b[..4] != sig { + self.problem(addr, format!("fractal heap {what} block: bad signature")); + return None; + } + if b[4] != 0 { + self.problem(addr, format!("fractal heap {what} block: version {}", b[4])); + return None; + } + let back = le(&b[5..5 + self.os]); + if back != self.heap { + self.problem( + addr, + format!( + "fractal heap {what} block points at heap header {back:#x}, not {:#x}", + self.heap + ), + ); + } + let boff = le(&b[5 + self.os..hdr_len]); + if boff != heap_offset { + self.problem( + addr, + format!("fractal heap {what} block has heap offset {boff}, expected {heap_offset}"), + ); + } + Some(start + hdr_len) + } + + fn direct(&mut self, addr: u64, size: u64, heap_offset: u64) { + let Some(pos) = self.prefix(addr, b"FHDB", "direct", heap_offset) else { + return; + }; + self.r.direct_blocks += 1; + if self.fh.filter_pipeline.is_some() { + return; // stored filtered: its size on disk is not the block size + } + let start = pos - (5 + self.os + self.boff_bytes); + let Some(end) = usize::try_from(size) + .ok() + .and_then(|s| start.checked_add(s)) + else { + self.problem(addr, "fractal heap direct block size out of range"); + return; + }; + let Some(block) = self.data.get(start..end) else { + self.problem( + addr, + "fractal heap direct block extends past the end of the file", + ); + return; + }; + if self.checksum_dblocks { + let Some(stored) = block.get(pos - start..pos - start + 4) else { + self.problem(addr, "fractal heap direct block too small for its checksum"); + return; + }; + let stored = u32::from_le_bytes([stored[0], stored[1], stored[2], stored[3]]); + let mut copy = block.to_vec(); + copy[pos - start..pos - start + 4].fill(0); + let computed = jenkins_lookup3(©); + self.r.checksums += 1; + if computed != stored { + self.problem( + addr, + format!( + "fractal heap direct block: checksum mismatch: stored {stored:#010x}, computed {computed:#010x}" + ), + ); + } + } + } + + fn indirect(&mut self, addr: u64, nrows: u16, heap_offset: u64, depth: u32) { + if depth > MAX_DEPTH { + self.problem(addr, "fractal heap indirect blocks nested too deeply"); + return; + } + let Some(mut pos) = self.prefix(addr, b"FHIB", "indirect", heap_offset) else { + return; + }; + self.r.indirect_blocks += 1; + let start = pos - (5 + self.os + self.boff_bytes); + let width = usize::from(self.fh.table_width); + let filtered = self.fh.filter_pipeline.is_some(); + let direct_rows = self.max_direct_rows(); + let mut children: Vec<(u64, bool, u64, u64)> = Vec::new(); // addr, direct, size/rows, offset + let mut off = heap_offset; + for row in 0..usize::from(nrows) { + let Some(rs) = self.row_size(row) else { + self.problem(addr, "fractal heap row size overflows"); + return; + }; + let direct = row < direct_rows; + for _ in 0..width { + let Some(b) = self.data.get(pos..pos + self.os) else { + self.problem( + addr, + "fractal heap indirect block extends past the end of the file", + ); + return; + }; + let child = le(b); + pos += self.os; + if direct && filtered { + pos += self.ls + 4; + } + if !undefined(child, self.os) { + children.push((child, direct, rs, off)); + } + off = off.saturating_add(rs); + } + } + let Some(stored) = self.data.get(pos..pos + 4) else { + self.problem( + addr, + "fractal heap indirect block extends past the end of the file", + ); + return; + }; + let stored = u32::from_le_bytes([stored[0], stored[1], stored[2], stored[3]]); + let computed = jenkins_lookup3(&self.data[start..pos]); + self.r.checksums += 1; + if computed != stored { + self.problem( + addr, + format!( + "fractal heap indirect block: checksum mismatch: stored {stored:#010x}, computed {computed:#010x}" + ), + ); + return; // its child pointers cannot be trusted + } + for (child, direct, size, off) in children { + if direct { + self.direct(child, size, off); + } else { + let rows = self.rows_for_size(size); + self.indirect(child, rows, off, depth + 1); + } + } + } +} diff --git a/crates/clawhdf5-tools/src/info.rs b/crates/clawhdf5-tools/src/info.rs new file mode 100644 index 0000000..606457b --- /dev/null +++ b/crates/clawhdf5-tools/src/info.rs @@ -0,0 +1,245 @@ +//! Dataset facts shared by `ls`, `dump`, `stat` and `check`: shape text, +//! layout, filters and storage. + +use clawhdf5_format::chunked_read::{ChunkInfo, list_chunks}; +use clawhdf5_format::data_layout::DataLayout; +use clawhdf5_format::dataspace::{Dataspace, DataspaceType}; +use clawhdf5_format::datatype::Datatype; +use clawhdf5_format::filter_pipeline::{FilterDescription, FilterPipeline}; +use clawhdf5_format::message_type::MessageType; +use clawhdf5_format::object_header::ObjectHeader; + +use crate::h5::{Error, H5, Result}; + +/// h5ls's `{10/Inf, 20}` shape text. `always_max` prints `cur/max` for every +/// dimension (h5ls -v). +pub fn shape_text(ds: &Dataspace, always_max: bool) -> String { + match ds.space_type { + DataspaceType::Null => "{NULL}".into(), + DataspaceType::Scalar => "{SCALAR}".into(), + DataspaceType::Simple => { + let dims: Vec = ds + .dimensions + .iter() + .enumerate() + .map(|(i, &d)| { + let m = ds.max_dimensions.as_ref().and_then(|m| m.get(i).copied()); + match m { + Some(u64::MAX) => format!("{d}/Inf"), + Some(m) if m != d || always_max => format!("{d}/{m}"), + None if always_max => format!("{d}/{d}"), + _ => d.to_string(), + } + }) + .collect(); + format!("{{{}}}", dims.join(", ")) + } + } +} + +/// h5dump's `SIMPLE { ( 3, 4 ) / ( 3, H5S_UNLIMITED ) }`. +pub fn dataspace_ddl(ds: &Dataspace) -> String { + match ds.space_type { + DataspaceType::Null => "NULL".into(), + DataspaceType::Scalar => "SCALAR".into(), + DataspaceType::Simple => { + let cur: Vec = ds.dimensions.iter().map(|d| d.to_string()).collect(); + let max: Vec = ds + .dimensions + .iter() + .enumerate() + .map( + |(i, &d)| match ds.max_dimensions.as_ref().and_then(|m| m.get(i).copied()) { + Some(u64::MAX) => "H5S_UNLIMITED".into(), + Some(m) => m.to_string(), + None => d.to_string(), + }, + ) + .collect(); + format!( + "SIMPLE {{ ( {} ) / ( {} ) }}", + cur.join(", "), + max.join(", ") + ) + } + } +} + +pub fn filter_name(f: &FilterDescription) -> String { + let known = match f.filter_id { + 1 => "deflate", + 2 => "shuffle", + 3 => "fletcher32", + 4 => "szip", + 5 => "nbit", + 6 => "scaleoffset", + 307 => "bzip2", + 32000 => "lzf", + 32001 => "blosc", + 32004 => "lz4", + 32008 => "bitshuffle", + 32013 => "zfp", + 32015 => "zstd", + 32026 => "blosc2", + _ => "", + }; + if !known.is_empty() { + return known.into(); + } + match &f.name { + Some(n) if !n.is_empty() => n.clone(), + _ => "user-defined".into(), + } +} + +/// `deflate-1 OPT {4}` as h5ls prints a filter. +pub fn filter_text(f: &FilterDescription) -> String { + let opt = if f.flags & 1 != 0 { " OPT" } else { "" }; + let cd = if f.client_data.is_empty() { + String::new() + } else { + format!( + " {{{}}}", + f.client_data + .iter() + .map(|c| c.to_string()) + .collect::>() + .join(", ") + ) + }; + format!("{}-{}{opt}{cd}", filter_name(f), f.filter_id) +} + +pub fn layout_name(l: &DataLayout) -> &'static str { + match l { + DataLayout::Compact { .. } => "compact", + DataLayout::Contiguous { .. } => "contiguous", + DataLayout::Chunked { .. } => "chunked", + DataLayout::Virtual { .. } => "virtual", + } +} + +/// Chunk index kind of a chunked layout. +pub fn chunk_index_name(l: &DataLayout) -> &'static str { + match l { + DataLayout::Chunked { + version, + chunk_index_type, + .. + } => { + if *version < 4 { + return "v1 B-tree"; + } + match chunk_index_type { + Some(1) => "single chunk", + Some(2) => "implicit", + Some(3) => "fixed array", + Some(4) => "extensible array", + Some(5) => "v2 B-tree", + _ => "unknown", + } + } + _ => "", + } +} + +/// Everything about one dataset that can be learned without reading its +/// values. +pub struct DsInfo { + pub dt: Result, + pub ds: Result, + pub layout: Result, + pub filters: Result>, + pub external: bool, +} + +impl DsInfo { + pub fn read(h5: &H5, path: &str, h: &ObjectHeader) -> DsInfo { + DsInfo { + dt: h5.datatype(h), + ds: h5.resolved_dataspace(path, h), + layout: h5.layout(h), + filters: h5.filters(h), + external: h + .messages + .iter() + .any(|m| m.msg_type == MessageType::ExternalDataFiles), + } + } + + pub fn logical_bytes(&self) -> Option { + let (Ok(dt), Ok(ds)) = (&self.dt, &self.ds) else { + return None; + }; + crate::h5::byte_len(ds, dt).ok() + } +} + +/// Every allocated chunk (empty when none are). Errors only for a corrupt +/// chunk index. +pub fn chunks( + h5: &H5, + layout: &DataLayout, + ds: &Dataspace, + dt: &Datatype, +) -> Result> { + let DataLayout::Chunked { btree_address, .. } = layout else { + return Ok(Vec::new()); + }; + let Some(addr) = *btree_address else { + return Ok(Vec::new()); + }; + list_chunks( + h5.data(), + layout, + ds, + dt.type_size() as usize, + h5.os(), + h5.ls(), + ) + .map(|(c, _)| c) + .map_err(|e| { + Error::at( + addr, + format!("chunk index ({}): {e}", chunk_index_name(layout)), + ) + }) +} + +/// Bytes of raw data the dataset has allocated in the file (what libhdf5's +/// `H5Dget_storage_size` reports). +pub fn allocated_bytes(h5: &H5, info: &DsInfo) -> Result { + let layout = info.layout.as_ref().map_err(Clone::clone)?; + Ok(match layout { + DataLayout::Compact { data } => data.len() as u64, + DataLayout::Contiguous { address, size } => { + if address.is_some() { + *size + } else { + 0 + } + } + DataLayout::Chunked { .. } => { + let dt = info.dt.as_ref().map_err(Clone::clone)?; + let ds = info.ds.as_ref().map_err(Clone::clone)?; + chunks(h5, layout, ds, dt)? + .iter() + .map(|c| u64::from(c.chunk_size)) + .sum() + } + DataLayout::Virtual { .. } => 0, + }) +} + +/// Number of hard links to an object, as its header records it. +pub fn link_count(h: &ObjectHeader) -> u64 { + if let Some(rc) = h.reference_count { + return u64::from(rc); + } + h.messages + .iter() + .find(|m| m.msg_type == MessageType::ObjectReferenceCount) + .and_then(|m| m.data.get(1..5)) + .map(|b| u64::from(u32::from_le_bytes([b[0], b[1], b[2], b[3]]))) + .unwrap_or(1) +} diff --git a/crates/clawhdf5-tools/src/lib.rs b/crates/clawhdf5-tools/src/lib.rs new file mode 100644 index 0000000..693ea9a --- /dev/null +++ b/crates/clawhdf5-tools/src/lib.rs @@ -0,0 +1,63 @@ +//! `h5rs`: HDF5 command-line tools built on clawhdf5 alone, no libhdf5. +//! +//! The binary's subcommands (`ls`, `dump`, `stat`, `diff`, `check`) live in +//! the modules below; [`run`] dispatches to them. See the crate README for +//! the command reference. + +pub mod check; +pub mod cli; +pub mod diff; +pub mod dtype; +pub mod dump; +pub mod h5; +pub mod heap_blocks; +pub mod info; +pub mod ls; +pub mod stat; +pub mod value; + +use cli::{Args, Out}; + +pub const USAGE: &str = "\ +h5rs: HDF5 tools in pure Rust (clawhdf5, no libhdf5) + +usage: h5rs [options] ... + +commands: + ls list objects (like h5ls) + dump print a file's structure and values as DDL or JSON (like h5dump) + stat object, layout, filter and storage statistics (like h5stat) + diff compare two files (like h5diff) + check validate a file's structure and checksums + +Run `h5rs --help` for a command's options."; + +/// Run `h5rs` with `argv` (without the program name). Returns the exit +/// status. +pub fn run(argv: Vec, out: &mut Out) -> std::io::Result { + let mut it = argv.into_iter(); + let Some(cmd) = it.next() else { + writeln!(out.e, "{USAGE}")?; + return Ok(2); + }; + let rest: Vec = it.collect(); + match cmd.as_str() { + "ls" => ls::run(&mut Args::new("ls", rest), out), + "dump" => dump::run(&mut Args::new("dump", rest), out), + "stat" => stat::run(&mut Args::new("stat", rest), out), + "diff" => diff::run(&mut Args::new("diff", rest), out), + "check" => check::run(&mut Args::new("check", rest), out), + "-h" | "--help" | "help" => { + writeln!(out.o, "{USAGE}")?; + Ok(0) + } + "-V" | "--version" => { + writeln!(out.o, "h5rs {}", env!("CARGO_PKG_VERSION"))?; + Ok(0) + } + other => { + writeln!(out.e, "h5rs: unknown command {other:?}\n\n{USAGE}")?; + Ok(2) + } + } +} diff --git a/crates/clawhdf5-tools/src/ls.rs b/crates/clawhdf5-tools/src/ls.rs new file mode 100644 index 0000000..ef966f6 --- /dev/null +++ b/crates/clawhdf5-tools/src/ls.rs @@ -0,0 +1,385 @@ +//! `h5rs ls`: list a file's objects like h5ls. + +use std::collections::HashMap; + +use clawhdf5_format::attribute::AttributeMessage; +use clawhdf5_format::dataspace::DataspaceType; +use clawhdf5_format::object_header::ObjectHeader; + +use crate::cli::{Args, Out}; +use crate::h5::{H5, Kind, Link, LinkKind, split_file_arg}; +use crate::info::{self, DsInfo}; +use crate::value::{self, Decoder}; + +pub const USAGE: &str = "\ +usage: h5rs ls [-r] [-v] [--max-bytes N] FILE[/OBJECT] + +List the objects in FILE (or in the group OBJECT), one per line, like h5ls: +name, kind, shape and, for datasets, the datatype. + + -r, --recursive list every object below, with full paths + -v, --verbose also print address, link count, layout, chunking, + storage, filters, datatype and attributes + --max-bytes N largest attribute value decoded for -v (default 1 GiB) + +Exit status: 0 listed, 1 some object could not be described, 2 error."; + +struct Ls<'a> { + h5: &'a H5, + verbose: bool, + kinds: HashMap, + problems: usize, +} + +pub fn run(args: &mut Args, out: &mut Out) -> std::io::Result { + let mut recursive = false; + let mut verbose = false; + let mut max_bytes = None; + let mut target = None; + while let Some(a) = args.next() { + match a.as_str() { + "-r" | "--recursive" => recursive = true, + "-v" | "--verbose" => verbose = true, + "-rv" | "-vr" => { + recursive = true; + verbose = true; + } + "--max-bytes" => match args.number() { + Some(n) => max_bytes = Some(n), + None => return args.usage_error(out, "--max-bytes needs a number", USAGE), + }, + "-h" | "--help" => { + writeln!(out.o, "{USAGE}")?; + return Ok(0); + } + s if s.starts_with('-') && s.len() > 1 => { + return args.usage_error(out, &format!("unknown option {a}"), USAGE); + } + _ if target.is_none() => target = Some(a), + _ => return args.usage_error(out, &format!("unexpected argument {a}"), USAGE), + } + } + let Some(target) = target else { + return args.usage_error(out, "missing FILE", USAGE); + }; + let (file, obj) = split_file_arg(&target); + let mut h5 = match H5::open(std::path::Path::new(&file)) { + Ok(h) => h, + Err(e) => { + writeln!(out.e, "h5rs ls: {e}")?; + return Ok(2); + } + }; + if let Some(m) = max_bytes { + h5.max_bytes = m; + } + let mut ls = Ls { + h5: &h5, + verbose, + kinds: HashMap::new(), + problems: 0, + }; + let obj_path = obj.unwrap_or_else(|| "/".into()); + let addr = match h5.resolve(&obj_path) { + Ok(a) => a, + Err(e) => { + writeln!(out.e, "h5rs ls: {obj_path}: not found: {e}")?; + return Ok(2); + } + }; + let header = match h5.header(addr) { + Ok(h) => h, + Err(e) => { + writeln!(out.e, "h5rs ls: {obj_path}: {e}")?; + return Ok(2); + } + }; + let is_root = obj_path.trim_matches('/').is_empty(); + if Kind::of(&header) != Kind::Group && !is_root { + // h5ls names a single object by its base name, or with -r by its + // path as given (without the leading slash). + let name = if recursive { + obj_path.trim_start_matches('/').to_string() + } else { + obj_path.rsplit('/').next().unwrap_or(&obj_path).to_string() + }; + ls.object_line(out, &name, &obj_path, addr, &Ok(header))?; + } else if recursive { + let start = if is_root { "/" } else { "" }; + let mut io_err = None; + let res = h5.walk_from(addr, start, &mut |item| { + if io_err.is_some() || (item.link.is_none() && !is_root) { + return; // listing a group's contents, not the group + } + let path = if item.path.is_empty() { "/" } else { item.path }; + let full = if is_root { + path.to_string() + } else { + format!("{}{path}", obj_path.trim_end_matches('/')) + }; + if let Err(e) = ls.item( + out, + path, + &full, + item.link, + item.addr, + item.first_path, + item.header, + ) { + io_err = Some(e); + } + }); + if let Some(e) = io_err { + return Err(e); + } + if let Err(e) = res { + writeln!(out.e, "h5rs ls: {e}")?; + ls.problems += 1; + } + } else { + match h5.links(&header) { + Ok(links) => { + for l in &links { + let (a, hdr) = match l.kind { + LinkKind::Hard(a) => (Some(a), Some(h5.header(a))), + _ => (None, None), + }; + let full = format!("{}/{}", obj_path.trim_end_matches('/'), l.name); + ls.item(out, &l.name, &full, Some(l), a, None, hdr.as_ref())?; + } + } + Err(e) => { + writeln!(out.e, "h5rs ls: {obj_path}: cannot list group: {e}")?; + ls.problems += 1; + } + } + } + Ok(if ls.problems > 0 { 1 } else { 0 }) +} + +impl Ls<'_> { + #[allow(clippy::too_many_arguments)] + fn item( + &mut self, + out: &mut Out, + name: &str, + full: &str, + link: Option<&Link>, + addr: Option, + first: Option<&str>, + header: Option<&crate::h5::Result>, + ) -> std::io::Result<()> { + if let Some(l) = link { + match &l.kind { + LinkKind::Soft(t) => return writeln!(out.o, "{name:<24} Soft Link {{{t}}}"), + LinkKind::External { file, path } => { + return writeln!(out.o, "{name:<24} External Link {{{file}/{path}}}"); + } + LinkKind::UserDefined(t) => { + return writeln!(out.o, "{name:<24} User-defined link (type {t})"); + } + LinkKind::Hard(_) => {} + } + } + let Some(addr) = addr else { return Ok(()) }; + if let Some(first) = first { + let k = self.kinds.get(&addr).copied().unwrap_or(Kind::Unknown); + return writeln!(out.o, "{name:<24} {}, same as {first}", k.name()); + } + match header { + Some(h) => self.object_line(out, name, full, addr, h), + None => Ok(()), + } + } + + fn object_line( + &mut self, + out: &mut Out, + name: &str, + full: &str, + addr: u64, + header: &crate::h5::Result, + ) -> std::io::Result<()> { + let h = match header { + Ok(h) => h, + Err(e) => { + self.problems += 1; + writeln!(out.o, "{name:<24} ** error **")?; + return writeln!(out.e, "h5rs ls: {name}: {e}"); + } + }; + let kind = Kind::of(h); + self.kinds.insert(addr, kind); + match kind { + Kind::Dataset => { + let info = DsInfo::read(self.h5, full, h); + let shape = match &info.ds { + Ok(ds) => info::shape_text(ds, self.verbose), + Err(_) => "{?}".into(), + }; + if self.verbose { + writeln!(out.o, "{name:<24} Dataset {shape}")?; + } else { + let t = match &info.dt { + Ok(dt) => crate::dtype::short(dt), + Err(_) => "?".into(), + }; + writeln!(out.o, "{name:<24} Dataset {shape} {t}")?; + } + for e in [ + info.dt.as_ref().err(), + info.ds.as_ref().err(), + info.layout.as_ref().err(), + info.filters.as_ref().err(), + ] + .into_iter() + .flatten() + { + self.problems += 1; + writeln!(out.e, "h5rs ls: {name}: {e}")?; + } + if self.verbose { + self.dataset_details(out, addr, h, &info)?; + } + } + k => { + writeln!(out.o, "{name:<24} {}", k.name())?; + if self.verbose { + writeln!(out.o, " Address: {addr}")?; + writeln!(out.o, " Links: {}", info::link_count(h))?; + if k == Kind::Datatype + && let Ok(dt) = self.h5.datatype(h) + { + writeln!(out.o, " Type: {}", crate::dtype::long(&dt))?; + } + self.attributes(out, name, h)?; + } + } + } + Ok(()) + } + + fn dataset_details( + &mut self, + out: &mut Out, + addr: u64, + h: &ObjectHeader, + info: &DsInfo, + ) -> std::io::Result<()> { + writeln!(out.o, " Address: {addr}")?; + writeln!(out.o, " Links: {}", info::link_count(h))?; + if let Ok(l) = &info.layout { + let idx = info::chunk_index_name(l); + if idx.is_empty() { + writeln!(out.o, " Layout: {}", info::layout_name(l))?; + } else { + writeln!( + out.o, + " Layout: {} ({idx} index)", + info::layout_name(l) + )?; + } + if let clawhdf5_format::data_layout::DataLayout::Chunked { + chunk_dimensions, .. + } = l + { + let rank = chunk_dimensions.len().saturating_sub(1); + let dims = &chunk_dimensions[..rank]; + let esize = info + .dt + .as_ref() + .map(|d| u64::from(d.type_size())) + .unwrap_or(0); + let bytes = dims + .iter() + .try_fold(esize, |a, &d| a.checked_mul(u64::from(d))) + .map(|b| b.to_string()) + .unwrap_or_else(|| "?".into()); + writeln!( + out.o, + " Chunks: {{{}}} {bytes} bytes", + dims.iter() + .map(|d| d.to_string()) + .collect::>() + .join(", ") + )?; + } + } + if info.external { + writeln!(out.o, " External: raw data in external files")?; + } + let logical = info.logical_bytes(); + match (logical, info::allocated_bytes(self.h5, info)) { + (Some(l), Ok(a)) => { + if a > 0 { + writeln!( + out.o, + " Storage: {l} logical bytes, {a} allocated bytes, {:.2}% utilization", + l as f64 * 100.0 / a as f64 + )?; + } else { + writeln!(out.o, " Storage: {l} logical bytes, 0 allocated bytes")?; + } + } + (_, Err(e)) => { + self.problems += 1; + writeln!(out.e, "h5rs ls: {e}")?; + } + _ => {} + } + if let Ok(Some(p)) = &info.filters { + for (i, f) in p.filters.iter().enumerate() { + writeln!(out.o, " Filter-{i}: {}", info::filter_text(f))?; + } + } + if let Ok(dt) = &info.dt { + writeln!(out.o, " Type: {}", crate::dtype::long(dt))?; + } + self.attributes(out, "", h) + } + + fn attributes(&mut self, out: &mut Out, name: &str, h: &ObjectHeader) -> std::io::Result<()> { + let (attrs, errs) = match self.h5.attributes(h) { + Ok(x) => x, + Err(e) => { + self.problems += 1; + return writeln!(out.e, "h5rs ls: {name}: {e}"); + } + }; + for e in errs { + self.problems += 1; + writeln!(out.e, "h5rs ls: {name}: attribute: {e}")?; + } + for a in &attrs { + self.attribute(out, a)?; + } + Ok(()) + } + + fn attribute(&mut self, out: &mut Out, a: &AttributeMessage) -> std::io::Result<()> { + let shape = match a.dataspace.space_type { + DataspaceType::Scalar => "scalar".to_string(), + DataspaceType::Null => "null".to_string(), + DataspaceType::Simple => info::shape_text(&a.dataspace, false), + }; + writeln!(out.o, " Attribute: {} {shape}", a.name)?; + writeln!( + out.o, + " Type: {}", + crate::dtype::long(&a.datatype) + )?; + let n = crate::h5::num_elements(&a.dataspace).unwrap_or(0); + if n == 0 { + return Ok(()); + } + let dec = Decoder::new(self.h5); + let shown = n.min(8) as usize; + let mut vals = Vec::with_capacity(shown); + for i in 0..shown { + let v = dec.element(&a.datatype, &a.raw_data, i); + vals.push(value::text(&v, &|_| None)); + } + let more = if n > 8 { ", ..." } else { "" }; + writeln!(out.o, " Data: {}{more}", vals.join(", ")) + } +} diff --git a/crates/clawhdf5-tools/src/main.rs b/crates/clawhdf5-tools/src/main.rs new file mode 100644 index 0000000..4e6d431 --- /dev/null +++ b/crates/clawhdf5-tools/src/main.rs @@ -0,0 +1,58 @@ +//! The `h5rs` binary. A panic anywhere is a bug: it is caught, reported as +//! an internal error and turned into exit status 3, never a crash. + +use std::io::Write; +use std::panic::{self, AssertUnwindSafe}; + +use clawhdf5_tools::cli::Out; + +/// Exit status for an internal error (a caught panic). +const INTERNAL_ERROR: i32 = 3; + +fn main() { + panic::set_hook(Box::new(|info| { + let msg = if let Some(s) = info.payload().downcast_ref::<&str>() { + (*s).to_string() + } else if let Some(s) = info.payload().downcast_ref::() { + s.clone() + } else { + "unknown panic".into() + }; + let at = info + .location() + .map(|l| format!(" at {}:{}", l.file(), l.line())) + .unwrap_or_default(); + eprintln!("h5rs: internal error (this is a bug; please report it): {msg}{at}"); + })); + let argv: Vec = std::env::args().skip(1).collect(); + let stdout = std::io::stdout(); + let mut o = std::io::BufWriter::new(stdout.lock()); + let mut e = std::io::stderr(); + let result = panic::catch_unwind(AssertUnwindSafe(|| { + let mut out = Out { + o: &mut o, + e: &mut e, + }; + clawhdf5_tools::run(argv, &mut out) + })); + let code = match result { + Ok(Ok(code)) => match o.flush() { + Ok(()) => code, + Err(err) if err.kind() == std::io::ErrorKind::BrokenPipe => code, + Err(err) => { + eprintln!("h5rs: {err}"); + 2 + } + }, + Ok(Err(err)) if err.kind() == std::io::ErrorKind::BrokenPipe => 0, + Ok(Err(err)) => { + eprintln!("h5rs: {err}"); + 2 + } + Err(_) => { + let _ = o.flush(); + INTERNAL_ERROR + } + }; + std::process::exit(code); +} diff --git a/crates/clawhdf5-tools/src/stat.rs b/crates/clawhdf5-tools/src/stat.rs new file mode 100644 index 0000000..c7e44da --- /dev/null +++ b/crates/clawhdf5-tools/src/stat.rs @@ -0,0 +1,305 @@ +//! `h5rs stat`: object, layout, filter and storage statistics, laid out like +//! h5stat's report. + +use std::collections::BTreeMap; + +use clawhdf5_format::data_layout::DataLayout; +use clawhdf5_format::dataspace::DataspaceType; + +use crate::cli::{Args, Out}; +use crate::h5::{H5, Kind, LinkKind}; +use crate::info::{self, DsInfo}; + +pub const USAGE: &str = "\ +usage: h5rs stat FILE + +Print statistics for FILE, in the layout of h5stat's report: object counts, +links, dataset ranks, layouts, filters, attribute counts, raw-data size and +a file-space summary. Metadata space is not broken down by structure (h5stat +does); it is reported as one figure together with free space. + +Exit status: 0 printed, 1 some object could not be read (counted as +\"unreadable\"), 2 error."; + +#[derive(Default)] +struct Stats { + groups: u64, + datasets: u64, + datatypes: u64, + other: u64, + unreadable: u64, + links: u64, + max_links_to_object: u64, + max_objects_in_group: u64, + group_sizes: BTreeMap, + ranks: BTreeMap, + max_1d: u64, + layout: [u64; 4], + external: u64, + no_filter: u64, + filter_counts: BTreeMap<&'static str, u64>, + raw_data: u64, + raw_errors: u64, + attr_objects: u64, + max_attrs: u64, + attr_counts: BTreeMap, +} + +const FILTER_KINDS: [&str; 8] = [ + "GZIP", + "SHUFFLE", + "FLETCHER32", + "SZIP", + "NBIT", + "SCALEOFFSET", + "USER-DEFINED", + "", +]; + +fn filter_kind(id: u16) -> &'static str { + match id { + 1 => "GZIP", + 2 => "SHUFFLE", + 3 => "FLETCHER32", + 4 => "SZIP", + 5 => "NBIT", + 6 => "SCALEOFFSET", + _ => "USER-DEFINED", + } +} + +pub fn run(args: &mut Args, out: &mut Out) -> std::io::Result { + let mut file = None; + while let Some(a) = args.next() { + match a.as_str() { + "-h" | "--help" => { + writeln!(out.o, "{USAGE}")?; + return Ok(0); + } + s if s.starts_with('-') && s.len() > 1 => { + return args.usage_error(out, &format!("unknown option {a}"), USAGE); + } + _ if file.is_none() => file = Some(a), + _ => return args.usage_error(out, &format!("unexpected argument {a}"), USAGE), + } + } + let Some(file) = file else { + return args.usage_error(out, "missing FILE", USAGE); + }; + let h5 = match H5::open(std::path::Path::new(&file)) { + Ok(h) => h, + Err(e) => { + writeln!(out.e, "h5rs stat: {e}")?; + return Ok(2); + } + }; + let mut s = Stats::default(); + let mut errors: Vec = Vec::new(); + let walk = h5.walk(|it| { + if let Some(l) = it.link + && !matches!(l.kind, LinkKind::Hard(_)) + { + s.links += 1; + return; + } + if it.first_path.is_some() { + return; + } + let Some(Ok(h)) = it.header else { + if let Some(Err(e)) = it.header { + s.unreadable += 1; + errors.push(format!("{}: {e}", it.path)); + } + return; + }; + s.max_links_to_object = s.max_links_to_object.max(info::link_count(h)); + match h5.attributes(h) { + Ok((a, errs)) => { + let n = a.len() + errs.len(); + if n > 0 { + s.attr_objects += 1; + s.max_attrs = s.max_attrs.max(n as u64); + *s.attr_counts.entry(n).or_default() += 1; + } + for e in errs { + errors.push(format!("{}: attribute: {e}", it.path)); + } + } + Err(e) => errors.push(format!("{}: {e}", it.path)), + } + match Kind::of(h) { + Kind::Group => { + s.groups += 1; + match h5.links(h) { + Ok(l) => { + s.max_objects_in_group = s.max_objects_in_group.max(l.len() as u64); + *s.group_sizes.entry(l.len()).or_default() += 1; + } + Err(e) => { + s.unreadable += 1; + errors.push(format!("{}: {e}", it.path)); + } + } + } + Kind::Datatype => s.datatypes += 1, + Kind::Unknown => { + // The root group of a file with no links at all has no + // group messages; count it as a group. + if it.link.is_none() { + s.groups += 1; + *s.group_sizes.entry(0).or_default() += 1; + } else { + s.other += 1; + } + } + Kind::Dataset => { + s.datasets += 1; + let info = DsInfo::read(&h5, it.path, h); + if let Ok(ds) = &info.ds { + let rank = match ds.space_type { + DataspaceType::Simple => ds.dimensions.len(), + _ => 0, + }; + *s.ranks.entry(rank).or_default() += 1; + if rank == 1 { + s.max_1d = s.max_1d.max(ds.dimensions[0]); + } + } + match &info.layout { + Ok(l) => { + let i = match l { + DataLayout::Compact { .. } => 0, + DataLayout::Contiguous { .. } => 1, + DataLayout::Chunked { .. } => 2, + DataLayout::Virtual { .. } => 3, + }; + s.layout[i] += 1; + } + Err(e) => errors.push(format!("{}: {e}", it.path)), + } + if info.external { + s.external += 1; + } + match &info.filters { + Ok(Some(p)) if !p.filters.is_empty() => { + let mut kinds: Vec<&'static str> = + p.filters.iter().map(|f| filter_kind(f.filter_id)).collect(); + kinds.sort_unstable(); + kinds.dedup(); + for k in kinds { + *s.filter_counts.entry(k).or_default() += 1; + } + } + Ok(_) => s.no_filter += 1, + Err(e) => errors.push(format!("{}: {e}", it.path)), + } + match info::allocated_bytes(&h5, &info) { + Ok(b) => s.raw_data = s.raw_data.saturating_add(b), + Err(e) => { + s.raw_errors += 1; + errors.push(format!("{}: storage size: {e}", it.path)); + } + } + } + } + }); + if let Err(e) = walk { + errors.push(e.to_string()); + } + report(&h5, &file, &s, out)?; + for e in &errors { + writeln!(out.e, "h5rs stat: {e}")?; + } + Ok(if errors.is_empty() { 0 } else { 1 }) +} + +fn report(h5: &H5, file: &str, s: &Stats, out: &mut Out) -> std::io::Result<()> { + let o = &mut out.o; + writeln!(o, "Filename: {file}")?; + writeln!(o, "File information")?; + writeln!(o, "\t# of unique groups: {}", s.groups)?; + writeln!(o, "\t# of unique datasets: {}", s.datasets)?; + writeln!(o, "\t# of unique named datatypes: {}", s.datatypes)?; + writeln!(o, "\t# of unique links: {}", s.links)?; + writeln!(o, "\t# of unique other: {}", s.other)?; + if s.unreadable > 0 { + writeln!(o, "\t# of unreadable objects: {}", s.unreadable)?; + } + writeln!(o, "\tMax. # of links to object: {}", s.max_links_to_object)?; + writeln!( + o, + "\tMax. # of objects in group: {}", + s.max_objects_in_group + )?; + let sb = h5.sb(); + writeln!(o, "Superblock:")?; + writeln!(o, "\tVersion: {}", sb.version)?; + writeln!(o, "\tSize of offsets: {} bytes", sb.offset_size)?; + writeln!(o, "\tSize of lengths: {} bytes", sb.length_size)?; + writeln!(o, "\tUser block: {} bytes", h5.file.user_block_size())?; + writeln!(o, "Group bins:")?; + for (n, c) in &s.group_sizes { + writeln!(o, "\t# of groups with {n} link(s): {c}")?; + } + writeln!(o, "\tTotal # of groups: {}", s.groups)?; + writeln!(o, "Dataset dimension information:")?; + writeln!( + o, + "\tMax. rank of datasets: {}", + s.ranks.keys().next_back().copied().unwrap_or(0) + )?; + writeln!(o, "\tDataset ranks:")?; + for (r, c) in &s.ranks { + writeln!(o, "\t\t# of dataset with rank {r}: {c}")?; + } + writeln!(o, "1-D Dataset information:")?; + writeln!(o, "\tMax. dimension size of 1-D datasets: {}", s.max_1d)?; + writeln!(o, "Dataset storage information:")?; + writeln!(o, "\tTotal raw data size: {}", s.raw_data)?; + if s.raw_errors > 0 { + writeln!( + o, + "\tDatasets whose storage size could not be read: {}", + s.raw_errors + )?; + } + writeln!(o, "Dataset layout information:")?; + for (i, n) in ["COMPACT", "CONTIG", "CHUNKED", "VIRTUAL"] + .iter() + .enumerate() + { + writeln!(o, "\tDataset layout counts[{n}]: {}", s.layout[i])?; + } + writeln!(o, "\tDatasets with external raw data: {}", s.external)?; + writeln!(o, "Dataset filters information:")?; + writeln!(o, "\tNumber of datasets with:")?; + writeln!(o, "\t\tNO filter: {}", s.no_filter)?; + for k in FILTER_KINDS.iter().filter(|k| !k.is_empty()) { + writeln!( + o, + "\t\t{k} filter: {}", + s.filter_counts.get(k).copied().unwrap_or(0) + )?; + } + writeln!(o, "Attribute information:")?; + for (n, c) in &s.attr_counts { + writeln!(o, "\t# of objects with {n} attribute(s): {c}")?; + } + writeln!( + o, + "\tTotal # of objects with attributes: {}", + s.attr_objects + )?; + writeln!(o, "\tMax. # of attributes to objects: {}", s.max_attrs)?; + let total = std::fs::metadata(&h5.path).map(|m| m.len()).unwrap_or(0); + let ub = h5.file.user_block_size(); + writeln!(o, "Summary of file space information:")?; + writeln!(o, " User block: {ub} bytes")?; + writeln!(o, " Raw data: {} bytes", s.raw_data)?; + writeln!( + o, + " Metadata and free space: {} bytes", + total.saturating_sub(s.raw_data).saturating_sub(ub) + )?; + writeln!(o, "Total space: {total} bytes") +} diff --git a/crates/clawhdf5-tools/src/value.rs b/crates/clawhdf5-tools/src/value.rs new file mode 100644 index 0000000..8160574 --- /dev/null +++ b/crates/clawhdf5-tools/src/value.rs @@ -0,0 +1,483 @@ +//! Decoding one element of any datatype into a [`Value`], and printing it. +//! +//! Decoding never panics: a short buffer, an unknown byte order or a +//! dangling heap reference becomes [`Value::Error`]. + +use clawhdf5_format::datatype::{Datatype, DatatypeByteOrder, ReferenceType, StringPadding}; +use serde_json::Value as J; + +use crate::dtype; +use crate::h5::H5; + +#[derive(Debug, Clone, PartialEq)] +pub enum Value { + Int(i128), + /// A float and the width (in bits) it was stored with, so it prints at + /// its own precision. + Float(f64, u8), + Str(String), + /// Opaque, bitfield, time and oversized integers. + Bytes(Vec), + /// An enum member (name, when the value matches one) and its value. + Enum(Option, i128), + Compound(Vec<(String, Value)>), + Array(Vec), + /// A variable-length sequence. + Seq(Vec), + /// A reference: the referenced object's address (`None` = null). + Ref(Option), + /// A region or attribute reference, kept as its bytes. + OtherRef(Vec), + Error(String), +} + +pub fn hex(b: &[u8]) -> String { + let mut s = String::from("0x"); + for x in b { + s.push_str(&format!("{x:02x}")); + } + s +} + +/// Element bytes as an unsigned integer (at most 16 bytes). +fn bits(b: &[u8], order: &DatatypeByteOrder) -> Option { + if b.len() > 16 { + return None; + } + let mut v = 0u128; + match order { + DatatypeByteOrder::LittleEndian => { + for (i, x) in b.iter().enumerate() { + v |= u128::from(*x) << (8 * i); + } + } + DatatypeByteOrder::BigEndian => { + for x in b { + v = (v << 8) | u128::from(*x); + } + } + DatatypeByteOrder::Vax => return None, + } + Some(v) +} + +/// The value of an integer (fixed-point) element, honouring its bit offset +/// and precision. `None` when it cannot be represented (over 16 bytes) or +/// `dt` is not an integer. +pub fn decode_int(dt: &Datatype, b: &[u8]) -> Option { + let Datatype::FixedPoint { + size, + byte_order, + signed, + bit_offset, + bit_precision, + } = dt + else { + return None; + }; + let size = usize::try_from(*size).ok()?; + let v = bits(b.get(..size)?, byte_order)?; + let off = u32::from(*bit_offset); + let prec = u32::from(*bit_precision).min(128); + if off >= 128 || prec == 0 { + return Some(0); + } + let mut x = v >> off; + if prec < 128 { + x &= (1u128 << prec) - 1; + } + if *signed && prec < 128 && (x >> (prec - 1)) & 1 == 1 { + x |= !0u128 << prec; + return Some(x as i128); + } + if !*signed && prec == 128 && x > i128::MAX as u128 { + return None; + } + Some(x as i128) +} + +fn decode_float(dt: &Datatype, b: &[u8]) -> Value { + let Datatype::FloatingPoint { + size, byte_order, .. + } = dt + else { + return Value::Error("not a float".into()); + }; + let Ok(n) = usize::try_from(*size) else { + return Value::Error("float size".into()); + }; + let Some(b) = b.get(..n) else { + return Value::Error("short element".into()); + }; + if dtype::is_ieee(dt) + && let Some(v) = bits(b, byte_order) + { + return match n { + 2 => Value::Float( + f64::from(clawhdf5_format::float16::f16_bits_to_f32(v as u16)), + 16, + ), + 4 => Value::Float(f64::from(f32::from_bits(v as u32)), 32), + _ => Value::Float(f64::from_bits(v as u64), 64), + }; + } + // Non-IEEE layouts (N-Bit floats, VAX order): the library converts. + match clawhdf5_format::data_read::read_as_f64(b, dt) { + Ok(v) if v.len() == 1 => Value::Float(v[0], (n * 8).min(64) as u8), + Ok(_) => Value::Error("float conversion".into()), + Err(e) => Value::Error(e.to_string()), + } +} + +fn trim_string(b: &[u8], pad: Option<&StringPadding>) -> String { + let cut = b.iter().position(|&c| c == 0).unwrap_or(b.len()); + let mut s = &b[..cut]; + if matches!(pad, Some(StringPadding::SpacePad)) { + while let [rest @ .., b' '] = s { + s = rest; + } + } + String::from_utf8_lossy(s).into_owned() +} + +/// Little-endian unsigned integer of `b` (up to 8 bytes). +fn le(b: &[u8]) -> u64 { + b.iter() + .take(8) + .enumerate() + .fold(0u64, |a, (i, &x)| a | (u64::from(x) << (8 * i))) +} + +/// Decodes elements of one file. +pub struct Decoder<'a> { + pub h5: &'a H5, +} + +impl<'a> Decoder<'a> { + pub fn new(h5: &'a H5) -> Self { + Self { h5 } + } + + /// Decode element `i` of `raw`, an array of `dt` elements. + pub fn element(&self, dt: &Datatype, raw: &[u8], i: usize) -> Value { + let size = dt.type_size() as usize; + match i + .checked_mul(size) + .and_then(|s| raw.get(s..s.checked_add(size)?)) + { + Some(b) => self.decode(dt, b, 0), + None => Value::Error("element out of range".into()), + } + } + + pub fn decode(&self, dt: &Datatype, b: &[u8], depth: u32) -> Value { + if depth > 32 { + return Value::Error("datatype nesting too deep".into()); + } + let size = dt.type_size() as usize; + let Some(b) = b.get(..size) else { + return Value::Error("short element".into()); + }; + match dt { + Datatype::FixedPoint { .. } => match decode_int(dt, b) { + Some(v) => Value::Int(v), + None => Value::Bytes(b.to_vec()), + }, + Datatype::FloatingPoint { .. } => decode_float(dt, b), + Datatype::Time { .. } | Datatype::BitField { .. } | Datatype::Opaque { .. } => { + Value::Bytes(b.to_vec()) + } + Datatype::String { padding, .. } => Value::Str(trim_string(b, Some(padding))), + Datatype::Compound { members, .. } => { + let mut out = Vec::with_capacity(members.len()); + for m in members { + let off = usize::try_from(m.byte_offset).unwrap_or(usize::MAX); + let v = match b.get(off..) { + Some(mb) => self.decode(&m.datatype, mb, depth + 1), + None => Value::Error("member out of bounds".into()), + }; + out.push((m.name.clone(), v)); + } + Value::Compound(out) + } + Datatype::Reference { ref_type, .. } => match ref_type { + ReferenceType::Object | ReferenceType::Object2 => { + match clawhdf5_format::data_read::read_object_references(b, dt, self.h5.os()) { + Ok(r) if r.len() == 1 => { + let a = r[0].address; + let undef = a == u64::MAX + || (self.h5.os() < 8 && a == (1u64 << (8 * self.h5.os())) - 1); + Value::Ref(if undef || a == 0 { None } else { Some(a) }) + } + Ok(_) => Value::Error("reference".into()), + Err(e) => Value::Error(e.to_string()), + } + } + _ => Value::OtherRef(b.to_vec()), + }, + Datatype::Enumeration { + base_type, members, .. + } => { + let Some(v) = decode_int(base_type, b) else { + return Value::Bytes(b.to_vec()); + }; + let bs = base_type.type_size() as usize; + let name = members + .iter() + .find(|m| m.value.get(..bs) == b.get(..bs)) + .map(|m| m.name.clone()); + Value::Enum(name, v) + } + Datatype::Array { + base_type, + dimensions, + } => { + let n = dimensions + .iter() + .try_fold(1usize, |a, &d| a.checked_mul(d as usize)); + let bs = base_type.type_size() as usize; + let Some(n) = n.filter(|n| n.checked_mul(bs).is_some_and(|t| t <= b.len())) else { + return Value::Error("array larger than its element".into()); + }; + let mut out = Vec::with_capacity(n); + for k in 0..n { + out.push(self.decode(base_type, &b[k * bs..], depth + 1)); + } + Value::Array(out) + } + Datatype::VariableLength { + is_string, + padding, + base_type, + .. + } => self.decode_vlen(*is_string, padding.as_ref(), base_type, b, depth), + } + } + + fn decode_vlen( + &self, + is_string: bool, + padding: Option<&StringPadding>, + base: &Datatype, + b: &[u8], + depth: u32, + ) -> Value { + let os = usize::from(self.h5.os()); + let (Some(lenb), Some(addrb), Some(idxb)) = + (b.get(..4), b.get(4..4 + os), b.get(4 + os..8 + os)) + else { + return Value::Error("short VL element".into()); + }; + let len = le(lenb) as usize; + let addr = le(addrb); + let idx = le(idxb) as u32; + let undef = if os >= 8 { + u64::MAX + } else { + (1u64 << (8 * os)) - 1 + }; + let obj = if len == 0 || addr == 0 || addr == undef { + Vec::new() + } else { + match self.h5.heap_object(addr, idx) { + Ok(o) => o, + Err(e) => return Value::Error(e.to_string()), + } + }; + if is_string { + let l = len.min(obj.len()); + return Value::Str(trim_string(&obj[..l], padding)); + } + let bs = base.type_size() as usize; + if bs == 0 { + return Value::Error("VL base type of size 0".into()); + } + match len.checked_mul(bs) { + Some(need) if need <= obj.len() => {} + _ => return Value::Error("VL sequence longer than its heap object".into()), + } + let mut out = Vec::with_capacity(len); + for k in 0..len { + out.push(self.decode(base, &obj[k * bs..], depth + 1)); + } + Value::Seq(out) + } +} + +/// Format a float like C's `%g` would at full round-trip precision: plain +/// digits for moderate magnitudes, an exponent otherwise. +pub fn fmt_float(v: f64, width: u8) -> String { + if v.is_nan() { + return "NaN".into(); + } + if v.is_infinite() { + return if v > 0.0 { "Inf".into() } else { "-Inf".into() }; + } + let a = v.abs(); + let plain = a == 0.0 || (1e-5..1e16).contains(&a); + match (width, plain) { + (16 | 32, true) => format!("{}", v as f32), + (16 | 32, false) => format!("{:e}", v as f32), + (_, true) => format!("{v}"), + (_, false) => format!("{v:e}"), + } +} + +fn escape(s: &str) -> String { + let mut o = String::with_capacity(s.len() + 2); + for c in s.chars() { + match c { + '"' => o.push_str("\\\""), + '\\' => o.push_str("\\\\"), + '\n' => o.push_str("\\n"), + '\r' => o.push_str("\\r"), + '\t' => o.push_str("\\t"), + c if (c as u32) < 0x20 => o.push_str(&format!("\\{:03o}", c as u32)), + c => o.push(c), + } + } + o +} + +/// A fixed-length string element's bytes quoted as h5dump prints a +/// null-padded string: every byte, NULs as `\000`. +pub fn quote_bytes(b: &[u8]) -> String { + format!("\"{}\"", escape(&String::from_utf8_lossy(b))) +} + +/// Text form, as in an h5dump DATA block. +pub fn text(v: &Value, h5paths: &dyn Fn(u64) -> Option) -> String { + match v { + Value::Int(i) => i.to_string(), + Value::Float(f, w) => fmt_float(*f, *w), + Value::Str(s) => format!("\"{}\"", escape(s)), + Value::Bytes(b) => hex(b), + Value::Enum(Some(n), _) => n.clone(), + Value::Enum(None, i) => i.to_string(), + Value::Compound(ms) => format!( + "{{ {} }}", + ms.iter() + .map(|(_, v)| text(v, h5paths)) + .collect::>() + .join(", ") + ), + Value::Array(vs) => format!( + "[ {} ]", + vs.iter() + .map(|v| text(v, h5paths)) + .collect::>() + .join(", ") + ), + Value::Seq(vs) => format!( + "({})", + vs.iter() + .map(|v| text(v, h5paths)) + .collect::>() + .join(", ") + ), + Value::Ref(None) => "NULL".into(), + Value::Ref(Some(a)) => match h5paths(*a) { + Some(p) => format!("\"{p}\""), + None => format!("{a:#x}"), + }, + Value::OtherRef(b) => hex(b), + Value::Error(e) => format!(""), + } +} + +/// hdf5-json value form. +pub fn to_json(v: &Value, h5paths: &dyn Fn(u64) -> Option) -> J { + match v { + Value::Int(i) => { + if let Ok(x) = i64::try_from(*i) { + J::from(x) + } else if let Ok(x) = u64::try_from(*i) { + J::from(x) + } else { + J::from(i.to_string()) + } + } + Value::Float(f, _) => { + if f.is_finite() { + serde_json::Number::from_f64(*f) + .map(J::Number) + .unwrap_or(J::Null) + } else if f.is_nan() { + J::from("NaN") + } else if *f > 0.0 { + J::from("Infinity") + } else { + J::from("-Infinity") + } + } + Value::Str(s) => J::from(s.as_str()), + Value::Bytes(b) | Value::OtherRef(b) => J::from(hex(b)), + Value::Enum(_, i) => to_json(&Value::Int(*i), h5paths), + Value::Compound(ms) => J::Array(ms.iter().map(|(_, v)| to_json(v, h5paths)).collect()), + Value::Array(vs) | Value::Seq(vs) => { + J::Array(vs.iter().map(|v| to_json(v, h5paths)).collect()) + } + Value::Ref(None) => J::Null, + Value::Ref(Some(a)) => J::from(h5paths(*a).unwrap_or_else(|| format!("{a:#x}"))), + Value::Error(e) => serde_json::json!({ "error": e }), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn int(size: u32, signed: bool, order: DatatypeByteOrder, off: u16, prec: u16) -> Datatype { + Datatype::FixedPoint { + size, + byte_order: order, + signed, + bit_offset: off, + bit_precision: prec, + } + } + + #[test] + fn integers_decode_with_order_offset_and_sign() { + let le = DatatypeByteOrder::LittleEndian; + let be = DatatypeByteOrder::BigEndian; + assert_eq!( + decode_int(&int(2, true, le.clone(), 0, 16), &[0xff, 0xff]), + Some(-1) + ); + assert_eq!( + decode_int(&int(2, false, le, 0, 16), &[0xff, 0xff]), + Some(65535) + ); + assert_eq!( + decode_int(&int(2, true, be.clone(), 0, 16), &[0x80, 0x00]), + Some(-32768) + ); + // 17-bit signed field at offset 4: -5 + let stored = (((-5i32) as u32) & 0x1_FFFF) << 4; + assert_eq!( + decode_int(&int(4, true, be, 4, 17), &stored.to_be_bytes()), + Some(-5) + ); + assert_eq!( + decode_int(&int(4, true, DatatypeByteOrder::Vax, 0, 32), &[0; 4]), + None + ); + assert_eq!( + decode_int( + &int(4, true, DatatypeByteOrder::LittleEndian, 0, 32), + &[0; 2] + ), + None + ); + } + + #[test] + fn floats_print_at_their_own_precision() { + assert_eq!(fmt_float(f64::from(0.1f32), 32), "0.1"); + assert_eq!(fmt_float(0.1, 64), "0.1"); + assert_eq!(fmt_float(1e20, 64), "1e20"); + assert_eq!(fmt_float(2.0, 64), "2"); + assert_eq!(fmt_float(f64::NAN, 64), "NaN"); + } +} diff --git a/crates/clawhdf5-tools/tests/gen_files.py b/crates/clawhdf5-tools/tests/gen_files.py new file mode 100644 index 0000000..988cb22 --- /dev/null +++ b/crates/clawhdf5-tools/tests/gen_files.py @@ -0,0 +1,115 @@ +"""Write the HDF5 files the h5rs interop tests run on. + +usage: gen_files.py OUTDIR + +Writes OUTDIR/{earliest,latest}.h5 (the same content with the oldest and the +newest file-format structures: symbol tables and v1 B-trees vs. v2 object +headers, fractal heaps, v2 B-trees and the chunk indexes of HDF5 1.10+), the +pairs the diff tests compare, and a file with a user block. Prints one JSON +object with the values h5py reads back, for the dump tests. +""" + +import json +import os +import sys + +import h5py +import numpy as np + +out = sys.argv[1] + + +def content(f, dense): + f.attrs["title"] = "h5rs test" + f.attrs["version"] = np.int64(3) + f.attrs["scale"] = np.array([0.5, 1.5], dtype="f4") + f["contig"] = np.arange(12, dtype="f8").reshape(3, 4) + dcpl = h5py.h5p.create(h5py.h5p.DATASET_CREATE) + dcpl.set_layout(h5py.h5d.COMPACT) + space = h5py.h5s.create_simple((4,)) + h5py.h5d.create(f.id, b"compact", h5py.h5t.STD_I16LE, space, dcpl).write( + h5py.h5s.ALL, h5py.h5s.ALL, np.arange(4, dtype="f4")]) + f.create_dataset( + "enum", data=np.array([0, 1, 1], dtype="u1"), + dtype=h5py.enum_dtype({"RED": 0, "GREEN": 1}, basetype="u1"), + ) + f["be"] = np.arange(5, dtype=">i4") + f["arr"] = np.array([([1, 2, 3],), ([4, 5, 6],)], dtype=[("v", "3i4")]) + f["named_t"] = np.dtype("i8") + f["soft"] = h5py.SoftLink("/contig") + f["dangling"] = h5py.SoftLink("/nowhere") + f["external"] = h5py.ExternalLink("other.h5", "/x") + f["hard2"] = g["sub"] + many = f.create_group("many") + for i in range(12 if dense else 4): + many[f"d{i:02}"] = np.int32(i) + many.attrs[f"a{i:02}"] = i + + +values = {} +# "latest" under HDF5 2.0 writes datatype messages that libhdf5 1.14 tools +# cannot read, so the newest format is taken as 1.14's. +for libver in ("earliest", "latest"): + path = os.path.join(out, f"{libver}.h5") + bounds = ("earliest", "v114") if libver == "earliest" else ("v114", "v114") + with h5py.File(path, "w", libver=bounds) as f: + content(f, True) + with h5py.File(path, "r") as f: + vals = {} + + def grab(name, obj): + if isinstance(obj, h5py.Dataset) and obj.dtype.kind in "iuf" and obj.shape is not None: + vals["/" + name] = obj[()].tolist() + + f.visititems(grab) + values[libver] = vals + +# diff pairs +def small(path, data=None, extra=False, attr=False, shape=(3, 4), dtype="f8"): + with h5py.File(os.path.join(out, path), "w") as f: + d = np.arange(12, dtype=dtype).reshape(shape) if data is None else data + f["d"] = d + f["g/x"] = np.arange(3) + if extra: + f["only_here"] = 1 + if attr: + f["d"].attrs["u"] = 1 + + +base = np.arange(12, dtype="f8").reshape(3, 4) +small("base.h5") +small("same.h5") +changed = base.copy() +changed[0, 2] += 0.001 +changed[2, 3] += 0.001 +small("changed.h5", data=changed) +small("extra.h5", extra=True) +small("attr.h5", attr=True) +small("reshaped.h5", shape=(4, 3)) +small("int.h5", dtype="i4") + +with h5py.File(os.path.join(out, "userblock.h5"), "w", userblock_size=1024) as f: + f["d"] = np.arange(10) + +print(json.dumps(values)) diff --git a/crates/clawhdf5-tools/tests/h5rs_interop.rs b/crates/clawhdf5-tools/tests/h5rs_interop.rs new file mode 100644 index 0000000..2e03b45 --- /dev/null +++ b/crates/clawhdf5-tools/tests/h5rs_interop.rs @@ -0,0 +1,569 @@ +//! `h5rs` against libhdf5's own tools, on files h5py writes +//! (`tests/gen_files.py`): `ls` against h5ls, `stat` against h5stat, `dump` +//! against h5dump, `diff` exit codes against h5diff, `dump --json` values +//! against h5py, and `check` on valid and deliberately corrupted files. +//! +//! Needs python3 with h5py (`CLAWHDF5_PYTHON`) and, for the comparisons, +//! the libhdf5 command-line tools (h5ls, h5stat, h5dump, h5diff) on `PATH`. +//! Each test skips when what it needs is missing, unless +//! `CLAWHDF5_REQUIRE_INTEROP=1`, which turns a skip into a failure. + +use std::collections::BTreeMap; +use std::path::{Path, PathBuf}; +use std::process::{Command, Output}; + +use clawhdf5_format::checksum::jenkins_lookup3; + +fn python() -> String { + std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string()) +} + +fn interop_required() -> bool { + std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1") +} + +fn python_available() -> bool { + Command::new(python()) + .args(["-c", "import h5py, numpy"]) + .output() + .map(|o| o.status.success()) + .unwrap_or(false) +} + +fn tool_available(name: &str) -> bool { + Command::new(name) + .arg("--version") + .output() + .map(|o| o.status.success()) + .unwrap_or(false) +} + +/// Skip (return true) when `what` is unavailable, or fail when interop is +/// required. +fn missing(ok: bool, what: &str) -> bool { + if ok { + return false; + } + assert!( + !interop_required(), + "CLAWHDF5_REQUIRE_INTEROP=1 but {what} is not available" + ); + eprintln!("SKIP: {what} not available"); + true +} + +fn h5rs(args: &[&str]) -> Output { + Command::new(env!("CARGO_BIN_EXE_h5rs")) + .args(args) + .output() + .expect("run h5rs") +} + +fn run(tool: &str, args: &[&str]) -> Output { + Command::new(tool).args(args).output().expect("run tool") +} + +fn stdout(o: &Output) -> String { + String::from_utf8_lossy(&o.stdout).into_owned() +} + +/// The generated files and what h5py read from them. +struct Files { + dir: tempfile::TempDir, + values: serde_json::Value, +} + +impl Files { + fn path(&self, name: &str) -> PathBuf { + self.dir.path().join(name) + } + + fn p(&self, name: &str) -> String { + self.path(name).to_string_lossy().into_owned() + } +} + +fn generate() -> Option { + if missing(python_available(), "python3 with h5py") { + return None; + } + let dir = tempfile::tempdir().unwrap(); + let script = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/gen_files.py"); + let out = Command::new(python()) + .arg(&script) + .arg(dir.path()) + .output() + .expect("run gen_files.py"); + assert!( + out.status.success(), + "gen_files.py failed:\n{}", + String::from_utf8_lossy(&out.stderr) + ); + let values = serde_json::from_slice(&out.stdout).expect("gen_files.py output"); + Some(Files { dir, values }) +} + +const LIBVERS: [&str; 2] = ["earliest.h5", "latest.h5"]; + +// --------------------------------------------------------------------------- +// ls +// --------------------------------------------------------------------------- + +/// Every h5ls line must be the start of the matching h5rs line (h5rs adds +/// the datatype after a dataset's shape). +fn assert_ls_like(ours: &str, reference: &str, what: &str) { + let ours: Vec<&str> = ours.lines().collect(); + let refs: Vec<&str> = reference.lines().collect(); + assert_eq!( + ours.len(), + refs.len(), + "{what}: line count\nh5rs:\n{}\nh5ls:\n{}", + ours.join("\n"), + refs.join("\n") + ); + for (o, r) in ours.iter().zip(&refs) { + let r = r.trim_end(); + assert!( + o.starts_with(r) && (o.len() == r.len() || r.contains("Dataset {")), + "{what}:\n h5rs: {o}\n h5ls: {r}" + ); + } +} + +#[test] +fn ls_matches_h5ls() { + let Some(f) = generate() else { return }; + if missing(tool_available("h5ls"), "h5ls") { + return; + } + for name in LIBVERS.iter().chain(&["userblock.h5"]) { + let p = f.p(name); + for (args, what) in [(vec!["-r"], "recursive"), (vec![], "root")] { + let mut a: Vec<&str> = args.clone(); + a.push(&p); + let ours = h5rs(&[&["ls"], a.as_slice()].concat()); + assert!(ours.status.success(), "{name} {what}: {:?}", ours); + let reference = run("h5ls", &a); + assert_ls_like( + &stdout(&ours), + &stdout(&reference), + &format!("{name} {what}"), + ); + } + } + // A group, and a dataset, named after the file. + for obj in ["/grp", "/grp/sub", "/contig", "/grp/ext2"] { + let arg = format!("{}{obj}", f.p("latest.h5")); + let ours = h5rs(&["ls", &arg]); + let reference = run("h5ls", &[&arg]); + assert_ls_like(&stdout(&ours), &stdout(&reference), &arg); + let ours = h5rs(&["ls", "-r", &arg]); + let reference = run("h5ls", &["-r", &arg]); + assert_ls_like(&stdout(&ours), &stdout(&reference), &format!("-r {arg}")); + } +} + +#[test] +fn ls_verbose_describes_layout_filters_and_attributes() { + let Some(f) = generate() else { return }; + let o = h5rs(&["ls", "-r", "-v", &f.p("latest.h5")]); + assert!(o.status.success()); + let s = stdout(&o); + for want in [ + "Layout: chunked (extensible array index)", + "Layout: chunked (v2 B-tree index)", + "Layout: chunked (fixed array index)", + "Layout: chunked (single chunk index)", + "Filter-0: shuffle-2", + "Filter-1: deflate-1", + "Filter-2: fletcher32-3", + "Chunks: {100} 400 bytes", + "Attribute: units scalar", + "Data: \"m\"", + ] { + assert!(s.contains(want), "missing {want:?} in\n{s}"); + } + let o = h5rs(&["ls", "-v", &format!("{}/grp", f.p("earliest.h5"))]); + assert!(stdout(&o).contains("Layout: chunked (v1 B-tree index)")); +} + +// --------------------------------------------------------------------------- +// stat +// --------------------------------------------------------------------------- + +/// `key: value` lines of an h5stat-style report. +fn stat_facts(s: &str) -> BTreeMap { + s.lines() + .filter_map(|l| { + let (k, v) = l.rsplit_once(':')?; + Some((k.trim().to_string(), v.trim().to_string())) + }) + .collect() +} + +#[test] +fn stat_matches_h5stat() { + let Some(f) = generate() else { return }; + if missing(tool_available("h5stat"), "h5stat") { + return; + } + let keys = [ + "# of unique groups", + "# of unique datasets", + "# of unique named datatypes", + "# of unique links", + "# of unique other", + "Max. # of links to object", + "Max. # of objects in group", + "Max. rank of datasets", + "Max. dimension size of 1-D datasets", + "Total raw data size", + "Dataset layout counts[COMPACT]", + "Dataset layout counts[CONTIG]", + "Dataset layout counts[CHUNKED]", + "Dataset layout counts[VIRTUAL]", + "NO filter", + "GZIP filter", + "SHUFFLE filter", + "FLETCHER32 filter", + "SZIP filter", + "NBIT filter", + "SCALEOFFSET filter", + "USER-DEFINED filter", + "Max. # of attributes to objects", + "Total space", + ]; + for name in LIBVERS.iter().chain(&["userblock.h5", "base.h5"]) { + let p = f.p(name); + let ours = h5rs(&["stat", &p]); + assert!(ours.status.success(), "{name}: {ours:?}"); + let reference = run("h5stat", &[&p]); + assert!(reference.status.success(), "h5stat {name} failed"); + let (o, r) = (stat_facts(&stdout(&ours)), stat_facts(&stdout(&reference))); + for k in keys { + // h5stat leaves out the attribute summary when nothing has one. + let Some(want) = r.get(k) else { continue }; + assert_eq!(o.get(k), Some(want), "{name}: {k}"); + } + } +} + +// --------------------------------------------------------------------------- +// dump +// --------------------------------------------------------------------------- + +#[test] +fn dump_matches_h5dump() { + let Some(f) = generate() else { return }; + if missing(tool_available("h5dump"), "h5dump") { + return; + } + for name in LIBVERS.iter().chain(&["userblock.h5", "changed.h5"]) { + let p = f.p(name); + for args in [vec![], vec!["-A"]] { + let mut a = args.clone(); + a.push(p.as_str()); + let ours = h5rs(&[&["dump"], a.as_slice()].concat()); + assert!(ours.status.success(), "{name} {args:?}: {ours:?}"); + let reference = run("h5dump", &a); + // h5dump names the file as given; h5rs by its file name. + let r = stdout(&reference).replacen(&p, name, 1); + assert_eq!(stdout(&ours), r, "{name} {args:?}"); + } + } + let p = f.p("latest.h5"); + let ours = h5rs(&["dump", "-d", "/grp/ext2", &p]); + let reference = run("h5dump", &["-d", "/grp/ext2", &p]); + assert_eq!( + stdout(&ours), + stdout(&reference).replacen(&p, "latest.h5", 1) + ); +} + +#[test] +fn dump_json_values_match_h5py() { + let Some(f) = generate() else { return }; + for name in LIBVERS { + let o = h5rs(&["dump", "--json", &f.p(name)]); + assert!(o.status.success(), "{name}: {o:?}"); + let doc: serde_json::Value = serde_json::from_slice(&o.stdout).unwrap(); + assert_eq!(doc["apiVersion"], "1.1.1"); + let root = doc["root"].as_str().unwrap(); + assert!(doc["groups"][root].is_object()); + let mut by_path = BTreeMap::new(); + for (_, d) in doc["datasets"].as_object().unwrap() { + for a in d["alias"].as_array().unwrap() { + by_path.insert(a.as_str().unwrap().to_string(), d.clone()); + } + } + let want = f.values[name.trim_end_matches(".h5")].as_object().unwrap(); + assert!(want.len() > 10); + for (path, v) in want { + let d = by_path + .get(path) + .unwrap_or_else(|| panic!("{name}: no {path}")); + let got = &d["value"]; + // JSON numbers: compare as f64 (h5py's ints and floats both + // round-trip exactly through JSON here). + assert_eq!(flatten(got), flatten(v), "{name}: {path}"); + } + let links = doc["groups"][root]["links"].as_array().unwrap(); + let soft = links.iter().find(|l| l["title"] == "soft").unwrap(); + assert_eq!(soft["class"], "H5L_TYPE_SOFT"); + assert_eq!(soft["h5path"], "/contig"); + let ext = links.iter().find(|l| l["title"] == "external").unwrap(); + assert_eq!(ext["class"], "H5L_TYPE_EXTERNAL"); + assert_eq!(ext["file"], "other.h5"); + } +} + +fn flatten(v: &serde_json::Value) -> Vec { + match v { + serde_json::Value::Array(a) => a.iter().flat_map(flatten).collect(), + serde_json::Value::Number(n) => vec![n.as_f64().unwrap()], + other => panic!("not numeric: {other}"), + } +} + +// --------------------------------------------------------------------------- +// diff +// --------------------------------------------------------------------------- + +fn code(o: &Output) -> i32 { + o.status.code().expect("exit code") +} + +#[test] +fn diff_exit_codes_match_h5diff() { + let Some(f) = generate() else { return }; + if missing(tool_available("h5diff"), "h5diff") { + return; + } + let (b, same, changed, extra, attr) = ( + f.p("base.h5"), + f.p("same.h5"), + f.p("changed.h5"), + f.p("extra.h5"), + f.p("attr.h5"), + ); + let missing_file = f.p("does-not-exist.h5"); + let cases: Vec<(Vec<&str>, i32)> = vec![ + (vec![&b, &same], 0), + (vec![&b, &b], 0), + (vec![&b, &changed], 1), + (vec![&b, &changed, "/d"], 1), + (vec![&b, &changed, "/g"], 0), + (vec!["-d", "0.01", &b, &changed], 0), + (vec!["-d", "0.0001", &b, &changed], 1), + (vec!["-p", "0.01", &b, &changed], 0), + (vec!["-p", "1e-9", &b, &changed], 1), + (vec![&b, &extra], 1), + (vec![&b, &attr], 1), + (vec![&b, &attr, "/g"], 0), + (vec![&b, &same, "/nope"], 2), + (vec![&b, &missing_file], 2), + (vec![&b, &same, "/d", "/d"], 0), + ]; + for (args, want) in cases { + let theirs = code(&run("h5diff", &args)); + assert_eq!(theirs, want, "h5diff {args:?} (test expectation)"); + let o = h5rs(&[&["diff"], args.as_slice()].concat()); + assert_eq!(code(&o), want, "h5rs diff {args:?}: {}", stdout(&o)); + let q = h5rs(&[&["diff", "-q"], args.as_slice()].concat()); + assert_eq!(code(&q), want, "h5rs diff -q {args:?}"); + assert!(q.stdout.is_empty(), "-q printed output"); + } + // The report lists the differing positions. + let o = h5rs(&["diff", "-r", &b, &changed, "/d"]); + let s = stdout(&o); + assert!(s.contains("[ 0 2 ]") && s.contains("[ 2 3 ]"), "{s}"); + assert!(s.contains("2 difference(s) found"), "{s}"); +} + +/// Where h5rs deliberately differs from h5diff: objects that cannot be +/// compared (another shape, another datatype class) are a difference. +#[test] +fn diff_counts_incomparable_objects_as_different() { + let Some(f) = generate() else { return }; + let b = f.p("base.h5"); + for other in ["reshaped.h5", "int.h5"] { + let o = h5rs(&["diff", &b, &f.p(other)]); + assert_eq!(code(&o), 1, "{other}: {}", stdout(&o)); + assert!(stdout(&o).contains("Not comparable"), "{other}"); + } +} + +// --------------------------------------------------------------------------- +// check +// --------------------------------------------------------------------------- + +#[test] +fn check_accepts_valid_files() { + let Some(f) = generate() else { return }; + for name in LIBVERS.iter().chain(&[ + "userblock.h5", + "base.h5", + "changed.h5", + "extra.h5", + "attr.h5", + ]) { + let o = h5rs(&["check", "--data", &f.p(name)]); + assert_eq!(code(&o), 0, "{name}:\n{}", stdout(&o)); + assert!(stdout(&o).contains("no problems found")); + } + // The newest format's checksummed structures were all verified. + let s = stdout(&h5rs(&["check", &f.p("latest.h5")])); + assert!(s.contains("superblock 1, v2 object headers 33,"), "{s}"); + assert!(s.contains("chunk indexes 5"), "{s}"); +} + +/// Signatures of checksummed structures, and where each one's checksum +/// is: found by trying every end offset (the checksum is the Jenkins +/// lookup3 hash of the bytes before it). A fractal heap direct block's +/// checksum is instead stored near its start, computed over the whole block +/// with the field zeroed. +fn checksummed_structures(data: &[u8]) -> Vec<(&'static str, usize, usize)> { + const SIGS: [&str; 13] = [ + "OHDR", "OCHK", "BTHD", "BTIN", "BTLF", "FRHP", "FHIB", "EAHD", "EAIB", "EASB", "EADB", + "FAHD", "FADB", + ]; + let mut found = Vec::new(); + for start in 0..data.len().saturating_sub(4) { + let sig = &data[start..start + 4]; + if let Some(name) = SIGS.iter().find(|s| s.as_bytes() == sig) { + let limit = data.len().min(start + 65536); + for end in start + 8..limit.saturating_sub(4) { + let stored = u32::from_le_bytes(data[end..end + 4].try_into().unwrap()); + if jenkins_lookup3(&data[start..end]) == stored { + found.push((*name, start, end)); + break; + } + } + } else if sig == b"FHDB" { + // version(1) heap address(8) block offset(1..8) checksum(4) + 'found: for boff in 1..=8usize { + let at = start + 5 + 8 + boff; + let Some(stored) = data.get(at..at + 4) else { + break; + }; + let stored = u32::from_le_bytes(stored.try_into().unwrap()); + for size in [512usize, 1024, 2048, 4096, 8192, 16384, 32768, 65536] { + let Some(block) = data.get(start..start + size) else { + break; + }; + let mut b = block.to_vec(); + b[at - start..at - start + 4].fill(0); + if jenkins_lookup3(&b) == stored { + found.push(("FHDB", start, at)); + break 'found; + } + } + } + } + } + found +} + +#[test] +fn check_flags_every_corrupted_checksum() { + let Some(f) = generate() else { return }; + let path = f.path("latest.h5"); + let data = std::fs::read(&path).unwrap(); + let structures = checksummed_structures(&data); + let kinds: std::collections::BTreeSet<&str> = structures.iter().map(|s| s.0).collect(); + for want in [ + "OHDR", "OCHK", "BTHD", "BTLF", "FRHP", "FHDB", "EAHD", "EAIB", "EADB", "FAHD", "FADB", + ] { + assert!(kinds.contains(want), "test file has no {want}: {kinds:?}"); + } + let bad = f.path("bad.h5"); + for (name, start, at) in structures { + let mut d = data.clone(); + d[at] ^= 0x01; + std::fs::write(&bad, &d).unwrap(); + let o = h5rs(&["check", &bad.to_string_lossy()]); + let s = stdout(&o); + assert_eq!( + code(&o), + 1, + "{name} at {start:#x}: checksum flip not flagged\n{s}" + ); + assert!( + s.lines() + .any(|l| l.starts_with("problem: 0x") && l.contains("checksum")), + "{name} at {start:#x}: no checksum problem reported\n{s}" + ); + } +} + +#[test] +fn check_flags_a_truncated_file() { + let Some(f) = generate() else { return }; + let data = std::fs::read(f.path("latest.h5")).unwrap(); + let bad = f.path("truncated.h5"); + std::fs::write(&bad, &data[..data.len() * 3 / 4]).unwrap(); + let o = h5rs(&["check", &bad.to_string_lossy()]); + assert_eq!(code(&o), 1); + assert!(stdout(&o).contains("file is truncated"), "{}", stdout(&o)); +} + +/// A v1 B-tree chunk index has no checksum; its keys are checked against +/// the dataset: a chunk offset that is not a multiple of the chunk size. +#[test] +fn check_flags_a_misaligned_chunk() { + let Some(f) = generate() else { return }; + let path = f.path("earliest.h5"); + let mut data = std::fs::read(&path).unwrap(); + // The leaf chunk B-tree of /grp/gz (10 deflated chunks of 100 i32, so + // each stored in under 400 bytes): "TREE" type(1)=1 level(1)=0 + // entries(2) left(8) right(8), then key, child, key, child, ... where a + // rank-1 chunk key is size(4) mask(4) offsets(2 x 8) and a child is 8. + // Key 1 starts at 24 + 24 + 8 = 56; its dimension-0 offset (100) at 64. + let mut patched = false; + for start in 0..data.len() - 72 { + if &data[start..start + 4] == b"TREE" && data[start + 4] == 1 && data[start + 5] == 0 { + let entries = u16::from_le_bytes(data[start + 6..start + 8].try_into().unwrap()); + let size0 = u32::from_le_bytes(data[start + 24..start + 28].try_into().unwrap()); + let off1 = u64::from_le_bytes(data[start + 64..start + 72].try_into().unwrap()); + if entries == 10 && size0 < 400 && off1 == 100 { + data[start + 64..start + 72].copy_from_slice(&103u64.to_le_bytes()); + patched = true; + break; + } + } + } + assert!(patched, "no chunk B-tree found for /grp/gz"); + let bad = f.path("misaligned.h5"); + std::fs::write(&bad, &data).unwrap(); + let o = h5rs(&["check", &bad.to_string_lossy()]); + let s = stdout(&o); + assert_eq!(code(&o), 1, "{s}"); + assert!( + s.contains("/grp/gz: chunk at [103, 0] offset 103 in dimension 0 is not a multiple of the chunk size 100"), + "{s}" + ); +} + +#[test] +fn every_subcommand_rejects_a_non_hdf5_file_cleanly() { + let dir = tempfile::tempdir().unwrap(); + let p = dir.path().join("junk.h5"); + std::fs::write(&p, b"this is not an HDF5 file at all").unwrap(); + let p = p.to_string_lossy().into_owned(); + for args in [ + vec!["ls", p.as_str()], + vec!["dump", p.as_str()], + vec!["stat", p.as_str()], + vec!["diff", p.as_str(), p.as_str()], + ] { + let o = h5rs(&args); + assert_eq!(code(&o), 2, "{args:?}"); + assert!(String::from_utf8_lossy(&o.stderr).contains("not an HDF5 file")); + } + let o = h5rs(&["check", &p]); + assert_eq!(code(&o), 1); + assert!(stdout(&o).contains("no HDF5 signature")); + assert_eq!(code(&h5rs(&["nonsense"])), 2); + assert_eq!(code(&h5rs(&["ls"])), 2); + assert_eq!(code(&h5rs(&["--help"])), 0); +}