diff --git a/.gitea/workflows/ci.yml b/.gitea/workflows/ci.yml index 1a7cdde..f57d8e4 100644 --- a/.gitea/workflows/ci.yml +++ b/.gitea/workflows/ci.yml @@ -31,12 +31,18 @@ jobs: # cmake builds libz-ng-sys for the opt-in `fast-deflate` (zlib-ng) # steps in ci-test.sh; rust:latest does not ship it. The default # build (pure-Rust zlib-rs) does not need it. - apt-get install -y --no-install-recommends python3 python3-venv cmake + # hdf5-tools: h5ls/h5stat/h5dump/h5diff, which the h5rs + # (clawhdf5-tools) interop tests compare against. + apt-get install -y --no-install-recommends python3 python3-venv cmake hdf5-tools python3 -m venv /opt/interop /opt/interop/bin/pip install --no-cache-dir h5py numpy netCDF4 xarray hdf5plugin echo "/opt/interop/bin" >> "$GITHUB_PATH" - name: Show interop library versions - run: /opt/interop/bin/python -c "import h5py, netCDF4, hdf5plugin; print('h5py', h5py.__version__, 'HDF5', h5py.version.hdf5_version, 'netCDF4', netCDF4.__version__, 'hdf5plugin', hdf5plugin.version)" + # h5dump's version too: the h5rs dump test requires its exact output + # (checked against Debian's 1.14.5 in rust:latest and 1.14.6). + run: | + /opt/interop/bin/python -c "import h5py, netCDF4, hdf5plugin; print('h5py', h5py.__version__, 'HDF5', h5py.version.hdf5_version, 'netCDF4', netCDF4.__version__, 'hdf5plugin', hdf5plugin.version)" + h5dump --version - name: Run CI script env: # Name the interpreter outright rather than relying on $GITHUB_PATH diff --git a/CHANGELOG.md b/CHANGELOG.md index 1223a46..44ca7c0 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -163,6 +163,65 @@ `quantized_index = false`, or pass `create --f32-index` to the CLI, to opt out. The CLI's `--quantized-index` is still accepted but is now a no-op. +### Tools +- New crate **`clawhdf5-tools`** with the binary **`h5rs`**: HDF5 + command-line tools without libhdf5, built only on the `clawhdf5` facade + and `clawhdf5-format` (no C, so it also builds as a static musl binary). + - `h5rs ls [-r] [-v] FILE[/path]` lists objects like h5ls (its first two + columns are h5ls's text on the test files) plus the datatype; `-v` adds + address, link count, layout and chunk index, chunk size, storage, + filters, datatype and attributes. + - `h5rs dump [--json] [-A] [-p] [-d PATH] FILE` prints DDL text that is + byte-identical to h5dump 1.14.6's (and to Debian's 1.14.5, which CI + uses) on the test files (all layouts and + chunk indexes, v1/v2 groups, compound, enum, strings, links, named + types, attributes; null-padded strings show their NULs at any depth), + or JSON in the HDF Group's hdf5-json layout (schema in the crate + README). Nested compounds print inline and `long double` values as + errors (exit 1); both are listed in the README. + - `h5rs stat FILE` reports h5stat's object, link, rank, layout, filter, + attribute, raw-data and file-size figures (equal to h5stat's on the test + files); metadata space is one figure, not broken down. + - `h5rs diff [-r] [-q] [-n N] [-d D] [-p R] [--follow-symlinks] A B [OBJ1 + [OBJ2]]` (option names as h5diff's: `-c` is `--compare`, the count is + `-n`/`--count=N`) compares objects, kinds, datatypes, shapes, attributes, values and link + targets; exit status 0/1/2 as h5diff's. Soft links are compared by + target path, as h5diff's default, or with `--follow-symlinks` by the + objects they lead to (external links are never followed). Every path is + compared, including every name of a hard-linked object and the members + of a hard-linked group; with a `-d`/`-p` tolerance, integers are + compared exactly in integer arithmetic (no loss above 2^53), and a `-p` + below the f64 epsilon compares exactly, as h5diff's. Objects that cannot + be compared count as a difference (h5diff exits 0 for them), and NaN + equals NaN. + - `h5rs check [--data] FILE` is a structural validator: it walks every + object, parses every header message, verifies the checksums of every + version 2+ structure it meets (superblock, object headers and + continuation chunks, v2 B-tree nodes, fractal heap headers and — which + the library's reads do not — every direct and indirect heap block, and + extensible/fixed array chunk indexes), checks each chunk index against + its dataset (aligned, in-extent, unique, plausibly sized chunks), and + that raw data lies inside the file without overlaps. Every problem is + printed with its address; exit 1 when there are any. libhdf5's h5check + reads only the 1.8 format. On the conformance corpus it passes all 418 + files that both clawhdf5 and h5py read in full, and `check --data` flags + 152 of the 180 files of the CVE corpus (tank, 2026-09-26). `--data` also + follows variable-length data into its global heap collections and + reports a damaged one at its address. It inherits the library's + tolerance, though: 21 of the 28 it passes are files + h5dump 1.14.6 rejects (see `docs/known-issues.md`, header checks). + - Values over `--max-bytes` (default 1 GiB) are reported instead of read; + a panic is caught and reported as an internal error (exit 3). + `scripts/h5rs-fuzz.sh` runs every subcommand over a corpus (default the + CVE reproducers, optionally with byte-flipped copies) with overflow + checks, a timeout and a memory limit, and fails on any panic, crash or + hang; `scripts/h5rs-check-ok-files.sh` runs `check --data` over the + fully-read conformance files. + - Because the library does not verify fractal heap block checksums when + it reads a dense group's links or dense attributes, `h5rs` verifies a + heap's blocks before reading from it and refuses a damaged one, as + libhdf5 does, instead of printing what the damaged block holds. + ### Signing - `clawhdf5-agent`: **Ed25519-signed checkpoints** — the README's "cryptographically verifiable memory", now true. With diff --git a/CLAUDE.md b/CLAUDE.md index 12a4b93..472ee3a 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -5,7 +5,7 @@ Pure-Rust HDF5 format implementation with HNSW vector search, WAL-backed persist ## Architecture -Cargo workspace with 16 crates under `crates/` (plus `libaec-sys`, an internal FFI bindings crate for the optional `szip` feature): +Cargo workspace with 17 crates under `crates/` (plus `libaec-sys`, an internal FFI bindings crate for the optional `szip` feature): | Crate | Role | |-------|------| @@ -21,7 +21,8 @@ Cargo workspace with 16 crates under `crates/` (plus `libaec-sys`, an internal F | `clawhdf5-accel` | CPU SIMD acceleration path | | `clawhdf5-migrate` | SQLite → HDF5 agent-memory migration | | `clawhdf5-android` | Android JNI bindings | -| `clawhdf5-cli` | Command-line interface | +| `clawhdf5-cli` | Command-line interface (agent memory) | +| `clawhdf5-tools` | `h5rs`: pure-Rust HDF5 tools — `ls`, `dump` (DDL / hdf5-json), `stat`, `diff`, `check` (structural + checksum validator) | | `clawhdf5-napi` | Node.js native addon bindings | | `clawhdf5-py` | PyO3 Python bindings | | `clawhdf5-bench` | Benchmark suite | @@ -188,6 +189,16 @@ cargo run -p clawhdf5-cli -- --help # create, save, search, recall, stats, flush-wal, agents-md, export, snapshot subcommands ``` +### HDF5 tools (`h5rs`, crate `clawhdf5-tools`) +```bash +cargo run -p clawhdf5-tools -- ls -r file.h5 # also dump [--json], stat, diff, check +bash scripts/h5rs-fuzz.sh # every subcommand over the CVE corpus: no panic/crash/hang +bash scripts/h5rs-check-ok-files.sh --data # check passes every fully-read conformance file +``` +Its interop tests compare against h5ls/h5stat/h5dump/h5diff (Debian +`hdf5-tools`, installed in CI); `dump` must stay byte-identical to h5dump on +the test files. + ### Python bindings ```bash cd crates/clawhdf5-py diff --git a/Cargo.toml b/Cargo.toml index 2e1f6fe..b86d8b6 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -16,6 +16,7 @@ members = [ "crates/clawhdf5-cli", "crates/clawhdf5-napi", "crates/clawhdf5-bench", + "crates/clawhdf5-tools", "crates/libaec-sys", ] resolver = "2" diff --git a/crates/clawhdf5-tools/Cargo.toml b/crates/clawhdf5-tools/Cargo.toml new file mode 100644 index 0000000..d54f39d --- /dev/null +++ b/crates/clawhdf5-tools/Cargo.toml @@ -0,0 +1,23 @@ +[package] +name = "clawhdf5-tools" +version = "2.7.0" +edition = "2024" +rust-version.workspace = true +license = "MIT" +description = "h5rs: pure-Rust HDF5 command-line tools (ls, dump, stat, diff, check) without libhdf5" +repository = "https://git.redclaw.dev/quantumclaw/clawhdf5" +keywords = ["hdf5", "h5dump", "h5ls", "cli", "science"] +categories = ["command-line-utilities", "science"] +readme = "README.md" + +[[bin]] +name = "h5rs" +path = "src/main.rs" + +[dependencies] +clawhdf5 = { path = "../clawhdf5", version = "2.7.0" } +clawhdf5-format = { path = "../clawhdf5-format", version = "2.7.0" } +serde_json = "1" + +[dev-dependencies] +tempfile = { workspace = true } diff --git a/crates/clawhdf5-tools/README.md b/crates/clawhdf5-tools/README.md new file mode 100644 index 0000000..2c1dab5 --- /dev/null +++ b/crates/clawhdf5-tools/README.md @@ -0,0 +1,276 @@ +# clawhdf5-tools: `h5rs` + +HDF5 command-line tools in pure Rust, built only on the `clawhdf5` facade and +`clawhdf5-format`. No libhdf5 and no C code, so the binary also builds as a +fully static executable: `cargo build --release -p clawhdf5-tools --target +x86_64-unknown-linux-musl` gives a static-pie `h5rs` of 1.8 MB with no +shared-library dependencies (built and run on tank, 2026-09-26, Rust 1.98). + +| Command | Modelled on | What it does | +|---------|-------------|--------------| +| `h5rs ls` | `h5ls` | list objects: name, kind, shape, datatype; `-v` adds layout, chunking, storage, filters, attributes | +| `h5rs dump` | `h5dump` | the file's structure and values as DDL text, or as JSON (hdf5-json layout) | +| `h5rs stat` | `h5stat` | object, link, rank, layout, filter and attribute counts; raw-data and file size | +| `h5rs diff` | `h5diff` | structural and value differences between two files or objects | +| `h5rs check` | `h5check` | structural validator: every object and header message, the checksums of version 2+ structures, chunk index consistency | + +```bash +cargo install --path crates/clawhdf5-tools # or: cargo build --release -p clawhdf5-tools +h5rs --help +h5rs --help +``` + +Every command takes `--max-bytes N` where it reads values (default 1 GiB): a +dataset whose dataspace claims more than that is reported instead of read, so +a corrupt size cannot exhaust memory. + +## `h5rs ls` + +```console +$ h5rs ls -r data.h5 # an extract of the output +/ Group +/external External Link {other.h5//x} +/grp Group +/grp/ext2 Dataset {6/Inf, 10/Inf} float32 +/grp/gz Dataset {1000} int32 +/hard2 Group, same as /grp/sub +/named_t Type +/soft Soft Link {/contig} +``` + +The first two columns are h5ls's (`h5ls -r` prints the same text); h5rs adds +the datatype. `FILE/path` lists a group's members or one dataset, as h5ls +does. `-v` prints, per object: + +```console +$ h5rs ls -v data.h5/grp/gz +gz Dataset {1000/1000} + Address: 1120 + Links: 1 + Layout: chunked (fixed array index) + Chunks: {100} 400 bytes + Storage: 4000 logical bytes, 1237 allocated bytes, 323.36% utilization + Filter-0: shuffle-2 OPT {4} + Filter-1: deflate-1 OPT {4} + Filter-2: fletcher32-3 + Type: 32-bit little-endian integer +``` + +## `h5rs dump` + +```console +$ h5rs dump data.h5 # DDL, like h5dump +$ h5rs dump -A data.h5 # no dataset values (attributes still shown), like h5dump -A +$ h5rs dump -d /grp/gz data.h5 # one dataset +$ h5rs dump -p data.h5 # also STORAGE_LAYOUT and FILTERS blocks +$ h5rs dump --json data.h5 # hdf5-json +``` + +The DDL output is h5dump's: on the test files of `tests/gen_files.py` +(compact, contiguous and chunked datasets with every chunk index; v1 and v2 +groups; integers of both byte orders, floats, compound with an array member, +enum, fixed and variable-length strings; soft, external and hard links; a +named datatype; compact and dense attributes) `h5rs dump` and `h5rs dump -A` +print the same bytes as h5dump 1.14.6 and as Debian's h5dump 1.14.5 (the +`hdf5-tools` package CI installs in `rust:latest`; the whole interop suite +was run in that image on 2026-09-26) — `dump_matches_h5dump` in +`tests/h5rs_interop.rs` checks this, and `dump_shows_nul_padding_in_nested_strings` +that null-padded strings show their NULs (`"a\000b"`) at any depth, as +h5dump's do. Not covered by those tests: references, opaque, bitfield, +variable-length sequences and virtual datasets. Known differences from +h5dump: + +- Floats print at their own precision (a `float32` 0.1 prints as `0.1`), + which for some values is more digits than h5dump's `%g`. +- A compound nested in a compound prints inline (`{ 1, 2.5 }`) where + h5dump prints it as an indented block, one member per line; only the + outer compound is a block. +- `long double` (x87 80-bit) and other floats wider than 64 bits: the + datatype is printed as an `H5T_FLOAT { ... }` block instead of h5dump's + one-line description, and each value as ``; `dump` then + exits 1. The library cannot convert them (see `docs/known-issues.md`). +- The `-p` block is h5rs's own (it names the chunk index), not h5dump's. + +### JSON schema + +`--json` follows the HDF Group's [hdf5-json](https://github.com/HDFGroup/hdf5-json) +layout: + +```json +{ + "apiVersion": "1.1.1", + "root": "g-0000000000000060", + "groups": { "": { "alias": ["/grp"], "attributes": [...], "links": [...] } }, + "datasets": { "": { "alias": [...], "attributes": [...], "shape": {...}, + "type": {...}, "creationProperties": {...}, "value": ... } }, + "datatypes": { "": { "alias": [...], "attributes": [...], "type": {...} } } +} +``` + +- **ids** are `g-`/`d-`/`t-` plus the object header address in 16 hex digits + (hdf5-json uses UUIDs; these are stable for a given file). `alias` lists + every path that reaches the object. +- **links**: `{"class": "H5L_TYPE_HARD", "title", "collection", "id"}`, + `{"class": "H5L_TYPE_SOFT", "title", "h5path"}`, + `{"class": "H5L_TYPE_EXTERNAL", "title", "file", "h5path"}`, + `{"class": "H5L_TYPE_USER_DEFINED", "title", "linkClass"}`. +- **shape**: `{"class": "H5S_NULL"}`, `{"class": "H5S_SCALAR"}` or + `{"class": "H5S_SIMPLE", "dims": [...], "maxdims": [...]}` with + `"H5S_UNLIMITED"` for an unlimited dimension. +- **type**: `{"class": "H5T_INTEGER" | "H5T_FLOAT" | "H5T_BITFIELD", "base": + "H5T_STD_I32LE" ...}`, `{"class": "H5T_STRING", "charSet", "strPad", + "length": n | "H5T_VARIABLE"}`, `{"class": "H5T_COMPOUND", "fields": [{"name", + "type"}]}`, `{"class": "H5T_ARRAY", "base", "dims"}`, `{"class": "H5T_ENUM", + "base", "mapping": {"NAME": value}}`, `{"class": "H5T_VLEN", "base"}`, + `{"class": "H5T_OPAQUE", "size", "tag"}`, `{"class": "H5T_REFERENCE", "base": + "H5T_STD_REF_OBJ" | "H5T_STD_REF_DSETREG" | "H5T_STD_REF"}`. +- **value**: nested lists in the dataset's shape (a scalar is the bare value, + a null dataspace `null`). A compound element is a list of its members, an + enum element its integer value, a string a JSON string, opaque/bitfield data + a `0x...` hex string, an object reference the referenced object's path, + NaN/infinities the strings `"NaN"`, `"Infinity"`, `"-Infinity"`, and an + integer beyond 64 bits a decimal string. +- **creationProperties**: `layout` (`{"class": "H5D_CHUNKED", "dims": [...]}` + etc.) and `filters` (`[{"id", "name", "class", "parameters"}]`). + +A value that cannot be read is replaced by `"value_error": ""` and +the command exits 1. + +## `h5rs stat` + +Prints h5stat's report sections with the same labels for the facts it +computes — object and link counts, max links to an object, max objects in a +group, dataset ranks, layout counts, filter counts, attribute counts, total +raw data size and total file size (all equal to h5stat's on the test files; +`stat_matches_h5stat` checks them). It does not break metadata space down by +structure as h5stat does; it reports metadata and free space as one figure. + +## `h5rs diff` + +```console +$ h5rs diff a.h5 b.h5 # whole files +$ h5rs diff a.h5 b.h5 /grp # one object and everything below it +$ h5rs diff a.h5 b.h5 /x /y # different paths in each +$ h5rs diff -r a.h5 b.h5 /d # list every differing element +$ h5rs diff -d 0.001 a.h5 b.h5 # |a - b| > 0.001 is a difference +$ h5rs diff -p 0.01 a.h5 b.h5 # |a - b| / |a| > 1% is a difference +$ h5rs diff --follow-symlinks a.h5 b.h5 /lnk # what the soft link /lnk leads to +$ h5rs diff -r -n 10 a.h5 b.h5 # list at most 10 differing elements per object +``` + +The options are named as h5diff's: `-n N`/`--count=N` limits the listed +elements, `-c`/`--compare` (list objects that are not comparable) is +accepted and always in effect, and `--delta=D`, `--relative=R` work too. + +Integers are compared in integer arithmetic, with or without a tolerance, so +64-bit values beyond 2^53 lose no precision (`-d 0` tells 2^60 from +2^60 + 1). A relative tolerance below the f64 epsilon (2.2e-16) compares +exactly, as h5diff's does. + +Exit status: 0 no differences, 1 differences, 2 error — the same as h5diff's +on the cases `diff_exit_codes_match_h5diff` runs. Compared: which objects +exist, their kinds, datatypes and shapes, attribute sets and values, dataset +values, and soft/external link targets. A soft link — including an OBJ +that is itself a soft link — is compared as a link, by its target path, as +h5diff does; `--follow-symlinks` compares the objects soft links lead to +instead (and walks into soft-linked groups), and two dangling links are then +the same, as in h5diff. External links are always compared by target (file +and path): `--follow-symlinks` does not open other files, where h5diff's +does. Two dangling soft links with different targets are a difference +(h5diff reports them with `-r`/`-v` but exits 0 without). Every path is compared: an object +hard-linked under two names is compared under both, with everything below +it, so a file that shares one object between two names equals a file that +stores two identical copies. Differences from h5diff, on purpose: objects +that cannot be compared (different shapes or datatype classes) count as a +difference (h5diff warns and exits 0); two NaNs are equal; and the members +of a group reached by a second hard link are compared (h5diff lists them in +one file only and exits 1). + +## `h5rs check` + +```console +$ h5rs check data.h5 +checked data.h5: superblock v3, 33 objects (4 groups, 28 datasets, 1 named datatypes), 139 header messages, 34 chunks +checksums verified: superblock 1, v2 object headers 33, v2 B-trees 3, fractal heaps 3 (+3 blocks), chunk indexes 5 +no problems found + +$ h5rs check damaged.h5 # one bit flipped in /grp/ext1's chunk index header +problem: 0x79c /grp/ext1: chunk index (extensible array): checksum mismatch: expected 0xcfe2391f, computed 0x2dd59bed +checked damaged.h5: superblock v3, 33 objects (4 groups, 28 datasets, 1 named datatypes), 139 header messages, 27 chunks +checksums verified: superblock 1, v2 object headers 33, v2 B-trees 3, fractal heaps 3 (+3 blocks), chunk indexes 4 +1 problem found +``` + +Each problem line is `problem:
: `; addresses +are HDF5 addresses (relative to the superblock, as h5dump and h5ls print +them). + +It walks every object reachable from the root group (and the superblock +extension) and checks: + +- the superblock (and its checksum, version 2+) and that the file is not + shorter than the superblock's end-of-file address; +- every object header, with the checksum of version 2 headers and of their + continuation chunks, and every header message parsed by type (dataspace, + datatype, fill value, layout, filter pipeline, attributes, link info, + links, group info, symbol table), including shared messages; +- groups: the symbol table (v1 B-tree, local heap, symbol nodes), or the + links, and for dense storage the fractal heap — header, and every direct + and indirect block with its checksum, back-pointer and heap offset — and + the v2 B-tree name and creation-order indexes (every node's checksum, and + the record count against the header's); +- dense attribute storage the same way; +- datasets: the layout against the dataspace and datatype (compact and + contiguous sizes, chunk rank), and for chunked datasets the whole chunk + index (v1 B-tree, single chunk, implicit, fixed array, extensible array, + v2 B-tree, with the checksums of the last three): every chunk's offset must + be a multiple of the chunk size and inside the extent, appear once, and + have a plausible size; +- that all raw data (contiguous blocks and chunks) lies inside the file and + no two pieces overlap. + +`--data` also reads every dataset, decoding every chunk through its filters +(which catches corrupt compressed data and Fletcher-32 mismatches), and +follows every variable-length element (strings and sequences, also inside +compounds and arrays) of every dataset and attribute into its global heap +collection: a collection that does not parse, a missing heap object, or a +sequence longer than its heap object is a problem at the collection's +address. Data the +tool cannot decode (a filter it does not implement, such as szip, or a +dataset over `--max-bytes`) is a `note:`, not a problem. Every problem is +printed with the address of the structure involved; the exit status is 0 +when there are none, 1 when there are, 2 for a usage error or a missing +file. + +libhdf5's h5check understands only the HDF5 1.8 file format; `h5rs check` +also covers the structures HDF5 1.10+ writes (fixed/extensible array and v2 +B-tree chunk indexes, and version 3 superblocks). + +What it does not check: free-space manager and shared-message (SOHM) table +checksums, global heap collections no variable-length value points into (and +none at all without `--data`), and objects reachable only by external links. It validates with +clawhdf5's parsers, so it accepts what they accept: some header damage that +libhdf5 refuses goes unreported (of the 180 files of the HDF Group's CVE +corpus, `check --data` passes 28, and h5dump 1.14.6 rejects 21 of those; +tank, 2026-09-26). + +## Robustness + +A panic is a bug: `h5rs` catches it, prints `internal error`, and exits 3 +(`check` records it against the object and carries on). `scripts/h5rs-fuzz.sh` +runs every subcommand over every file of a corpus (by default the HDF Group's +CVE reproducers, fetched by `conformance/fetch-corpus.sh`), optionally with +byte-flipped copies (`MUTATE=N`), under a timeout and a memory limit, with +overflow checks on, and fails on any panic, crash or hang. +`scripts/h5rs-check-ok-files.sh` runs `check --data` over the conformance +files that both clawhdf5 and h5py read in full, which must all pass. + +## Tests + +```bash +CLAWHDF5_PYTHON=.venv/bin/python CLAWHDF5_REQUIRE_INTEROP=1 cargo test -p clawhdf5-tools +``` + +The interop tests write their files with h5py and compare with h5ls, h5stat, +h5dump and h5diff; each skips when what it needs is missing unless +`CLAWHDF5_REQUIRE_INTEROP=1`. diff --git a/crates/clawhdf5-tools/src/check.rs b/crates/clawhdf5-tools/src/check.rs new file mode 100644 index 0000000..b5dd3c5 --- /dev/null +++ b/crates/clawhdf5-tools/src/check.rs @@ -0,0 +1,995 @@ +//! `h5rs check`: a structural validator. +//! +//! It walks every object reachable from the root group (and the superblock +//! extension), parses every header message, verifies the checksum of every +//! checksummed (version 2+) structure it meets, checks each chunked +//! dataset's chunk index against the dataset's shape, and checks that raw +//! data lies inside the file without overlapping other raw data. Every +//! problem is reported with the address of the structure involved. + +use std::collections::{BTreeMap, HashSet}; +use std::panic::{self, AssertUnwindSafe}; + +use clawhdf5_format::attribute_info::AttributeInfoMessage; +use clawhdf5_format::btree_v2::{BTreeV2Header, collect_btree_v2_records}; +use clawhdf5_format::data_layout::DataLayout; +use clawhdf5_format::dataspace::{Dataspace, DataspaceType}; +use clawhdf5_format::datatype::Datatype; +use clawhdf5_format::group_info::GroupInfoMessage; +use clawhdf5_format::link_info::LinkInfoMessage; +use clawhdf5_format::message_type::MessageType; +use clawhdf5_format::object_header::ObjectHeader; +use clawhdf5_format::symbol_table::SymbolTableMessage; + +use crate::cli::{Args, Out}; +use crate::h5::{Error, ErrorKind, H5, Kind}; +use crate::info::{self, DsInfo}; + +pub const USAGE: &str = "\ +usage: h5rs check [--data] [-q] [--max-bytes N] FILE + +Validate FILE's structure: walk every object from the root group, parse +every header message, verify the checksums of version 2+ structures +(superblock, object headers and continuation chunks, v2 B-tree nodes, +fractal heap headers and blocks, extensible/fixed array chunk indexes), +check each chunked dataset's chunk index against its shape (offsets aligned +to the chunk size and inside the extent, no duplicates, sizes plausible), +and check that all raw data lies inside the file without overlaps. Every +problem is printed with the address of the structure involved. + + --data also read every dataset, decoding every chunk through its + filters (catches corrupt compressed data and Fletcher-32 + mismatches), and follow every variable-length element of + datasets and attributes into its global heap collection + -q, --quiet print only the problems, not the summary + --max-bytes N largest dataset read by --data (default 1 GiB) + +Exit status: 0 no problems, 1 problems found, 2 usage error or file not +found, 3 internal error."; + +const MAX_CHUNKS_CHECKED: usize = 10_000_000; + +/// Whether values of `dt` hold variable-length data (in the global heap). +fn has_vl(dt: &Datatype, depth: u32) -> bool { + if depth > 32 { + return false; + } + match dt { + Datatype::VariableLength { .. } => true, + Datatype::Compound { members, .. } => { + members.iter().any(|m| has_vl(&m.datatype, depth + 1)) + } + Datatype::Array { base_type, .. } => has_vl(base_type, depth + 1), + _ => false, + } +} + +#[derive(Default)] +struct Counts { + objects: u64, + groups: u64, + datasets: u64, + datatypes: u64, + messages: u64, + chunks: u64, + datasets_read: u64, + /// Global heap collections that variable-length data points into, + /// parsed without error (with --data). + global_heaps: u64, + sb_checksum: u64, + ohdr_v2: u64, + btree_v2: u64, + heaps: u64, + heap_block_checksums: u64, + chunk_index_checksummed: u64, +} + +struct Problem { + addr: u64, + path: String, + msg: String, +} + +struct Checker<'a> { + h5: &'a H5, + read_data: bool, + eof: u64, + problems: Vec, + /// Things not checked, which are not problems with the file. + notes: Vec, + counts: Counts, + /// Raw data extents: (start, end, owner path). + extents: Vec<(u64, u64, String)>, + heaps_seen: HashSet, + btrees_seen: HashSet, + /// Global heap collections already read (with --data). + gcols_seen: HashSet, + panicked: bool, +} + +pub fn run(args: &mut Args, out: &mut Out) -> std::io::Result { + let mut read_data = false; + let mut quiet = false; + let mut max_bytes = None; + let mut file = None; + while let Some(a) = args.next() { + match a.as_str() { + "--data" => read_data = true, + "-q" | "--quiet" => quiet = true, + "--max-bytes" => match args.number() { + Some(n) => max_bytes = Some(n), + None => return args.usage_error(out, "--max-bytes needs a number", USAGE), + }, + "-h" | "--help" => { + writeln!(out.o, "{USAGE}")?; + return Ok(0); + } + s if s.starts_with('-') && s.len() > 1 => { + return args.usage_error(out, &format!("unknown option {a}"), USAGE); + } + _ if file.is_none() => file = Some(a), + _ => return args.usage_error(out, &format!("unexpected argument {a}"), USAGE), + } + } + let Some(file) = file else { + return args.usage_error(out, "missing FILE", USAGE); + }; + let path = std::path::Path::new(&file); + if !path.is_file() { + writeln!(out.e, "h5rs check: {file}: no such file")?; + return Ok(2); + } + let mut h5 = match H5::open(path) { + Ok(h) => h, + Err(_) => return unopenable(path, out), + }; + if let Some(m) = max_bytes { + h5.max_bytes = m; + } + let mut c = Checker { + h5: &h5, + read_data, + eof: h5.data().len() as u64, + problems: Vec::new(), + notes: Vec::new(), + counts: Counts::default(), + extents: Vec::new(), + heaps_seen: HashSet::new(), + btrees_seen: HashSet::new(), + gcols_seen: HashSet::new(), + panicked: false, + }; + c.superblock(); + c.objects(); + c.overlaps(); + for p in &c.problems { + writeln!(out.o, "problem: {:#x} {}: {}", p.addr, p.path, p.msg)?; + } + if !quiet { + for n in &c.notes { + writeln!(out.o, "note: {:#x} {}: {}", n.addr, n.path, n.msg)?; + } + } + if !quiet { + c.summary(&file, out)?; + } + Ok(if c.panicked { + 3 + } else if c.problems.is_empty() { + 0 + } else { + 1 + }) +} + +/// The file could not be opened at all: say why, as precisely as possible. +fn unopenable(path: &std::path::Path, out: &mut Out) -> std::io::Result { + let data = match std::fs::read(path) { + Ok(d) => d, + Err(e) => { + writeln!(out.e, "h5rs check: {}: {e}", path.display())?; + return Ok(2); + } + }; + let msg = match clawhdf5_format::signature::find_signature(&data) { + Err(_) => ( + 0, + "no HDF5 signature at offset 0 or any power of two from 512".to_string(), + ), + Ok(off) => match clawhdf5_format::superblock::Superblock::parse(&data[off..], 0) { + Ok(_) => (off as u64, "file cannot be opened".to_string()), + Err(e) => (off as u64, format!("superblock: {e}")), + }, + }; + writeln!(out.o, "problem: {:#x} /: {}", msg.0, msg.1)?; + writeln!( + out.o, + "checked {}: 1 problem found (file cannot be opened)", + path.display() + )?; + Ok(1) +} + +fn msg_name(t: MessageType) -> String { + match t { + MessageType::Unknown(n) => format!("message type {n:#x}"), + t => format!("{t:?} message"), + } +} + +impl Checker<'_> { + fn problem(&mut self, addr: u64, path: &str, msg: impl Into) { + self.problems.push(Problem { + addr, + path: path.to_string(), + msg: msg.into(), + }); + } + + fn err(&mut self, default_addr: u64, path: &str, e: &Error) { + // Reading attributes or links fails on a damaged heap or B-tree that + // the structure checks have already reported at the same address. + if let Some(a) = e.addr + && self.problems.iter().any(|p| p.addr == a && p.path == path) + { + return; + } + self.problem(e.addr.unwrap_or(default_addr), path, e.msg.clone()); + } + + /// Run `f`; a panic inside it becomes a problem instead of aborting the + /// whole check. + fn guarded(&mut self, addr: u64, path: &str, f: impl FnOnce(&mut Self)) { + let r = panic::catch_unwind(AssertUnwindSafe(|| f(self))); + if r.is_err() { + self.panicked = true; + self.problem( + addr, + path, + "internal error while checking this object (see stderr)", + ); + } + } + + fn superblock(&mut self) { + let sb = self.h5.sb().clone(); + if sb.version >= 2 { + // Superblock::parse verified it, or the file would not be open. + self.counts.sb_checksum += 1; + } + // libhdf5 stores the end-of-file address counting the user block + // (h5py's `userblock_size` files show it), unlike every other + // address, which is relative to the superblock. + let file_len = self.eof.saturating_add(self.h5.file.user_block_size()); + if sb.eof_address > file_len { + self.problem( + 0, + "/", + format!( + "file is truncated: the superblock's end-of-file address is {:#x} but the file \ + is {file_len:#x} bytes long", + sb.eof_address + ), + ); + } + if sb.root_group_address >= self.eof { + self.problem( + 0, + "/", + format!( + "root group address {:#x} is past the end of the file", + sb.root_group_address + ), + ); + } + let undef = if sb.offset_size >= 8 { + u64::MAX + } else { + (1u64 << (8 * u32::from(sb.offset_size))) - 1 + }; + if let Some(ext) = sb.superblock_extension_address.filter(|&a| a != undef) { + self.guarded(ext, "", |c| match c.h5.header(ext) { + Ok(h) => { + if h.version >= 2 { + c.counts.ohdr_v2 += 1; + } + c.messages(ext, "", &h); + } + Err(e) => c.err(ext, "", &e), + }); + } + } + + fn objects(&mut self) { + let h5 = self.h5; + // Collect first, then check: the walk borrows the file, the checks + // borrow `self` mutably. + let (items, walk) = h5.walk_collect(); + if let Err(e) = walk { + self.err(h5.root(), "/", &e); + } + for it in items { + let path = it.path; + // Soft, external and user-defined links have no header here. + let (Some(addr), None, Some(header)) = (it.addr, it.first_path, it.header) else { + continue; + }; + if addr >= self.eof { + self.problem( + addr, + &path, + "object header address is past the end of the file", + ); + continue; + } + let h = match header { + Ok(h) => h, + Err(e) => { + self.err(addr, &path, &e); + continue; + } + }; + self.counts.objects += 1; + self.guarded(addr, &path, |c| c.object(addr, &path, &h)); + } + } + + fn object(&mut self, addr: u64, path: &str, h: &ObjectHeader) { + if h.version >= 2 { + self.counts.ohdr_v2 += 1; + } + self.messages(addr, path, h); + match self.h5.attributes(h) { + Ok((attrs, errs)) => { + for e in errs { + self.problem(addr, path, format!("attribute: {e}")); + } + if self.read_data { + for a in &attrs { + let what = format!("attribute \"{}\": ", a.name); + self.vl_data(addr, path, &what, &a.datatype, &a.dataspace, &a.raw_data); + } + } + } + Err(e) => self.err(addr, path, &e), + } + let is_root = addr == self.h5.root(); + match Kind::of(h) { + Kind::Group => { + self.counts.groups += 1; + self.group(addr, path, h); + } + Kind::Dataset => { + self.counts.datasets += 1; + self.dataset(addr, path, h); + } + Kind::Datatype => self.counts.datatypes += 1, + Kind::Unknown if is_root => self.counts.groups += 1, + Kind::Unknown => { + self.problem( + addr, + path, + "object header describes no group, dataset or named datatype", + ); + } + } + } + + /// Parse every message of a header by type. + fn messages(&mut self, addr: u64, path: &str, h: &ObjectHeader) { + let (os, ls) = (self.h5.os(), self.h5.ls()); + for m in &h.messages { + self.counts.messages += 1; + let data = + match clawhdf5_format::shared_message::message_data(self.h5.data(), m, os, ls) { + Ok(d) => d.into_owned(), + Err(e) => { + self.problem(addr, path, format!("shared {}: {e}", msg_name(m.msg_type))); + continue; + } + }; + let r: Result<(), String> = match m.msg_type { + MessageType::Dataspace => Dataspace::parse(&data, ls) + .map(drop) + .map_err(|e| e.to_string()), + MessageType::Datatype => { + Datatype::parse(&data).map(drop).map_err(|e| e.to_string()) + } + MessageType::FillValue | MessageType::FillValueOld => { + let mut mm = m.clone(); + mm.data = data.clone(); + clawhdf5_format::fill_value::parse_fill_value(&mm) + .map(drop) + .map_err(|e| e.to_string()) + } + MessageType::DataLayout => DataLayout::parse(&data, os, ls) + .map(drop) + .map_err(|e| e.to_string()), + MessageType::FilterPipeline => { + clawhdf5_format::filter_pipeline::FilterPipeline::parse(&data) + .map(drop) + .map_err(|e| e.to_string()) + } + MessageType::Attribute => { + clawhdf5_format::attribute::AttributeMessage::parse_in_file( + &data, + self.h5.data(), + os, + ls, + ) + .map(drop) + .map_err(|e| e.to_string()) + } + MessageType::AttributeInfo => match AttributeInfoMessage::parse(&data, os) { + Ok(ai) => { + self.dense_storage( + addr, + path, + "attribute", + ai.fractal_heap_address, + [ai.btree_name_index_address, ai.btree_creation_order_address], + ); + Ok(()) + } + Err(e) => Err(e.to_string()), + }, + MessageType::LinkInfo => LinkInfoMessage::parse(&data, os) + .map(drop) + .map_err(|e| e.to_string()), + MessageType::Link => clawhdf5_format::link_message::LinkMessage::parse(&data, os) + .map(drop) + .or_else(|e| match e { + clawhdf5_format::error::FormatError::InvalidLinkType(t) if t >= 65 => { + Ok(()) + } + e => Err(e.to_string()), + }), + MessageType::GroupInfo => GroupInfoMessage::parse(&data) + .map(drop) + .map_err(|e| e.to_string()), + MessageType::SymbolTable => SymbolTableMessage::parse(&data, os) + .map(drop) + .map_err(|e| e.to_string()), + _ => Ok(()), + }; + if let Err(e) = r { + self.problem(addr, path, format!("{}: {e}", msg_name(m.msg_type))); + } + } + } + + /// Dense (fractal heap + v2 B-tree) link or attribute storage. + fn dense_storage( + &mut self, + addr: u64, + path: &str, + what: &str, + heap: Option, + btrees: [Option; 2], + ) { + if let Some(fh) = heap + && self.heaps_seen.insert(fh) + { + self.counts.heaps += 1; + let r = crate::heap_blocks::verify(self.h5, fh); + self.counts.heap_block_checksums += r.checksums as u64; + for e in r.problems { + self.err(fh, path, &e.context(&format!("dense {what} storage"))); + } + } + for bt in btrees.into_iter().flatten() { + if !self.btrees_seen.insert(bt) { + continue; + } + let (os, ls) = (self.h5.os(), self.h5.ls()); + let Ok(off) = usize::try_from(bt) else { + self.problem(bt, path, format!("{what} index address out of range")); + continue; + }; + match BTreeV2Header::parse(self.h5.data(), off, os, ls) { + Ok(hdr) => { + self.counts.btree_v2 += 1; + match collect_btree_v2_records(self.h5.data(), &hdr, os, ls) { + Ok(recs) => { + if recs.len() as u64 != hdr.total_records { + self.problem( + bt, + path, + format!( + "{what} index: v2 B-tree header counts {} records, the tree holds {}", + hdr.total_records, + recs.len() + ), + ); + } + } + Err(e) => self.problem(bt, path, format!("{what} index (v2 B-tree): {e}")), + } + } + Err(e) => self.problem(bt, path, format!("{what} index (v2 B-tree header): {e}")), + } + } + let _ = addr; + } + + fn group(&mut self, addr: u64, path: &str, h: &ObjectHeader) { + if let Some(m) = h + .messages + .iter() + .find(|m| m.msg_type == MessageType::LinkInfo) + && let Ok(li) = LinkInfoMessage::parse(&m.data, self.h5.os()) + { + self.dense_storage( + addr, + path, + "link", + li.fractal_heap_address, + [li.btree_name_index_address, li.btree_creation_order_address], + ); + } + if let Err(e) = self.h5.links(h) { + self.err(addr, path, &e.context("cannot list the group")); + } + if let Some(m) = h + .messages + .iter() + .find(|m| m.msg_type == MessageType::SymbolTable) + && let Ok(st) = SymbolTableMessage::parse(&m.data, self.h5.os()) + { + for (a, what) in [ + (st.btree_address, "B-tree"), + (st.local_heap_address, "local heap"), + ] { + if a >= self.eof { + self.problem( + addr, + path, + format!("symbol table {what} address {a:#x} is past the end of the file"), + ); + } + } + } + } + + fn dataset(&mut self, addr: u64, path: &str, h: &ObjectHeader) { + let info = DsInfo::read(self.h5, path, h); + for e in [ + info.dt.as_ref().err(), + info.ds.as_ref().err(), + info.layout.as_ref().err(), + info.filters.as_ref().err(), + ] + .into_iter() + .flatten() + { + let e = e.clone(); + self.err(addr, path, &e); + } + let (Ok(dt), Ok(ds), Ok(layout)) = (&info.dt, &info.ds, &info.layout) else { + return; + }; + let esize = u64::from(dt.type_size()); + if esize == 0 { + self.problem(addr, path, "datatype has size 0"); + return; + } + let need = match crate::h5::byte_len(ds, dt) { + Ok(n) => n, + Err(e) => { + self.err(addr, path, &e); + return; + } + }; + if let Some(max) = &ds.max_dimensions { + for (i, (&d, &m)) in ds.dimensions.iter().zip(max).enumerate() { + if m != u64::MAX && d > m { + self.problem( + addr, + path, + format!("dimension {i} is {d}, above its maximum {m}"), + ); + } + } + } + match layout { + DataLayout::Compact { data } => { + if (data.len() as u64) < need { + self.problem( + addr, + path, + format!( + "compact data holds {} bytes; the dataset needs {need}", + data.len() + ), + ); + } + } + DataLayout::Contiguous { address, size } => { + if let Some(a) = address { + if !info.external && *size < need { + self.problem( + *a, + path, + format!("contiguous storage is {size} bytes; the dataset needs {need}"), + ); + } + self.extent(*a, *size, path); + } + } + DataLayout::Chunked { .. } => self.chunked(addr, path, &info, dt, ds, layout), + DataLayout::Virtual { .. } => { + let mut l = layout.clone(); + if let Err(e) = l.resolve_vds_mappings(self.h5.data(), self.h5.ls()) { + self.problem(addr, path, format!("virtual dataset mappings: {e}")); + } + } + } + if self.read_data { + match self.h5.read_dataset(path, dt, ds) { + Ok(raw) => { + self.counts.datasets_read += 1; + self.vl_data(addr, path, "", dt, ds, &raw); + } + // Valid data this tool cannot decode is not a problem with + // the file. + Err(e) if e.kind != ErrorKind::Corrupt => self.notes.push(Problem { + addr, + path: path.to_string(), + msg: format!("data not read: {}", e.msg), + }), + Err(e) => self.problem(addr, path, format!("reading the data: {}", e.msg)), + } + } + } + + /// With --data: follow every variable-length element of `raw` (the + /// values of a dataset or attribute) into the global heap, so a damaged + /// collection, a missing heap object or a sequence longer than its heap + /// object is reported at the collection's address. Each bad collection + /// is reported once per object. + fn vl_data( + &mut self, + addr: u64, + path: &str, + what: &str, + dt: &Datatype, + ds: &Dataspace, + raw: &[u8], + ) { + if !has_vl(dt, 0) { + return; + } + let n = crate::h5::num_elements(ds).unwrap_or(0); + let size = dt.type_size() as usize; + let mut bad: BTreeMap = BTreeMap::new(); + for i in 0..n { + let Some(b) = usize::try_from(i) + .ok() + .and_then(|i| i.checked_mul(size)) + .and_then(|s| raw.get(s..s.checked_add(size)?)) + else { + self.problem( + addr, + path, + format!("{what}element {i} is past the data read"), + ); + break; + }; + self.vl_element(dt, b, 0, &mut bad); + if bad.len() >= 100 { + break; + } + } + for (a, msg) in bad { + self.problem(a, path, format!("{what}variable-length data: {msg}")); + } + } + + fn vl_element(&mut self, dt: &Datatype, b: &[u8], depth: u32, bad: &mut BTreeMap) { + if depth > 32 { + return; + } + match dt { + Datatype::VariableLength { + is_string, + base_type, + .. + } => { + let os = usize::from(self.h5.os()); + let (Some(lenb), Some(addrb), Some(idxb)) = + (b.get(..4), b.get(4..4 + os), b.get(4 + os..8 + os)) + else { + return; + }; + let le = |x: &[u8]| { + x.iter() + .enumerate() + .fold(0u64, |a, (i, &v)| a | (u64::from(v) << (8 * i))) + }; + let (len, gcol, idx) = (le(lenb), le(addrb), le(idxb)); + let undef = if os >= 8 { + u64::MAX + } else { + (1u64 << (8 * os)) - 1 + }; + if len == 0 || gcol == 0 || gcol == undef || bad.contains_key(&gcol) { + return; + } + let obj = match self.h5.heap_object(gcol, idx as u32) { + Ok(o) => o, + Err(e) => { + bad.insert(e.addr.unwrap_or(gcol), e.msg); + return; + } + }; + if self.gcols_seen.insert(gcol) { + self.counts.global_heaps += 1; + } + let bs = if *is_string { + 1 + } else { + u64::from(base_type.type_size()) + }; + if len + .checked_mul(bs) + .is_none_or(|need| need > obj.len() as u64) + { + bad.insert( + gcol, + format!( + "global heap object {idx} holds {} bytes; the element needs {len} x {bs}", + obj.len() + ), + ); + return; + } + if !*is_string && bs > 0 && has_vl(base_type, depth + 1) { + let bs = bs as usize; + for k in 0..len as usize { + self.vl_element(base_type, &obj[k * bs..(k + 1) * bs], depth + 1, bad); + } + } + } + Datatype::Compound { members, .. } => { + for m in members { + if let Some(mb) = usize::try_from(m.byte_offset).ok().and_then(|o| b.get(o..)) { + self.vl_element(&m.datatype, mb, depth + 1, bad); + } + } + } + Datatype::Array { + base_type, + dimensions, + } => { + let bs = base_type.type_size() as usize; + let n = dimensions + .iter() + .try_fold(1usize, |a, &d| a.checked_mul(d as usize)) + .unwrap_or(usize::MAX); + for k in 0..n { + match b.get(k * bs..(k + 1) * bs) { + Some(eb) => self.vl_element(base_type, eb, depth + 1, bad), + None => break, + } + } + } + _ => {} + } + } + + fn chunked( + &mut self, + addr: u64, + path: &str, + info: &DsInfo, + dt: &Datatype, + ds: &Dataspace, + layout: &DataLayout, + ) { + let DataLayout::Chunked { + chunk_dimensions, + btree_address, + chunk_index_type, + .. + } = layout + else { + return; + }; + let rank = match ds.space_type { + DataspaceType::Simple => ds.dimensions.len(), + _ => 0, + }; + if chunk_dimensions.len() != rank + 1 { + self.problem( + addr, + path, + format!( + "chunked layout has {} chunk dimensions for a rank-{rank} dataset (expected {})", + chunk_dimensions.len(), + rank + 1 + ), + ); + return; + } + let cdims = &chunk_dimensions[..rank]; + if cdims.contains(&0) { + self.problem(addr, path, "a chunk dimension is 0"); + return; + } + let Some(index_addr) = *btree_address else { + return; // no chunk allocated yet + }; + if index_addr >= self.eof { + self.problem( + addr, + path, + format!("chunk index address {index_addr:#x} is past the end of the file"), + ); + return; + } + let chunks = match info::chunks(self.h5, layout, ds, dt) { + Ok(c) => c, + Err(e) => { + self.err(index_addr, path, &e); + return; + } + }; + if matches!(chunk_index_type, Some(3..=5)) { + self.counts.chunk_index_checksummed += 1; + } + let filtered = matches!(&info.filters, Ok(Some(p)) if !p.filters.is_empty()); + let chunk_bytes = cdims.iter().try_fold(u64::from(dt.type_size()), |a, &d| { + a.checked_mul(u64::from(d)) + }); + let max = ds.max_dimensions.clone(); + let mut seen: HashSet> = HashSet::with_capacity(chunks.len().min(1 << 20)); + let mut reported = 0usize; + for (n, c) in chunks.iter().enumerate() { + if n >= MAX_CHUNKS_CHECKED { + self.problem(index_addr, path, "too many chunks; stopped checking them"); + break; + } + self.counts.chunks += 1; + let mut bad = Vec::new(); + if c.offsets.len() < rank { + bad.push(format!( + "has {} coordinates for a rank-{rank} dataset", + c.offsets.len() + )); + } else { + for (i, (&o, &cd)) in c.offsets.iter().zip(cdims).enumerate() { + if o % u64::from(cd) != 0 { + bad.push(format!( + "offset {o} in dimension {i} is not a multiple of the chunk size {cd}" + )); + } + let limit = match max.as_ref().and_then(|m| m.get(i)) { + Some(&u64::MAX) | None => ds.dimensions[i], + Some(&m) => m.max(ds.dimensions[i]), + }; + if o >= limit && !(limit == 0 && o == 0) { + bad.push(format!( + "offset {o} in dimension {i} is outside the extent {limit}" + )); + } + } + if !seen.insert(c.offsets[..rank].to_vec()) { + bad.push("appears twice in the chunk index".into()); + } + } + if c.chunk_size == 0 { + bad.push("has size 0".into()); + } else if !filtered + && let Some(cb) = chunk_bytes + && u64::from(c.chunk_size) != cb + { + bad.push(format!( + "is {} bytes; an unfiltered chunk is {cb}", + c.chunk_size + )); + } + if !bad.is_empty() { + reported += 1; + if reported <= 50 { + self.problem( + c.address, + path, + format!("chunk at {:?} {}", c.offsets, bad.join("; ")), + ); + } + } + self.extent(c.address, u64::from(c.chunk_size), path); + } + if reported > 50 { + self.problem( + index_addr, + path, + format!("{} more bad chunks not listed", reported - 50), + ); + } + } + + /// Record raw data at `[start, start + len)`, checking it is in the file. + fn extent(&mut self, start: u64, len: u64, path: &str) { + if len == 0 { + return; + } + match start.checked_add(len) { + Some(end) if end <= self.eof => self.extents.push((start, end, path.to_string())), + _ => self.problem( + start, + path, + format!("raw data ({len} bytes) extends past the end of the file"), + ), + } + } + + fn overlaps(&mut self) { + let mut ext = std::mem::take(&mut self.extents); + ext.sort_unstable_by_key(|e| (e.0, e.1)); + let mut reported = 0usize; + let mut far: Option<(u64, String)> = None; + let mut msgs = Vec::new(); + for (s, e, p) in &ext { + if let Some((end, owner)) = &far + && s < end + { + reported += 1; + if reported <= 50 { + msgs.push(( + *s, + p.clone(), + format!("raw data at {s:#x} overlaps raw data of {owner}"), + )); + } + } + if far.as_ref().is_none_or(|(end, _)| e > end) { + far = Some((*e, p.clone())); + } + } + for (a, p, m) in msgs { + self.problem(a, &p, m); + } + if reported > 50 { + self.problem( + 0, + "/", + format!("{} more raw data overlaps not listed", reported - 50), + ); + } + } + + fn summary(&self, file: &str, out: &mut Out) -> std::io::Result<()> { + let c = &self.counts; + let sb = self.h5.sb(); + writeln!( + out.o, + "checked {file}: superblock v{}, {} objects ({} groups, {} datasets, {} named datatypes), \ + {} header messages, {} chunks", + sb.version, c.objects, c.groups, c.datasets, c.datatypes, c.messages, c.chunks + )?; + writeln!( + out.o, + "checksums verified: superblock {}, v2 object headers {}, v2 B-trees {}, \ + fractal heaps {} (+{} blocks), chunk indexes {}", + c.sb_checksum, + c.ohdr_v2, + c.btree_v2, + c.heaps, + c.heap_block_checksums, + c.chunk_index_checksummed + )?; + if self.read_data { + writeln!( + out.o, + "datasets read: {}, global heap collections read: {}", + c.datasets_read, c.global_heaps + )?; + } + match self.problems.len() { + 0 => writeln!(out.o, "no problems found"), + 1 => writeln!(out.o, "1 problem found"), + n => writeln!(out.o, "{n} problems found"), + } + } +} diff --git a/crates/clawhdf5-tools/src/cli.rs b/crates/clawhdf5-tools/src/cli.rs new file mode 100644 index 0000000..a4f6cde --- /dev/null +++ b/crates/clawhdf5-tools/src/cli.rs @@ -0,0 +1,52 @@ +//! Argument handling and output plumbing shared by the subcommands. + +use std::io::Write; + +/// Where a subcommand writes: `o` for results, `e` for diagnostics. +pub struct Out<'a> { + pub o: &'a mut dyn Write, + pub e: &'a mut dyn Write, +} + +/// The arguments after the subcommand name. +pub struct Args { + cmd: &'static str, + rest: std::collections::VecDeque, +} + +impl Args { + pub fn new(cmd: &'static str, rest: Vec) -> Self { + Self { + cmd, + rest: rest.into(), + } + } + + #[allow(clippy::should_implement_trait)] + pub fn next(&mut self) -> Option { + self.rest.pop_front() + } + + /// Put an option and its value back, to be read next (for the + /// `--name=value` form). + pub fn push_front(&mut self, name: String, value: String) { + self.rest.push_front(value); + self.rest.push_front(name); + } + + /// The value after an option such as `--max-bytes`. + pub fn value(&mut self) -> Option { + self.rest.pop_front() + } + + /// A numeric option value. + pub fn number(&mut self) -> Option { + self.rest.pop_front().and_then(|s| s.parse().ok()) + } + + /// Report a usage problem; exit status 2. + pub fn usage_error(&self, out: &mut Out, msg: &str, usage: &str) -> std::io::Result { + writeln!(out.e, "h5rs {}: {msg}\n\n{usage}", self.cmd)?; + Ok(2) + } +} diff --git a/crates/clawhdf5-tools/src/diff.rs b/crates/clawhdf5-tools/src/diff.rs new file mode 100644 index 0000000..c7ad723 --- /dev/null +++ b/crates/clawhdf5-tools/src/diff.rs @@ -0,0 +1,904 @@ +//! `h5rs diff`: compare two files (or two objects) like h5diff. + +use std::cell::OnceCell; +use std::collections::{BTreeMap, HashMap}; +use std::rc::Rc; + +use clawhdf5_format::attribute::AttributeMessage; +use clawhdf5_format::dataspace::{Dataspace, DataspaceType}; +use clawhdf5_format::datatype::Datatype; +use clawhdf5_format::object_header::ObjectHeader; + +use crate::cli::{Args, Out}; +use crate::h5::{H5, Kind, Link, LinkKind}; +use crate::value::{self, Decoder, Value}; + +pub const USAGE: &str = "\ +usage: h5rs diff [options] FILE1 FILE2 [OBJ1 [OBJ2]] + +Compare FILE1 and FILE2 (or OBJ1 in FILE1 with OBJ2 in FILE2, and everything +below them): the objects present, their kinds, datatypes, shapes, attribute +sets and values, and soft/external link targets (an OBJ that is itself a +soft link is compared as a link, as h5diff does, unless --follow-symlinks). + + -r, --report list every differing element (position, values, difference) + -q, --quiet print nothing; only the exit status + --follow-symlinks compare the objects soft links lead to (and walk into + soft-linked groups) instead of the links' target paths; + two dangling links are the same + -d, --delta D numbers differ only when |a - b| > D + -p, --relative R numbers differ only when |a - b| / |a| > R (an R below + the f64 epsilon, 2.2e-16, compares exactly, as h5diff) + -n, --count N list at most N differing elements per object with -r + -c, --compare list objects that are not comparable (always done; + accepted for h5diff compatibility) + --max-bytes N largest dataset read (default 1 GiB); a larger one is an error + +Two NaNs compare equal. Unlike h5diff, objects that cannot be compared +(different kinds, datatype classes or shapes) count as a difference. + +Exit status: 0 no differences, 1 differences found, 2 error (a file or +object could not be opened or read)."; + +#[derive(Clone, Copy)] +enum Tol { + Exact, + Delta(f64), + Relative(f64), +} + +struct Opts { + report: bool, + quiet: bool, + tol: Tol, + count: usize, + /// Compare the objects soft links lead to, not the links' targets. + follow: bool, +} + +/// What a relative path names in one file. +#[derive(Clone)] +enum Entry { + Obj(u64, Kind), + Soft(String), + External(String, String), + UserDefined(u8), + Broken(String), +} + +struct Side<'a> { + h5: &'a H5, + label: String, + base: String, + /// Relative path -> what it is. + entries: BTreeMap, + /// Address -> relative path, for comparing references. + rel_of: OnceCell>, +} + +impl Side<'_> { + fn full(&self, rel: &str) -> String { + if rel.is_empty() { + if self.base.is_empty() { + "/".into() + } else { + self.base.clone() + } + } else { + format!("{}{rel}", self.base) + } + } + + fn rel_paths(&self) -> &HashMap { + self.rel_of.get_or_init(|| { + let mut m = HashMap::new(); + let _ = self.h5.walk(|it| { + if let (Some(a), None) = (it.addr, it.first_path) { + m.entry(a).or_insert_with(|| it.path.to_string()); + } + }); + m + }) + } +} + +struct Diff { + opts: Opts, + diffs: u64, + /// Differences already reported under an object heading. + per_object: u64, + errors: u64, +} + +pub fn run(args: &mut Args, out: &mut Out) -> std::io::Result { + let mut opts = Opts { + report: false, + quiet: false, + tol: Tol::Exact, + count: usize::MAX, + follow: false, + }; + let mut max_bytes = None; + let mut pos = Vec::new(); + while let Some(a) = args.next() { + match a.as_str() { + "-r" | "--report" => opts.report = true, + "-q" | "--quiet" => opts.quiet = true, + "--follow-symlinks" => opts.follow = true, + "-d" | "--delta" => match args.number::() { + Some(d) if d >= 0.0 => opts.tol = Tol::Delta(d), + _ => return args.usage_error(out, "--delta needs a number >= 0", USAGE), + }, + "-p" | "--relative" => match args.number::() { + // As h5diff: a relative tolerance below the f64 epsilon + // cannot be told from rounding, so it compares exactly. + Some(r) if r >= 0.0 => { + opts.tol = if r < f64::EPSILON { + Tol::Exact + } else { + Tol::Relative(r) + } + } + _ => return args.usage_error(out, "--relative needs a number >= 0", USAGE), + }, + // h5diff's -c: h5rs always lists them. + "-c" | "--compare" => {} + "-n" | "--count" => match args.number() { + Some(n) => opts.count = n, + None => return args.usage_error(out, "--count needs a number", USAGE), + }, + "--max-bytes" => match args.number() { + Some(n) => max_bytes = Some(n), + None => return args.usage_error(out, "--max-bytes needs a number", USAGE), + }, + "-h" | "--help" => { + writeln!(out.o, "{USAGE}")?; + return Ok(0); + } + // h5diff's --count=N, --delta=D, --relative=R forms. + s if s.starts_with("--") && s.contains('=') => { + let (k, v) = s.split_once('=').unwrap_or((s, "")); + if !matches!(k, "--count" | "--delta" | "--relative" | "--max-bytes") { + return args.usage_error(out, &format!("unknown option {a}"), USAGE); + } + args.push_front(k.to_string(), v.to_string()); + } + s if s.starts_with('-') && s.len() > 1 => { + return args.usage_error(out, &format!("unknown option {a}"), USAGE); + } + _ => pos.push(a), + } + } + if pos.len() < 2 || pos.len() > 4 { + return args.usage_error(out, "expected FILE1 FILE2 [OBJ1 [OBJ2]]", USAGE); + } + let mut files = Vec::new(); + for f in &pos[..2] { + match H5::open(std::path::Path::new(f)) { + Ok(mut h) => { + if let Some(m) = max_bytes { + h.max_bytes = m; + } + files.push(h); + } + Err(e) => { + writeln!(out.e, "h5rs diff: {e}")?; + return Ok(2); + } + } + } + let obj1 = pos.get(2).cloned().unwrap_or_else(|| "/".into()); + let obj2 = pos.get(3).cloned().unwrap_or_else(|| obj1.clone()); + let mut sides = Vec::new(); + for (h5, (obj, f)) in files.iter().zip([(&obj1, &pos[0]), (&obj2, &pos[1])]) { + let start = match start(h5, obj, opts.follow) { + Ok(x) => x, + Err(_) => { + writeln!( + out.e, + "h5rs diff: object <{obj}> could not be found in <{f}>" + )?; + return Ok(2); + } + }; + let base = obj.trim_end_matches('/').to_string(); + let base = if base.is_empty() || base.starts_with('/') { + base + } else { + format!("/{base}") + }; + let collected = match start { + Start::Obj(addr) => collect(h5, addr, &base, opts.follow), + Start::Link(e) => Ok(BTreeMap::from([(String::new(), e)])), + }; + let entries = match collected { + Ok(e) => e, + Err(e) => { + writeln!(out.e, "h5rs diff: {f}: {e}")?; + return Ok(2); + } + }; + sides.push(Side { + h5, + label: f.clone(), + base, + entries, + rel_of: OnceCell::new(), + }); + } + let (a, b) = (&sides[0], &sides[1]); + let mut d = Diff { + opts, + diffs: 0, + per_object: 0, + errors: 0, + }; + let mut names: Vec<&String> = a.entries.keys().chain(b.entries.keys()).collect(); + names.sort(); + names.dedup(); + for rel in names { + match (a.entries.get(rel), b.entries.get(rel)) { + (Some(_), None) => { + d.diffs += 1; + d.say( + out, + &format!("<{}> exists only in <{}>", a.full(rel), a.label), + )?; + } + (None, Some(_)) => { + d.diffs += 1; + d.say( + out, + &format!("<{}> exists only in <{}>", b.full(rel), b.label), + )?; + } + (Some(ea), Some(eb)) => d.entry(out, a, b, rel, ea, eb)?, + (None, None) => {} + } + } + // Per-object counts were printed with each object; add a total when + // they do not already tell the whole story. + if !d.opts.quiet && d.diffs > 0 && d.diffs != d.per_object { + writeln!(out.o, "{} difference(s) found in total", d.diffs)?; + } + Ok(if d.errors > 0 { + 2 + } else if d.diffs > 0 { + 1 + } else { + 0 + }) +} + +/// What OBJ1/OBJ2 names. +enum Start { + Obj(u64), + /// A soft (not followed, or dangling), external or user-defined link. + Link(Entry), +} + +/// Resolve OBJ1/OBJ2. Soft links on the way to it are followed, as in any +/// HDF5 path; a soft link that *is* the named object is compared as a link +/// (its target path), as h5diff does, unless `follow`. +fn start(h5: &H5, obj: &str, follow: bool) -> crate::h5::Result { + let p = obj.trim_matches('/'); + if !p.is_empty() { + let (parent, name) = p.rsplit_once('/').unwrap_or(("", p)); + let link = h5 + .resolve(parent) + .and_then(|a| h5.header(a)) + .and_then(|h| h5.links(&h)) + .ok() + .and_then(|ls| ls.into_iter().find(|l| l.name == name)); + if let Some(l) = link { + return Ok(match l.kind { + LinkKind::Hard(a) => Start::Obj(a), + LinkKind::Soft(t) if follow => match soft_target(h5, &format!("/{parent}"), &t) { + Ok(a) => Start::Obj(a), + Err(_) => Start::Link(Entry::Soft(t)), + }, + LinkKind::Soft(t) => Start::Link(Entry::Soft(t)), + LinkKind::External { file, path } => Start::Link(Entry::External(file, path)), + LinkKind::UserDefined(t) => Start::Link(Entry::UserDefined(t)), + }); + } + } + h5.resolve(obj).map(Start::Obj) +} + +/// The object a soft link in group `parent` (an absolute path) leads to. +fn soft_target(h5: &H5, parent: &str, target: &str) -> crate::h5::Result { + if target.starts_with('/') { + h5.resolve(target) + } else { + h5.resolve(&format!("{}/{target}", parent.trim_end_matches('/'))) + } +} + +/// One object as the path walk sees it: what its paths compare as, and +/// (for a group) its links. +struct Node { + entry: Entry, + links: Vec, +} + +/// Every path below the object at `start`, relative to it (`""` is `start` +/// itself). Each hard link is its own path, so an object linked under two +/// names is compared under both, with everything below it: two files that +/// hold the same values are equal whether one shares an object between +/// names and the other stores copies. A hard link back to an ancestor (a +/// cycle) is recorded, not descended into. With `follow`, a soft link is +/// walked like a hard link to its target (a dangling one stays a link); +/// `base` is the absolute path of `start`, for relative link targets. +fn collect( + h5: &H5, + start: u64, + base: &str, + follow: bool, +) -> crate::h5::Result> { + let mut cache: HashMap> = HashMap::new(); + let mut node = |addr: u64| -> Rc { + cache + .entry(addr) + .or_insert_with(|| { + Rc::new(match h5.header(addr) { + Err(e) => Node { + entry: Entry::Broken(e.to_string()), + links: Vec::new(), + }, + Ok(h) => match Kind::of(&h) { + Kind::Group => match h5.links(&h) { + Ok(links) => Node { + entry: Entry::Obj(addr, Kind::Group), + links, + }, + Err(e) => Node { + entry: Entry::Broken(format!("links: {e}")), + links: Vec::new(), + }, + }, + k => Node { + entry: Entry::Obj(addr, k), + links: Vec::new(), + }, + }, + }) + }) + .clone() + }; + let mut entries = BTreeMap::new(); + // (address, relative path, addresses of the groups above it) + let mut stack: Vec<(u64, String, Rc>)> = + vec![(start, String::new(), Rc::new(Vec::new()))]; + while let Some((addr, path, above)) = stack.pop() { + if entries.len() >= crate::h5::MAX_OBJECTS { + return Err(crate::h5::Error::new(format!( + "more than {} paths; stopped walking", + crate::h5::MAX_OBJECTS + ))); + } + let n = node(addr); + entries.insert(path.clone(), n.entry.clone()); + if n.links.is_empty() || above.contains(&addr) { + continue; + } + let mut chain = (*above).clone(); + chain.push(addr); + let chain = Rc::new(chain); + for l in &n.links { + let child = format!("{path}/{}", l.name); + match &l.kind { + LinkKind::Hard(a) => stack.push((*a, child, chain.clone())), + LinkKind::Soft(t) if follow => match soft_target(h5, &format!("{base}{path}"), t) { + Ok(a) => stack.push((a, child, chain.clone())), + Err(_) => { + entries.insert(child, Entry::Soft(t.clone())); + } + }, + LinkKind::Soft(t) => { + entries.insert(child, Entry::Soft(t.clone())); + } + LinkKind::External { file, path } => { + entries.insert(child, Entry::External(file.clone(), path.clone())); + } + LinkKind::UserDefined(t) => { + entries.insert(child, Entry::UserDefined(*t)); + } + } + } + } + Ok(entries) +} + +fn kind_word(k: Kind) -> &'static str { + match k { + Kind::Group => "group", + Kind::Dataset => "dataset", + Kind::Datatype => "datatype", + Kind::Unknown => "object", + } +} + +impl Diff { + fn say(&self, out: &mut Out, msg: &str) -> std::io::Result<()> { + if self.opts.quiet { + Ok(()) + } else { + writeln!(out.o, "{msg}") + } + } + + fn error(&mut self, out: &mut Out, msg: &str) -> std::io::Result<()> { + self.errors += 1; + writeln!(out.e, "h5rs diff: {msg}") + } + + fn entry( + &mut self, + out: &mut Out, + a: &Side, + b: &Side, + rel: &str, + ea: &Entry, + eb: &Entry, + ) -> std::io::Result<()> { + let (pa, pb) = (a.full(rel), b.full(rel)); + match (ea, eb) { + (Entry::Broken(e), _) => self.error(out, &format!("<{pa}> in <{}>: {e}", a.label)), + (_, Entry::Broken(e)) => self.error(out, &format!("<{pb}> in <{}>: {e}", b.label)), + (Entry::Soft(x), Entry::Soft(y)) => { + // Followed links that are left are dangling on both sides, + // which h5diff counts as the same. + if x != y && !self.opts.follow { + self.diffs += 1; + self.say(out, &format!("soft link: <{pa}> -> {x} and <{pb}> -> {y}"))?; + } + Ok(()) + } + (Entry::External(f1, p1), Entry::External(f2, p2)) => { + if (f1, p1) != (f2, p2) { + self.diffs += 1; + self.say( + out, + &format!("external link: <{pa}> -> {f1}:{p1} and <{pb}> -> {f2}:{p2}"), + )?; + } + Ok(()) + } + (Entry::UserDefined(x), Entry::UserDefined(y)) => { + if x != y { + self.diffs += 1; + self.say(out, &format!("user-defined link: <{pa}> and <{pb}> differ"))?; + } + Ok(()) + } + (Entry::Obj(aa, ka), Entry::Obj(ab, kb)) if ka == kb => { + let (ha, hb) = match (a.h5.header(*aa), b.h5.header(*ab)) { + (Ok(x), Ok(y)) => (x, y), + (Err(e), _) | (_, Err(e)) => return self.error(out, &format!("<{pa}>: {e}")), + }; + let before = self.diffs; + let mut rows = Vec::new(); + match ka { + Kind::Dataset => self.dataset(out, a, b, &pa, &pb, &ha, &hb, &mut rows)?, + Kind::Datatype => match (a.h5.datatype(&ha), b.h5.datatype(&hb)) { + (Ok(x), Ok(y)) => { + if x != y { + self.diffs += 1; + rows.push("datatypes differ".to_string()); + } + } + (Err(e), _) | (_, Err(e)) => self.error(out, &format!("<{pa}>: {e}"))?, + }, + _ => {} + } + self.attributes(out, a, b, &pa, &ha, &hb, &mut rows)?; + let n = self.diffs - before; + if n > 0 && !self.opts.quiet { + writeln!(out.o, "{}: <{pa}> and <{pb}>", kind_word(*ka))?; + for r in &rows { + writeln!(out.o, "{r}")?; + } + writeln!(out.o, "{n} difference(s) found")?; + self.per_object += n; + } + Ok(()) + } + _ => { + self.diffs += 1; + self.say( + out, + &format!( + "Not comparable: <{pa}> is a {} and <{pb}> is a {}", + entry_word(ea), + entry_word(eb) + ), + ) + } + } + } + + #[allow(clippy::too_many_arguments)] + fn dataset( + &mut self, + out: &mut Out, + a: &Side, + b: &Side, + pa: &str, + pb: &str, + ha: &ObjectHeader, + hb: &ObjectHeader, + rows: &mut Vec, + ) -> std::io::Result<()> { + let got = (|| -> crate::h5::Result<_> { + Ok(( + a.h5.datatype(ha)?, + a.h5.resolved_dataspace(pa, ha)?, + b.h5.datatype(hb)?, + b.h5.resolved_dataspace(pb, hb)?, + )) + })(); + let (dta, dsa, dtb, dsb) = match got { + Ok(x) => x, + Err(e) => return self.error(out, &format!("<{pa}>: {e}")), + }; + if let Some(why) = not_comparable(&dta, &dsa, &dtb, &dsb) { + self.diffs += 1; + rows.push(format!("Not comparable: {why}")); + return Ok(()); + } + let raw = + a.h5.read_dataset(pa, &dta, &dsa) + .and_then(|x| Ok((x, b.h5.read_dataset(pb, &dtb, &dsb)?))); + let (ra, rb) = match raw { + Ok(x) => x, + Err(e) => return self.error(out, &format!("<{pa}>: {e}")), + }; + self.values(out, a, b, pa, (&dta, &ra), (&dtb, &rb), &dsa, rows) + } + + #[allow(clippy::too_many_arguments)] + fn values( + &mut self, + out: &mut Out, + a: &Side, + b: &Side, + what: &str, + (dta, ra): (&Datatype, &[u8]), + (dtb, rb): (&Datatype, &[u8]), + ds: &Dataspace, + rows: &mut Vec, + ) -> std::io::Result<()> { + let n = crate::h5::num_elements(ds).unwrap_or(0) as usize; + let (da, db) = (Decoder::new(a.h5), Decoder::new(b.h5)); + let dims: Vec = match ds.space_type { + DataspaceType::Simple => ds.dimensions.clone(), + _ => vec![1], + }; + let mut found = 0u64; + let mut header = false; + for i in 0..n { + let va = da.element(dta, ra, i); + let vb = db.element(dtb, rb, i); + if let (Value::Error(e), _) | (_, Value::Error(e)) = (&va, &vb) { + return self.error(out, &format!("<{what}> element {i}: {e}")); + } + if self.equal(a, b, &va, &vb) { + continue; + } + found += 1; + if self.opts.report && (found as usize) <= self.opts.count { + if !header { + header = true; + rows.push(format!( + "{:<24}{:<24}{:<24}{}", + "position", "value 1", "value 2", "difference" + )); + rows.push("-".repeat(80)); + } + let pos = format!("[ {} ]", index(i as u64, &dims)); + let ta = value::text(&va, &|_| None); + let tb = value::text(&vb, &|_| None); + let dif = match (int_of(&va), int_of(&vb), number(&va), number(&vb)) { + (Some(x), Some(y), ..) => x.abs_diff(y).to_string(), + (_, _, Some(x), Some(y)) => value::fmt_float((x - y).abs(), 64), + _ => String::new(), + }; + rows.push(format!("{pos:<24}{ta:<24}{tb:<24}{dif}")); + } + } + self.diffs += found; + Ok(()) + } + + #[allow(clippy::too_many_arguments)] + fn attributes( + &mut self, + out: &mut Out, + a: &Side, + b: &Side, + path: &str, + ha: &ObjectHeader, + hb: &ObjectHeader, + rows: &mut Vec, + ) -> std::io::Result<()> { + let (la, lb) = match (a.h5.attributes(ha), b.h5.attributes(hb)) { + (Ok(x), Ok(y)) => (x, y), + (Err(e), _) | (_, Err(e)) => return self.error(out, &format!("<{path}>: {e}")), + }; + for e in la.1.iter().chain(lb.1.iter()) { + self.error(out, &format!("<{path}>: attribute: {e}"))?; + } + let ma: BTreeMap<&str, &AttributeMessage> = + la.0.iter().map(|x| (x.name.as_str(), x)).collect(); + let mb: BTreeMap<&str, &AttributeMessage> = + lb.0.iter().map(|x| (x.name.as_str(), x)).collect(); + let mut names: Vec<&str> = ma.keys().chain(mb.keys()).copied().collect(); + names.sort_unstable(); + names.dedup(); + for n in names { + match (ma.get(n), mb.get(n)) { + (Some(x), Some(y)) => { + if let Some(why) = + not_comparable(&x.datatype, &x.dataspace, &y.datatype, &y.dataspace) + { + self.diffs += 1; + rows.push(format!("attribute \"{n}\": not comparable: {why}")); + continue; + } + for (m, h5) in [(x, a.h5), (y, b.h5)] { + let need = + crate::h5::byte_len(&m.dataspace, &m.datatype).unwrap_or(u64::MAX); + if need > h5.max_bytes || (m.raw_data.len() as u64) < need { + return self.error( + out, + &format!( + "<{path}> attribute \"{n}\": value is truncated or too large" + ), + ); + } + } + let before = rows.len(); + let what = format!("{path}\" attribute \"{n}"); + self.values( + out, + a, + b, + &what, + (&x.datatype, &x.raw_data), + (&y.datatype, &y.raw_data), + &x.dataspace, + rows, + )?; + if rows.len() > before { + rows.insert(before, format!("attribute \"{n}\":")); + } + } + (Some(_), None) => { + self.diffs += 1; + rows.push(format!("attribute \"{n}\" exists only in <{}>", a.label)); + } + (None, Some(_)) => { + self.diffs += 1; + rows.push(format!("attribute \"{n}\" exists only in <{}>", b.label)); + } + (None, None) => {} + } + } + Ok(()) + } + + fn equal(&self, a: &Side, b: &Side, x: &Value, y: &Value) -> bool { + // Integers in integer arithmetic: through f64 they lose precision + // above 2^53, and values that differ would compare equal. + if let (Some(p), Some(q)) = (int_of(x), int_of(y)) { + return int_close(p, q, self.opts.tol); + } + if let (Some(p), Some(q)) = (number(x), number(y)) { + return close(p, q, self.opts.tol); + } + match (x, y) { + (Value::Str(p), Value::Str(q)) => p == q, + (Value::Bytes(p), Value::Bytes(q)) | (Value::OtherRef(p), Value::OtherRef(q)) => p == q, + (Value::Compound(p), Value::Compound(q)) => { + p.len() == q.len() + && p.iter() + .zip(q) + .all(|((_, u), (_, v))| self.equal(a, b, u, v)) + } + (Value::Array(p), Value::Array(q)) | (Value::Seq(p), Value::Seq(q)) => { + p.len() == q.len() && p.iter().zip(q).all(|(u, v)| self.equal(a, b, u, v)) + } + (Value::Ref(None), Value::Ref(None)) => true, + (Value::Ref(Some(p)), Value::Ref(Some(q))) => { + // Addresses mean nothing across files: compare the paths the + // references lead to. + let (pp, qq) = (a.rel_paths().get(p), b.rel_paths().get(q)); + pp.is_some() && pp == qq + } + _ => false, + } + } +} + +fn int_of(v: &Value) -> Option { + match v { + Value::Int(i) | Value::Enum(_, i) => Some(*i), + _ => None, + } +} + +fn number(v: &Value) -> Option { + match v { + Value::Int(i) | Value::Enum(_, i) => Some(*i as f64), + Value::Float(f, _) => Some(*f), + _ => None, + } +} + +fn close(a: f64, b: f64, tol: Tol) -> bool { + if a.is_nan() || b.is_nan() { + return a.is_nan() && b.is_nan(); + } + if a == b { + return true; + } + let d = (a - b).abs(); + match tol { + Tol::Exact => false, + Tol::Delta(t) => d <= t, + Tol::Relative(r) => a != 0.0 && d / a.abs() <= r, + } +} + +/// `close` for integers, exactly: the difference is taken in i128, so no +/// precision is lost at any 64-bit magnitude. +fn int_close(a: i128, b: i128, tol: Tol) -> bool { + if a == b { + return true; + } + let d = a.abs_diff(b); + match tol { + Tol::Exact => false, + // d is whole, so d <= t exactly when d <= floor(t) (saturating). + Tol::Delta(t) => d <= t.floor() as u128, + // d >= 1 here, so the quotient is never rounded to 0. + Tol::Relative(r) => a != 0 && d as f64 / a.unsigned_abs() as f64 <= r, + } +} + +fn entry_word(e: &Entry) -> &'static str { + match e { + Entry::Obj(_, k) => kind_word(*k), + Entry::Soft(_) => "soft link", + Entry::External(..) => "external link", + Entry::UserDefined(_) => "user-defined link", + Entry::Broken(_) => "unreadable object", + } +} + +/// Why two datasets/attributes cannot be compared element by element. +fn not_comparable( + dta: &Datatype, + dsa: &Dataspace, + dtb: &Datatype, + dsb: &Dataspace, +) -> Option { + let rank = |d: &Dataspace| match d.space_type { + DataspaceType::Simple => Some(d.dimensions.clone()), + DataspaceType::Scalar => Some(Vec::new()), + DataspaceType::Null => None, + }; + let (sa, sb) = (rank(dsa), rank(dsb)); + if sa != sb { + let show = |s: &Option>| match s { + None => "null".to_string(), + Some(d) if d.is_empty() => "scalar".to_string(), + Some(d) => format!("{d:?}"), + }; + return Some(format!("shapes differ: {} and {}", show(&sa), show(&sb))); + } + if !types_comparable(dta, dtb) { + return Some(format!( + "datatypes differ: {} and {}", + crate::dtype::short(dta), + crate::dtype::short(dtb) + )); + } + None +} + +fn types_comparable(a: &Datatype, b: &Datatype) -> bool { + use crate::dtype::class; + let numeric = |t: &Datatype| matches!(class(t), "integer" | "float"); + if numeric(a) && numeric(b) { + return class(a) == class(b); + } + if class(a) != class(b) { + return false; + } + match (a, b) { + (Datatype::Compound { members: ma, .. }, Datatype::Compound { members: mb, .. }) => { + ma.len() == mb.len() + && ma + .iter() + .zip(mb) + .all(|(x, y)| x.name == y.name && types_comparable(&x.datatype, &y.datatype)) + } + ( + Datatype::Array { + base_type: x, + dimensions: dx, + }, + Datatype::Array { + base_type: y, + dimensions: dy, + }, + ) => dx == dy && types_comparable(x, y), + ( + Datatype::VariableLength { base_type: x, .. }, + Datatype::VariableLength { base_type: y, .. }, + ) => class(a) == "string" || types_comparable(x, y), + ( + Datatype::Enumeration { base_type: x, .. }, + Datatype::Enumeration { base_type: y, .. }, + ) => types_comparable(x, y), + _ => true, + } +} + +fn index(mut i: u64, dims: &[u64]) -> String { + let mut idx = vec![0u64; dims.len()]; + for (k, &d) in dims.iter().enumerate().rev() { + let d = d.max(1); + idx[k] = i % d; + i /= d; + } + idx.iter() + .map(|x| x.to_string()) + .collect::>() + .join(" ") +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn tolerances() { + assert!(close(1.0, 1.0, Tol::Exact)); + assert!(!close(1.0, 1.001, Tol::Exact)); + assert!(close(1.0, 1.001, Tol::Delta(0.01))); + assert!(!close(1.0, 1.1, Tol::Delta(0.01))); + assert!(close(100.0, 101.0, Tol::Relative(0.02))); + assert!(!close(0.0, 1e-9, Tol::Relative(0.5))); + assert!(close(f64::NAN, f64::NAN, Tol::Exact)); + assert!(!close(f64::NAN, 1.0, Tol::Delta(1e9))); + } + + #[test] + fn integer_tolerances_are_exact_above_2_pow_53() { + let big = 1i128 << 60; + assert!(!int_close(big, big + 1, Tol::Delta(0.0))); + assert!(!int_close(big, big + 1, Tol::Delta(0.5))); + assert!(int_close(big, big + 1, Tol::Delta(1.0))); + assert!(!int_close(big, big + 200, Tol::Delta(199.9))); + assert!(!int_close(big, big + 1, Tol::Relative(0.0))); + assert!(!int_close(big, big + 1, Tol::Relative(1e-19))); + assert!(int_close(big, big + 1, Tol::Relative(1e-18))); + let umax = i128::from(u64::MAX); + assert!(!int_close(umax, umax - 1, Tol::Delta(0.0))); + assert!(int_close( + i128::from(i64::MIN), + umax, + Tol::Delta(f64::INFINITY) + )); + assert!(!int_close(0, 1, Tol::Relative(1e9))); + } + + #[test] + fn positions() { + assert_eq!(index(5, &[3, 4]), "1 1"); + assert_eq!(index(0, &[1]), "0"); + } +} diff --git a/crates/clawhdf5-tools/src/dtype.rs b/crates/clawhdf5-tools/src/dtype.rs new file mode 100644 index 0000000..c42c4eb --- /dev/null +++ b/crates/clawhdf5-tools/src/dtype.rs @@ -0,0 +1,529 @@ +//! Names for datatypes: a short one for listings, h5ls's long form, h5dump's +//! DDL and the HDF Group's hdf5-json type objects. + +use clawhdf5_format::datatype::{ + CharacterSet, Datatype, DatatypeByteOrder, ReferenceType, StringPadding, +}; +use serde_json::{Value as J, json}; + +fn be(o: &DatatypeByteOrder) -> bool { + matches!(o, DatatypeByteOrder::BigEndian) +} + +fn order_suffix(o: &DatatypeByteOrder) -> &'static str { + match o { + DatatypeByteOrder::LittleEndian => "LE", + DatatypeByteOrder::BigEndian => "BE", + DatatypeByteOrder::Vax => "VAX", + } +} + +fn order_word(o: &DatatypeByteOrder) -> &'static str { + match o { + DatatypeByteOrder::LittleEndian => "little-endian", + DatatypeByteOrder::BigEndian => "big-endian", + DatatypeByteOrder::Vax => "VAX-order", + } +} + +/// True when a float is laid out exactly as IEEE 754 binary16/32/64. +pub fn is_ieee(dt: &Datatype) -> bool { + let Datatype::FloatingPoint { + size, + bit_offset, + bit_precision, + exponent_location, + exponent_size, + mantissa_location, + mantissa_size, + exponent_bias, + .. + } = dt + else { + return false; + }; + let std = match size { + 2 => (16, 10, 5, 10, 15), + 4 => (32, 23, 8, 23, 127), + 8 => (64, 52, 11, 52, 1023), + _ => return false, + }; + *bit_offset == 0 + && *mantissa_location == 0 + && ( + *bit_precision, + *exponent_location, + *exponent_size, + *mantissa_size, + *exponent_bias, + ) == std +} + +/// Short name used by `ls`: `int32`, `float64-be`, `string[3]`, ... +pub fn short(dt: &Datatype) -> String { + match dt { + Datatype::FixedPoint { + size, + signed, + byte_order, + .. + } => format!( + "{}int{}{}", + if *signed { "" } else { "u" }, + u64::from(*size) * 8, + if be(byte_order) { "-be" } else { "" } + ), + Datatype::FloatingPoint { + size, byte_order, .. + } => format!( + "float{}{}", + u64::from(*size) * 8, + if be(byte_order) { "-be" } else { "" } + ), + Datatype::Time { size, .. } => format!("time{}", u64::from(*size) * 8), + Datatype::String { size, .. } => format!("string[{size}]"), + Datatype::BitField { size, .. } => format!("bitfield{}", u64::from(*size) * 8), + Datatype::Opaque { size, .. } => format!("opaque[{size}]"), + Datatype::Compound { members, .. } => format!( + "compound{{{}}}", + members + .iter() + .map(|m| format!("{}: {}", m.name, short(&m.datatype))) + .collect::>() + .join(", ") + ), + Datatype::Reference { ref_type, .. } => match ref_type { + ReferenceType::Object | ReferenceType::Object2 => "object-reference".into(), + ReferenceType::DatasetRegion | ReferenceType::DatasetRegion2 => { + "region-reference".into() + } + ReferenceType::Attribute => "attribute-reference".into(), + }, + Datatype::Enumeration { base_type, .. } => format!("enum<{}>", short(base_type)), + Datatype::VariableLength { + is_string: true, .. + } => "vlen-string".into(), + Datatype::VariableLength { base_type, .. } => format!("vlen<{}>", short(base_type)), + Datatype::Array { + base_type, + dimensions, + } => format!( + "array[{}]<{}>", + dimensions + .iter() + .map(|d| d.to_string()) + .collect::>() + .join(","), + short(base_type) + ), + } +} + +fn pad_word(p: &StringPadding) -> &'static str { + match p { + StringPadding::NullTerminate => "null-terminated", + StringPadding::NullPad => "null-padded", + StringPadding::SpacePad => "space-padded", + } +} + +fn cset_word(c: &CharacterSet) -> &'static str { + match c { + CharacterSet::Ascii => "ASCII", + CharacterSet::Utf8 => "UTF-8", + } +} + +/// h5ls -v style description. +pub fn long(dt: &Datatype) -> String { + match dt { + Datatype::FixedPoint { + size, + signed, + byte_order, + .. + } => format!( + "{}-bit {} {}integer", + u64::from(*size) * 8, + order_word(byte_order), + if *signed { "" } else { "unsigned " } + ), + Datatype::FloatingPoint { + size, byte_order, .. + } => { + if is_ieee(dt) { + format!( + "IEEE {}-bit {} float", + u64::from(*size) * 8, + order_word(byte_order) + ) + } else { + format!( + "{}-bit {} non-IEEE float", + u64::from(*size) * 8, + order_word(byte_order) + ) + } + } + Datatype::Time { size, .. } => format!("{}-bit time", u64::from(*size) * 8), + Datatype::String { + size, + padding, + charset, + } => format!( + "{size}-byte {} {} string", + pad_word(padding), + cset_word(charset) + ), + Datatype::BitField { + size, byte_order, .. + } => format!( + "{}-bit {} bitfield", + u64::from(*size) * 8, + order_word(byte_order) + ), + Datatype::Opaque { size, tag } => { + let tag = String::from_utf8_lossy(tag); + let tag = tag.trim_end_matches('\0'); + if tag.is_empty() { + format!("{size}-byte opaque type") + } else { + format!("{size}-byte opaque type (tag \"{tag}\")") + } + } + Datatype::Compound { size, members } => { + let mut s = String::from("struct {"); + for m in members { + s.push_str(&format!( + "\n \"{}\" +{} {}", + m.name, + m.byte_offset, + long(&m.datatype) + )); + } + s.push_str(&format!("\n }} {size} bytes")); + s + } + Datatype::Reference { ref_type, .. } => match ref_type { + ReferenceType::Object | ReferenceType::Object2 => "object reference".into(), + ReferenceType::DatasetRegion | ReferenceType::DatasetRegion2 => { + "dataset region reference".into() + } + ReferenceType::Attribute => "attribute reference".into(), + }, + Datatype::Enumeration { + base_type, members, .. + } => { + let vals: Vec = members + .iter() + .map(|m| format!("{} = {}", m.name, enum_value_text(base_type, &m.value))) + .collect(); + format!("enum {} {{{}}}", long(base_type), vals.join(", ")) + } + Datatype::VariableLength { + is_string: true, + padding, + charset, + .. + } => format!( + "variable-length {} {} string", + padding.as_ref().map(pad_word).unwrap_or("null-terminated"), + charset.as_ref().map(cset_word).unwrap_or("ASCII") + ), + Datatype::VariableLength { base_type, .. } => { + format!("variable length of {}", long(base_type)) + } + Datatype::Array { + base_type, + dimensions, + } => format!( + "[{}] {}", + dimensions + .iter() + .map(|d| d.to_string()) + .collect::>() + .join(","), + long(base_type) + ), + } +} + +/// The integer an enum member's value bytes hold, as text. +pub fn enum_value_text(base: &Datatype, bytes: &[u8]) -> String { + match crate::value::decode_int(base, bytes) { + Some(v) => v.to_string(), + None => crate::value::hex(bytes), + } +} + +/// h5dump's predefined name for an atomic type, if it has one. +fn atomic_ddl(dt: &Datatype) -> Option { + match dt { + Datatype::FixedPoint { + size, + signed, + byte_order, + .. + } => Some(format!( + "H5T_STD_{}{}{}", + if *signed { "I" } else { "U" }, + u64::from(*size) * 8, + order_suffix(byte_order) + )), + Datatype::FloatingPoint { + size, byte_order, .. + } if is_ieee(dt) => Some(format!( + "H5T_IEEE_F{}{}", + u64::from(*size) * 8, + order_suffix(byte_order) + )), + Datatype::BitField { + size, byte_order, .. + } => Some(format!( + "H5T_STD_B{}{}", + u64::from(*size) * 8, + order_suffix(byte_order) + )), + Datatype::Time { .. } => Some("H5T_TIME".into()), + _ => None, + } +} + +/// h5dump DDL for a datatype; `ind` is the indentation of the line the type +/// starts on (continuation lines are indented relative to it). +pub fn ddl(dt: &Datatype, ind: usize) -> String { + let pad = " ".repeat(ind + 3); + let end = " ".repeat(ind); + if let Some(s) = atomic_ddl(dt) { + return s; + } + match dt { + Datatype::FloatingPoint { + size, + byte_order, + bit_offset, + bit_precision, + exponent_location, + exponent_size, + mantissa_location, + mantissa_size, + exponent_bias, + } => format!( + "H5T_FLOAT {{\n{pad}SIZE {size};\n{pad}ORDER H5T_ORDER_{};\n{pad}OFFSET {bit_offset};\n\ + {pad}PRECISION {bit_precision};\n{pad}EXPONENT {exponent_location} {exponent_size} \ + BIAS {exponent_bias};\n{pad}MANTISSA {mantissa_location} {mantissa_size};\n{end}}}", + order_suffix(byte_order) + ), + Datatype::String { + size, + padding, + charset, + } => string_ddl(&size.to_string(), padding, charset, &pad, &end), + Datatype::VariableLength { + is_string: true, + padding, + charset, + .. + } => string_ddl( + "H5T_VARIABLE", + padding.as_ref().unwrap_or(&StringPadding::NullTerminate), + charset.as_ref().unwrap_or(&CharacterSet::Ascii), + &pad, + &end, + ), + Datatype::VariableLength { base_type, .. } => { + format!("H5T_VLEN {{ {} }}", ddl(base_type, ind)) + } + Datatype::Opaque { size, tag } => { + let tag = String::from_utf8_lossy(tag); + format!( + "H5T_OPAQUE {{\n{pad}OPAQUE_SIZE {size};\n{pad}OPAQUE_TAG \"{}\";\n{end}}}", + tag.trim_end_matches('\0') + ) + } + Datatype::Compound { members, .. } => { + let mut s = String::from("H5T_COMPOUND {\n"); + for m in members { + s.push_str(&format!( + "{pad}{} \"{}\";\n", + ddl(&m.datatype, ind + 3), + m.name + )); + } + s.push_str(&end); + s.push('}'); + s + } + Datatype::Reference { ref_type, .. } => match ref_type { + ReferenceType::Object => "H5T_REFERENCE { H5T_STD_REF_OBJECT }".into(), + ReferenceType::DatasetRegion => "H5T_REFERENCE { H5T_STD_REF_DSETREG }".into(), + _ => "H5T_REFERENCE { H5T_STD_REF }".into(), + }, + Datatype::Enumeration { + base_type, members, .. + } => { + let mut s = format!("H5T_ENUM {{\n{pad}{};\n", ddl(base_type, ind + 3)); + for m in members { + let name = format!("\"{}\"", m.name); + s.push_str(&format!( + "{pad}{name:<18} {};\n", + enum_value_text(base_type, &m.value) + )); + } + s.push_str(&end); + s.push('}'); + s + } + Datatype::Array { + base_type, + dimensions, + } => format!( + "H5T_ARRAY {{ {} {} }}", + dimensions + .iter() + .map(|d| format!("[{d}]")) + .collect::(), + ddl(base_type, ind) + ), + // Atomic types were handled above. + _ => short(dt), + } +} + +fn string_ddl( + size: &str, + padding: &StringPadding, + charset: &CharacterSet, + pad: &str, + end: &str, +) -> String { + let p = match padding { + StringPadding::NullTerminate => "H5T_STR_NULLTERM", + StringPadding::NullPad => "H5T_STR_NULLPAD", + StringPadding::SpacePad => "H5T_STR_SPACEPAD", + }; + let c = match charset { + CharacterSet::Ascii => "H5T_CSET_ASCII", + CharacterSet::Utf8 => "H5T_CSET_UTF8", + }; + format!( + "H5T_STRING {{\n{pad}STRSIZE {size};\n{pad}STRPAD {p};\n{pad}CSET {c};\n{pad}CTYPE H5T_C_S1;\n{end}}}" + ) +} + +/// hdf5-json type object. +pub fn json(dt: &Datatype) -> J { + match dt { + Datatype::FixedPoint { .. } => { + json!({"class": "H5T_INTEGER", "base": atomic_ddl(dt)}) + } + Datatype::FloatingPoint { .. } if is_ieee(dt) => { + json!({"class": "H5T_FLOAT", "base": atomic_ddl(dt)}) + } + Datatype::FloatingPoint { + size, byte_order, .. + } => json!({ + "class": "H5T_FLOAT", + "size": size, + "order": format!("H5T_ORDER_{}", order_suffix(byte_order)), + }), + Datatype::BitField { .. } => json!({"class": "H5T_BITFIELD", "base": atomic_ddl(dt)}), + Datatype::Time { size, .. } => json!({"class": "H5T_TIME", "size": size}), + Datatype::String { + size, + padding, + charset, + } => json!({ + "class": "H5T_STRING", + "charSet": cset_json(charset), + "strPad": pad_json(padding), + "length": size, + }), + Datatype::VariableLength { + is_string: true, + padding, + charset, + .. + } => json!({ + "class": "H5T_STRING", + "charSet": cset_json(charset.as_ref().unwrap_or(&CharacterSet::Ascii)), + "strPad": pad_json(padding.as_ref().unwrap_or(&StringPadding::NullTerminate)), + "length": "H5T_VARIABLE", + }), + Datatype::VariableLength { base_type, .. } => { + json!({"class": "H5T_VLEN", "base": json(base_type)}) + } + Datatype::Opaque { size, tag } => json!({ + "class": "H5T_OPAQUE", + "size": size, + "tag": String::from_utf8_lossy(tag).trim_end_matches('\0'), + }), + Datatype::Compound { members, .. } => json!({ + "class": "H5T_COMPOUND", + "fields": members + .iter() + .map(|m| json!({"name": m.name, "type": json(&m.datatype)})) + .collect::>(), + }), + Datatype::Reference { ref_type, .. } => json!({ + "class": "H5T_REFERENCE", + "base": match ref_type { + ReferenceType::Object => "H5T_STD_REF_OBJ", + ReferenceType::DatasetRegion => "H5T_STD_REF_DSETREG", + _ => "H5T_STD_REF", + }, + }), + Datatype::Enumeration { + base_type, members, .. + } => { + let mut mapping = serde_json::Map::new(); + for m in members { + let v = crate::value::decode_int(base_type, &m.value) + .and_then(|v| i64::try_from(v).ok()) + .map(J::from) + .unwrap_or_else(|| J::from(crate::value::hex(&m.value))); + mapping.insert(m.name.clone(), v); + } + json!({"class": "H5T_ENUM", "base": json(base_type), "mapping": mapping}) + } + Datatype::Array { + base_type, + dimensions, + } => json!({"class": "H5T_ARRAY", "base": json(base_type), "dims": dimensions}), + } +} + +fn cset_json(c: &CharacterSet) -> &'static str { + match c { + CharacterSet::Ascii => "H5T_CSET_ASCII", + CharacterSet::Utf8 => "H5T_CSET_UTF8", + } +} + +fn pad_json(p: &StringPadding) -> &'static str { + match p { + StringPadding::NullTerminate => "H5T_STR_NULLTERM", + StringPadding::NullPad => "H5T_STR_NULLPAD", + StringPadding::SpacePad => "H5T_STR_SPACEPAD", + } +} + +/// Class name used to decide whether two datatypes can be compared. +pub fn class(dt: &Datatype) -> &'static str { + match dt { + Datatype::FixedPoint { .. } => "integer", + Datatype::FloatingPoint { .. } => "float", + Datatype::Time { .. } => "time", + Datatype::String { .. } => "string", + Datatype::BitField { .. } => "bitfield", + Datatype::Opaque { .. } => "opaque", + Datatype::Compound { .. } => "compound", + Datatype::Reference { .. } => "reference", + Datatype::Enumeration { .. } => "enum", + Datatype::VariableLength { + is_string: true, .. + } => "string", + Datatype::VariableLength { .. } => "vlen", + Datatype::Array { .. } => "array", + } +} diff --git a/crates/clawhdf5-tools/src/dump.rs b/crates/clawhdf5-tools/src/dump.rs new file mode 100644 index 0000000..7b1d8e7 --- /dev/null +++ b/crates/clawhdf5-tools/src/dump.rs @@ -0,0 +1,1002 @@ +//! `h5rs dump`: the whole file (or one dataset) as h5dump-style DDL text or +//! as hdf5-json. + +use std::cell::OnceCell; +use std::collections::HashMap; + +use clawhdf5_format::attribute::AttributeMessage; +use clawhdf5_format::data_layout::DataLayout; +use clawhdf5_format::dataspace::{Dataspace, DataspaceType}; +use clawhdf5_format::datatype::{Datatype, StringPadding}; +use clawhdf5_format::object_header::ObjectHeader; +use serde_json::{Map, Value as J, json}; + +use crate::cli::{Args, Out}; +use crate::h5::{Error, H5, Kind, Link, LinkKind}; +use crate::info::{self, DsInfo}; +use crate::value::{self, Decoder, Value}; + +pub const USAGE: &str = "\ +usage: h5rs dump [--json] [-A] [-p] [-d PATH] [--max-bytes N] FILE + +Print FILE's groups, datasets, named datatypes, links and attributes, with +their values, as h5dump-style DDL text (default) or as JSON. + + --json hdf5-json layout (see the crate README for the schema) + -A, --header no dataset values (attribute values are still printed, + as with h5dump -A) + -p, --properties also print each dataset's storage layout and filters + -d, --dataset P dump only the dataset at path P + --max-bytes N largest dataset or attribute decoded (default 1 GiB); + a larger one is reported instead of read + +Exit status: 0 dumped, 1 something could not be read (reported on stderr +and marked in the output), 2 error."; + +struct Opts { + json: bool, + header_only: bool, + props: bool, +} + +struct Dump<'a> { + h5: &'a H5, + opts: Opts, + problems: usize, + paths: OnceCell>, +} + +pub fn run(args: &mut Args, out: &mut Out) -> std::io::Result { + let mut opts = Opts { + json: false, + header_only: false, + props: false, + }; + let mut dataset = None; + let mut max_bytes = None; + let mut file = None; + while let Some(a) = args.next() { + match a.as_str() { + "--json" | "-j" => opts.json = true, + "-A" | "--header" => opts.header_only = true, + "-p" | "--properties" => opts.props = true, + "-d" | "--dataset" => match args.value() { + Some(p) => dataset = Some(p), + None => return args.usage_error(out, "-d needs a path", USAGE), + }, + "--max-bytes" => match args.number() { + Some(n) => max_bytes = Some(n), + None => return args.usage_error(out, "--max-bytes needs a number", USAGE), + }, + "-h" | "--help" => { + writeln!(out.o, "{USAGE}")?; + return Ok(0); + } + s if s.starts_with('-') && s.len() > 1 => { + return args.usage_error(out, &format!("unknown option {a}"), USAGE); + } + _ if file.is_none() => file = Some(a), + _ => return args.usage_error(out, &format!("unexpected argument {a}"), USAGE), + } + } + let Some(file) = file else { + return args.usage_error(out, "missing FILE", USAGE); + }; + let mut h5 = match H5::open(std::path::Path::new(&file)) { + Ok(h) => h, + Err(e) => { + writeln!(out.e, "h5rs dump: {e}")?; + return Ok(2); + } + }; + if let Some(m) = max_bytes { + h5.max_bytes = m; + } + let mut d = Dump { + h5: &h5, + opts, + problems: 0, + paths: OnceCell::new(), + }; + let fname = std::path::Path::new(&file) + .file_name() + .map(|s| s.to_string_lossy().into_owned()) + .unwrap_or(file.clone()); + let code = if d.opts.json { + d.json(out, dataset.as_deref())? + } else { + d.ddl(out, &fname, dataset.as_deref())? + }; + if code != 0 { + return Ok(code); + } + Ok(if d.problems > 0 { 1 } else { 0 }) +} + +/// A full path for `name` inside the group at `base`. +fn join(base: &str, name: &str) -> String { + if base == "/" { + format!("/{name}") + } else { + format!("{base}/{name}") + } +} + +fn quote(s: &str) -> String { + s.replace('\\', "\\\\").replace('"', "\\\"") +} + +impl Dump<'_> { + /// Object address -> first path, for printing references. + fn paths(&self) -> &HashMap { + self.paths.get_or_init(|| { + let mut m = HashMap::new(); + let _ = self.h5.walk(|it| { + if let (Some(a), None) = (it.addr, it.first_path) { + m.entry(a).or_insert_with(|| it.path.to_string()); + } + }); + m + }) + } + + fn problem(&mut self, out: &mut Out, what: &str, e: &Error) -> std::io::Result<()> { + self.problems += 1; + writeln!(out.e, "h5rs dump: {what}: {e}") + } + + // ------------------------------------------------------------------ + // DDL + // ------------------------------------------------------------------ + + fn ddl(&mut self, out: &mut Out, fname: &str, only: Option<&str>) -> std::io::Result { + writeln!(out.o, "HDF5 \"{}\" {{", quote(fname))?; + if let Some(p) = only { + let h = match self.h5.resolve(p).and_then(|a| self.h5.header(a)) { + Ok(h) => h, + Err(e) => { + writeln!(out.o, "}}")?; + writeln!(out.e, "h5rs dump: {p}: {e}")?; + return Ok(2); + } + }; + if Kind::of(&h) != Kind::Dataset { + writeln!(out.o, "}}")?; + writeln!(out.e, "h5rs dump: {p}: not a dataset")?; + return Ok(2); + } + let full = if p.starts_with('/') { + p.to_string() + } else { + format!("/{p}") + }; + self.ddl_dataset(out, &full, &full, &h, 0)?; + } else { + let root = self.h5.root(); + let mut seen = HashMap::new(); + match self.h5.header(root) { + Ok(h) => self.ddl_group(out, "/", "/", root, &h, 0, &mut seen)?, + Err(e) => { + writeln!(out.o, "GROUP \"/\" {{\n}}")?; + self.problem(out, "/", &e)?; + } + } + } + writeln!(out.o, "}}")?; + Ok(0) + } + + #[allow(clippy::too_many_arguments)] + fn ddl_group( + &mut self, + out: &mut Out, + name: &str, + path: &str, + addr: u64, + h: &ObjectHeader, + ind: usize, + seen: &mut HashMap, + ) -> std::io::Result<()> { + let pad = " ".repeat(ind); + seen.insert(addr, path.to_string()); + writeln!(out.o, "{pad}GROUP \"{}\" {{", quote(name))?; + self.ddl_attributes(out, path, h, ind + 3)?; + let links = match self.h5.links(h) { + Ok(l) => l, + Err(e) => { + self.problem(out, path, &e)?; + Vec::new() + } + }; + // The DDL recursion follows the group nesting; bound it (the walk used + // by the other commands is iterative). + if seen.len() > crate::h5::MAX_OBJECTS || ind > 3 * MAX_DDL_DEPTH { + let e = Error::new("too many objects, or groups nested too deeply; stopped"); + self.problem(out, path, &e)?; + return writeln!(out.o, "{pad}}}"); + } + for l in &links { + self.ddl_link(out, path, l, ind + 3, seen)?; + } + writeln!(out.o, "{pad}}}") + } + + fn ddl_link( + &mut self, + out: &mut Out, + base: &str, + l: &Link, + ind: usize, + seen: &mut HashMap, + ) -> std::io::Result<()> { + let pad = " ".repeat(ind); + let inner = " ".repeat(ind + 3); + let name = quote(&l.name); + let path = join(base, &l.name); + match &l.kind { + LinkKind::Soft(t) => writeln!( + out.o, + "{pad}SOFTLINK \"{name}\" {{\n{inner}LINKTARGET \"{}\"\n{pad}}}", + quote(t) + ), + LinkKind::External { file, path: p } => writeln!( + out.o, + "{pad}EXTERNAL_LINK \"{name}\" {{\n{inner}TARGETFILE \"{}\"\n{inner}TARGETPATH \"{}\"\n{pad}}}", + quote(file), + quote(p) + ), + LinkKind::UserDefined(t) => writeln!( + out.o, + "{pad}USERDEFINED_LINK \"{name}\" {{\n{inner}LINKCLASS {t}\n{pad}}}" + ), + LinkKind::Hard(a) => { + let h = match self.h5.header(*a) { + Ok(h) => h, + Err(e) => { + self.problem(out, &path, &e)?; + return writeln!(out.o, "{pad}UNKNOWN_OBJECT \"{name}\" {{\n{pad}}}"); + } + }; + let kind = Kind::of(&h); + if let Some(first) = seen.get(a) { + let word = match kind { + Kind::Group => "GROUP", + Kind::Dataset => "DATASET", + Kind::Datatype => "DATATYPE", + Kind::Unknown => "OBJECT", + }; + return writeln!( + out.o, + "{pad}{word} \"{name}\" {{\n{inner}HARDLINK \"{}\"\n{pad}}}", + quote(first) + ); + } + match kind { + Kind::Group => self.ddl_group(out, &l.name, &path, *a, &h, ind, seen), + Kind::Dataset => { + seen.insert(*a, path.clone()); + self.ddl_dataset(out, &l.name, &path, &h, ind) + } + Kind::Datatype => { + seen.insert(*a, path.clone()); + match self.h5.datatype(&h) { + Ok(dt) => writeln!( + out.o, + "{pad}DATATYPE \"{name}\" {};", + crate::dtype::ddl(&dt, ind) + ), + Err(e) => { + self.problem(out, &path, &e)?; + writeln!(out.o, "{pad}DATATYPE \"{name}\" ?;") + } + } + } + Kind::Unknown => { + seen.insert(*a, path.clone()); + writeln!(out.o, "{pad}UNKNOWN_OBJECT \"{name}\" {{\n{pad}}}") + } + } + } + } + } + + fn ddl_dataset( + &mut self, + out: &mut Out, + name: &str, + path: &str, + h: &ObjectHeader, + ind: usize, + ) -> std::io::Result<()> { + let pad = " ".repeat(ind); + let inner = " ".repeat(ind + 3); + writeln!(out.o, "{pad}DATASET \"{}\" {{", quote(name))?; + let info = DsInfo::read(self.h5, path, h); + match &info.dt { + Ok(dt) => writeln!(out.o, "{inner}DATATYPE {}", crate::dtype::ddl(dt, ind + 3))?, + Err(e) => self.problem(out, path, e)?, + } + match &info.ds { + Ok(ds) => writeln!(out.o, "{inner}DATASPACE {}", info::dataspace_ddl(ds))?, + Err(e) => self.problem(out, path, e)?, + } + if self.opts.props { + self.ddl_properties(out, path, &info, ind + 3)?; + } + if !self.opts.header_only + && let (Ok(dt), Ok(ds)) = (&info.dt, &info.ds) + { + match self.h5.read_dataset(path, dt, ds) { + Ok(raw) => self.ddl_data(out, dt, ds, &raw, ind + 3)?, + Err(e) => { + writeln!(out.o, "{inner}DATA {{\n{inner}\n{inner}}}")?; + self.problem(out, path, &e)?; + } + } + } + self.ddl_attributes(out, path, h, ind + 3)?; + writeln!(out.o, "{pad}}}") + } + + fn ddl_properties( + &mut self, + out: &mut Out, + path: &str, + info: &DsInfo, + ind: usize, + ) -> std::io::Result<()> { + let pad = " ".repeat(ind); + let inner = " ".repeat(ind + 3); + let Ok(layout) = &info.layout else { + if let Err(e) = &info.layout { + self.problem(out, path, e)?; + } + return Ok(()); + }; + writeln!(out.o, "{pad}STORAGE_LAYOUT {{")?; + match layout { + DataLayout::Chunked { + chunk_dimensions, .. + } => { + let rank = chunk_dimensions.len().saturating_sub(1); + writeln!( + out.o, + "{inner}CHUNKED ( {} )", + chunk_dimensions[..rank] + .iter() + .map(|d| d.to_string()) + .collect::>() + .join(", ") + )?; + writeln!( + out.o, + "{inner}INDEX {}", + info::chunk_index_name(layout).to_uppercase() + )?; + } + DataLayout::Contiguous { address, size } => { + writeln!(out.o, "{inner}CONTIGUOUS")?; + match address { + Some(a) => writeln!(out.o, "{inner}OFFSET {a}")?, + None => writeln!(out.o, "{inner}NOT ALLOCATED")?, + } + writeln!(out.o, "{inner}SIZE {size}")?; + } + DataLayout::Compact { data } => { + writeln!(out.o, "{inner}COMPACT")?; + writeln!(out.o, "{inner}SIZE {}", data.len())?; + } + DataLayout::Virtual { .. } => writeln!(out.o, "{inner}VIRTUAL")?, + } + if let Ok(a) = info::allocated_bytes(self.h5, info) + && matches!(layout, DataLayout::Chunked { .. }) + { + writeln!(out.o, "{inner}SIZE {a}")?; + } + writeln!(out.o, "{pad}}}")?; + writeln!(out.o, "{pad}FILTERS {{")?; + match &info.filters { + Ok(Some(p)) if !p.filters.is_empty() => { + for f in &p.filters { + writeln!( + out.o, + "{inner}{} {{ ID {}; PARAMS {{ {} }} }}", + info::filter_name(f).to_uppercase(), + f.filter_id, + f.client_data + .iter() + .map(|c| c.to_string()) + .collect::>() + .join(" ") + )?; + } + } + Ok(_) => writeln!(out.o, "{inner}NONE")?, + Err(e) => self.problem(out, path, e)?, + } + writeln!(out.o, "{pad}}}") + } + + fn ddl_attributes( + &mut self, + out: &mut Out, + path: &str, + h: &ObjectHeader, + ind: usize, + ) -> std::io::Result<()> { + let (attrs, errs) = match self.h5.attributes(h) { + Ok(x) => x, + Err(e) => return self.problem(out, path, &e), + }; + for e in errs { + self.problem(out, path, &Error::new(format!("attribute: {e}")))?; + } + for a in &attrs { + self.ddl_attribute(out, path, a, ind)?; + } + Ok(()) + } + + fn ddl_attribute( + &mut self, + out: &mut Out, + path: &str, + a: &AttributeMessage, + ind: usize, + ) -> std::io::Result<()> { + let pad = " ".repeat(ind); + let inner = " ".repeat(ind + 3); + writeln!(out.o, "{pad}ATTRIBUTE \"{}\" {{", quote(&a.name))?; + writeln!( + out.o, + "{inner}DATATYPE {}", + crate::dtype::ddl(&a.datatype, ind + 3) + )?; + writeln!( + out.o, + "{inner}DATASPACE {}", + info::dataspace_ddl(&a.dataspace) + )?; + { + match self.checked_attr(a) { + Ok(()) => self.ddl_data(out, &a.datatype, &a.dataspace, &a.raw_data, ind + 3)?, + Err(e) => { + writeln!(out.o, "{inner}DATA {{\n{inner}\n{inner}}}")?; + self.problem(out, &format!("{path} attribute {}", a.name), &e)?; + } + } + } + writeln!(out.o, "{pad}}}") + } + + /// An attribute's value holds as many bytes as its dataspace says. + fn checked_attr(&self, a: &AttributeMessage) -> crate::h5::Result<()> { + let need = crate::h5::byte_len(&a.dataspace, &a.datatype)?; + if need > self.h5.max_bytes { + return Err(Error::new(format!( + "attribute is {need} bytes, over the {} byte limit (--max-bytes)", + self.h5.max_bytes + ))); + } + if (a.raw_data.len() as u64) < need { + return Err(Error::new(format!( + "attribute holds {} bytes, its dataspace needs {need}", + a.raw_data.len() + ))); + } + Ok(()) + } + + /// An h5dump DATA block: elements row by row, each line starting with + /// the index of its first element, wrapped where h5dump wraps (78 columns). + fn ddl_data( + &mut self, + out: &mut Out, + dt: &Datatype, + ds: &Dataspace, + raw: &[u8], + ind: usize, + ) -> std::io::Result<()> { + let pad = " ".repeat(ind); + writeln!(out.o, "{pad}DATA {{")?; + let dims: Vec = match ds.space_type { + DataspaceType::Null => { + return writeln!(out.o, "{pad}}}"); + } + DataspaceType::Scalar => vec![1], + DataspaceType::Simple => ds.dimensions.clone(), + }; + let n = crate::h5::num_elements(ds).unwrap_or(0) as usize; + let dec = Decoder::new(self.h5); + let last = dims.last().copied().unwrap_or(1).max(1) as usize; + let paths = |a: u64| self.paths().get(&a).cloned(); + let mut line = String::new(); + let mut errors = 0usize; + for i in 0..n { + let v = dec.element(dt, raw, i); + if matches!(v, Value::Error(_)) { + errors += 1; + } + let comma = if i + 1 < n { "," } else { "" }; + let esize = dt.type_size() as usize; + let eb = raw.get(i * esize..(i + 1) * esize).unwrap_or_default(); + if let (Value::Compound(_), Datatype::Compound { members, .. }) = (&v, dt) { + // h5dump prints each compound element as a block, one + // member per line. + if !line.is_empty() { + writeln!(out.o, "{line}")?; + line.clear(); + } + writeln!(out.o, "{pad}({}): {{", index_text(i as u64, &dims))?; + for (k, m) in members.iter().enumerate() { + let sep = if k + 1 < members.len() { "," } else { "" }; + let mb = usize::try_from(m.byte_offset) + .ok() + .and_then(|o| eb.get(o..)) + .unwrap_or_default(); + let t = ddl_text(&dec, &m.datatype, mb, &paths, 1); + writeln!(out.o, "{pad} {t}{sep}")?; + } + writeln!(out.o, "{pad} }}{comma}")?; + continue; + } + let t = if matches!(v, Value::Error(_)) { + format!("{}{comma}", value::text(&v, &paths)) + } else { + format!("{}{comma}", ddl_text(&dec, dt, eb, &paths, 0)) + }; + let row_start = i % last == 0; + if row_start || line.len() + 1 + t.len() > 77 { + if !line.is_empty() { + writeln!(out.o, "{line}")?; + } + line = format!("{pad}({}): {t}", index_text(i as u64, &dims)); + } else { + line.push(' '); + line.push_str(&t); + } + } + if !line.is_empty() { + writeln!(out.o, "{line}")?; + } + if errors > 0 { + self.problems += 1; + writeln!(out.e, "h5rs dump: {errors} element(s) could not be decoded")?; + } + writeln!(out.o, "{pad}}}") + } + + // ------------------------------------------------------------------ + // JSON + // ------------------------------------------------------------------ + + fn json(&mut self, out: &mut Out, only: Option<&str>) -> std::io::Result { + let mut groups = Map::new(); + let mut datasets = Map::new(); + let mut datatypes = Map::new(); + let mut doc = Map::new(); + doc.insert("apiVersion".into(), json!("1.1.1")); + if let Some(p) = only { + let full = if p.starts_with('/') { + p.to_string() + } else { + format!("/{p}") + }; + let r = self + .h5 + .resolve(&full) + .and_then(|a| Ok((a, self.h5.header(a)?))); + match r { + Ok((a, h)) if Kind::of(&h) == Kind::Dataset => { + let obj = self.json_dataset(out, &full, &h)?; + datasets.insert(obj_id(Kind::Dataset, a), obj); + } + Ok(_) => { + writeln!(out.e, "h5rs dump: {p}: not a dataset")?; + return Ok(2); + } + Err(e) => { + writeln!(out.e, "h5rs dump: {p}: {e}")?; + return Ok(2); + } + } + } else { + doc.insert("root".into(), json!(obj_id(Kind::Group, self.h5.root()))); + let (items, walk) = self.h5.walk_collect(); + if let Err(e) = walk { + self.problem(out, "/", &e)?; + } + // Aliases: every path an object is reachable by. + let mut aliases: HashMap> = HashMap::new(); + for it in &items { + if let Some(a) = it.addr { + aliases.entry(a).or_default().push(it.path.clone()); + } + } + for it in items { + let path = it.path; + let (Some(addr), None, Some(header)) = (it.addr, it.first_path, it.header) else { + continue; + }; + let h = match header { + Ok(h) => h, + Err(e) => { + self.problem(out, &path, &e)?; + continue; + } + }; + let kind = Kind::of(&h); + let alias = aliases.remove(&addr).unwrap_or_default(); + let mut obj = match kind { + Kind::Dataset => self.json_dataset(out, &path, &h)?, + Kind::Datatype => { + let mut m = Map::new(); + match self.h5.datatype(&h) { + Ok(dt) => { + m.insert("type".into(), crate::dtype::json(&dt)); + } + Err(e) => self.problem(out, &path, &e)?, + } + m.insert("attributes".into(), self.json_attributes(out, &path, &h)?); + J::Object(m) + } + _ => { + let mut m = Map::new(); + m.insert("attributes".into(), self.json_attributes(out, &path, &h)?); + let links = match self.h5.links(&h) { + Ok(l) => l, + Err(e) => { + if kind == Kind::Group { + self.problem(out, &path, &e)?; + } + Vec::new() + } + }; + let mut jl = Vec::new(); + for l in &links { + jl.push(self.json_link(l)); + } + m.insert("links".into(), J::Array(jl)); + J::Object(m) + } + }; + if let J::Object(m) = &mut obj { + m.insert("alias".into(), json!(alias)); + } + let id = obj_id(kind, addr); + match kind { + Kind::Dataset => datasets.insert(id, obj), + Kind::Datatype => datatypes.insert(id, obj), + _ => groups.insert(id, obj), + }; + } + doc.insert("groups".into(), J::Object(groups)); + } + doc.insert("datasets".into(), J::Object(datasets)); + if only.is_none() { + doc.insert("datatypes".into(), J::Object(datatypes)); + } + let text = serde_json::to_string_pretty(&J::Object(doc)).map_err(std::io::Error::other)?; + writeln!(out.o, "{text}")?; + Ok(0) + } + + fn json_link(&self, l: &Link) -> J { + match &l.kind { + LinkKind::Hard(a) => { + let kind = self + .h5 + .header(*a) + .map(|h| Kind::of(&h)) + .unwrap_or(Kind::Unknown); + let coll = match kind { + Kind::Dataset => "datasets", + Kind::Datatype => "datatypes", + _ => "groups", + }; + json!({"class": "H5L_TYPE_HARD", "title": l.name, "collection": coll, + "id": obj_id(kind, *a)}) + } + LinkKind::Soft(t) => json!({"class": "H5L_TYPE_SOFT", "title": l.name, "h5path": t}), + LinkKind::External { file, path } => json!({"class": "H5L_TYPE_EXTERNAL", + "title": l.name, "file": file, "h5path": path}), + LinkKind::UserDefined(t) => json!({"class": "H5L_TYPE_USER_DEFINED", + "title": l.name, "linkClass": t}), + } + } + + fn json_dataset(&mut self, out: &mut Out, path: &str, h: &ObjectHeader) -> std::io::Result { + let info = DsInfo::read(self.h5, path, h); + let mut m = Map::new(); + match &info.dt { + Ok(dt) => { + m.insert("type".into(), crate::dtype::json(dt)); + } + Err(e) => self.problem(out, path, e)?, + } + match &info.ds { + Ok(ds) => { + m.insert("shape".into(), shape_json(ds)); + } + Err(e) => self.problem(out, path, e)?, + } + let mut cp = Map::new(); + if let Ok(l) = &info.layout { + let mut lj = Map::new(); + lj.insert( + "class".into(), + json!(match l { + DataLayout::Compact { .. } => "H5D_COMPACT", + DataLayout::Contiguous { .. } => "H5D_CONTIGUOUS", + DataLayout::Chunked { .. } => "H5D_CHUNKED", + DataLayout::Virtual { .. } => "H5D_VIRTUAL", + }), + ); + if let DataLayout::Chunked { + chunk_dimensions, .. + } = l + { + let rank = chunk_dimensions.len().saturating_sub(1); + lj.insert("dims".into(), json!(chunk_dimensions[..rank])); + } + cp.insert("layout".into(), J::Object(lj)); + } else if let Err(e) = &info.layout { + self.problem(out, path, e)?; + } + if let Ok(Some(p)) = &info.filters { + let fl: Vec = p + .filters + .iter() + .map(|f| { + json!({"id": f.filter_id, "name": info::filter_name(f), + "class": filter_class(f.filter_id), "parameters": f.client_data}) + }) + .collect(); + cp.insert("filters".into(), J::Array(fl)); + } + m.insert("creationProperties".into(), J::Object(cp)); + if !self.opts.header_only + && let (Ok(dt), Ok(ds)) = (&info.dt, &info.ds) + { + // JSON values are built in memory, some 32+ bytes per element: + // hold them to the same budget as the raw data. + let n = crate::h5::num_elements(ds).unwrap_or(u64::MAX); + let read = if n.saturating_mul(JSON_BYTES_PER_ELEMENT) > self.h5.max_bytes { + Err(Error::new(format!( + "{n} elements are too many to hold as JSON within --max-bytes {}", + self.h5.max_bytes + )) + .with_kind(crate::h5::ErrorKind::Limit)) + } else { + self.h5.read_dataset(path, dt, ds) + }; + match read { + Ok(raw) => { + let v = self.json_values(out, dt, ds, &raw)?; + m.insert("value".into(), v); + } + Err(e) => { + m.insert("value_error".into(), json!(e.to_string())); + self.problem(out, path, &e)?; + } + } + } + m.insert("attributes".into(), self.json_attributes(out, path, h)?); + Ok(J::Object(m)) + } + + fn json_attributes( + &mut self, + out: &mut Out, + path: &str, + h: &ObjectHeader, + ) -> std::io::Result { + let (attrs, errs) = match self.h5.attributes(h) { + Ok(x) => x, + Err(e) => { + self.problem(out, path, &e)?; + return Ok(J::Array(Vec::new())); + } + }; + for e in errs { + self.problem(out, path, &Error::new(format!("attribute: {e}")))?; + } + let mut v = Vec::new(); + for a in &attrs { + let mut m = Map::new(); + m.insert("name".into(), json!(a.name)); + m.insert("shape".into(), shape_json(&a.dataspace)); + m.insert("type".into(), crate::dtype::json(&a.datatype)); + { + match self.checked_attr(a) { + Ok(()) => { + let val = self.json_values(out, &a.datatype, &a.dataspace, &a.raw_data)?; + m.insert("value".into(), val); + } + Err(e) => { + m.insert("value_error".into(), json!(e.to_string())); + self.problem(out, &format!("{path} attribute {}", a.name), &e)?; + } + } + } + v.push(J::Object(m)); + } + Ok(J::Array(v)) + } + + fn json_values( + &mut self, + out: &mut Out, + dt: &Datatype, + ds: &Dataspace, + raw: &[u8], + ) -> std::io::Result { + let n = crate::h5::num_elements(ds).unwrap_or(0) as usize; + let dec = Decoder::new(self.h5); + let paths = |a: u64| self.paths().get(&a).cloned(); + let mut flat = Vec::with_capacity(n); + let mut errors = 0usize; + for i in 0..n { + let v = dec.element(dt, raw, i); + if matches!(v, Value::Error(_)) { + errors += 1; + } + flat.push(value::to_json(&v, &paths)); + } + if errors > 0 { + self.problems += 1; + writeln!(out.e, "h5rs dump: {errors} element(s) could not be decoded")?; + } + Ok(match ds.space_type { + DataspaceType::Null => J::Null, + DataspaceType::Scalar => flat.into_iter().next().unwrap_or(J::Null), + DataspaceType::Simple => nest(flat, &ds.dimensions), + }) + } +} + +/// Deepest group nesting the DDL output follows. +const MAX_DDL_DEPTH: usize = 256; + +/// Memory budgeted per element for a dataset's value held as JSON. +const JSON_BYTES_PER_ELEMENT: u64 = 64; + +/// `(i,j,k)` index of flat element `i` of an array with `dims`. +/// One element as h5dump prints it in a DATA block: [`value::text`], except +/// that a null-padded fixed string shows its padding (every byte, NULs as +/// `\000`) at any depth, in compound members and array elements too. +fn ddl_text( + dec: &Decoder, + dt: &Datatype, + b: &[u8], + paths: &dyn Fn(u64) -> Option, + depth: u32, +) -> String { + let err = || value::text(&Value::Error("short element".into()), paths); + if depth > 32 { + return value::text(&dec.decode(dt, b, depth), paths); + } + match dt { + Datatype::String { + padding: StringPadding::NullPad, + size, + .. + } => match b.get(..*size as usize) { + Some(s) => value::quote_bytes(s), + None => err(), + }, + Datatype::Compound { members, .. } => format!( + "{{ {} }}", + members + .iter() + .map( + |m| match usize::try_from(m.byte_offset).ok().and_then(|o| b.get(o..)) { + Some(mb) => ddl_text(dec, &m.datatype, mb, paths, depth + 1), + None => err(), + } + ) + .collect::>() + .join(", ") + ), + Datatype::Array { + base_type, + dimensions, + } => { + let bs = base_type.type_size() as usize; + let n = dimensions + .iter() + .try_fold(1usize, |a, &d| a.checked_mul(d as usize)); + match n.filter(|n| n.checked_mul(bs).is_some_and(|t| t <= b.len())) { + Some(n) => format!( + "[ {} ]", + (0..n) + .map(|k| ddl_text(dec, base_type, &b[k * bs..], paths, depth + 1)) + .collect::>() + .join(", ") + ), + None => value::text(&dec.decode(dt, b, depth), paths), + } + } + _ => value::text(&dec.decode(dt, b, depth), paths), + } +} + +fn index_text(mut i: u64, dims: &[u64]) -> String { + let mut idx = vec![0u64; dims.len()]; + for (k, &d) in dims.iter().enumerate().rev() { + let d = d.max(1); + idx[k] = i % d; + i /= d; + } + idx.iter() + .map(|x| x.to_string()) + .collect::>() + .join(",") +} + +/// Turn a flat row-major list into nested lists of shape `dims`. +fn nest(flat: Vec, dims: &[u64]) -> J { + if dims.len() <= 1 { + return J::Array(flat); + } + let inner: usize = dims[1..].iter().map(|&d| d as usize).product(); + if inner == 0 { + return J::Array((0..dims[0]).map(|_| nest(Vec::new(), &dims[1..])).collect()); + } + let mut it = flat.into_iter(); + let mut outer = Vec::with_capacity(dims[0] as usize); + for _ in 0..dims[0] { + let part: Vec = it.by_ref().take(inner).collect(); + outer.push(nest(part, &dims[1..])); + } + J::Array(outer) +} + +fn shape_json(ds: &Dataspace) -> J { + match ds.space_type { + DataspaceType::Null => json!({"class": "H5S_NULL"}), + DataspaceType::Scalar => json!({"class": "H5S_SCALAR"}), + DataspaceType::Simple => { + let mut m = Map::new(); + m.insert("class".into(), json!("H5S_SIMPLE")); + m.insert("dims".into(), json!(ds.dimensions)); + if let Some(max) = &ds.max_dimensions { + let mx: Vec = max + .iter() + .map(|&d| { + if d == u64::MAX { + json!("H5S_UNLIMITED") + } else { + json!(d) + } + }) + .collect(); + m.insert("maxdims".into(), J::Array(mx)); + } + J::Object(m) + } + } +} + +fn filter_class(id: u16) -> &'static str { + match id { + 1 => "H5Z_FILTER_DEFLATE", + 2 => "H5Z_FILTER_SHUFFLE", + 3 => "H5Z_FILTER_FLETCHER32", + 4 => "H5Z_FILTER_SZIP", + 5 => "H5Z_FILTER_NBIT", + 6 => "H5Z_FILTER_SCALEOFFSET", + _ => "H5Z_FILTER_USER", + } +} + +/// Deterministic object id: kind prefix + header address (hdf5-json uses +/// UUIDs; these are stable for a given file instead). +pub fn obj_id(kind: Kind, addr: u64) -> String { + let p = match kind { + Kind::Dataset => 'd', + Kind::Datatype => 't', + _ => 'g', + }; + format!("{p}-{addr:016x}") +} diff --git a/crates/clawhdf5-tools/src/h5.rs b/crates/clawhdf5-tools/src/h5.rs new file mode 100644 index 0000000..e2e7931 --- /dev/null +++ b/crates/clawhdf5-tools/src/h5.rs @@ -0,0 +1,718 @@ +//! The file model the tools share: an open file, its objects, their links +//! and the messages that describe a dataset, all read through +//! `clawhdf5-format` (and the `clawhdf5` facade for dataset values). +//! +//! Nothing here trusts the file: every size is checked before it is used, +//! and every failure is an [`Error`] carrying the address it happened at. + +use std::cell::RefCell; +use std::collections::HashMap; +use std::path::{Path, PathBuf}; +use std::rc::Rc; + +use clawhdf5::File; +use clawhdf5_format::attribute::{AttributeMessage, extract_attributes_tolerant}; +use clawhdf5_format::attribute_info::AttributeInfoMessage; +use clawhdf5_format::btree_v2::{BTreeV2Header, collect_btree_v2_records}; +use clawhdf5_format::data_layout::DataLayout; +use clawhdf5_format::dataspace::{Dataspace, DataspaceType}; +use clawhdf5_format::datatype::Datatype; +use clawhdf5_format::error::FormatError; +use clawhdf5_format::filter_pipeline::FilterPipeline; +use clawhdf5_format::fractal_heap::FractalHeapHeader; +use clawhdf5_format::global_heap::GlobalHeapCollection; +use clawhdf5_format::group_v1; +use clawhdf5_format::link_info::LinkInfoMessage; +use clawhdf5_format::link_message::{LinkMessage, LinkTarget}; +use clawhdf5_format::message_type::MessageType; +use clawhdf5_format::object_header::ObjectHeader; +use clawhdf5_format::superblock::Superblock; +use clawhdf5_format::symbol_table::SymbolTableMessage; + +/// Largest dataset or attribute (in bytes) the tools will decode by default. +/// A corrupt dataspace can claim far more elements than the file holds; the +/// limit keeps such a file from exhausting memory. `--max-bytes` changes it. +pub const DEFAULT_MAX_BYTES: u64 = 1 << 30; + +/// Upper bound on the objects one traversal visits, so a crafted file with +/// millions of links cannot make a walk run for ever. +pub const MAX_OBJECTS: usize = 1_000_000; + +/// A tool error: what went wrong and, when known, the file address (relative +/// to the superblock, as every HDF5 address is) of the structure involved. +#[derive(Debug, Clone)] +pub struct Error { + pub addr: Option, + pub msg: String, + pub kind: ErrorKind, +} + +/// Whether an error says something is wrong with the file. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum ErrorKind { + /// The file is damaged or not valid HDF5 (as far as clawhdf5 knows). + Corrupt, + /// Valid HDF5 that clawhdf5 cannot decode (a filter it does not + /// implement, external raw data files, ...). + Unsupported, + /// Refused by a limit of the tool (`--max-bytes`). + Limit, +} + +impl Error { + pub fn new(msg: impl Into) -> Self { + Self { + addr: None, + msg: msg.into(), + kind: ErrorKind::Corrupt, + } + } + + pub fn at(addr: u64, msg: impl Into) -> Self { + Self { + addr: Some(addr), + msg: msg.into(), + kind: ErrorKind::Corrupt, + } + } + + pub fn with_kind(mut self, kind: ErrorKind) -> Self { + self.kind = kind; + self + } + + /// The same error with `prefix: ` in front of its message. + pub fn context(mut self, prefix: &str) -> Self { + self.msg = format!("{prefix}: {}", self.msg); + self + } +} + +fn format_error_kind(e: &FormatError) -> ErrorKind { + match e { + FormatError::UnsupportedFilter(_) + | FormatError::ExternalDataFilesUnsupported + | FormatError::ExternalLinkUnsupported { .. } => ErrorKind::Unsupported, + _ => ErrorKind::Corrupt, + } +} + +impl std::fmt::Display for Error { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self.addr { + Some(a) => write!(f, "{} (at address {a:#x})", self.msg), + None => write!(f, "{}", self.msg), + } + } +} + +impl From for Error { + fn from(e: FormatError) -> Self { + Error::new(e.to_string()).with_kind(format_error_kind(&e)) + } +} + +impl From for Error { + fn from(e: clawhdf5::Error) -> Self { + let kind = match &e { + clawhdf5::Error::Format(f) => format_error_kind(f), + _ => ErrorKind::Corrupt, + }; + Error::new(e.to_string()).with_kind(kind) + } +} + +pub type Result = std::result::Result; + +/// Attach an address to a format error. +pub fn fe_at(addr: u64) -> impl Fn(FormatError) -> Error { + move |e| Error::at(addr, e.to_string()) +} + +/// What an object header describes. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Kind { + Group, + Dataset, + Datatype, + Unknown, +} + +impl Kind { + pub fn of(h: &ObjectHeader) -> Kind { + let has = |t: MessageType| h.messages.iter().any(|m| m.msg_type == t); + if has(MessageType::DataLayout) { + Kind::Dataset + } else if has(MessageType::SymbolTable) + || has(MessageType::LinkInfo) + || has(MessageType::Link) + || has(MessageType::GroupInfo) + { + Kind::Group + } else if has(MessageType::Datatype) { + Kind::Datatype + } else { + Kind::Unknown + } + } + + pub fn name(self) -> &'static str { + match self { + Kind::Group => "Group", + Kind::Dataset => "Dataset", + Kind::Datatype => "Type", + Kind::Unknown => "Unknown", + } + } +} + +/// Where a link points. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum LinkKind { + Hard(u64), + Soft(String), + External { + file: String, + path: String, + }, + /// A user-defined link class (type 65-255): its target means something + /// only to the application that registered the class. + UserDefined(u8), +} + +#[derive(Debug, Clone)] +pub struct Link { + pub name: String, + pub kind: LinkKind, +} + +/// An open file. +pub struct H5 { + pub path: PathBuf, + pub file: File, + pub max_bytes: u64, + heaps: RefCell, String>>>, + /// Fractal heaps whose blocks were verified: `None` = sound. + verified_heaps: RefCell>>, +} + +impl H5 { + pub fn open(path: &Path) -> Result
{ + if !path.exists() { + return Err(Error::new(format!("{}: no such file", path.display()))); + } + let file = File::open(path).map_err(|e| { + Error::new(format!( + "{}: not an HDF5 file this tool can open: {e}", + path.display() + )) + })?; + Ok(H5 { + path: path.to_path_buf(), + file, + max_bytes: DEFAULT_MAX_BYTES, + heaps: RefCell::new(HashMap::new()), + verified_heaps: RefCell::new(HashMap::new()), + }) + } + + /// The file's bytes from the superblock on: what every address indexes. + pub fn data(&self) -> &[u8] { + self.file.as_bytes() + } + + pub fn sb(&self) -> &Superblock { + self.file.superblock() + } + + pub fn os(&self) -> u8 { + self.sb().offset_size + } + + pub fn ls(&self) -> u8 { + self.sb().length_size + } + + pub fn root(&self) -> u64 { + self.sb().root_group_address + } + + pub fn header(&self, addr: u64) -> Result { + let off = usize::try_from(addr).map_err(|_| Error::at(addr, "address out of range"))?; + ObjectHeader::parse(self.data(), off, self.os(), self.ls()) + .map_err(|e| Error::at(addr, format!("object header: {e}"))) + } + + /// The payload of the first message of type `t`, resolving a shared + /// message to the message it points at. + pub fn payload(&self, h: &ObjectHeader, t: MessageType) -> Result>> { + match h.messages.iter().find(|m| m.msg_type == t) { + None => Ok(None), + Some(m) => { + clawhdf5_format::shared_message::message_data(self.data(), m, self.os(), self.ls()) + .map(|c| Some(c.into_owned())) + .map_err(|e| Error::new(format!("{t:?} message: {e}"))) + } + } + } + + pub fn datatype(&self, h: &ObjectHeader) -> Result { + let b = self + .payload(h, MessageType::Datatype)? + .ok_or_else(|| Error::new("no datatype message"))?; + Datatype::parse(&b) + .map(|(d, _)| d) + .map_err(|e| Error::new(format!("datatype message: {e}"))) + } + + pub fn dataspace(&self, h: &ObjectHeader) -> Result { + let b = self + .payload(h, MessageType::Dataspace)? + .ok_or_else(|| Error::new("no dataspace message"))?; + Dataspace::parse(&b, self.ls()).map_err(|e| Error::new(format!("dataspace message: {e}"))) + } + + pub fn layout(&self, h: &ObjectHeader) -> Result { + let m = h + .messages + .iter() + .find(|m| m.msg_type == MessageType::DataLayout) + .ok_or_else(|| Error::new("no layout message"))?; + DataLayout::parse(&m.data, self.os(), self.ls()) + .map_err(|e| Error::new(format!("layout message: {e}"))) + } + + pub fn filters(&self, h: &ObjectHeader) -> Result> { + match self.payload(h, MessageType::FilterPipeline)? { + None => Ok(None), + Some(b) => FilterPipeline::parse(&b) + .map(Some) + .map_err(|e| Error::new(format!("filter pipeline message: {e}"))), + } + } + + /// Every attribute that can be read, plus one error per attribute that + /// cannot. + pub fn attributes(&self, h: &ObjectHeader) -> Result<(Vec, Vec)> { + if let Some(m) = h + .messages + .iter() + .find(|m| m.msg_type == MessageType::AttributeInfo) + && let Ok(ai) = AttributeInfoMessage::parse(&m.data, self.os()) + && let Some(fh) = ai.fractal_heap_address + { + self.verified_heap(fh) + .map_err(|e| e.context("dense attribute storage"))?; + } + let (mut attrs, errs) = extract_attributes_tolerant(self.data(), h, self.os(), self.ls()) + .map_err(|e| Error::new(format!("attributes: {e}")))?; + attrs.sort_by(|a, b| a.name.cmp(&b.name)); + Ok((attrs, errs.iter().map(|e| e.to_string()).collect())) + } + + /// Every link of the group whose header is `h`, sorted by name. An + /// object that is not a group has none. + pub fn links(&self, h: &ObjectHeader) -> Result> { + let os = self.os(); + let ls = self.ls(); + let data = self.data(); + let mut out = Vec::new(); + if let Some(m) = h + .messages + .iter() + .find(|m| m.msg_type == MessageType::SymbolTable) + { + let stm = SymbolTableMessage::parse(&m.data, os) + .map_err(|e| Error::new(format!("symbol table message: {e}")))?; + let entries = group_v1::resolve_v1_group_entries(data, &stm, os, ls) + .map_err(|e| Error::at(stm.btree_address, format!("symbol table: {e}")))?; + let has_soft = entries.iter().any(group_v1::is_v1_soft_link); + for e in entries { + if !group_v1::is_v1_soft_link(&e) { + out.push(Link { + name: e.name, + kind: LinkKind::Hard(e.object_header_address), + }); + } + } + if has_soft { + let soft = group_v1::v1_soft_links(data, &stm, os, ls) + .map_err(|e| Error::at(stm.btree_address, format!("soft links: {e}")))?; + for (name, target) in soft { + out.push(Link { + name, + kind: LinkKind::Soft(target), + }); + } + } + } else { + let li = match h + .messages + .iter() + .find(|m| m.msg_type == MessageType::LinkInfo) + { + Some(m) => Some( + LinkInfoMessage::parse(&m.data, os) + .map_err(|e| Error::new(format!("link info message: {e}")))?, + ), + None => None, + }; + for m in h + .messages + .iter() + .filter(|m| m.msg_type == MessageType::Link) + { + out.push(link_from_message(&m.data, os)?); + } + if let Some(li) = li + && let Some(fh) = li.fractal_heap_address + { + for bytes in self.dense_heap_objects(fh, li.btree_name_index_address, 5)? { + out.push(link_from_message(&bytes, os)?); + } + } + } + out.sort_by(|a, b| a.name.cmp(&b.name)); + Ok(out) + } + + /// The objects a dense-storage B-tree (`btree`, of record type + /// `name_type`: 5 for links, 8 for attributes) indexes in the fractal + /// heap at `heap`. + pub fn dense_heap_objects( + &self, + heap: u64, + btree: Option, + name_type: u8, + ) -> Result>> { + self.verified_heap(heap)?; + let data = self.data(); + let os = self.os(); + let ls = self.ls(); + let fh = FractalHeapHeader::parse(data, to_usize(heap)?, os, ls) + .map_err(|e| Error::at(heap, format!("fractal heap header: {e}")))?; + let bt = btree.ok_or_else(|| Error::at(heap, "dense storage without a name index"))?; + let hdr = BTreeV2Header::parse(data, to_usize(bt)?, os, ls) + .map_err(|e| Error::at(bt, format!("v2 B-tree header: {e}")))?; + let recs = collect_btree_v2_records(data, &hdr, os, ls) + .map_err(|e| Error::at(bt, format!("v2 B-tree: {e}")))?; + // Name-index records: hash(4) + heap ID; creation-order ones: order(8) + heap ID. + let skip = if hdr.tree_type == name_type { 4 } else { 8 }; + let idlen = usize::from(fh.heap_id_length); + let mut out = Vec::with_capacity(recs.len()); + for r in &recs { + let id = r + .data + .get(skip..skip + idlen) + .ok_or_else(|| Error::at(bt, "v2 B-tree record shorter than a heap ID"))?; + let obj = fh + .read_managed_object(data, id, os) + .map_err(|e| Error::at(heap, format!("fractal heap object: {e}")))?; + out.push(obj); + } + Ok(out) + } + + /// Verify every block of the fractal heap at `addr` (once per heap): + /// the library reads heap objects without checking the blocks' + /// checksums, which libhdf5 does, so a damaged block would otherwise be + /// read as if it were sound. + pub fn verified_heap(&self, addr: u64) -> Result<()> { + if let Some(r) = self.verified_heaps.borrow().get(&addr) { + return r.clone().map_or(Ok(()), Err); + } + let report = crate::heap_blocks::verify(self, addr); + let n = report.problems.len(); + let r = report.problems.into_iter().next().map(|mut e| { + if n > 1 { + e.msg = format!("{} (and {} more problems in this heap)", e.msg, n - 1); + } + e + }); + self.verified_heaps.borrow_mut().insert(addr, r.clone()); + r.map_or(Ok(()), Err) + } + + /// The global heap object `idx` of the collection at `addr` (cached per + /// collection). + pub fn heap_object(&self, addr: u64, idx: u32) -> Result> { + let coll = { + let mut cache = self.heaps.borrow_mut(); + cache + .entry(addr) + .or_insert_with(|| match usize::try_from(addr) { + Ok(a) => GlobalHeapCollection::parse(self.data(), a, self.ls()) + .map(Rc::new) + .map_err(|e| e.to_string()), + Err(_) => Err("address out of range".into()), + }) + .clone() + .map_err(|e| Error::at(addr, format!("global heap: {e}")))? + }; + let idx16 = u16::try_from(idx) + .map_err(|_| Error::at(addr, format!("global heap object index {idx} out of range")))?; + coll.get_object(idx16) + .map(|o| o.data.clone()) + .ok_or_else(|| Error::at(addr, format!("global heap has no object {idx}"))) + } + + /// The dataspace of the dataset at `path` with a virtual dataset's + /// extent resolved from its sources (as libhdf5 reports it) instead of + /// the stored one. + pub fn resolved_dataspace(&self, path: &str, h: &ObjectHeader) -> Result { + let mut ds = self.dataspace(h)?; + if matches!(self.layout(h), Ok(DataLayout::Virtual { .. })) { + let d = self.file.dataset(path)?; + ds.dimensions = d.shape()?; + if ds.space_type == DataspaceType::Simple + && let Some(m) = &ds.max_dimensions + && m.len() != ds.dimensions.len() + { + ds.max_dimensions = None; + } + } + Ok(ds) + } + + /// Read a dataset's values (raw, in file byte order) through the + /// `clawhdf5` facade, refusing one larger than `max_bytes`. `ds` must be + /// the [resolved](Self::resolved_dataspace) dataspace. + pub fn read_dataset(&self, path: &str, dt: &Datatype, ds: &Dataspace) -> Result> { + let need = byte_len(ds, dt)?; + if need > self.max_bytes { + return Err(Error::new(format!( + "dataset is {need} bytes, over the {} byte limit (--max-bytes)", + self.max_bytes + )) + .with_kind(ErrorKind::Limit)); + } + let d = self.file.dataset(path)?; + let raw = d.read_selection(&clawhdf5::Selection::All)?; + if raw.len() as u64 != need { + return Err(Error::new(format!( + "read {} bytes, expected {need}", + raw.len() + ))); + } + Ok(raw) + } + + /// Walk every object reachable by hard links from the root, depth first + /// in name order, calling `visit` once per link (the root is visited + /// first with an empty link name). Each object is described once; later + /// hard links to it are reported with `first_path` set. + pub fn walk(&self, mut visit: impl FnMut(&WalkItem<'_>)) -> Result<()> { + self.walk_from(self.root(), "/", &mut visit) + } + + pub fn walk_from( + &self, + start: u64, + start_path: &str, + visit: &mut dyn FnMut(&WalkItem<'_>), + ) -> Result<()> { + let mut seen: HashMap = HashMap::new(); + let mut stack: Vec<(u64, String, Option, usize)> = + vec![(start, start_path.to_string(), None, 0)]; + let mut count = 0usize; + while let Some((addr, path, link, depth)) = stack.pop() { + count += 1; + if count > MAX_OBJECTS { + return Err(Error::new(format!( + "more than {MAX_OBJECTS} links; stopped walking" + ))); + } + // Soft, external and user-defined links have no address. + if let Some(Link { kind, .. }) = &link + && !matches!(kind, LinkKind::Hard(_)) + { + visit(&WalkItem { + path: &path, + link: link.as_ref(), + addr: None, + first_path: None, + header: None, + depth, + }); + continue; + } + if let Some(first) = seen.get(&addr) { + visit(&WalkItem { + path: &path, + link: link.as_ref(), + addr: Some(addr), + first_path: Some(first), + header: None, + depth, + }); + continue; + } + seen.insert(addr, path.clone()); + let header = self.header(addr); + visit(&WalkItem { + path: &path, + link: link.as_ref(), + addr: Some(addr), + first_path: None, + header: Some(&header), + depth, + }); + let Ok(h) = &header else { continue }; + if Kind::of(h) != Kind::Group { + continue; + } + // Link errors are the visitor's to report (it sees the header). + let Ok(links) = self.links(h) else { continue }; + let base = if path == "/" { "" } else { path.as_str() }; + for l in links.into_iter().rev() { + let child = format!("{base}/{}", l.name); + let a = match l.kind { + LinkKind::Hard(a) => a, + _ => 0, + }; + stack.push((a, child, Some(l), depth + 1)); + } + } + Ok(()) + } + + /// Resolve an absolute object path to its header address, following + /// soft links like the library does. + pub fn resolve(&self, path: &str) -> Result { + let p = path.trim_matches('/'); + if p.is_empty() { + return Ok(self.root()); + } + clawhdf5_format::group_v2::resolve_path_any(self.data(), self.sb(), p) + .map_err(|e| Error::new(format!("{path}: {e}"))) + } +} + +/// An owned copy of a [`WalkItem`], from [`H5::walk_collect`]. +pub struct Visited { + pub path: String, + pub addr: Option, + pub first_path: Option, + pub header: Option>, +} + +impl H5 { + /// Every step of [`H5::walk`], collected, and the walk's own error. + pub fn walk_collect(&self) -> (Vec, Result<()>) { + let mut items = Vec::new(); + let r = self.walk(|it| { + items.push(Visited { + path: it.path.to_string(), + addr: it.addr, + first_path: it.first_path.map(str::to_string), + header: it.header.cloned(), + }); + }); + (items, r) + } +} + +/// One step of [`H5::walk`]. +pub struct WalkItem<'a> { + pub path: &'a str, + /// The link that led here (`None` for the start object). + pub link: Option<&'a Link>, + /// Header address (`None` for a soft/external/user-defined link). + pub addr: Option, + /// Set when this object was already visited under another path. + pub first_path: Option<&'a str>, + /// The parsed header, for an object seen for the first time. + pub header: Option<&'a Result>, + pub depth: usize, +} + +fn link_from_message(data: &[u8], os: u8) -> Result { + match LinkMessage::parse(data, os) { + Ok(l) => Ok(Link { + name: l.name, + kind: match l.link_target { + LinkTarget::Hard { + object_header_address, + } => LinkKind::Hard(object_header_address), + LinkTarget::Soft { target_path } => LinkKind::Soft(target_path), + LinkTarget::External { + filename, + object_path, + } => LinkKind::External { + file: filename, + path: object_path, + }, + }, + }), + Err(FormatError::InvalidLinkType(t)) if t >= 65 => Ok(Link { + name: user_defined_link_name(data).unwrap_or_else(|| "?".into()), + kind: LinkKind::UserDefined(t), + }), + Err(e) => Err(Error::new(format!("link message: {e}"))), + } +} + +/// The name of a user-defined link, which `LinkMessage::parse` refuses. +fn user_defined_link_name(d: &[u8]) -> Option { + // version(1) flags(1) [type(1)] [corder(8)] [cset(1)] len(1|2|4|8) name + let flags = *d.get(1)?; + let mut p = 2usize; + if flags & 0x08 != 0 { + p += 1; + } + if flags & 0x04 != 0 { + p += 8; + } + if flags & 0x10 != 0 { + p += 1; + } + let w = 1usize << (flags & 0x03); + let mut n = 0usize; + for i in 0..w { + n |= usize::from(*d.get(p + i)?) << (8 * i); + } + p += w; + let name = d.get(p..p.checked_add(n)?)?; + Some(String::from_utf8_lossy(name).into_owned()) +} + +pub fn to_usize(a: u64) -> Result { + usize::try_from(a).map_err(|_| Error::at(a, "address out of range")) +} + +/// Number of elements a dataspace holds (0 for a null dataspace). +pub fn num_elements(ds: &Dataspace) -> Result { + match ds.space_type { + DataspaceType::Null => Ok(0), + DataspaceType::Scalar => Ok(1), + DataspaceType::Simple => ds + .dimensions + .iter() + .try_fold(1u64, |a, &d| a.checked_mul(d)) + .ok_or_else(|| Error::new("dataspace element count overflows")), + } +} + +/// Bytes needed for all of a dataspace's elements of type `dt`. +pub fn byte_len(ds: &Dataspace, dt: &Datatype) -> Result { + num_elements(ds)? + .checked_mul(u64::from(dt.type_size())) + .ok_or_else(|| Error::new("dataset byte size overflows")) +} + +/// Split a `FILE[/object/path]` argument the way h5ls does: the longest +/// prefix that is an existing file is the file. +pub fn split_file_arg(arg: &str) -> (String, Option) { + if Path::new(arg).is_file() { + return (arg.to_string(), None); + } + let mut idx: Vec = arg.match_indices('/').map(|(i, _)| i).collect(); + idx.reverse(); + for i in idx { + let (f, rest) = arg.split_at(i); + if !f.is_empty() && Path::new(f).is_file() { + return (f.to_string(), Some(rest.to_string())); + } + } + (arg.to_string(), None) +} diff --git a/crates/clawhdf5-tools/src/heap_blocks.rs b/crates/clawhdf5-tools/src/heap_blocks.rs new file mode 100644 index 0000000..a1c40cd --- /dev/null +++ b/crates/clawhdf5-tools/src/heap_blocks.rs @@ -0,0 +1,308 @@ +//! Walk every block of a fractal heap and verify it: signatures, the +//! back-pointer to the heap header, each block's heap offset, and the +//! checksums (always present on indirect blocks; on direct blocks when the +//! heap header's flag says so). The library reads only the blocks an object +//! lives in and does not verify block checksums, so `check` does it here. + +use std::collections::HashSet; + +use clawhdf5_format::checksum::jenkins_lookup3; +use clawhdf5_format::fractal_heap::FractalHeapHeader; + +use crate::h5::{Error, H5}; + +/// Heap header flag bit 1: direct blocks carry a checksum. +const FLAG_CHECKSUM_DBLOCKS: u8 = 0x02; +const MAX_DEPTH: u32 = 16; +const MAX_BLOCKS: usize = 1 << 20; + +#[derive(Default, Debug)] +pub struct HeapReport { + pub direct_blocks: usize, + pub indirect_blocks: usize, + /// Blocks whose checksum was verified. + pub checksums: usize, + pub problems: Vec, +} + +struct Walk<'a> { + data: &'a [u8], + heap: u64, + fh: FractalHeapHeader, + checksum_dblocks: bool, + boff_bytes: usize, + os: usize, + ls: usize, + seen: HashSet, + r: HeapReport, +} + +fn le(b: &[u8]) -> u64 { + b.iter() + .take(8) + .enumerate() + .fold(0u64, |a, (i, &x)| a | (u64::from(x) << (8 * i))) +} + +fn undefined(v: u64, os: usize) -> bool { + if os >= 8 { + v == u64::MAX + } else { + v == (1u64 << (8 * os)) - 1 + } +} + +fn log2(v: u64) -> u32 { + 63u32.saturating_sub(v.max(1).leading_zeros()) +} + +/// Verify the fractal heap whose header is at `heap`. The header itself is +/// parsed (and its checksum verified) by the library; an error there is +/// returned as the only problem. +pub fn verify(h5: &H5, heap: u64) -> HeapReport { + let data = h5.data(); + let Ok(off) = usize::try_from(heap) else { + return HeapReport { + problems: vec![Error::at(heap, "fractal heap address out of range")], + ..Default::default() + }; + }; + let fh = match FractalHeapHeader::parse(data, off, h5.os(), h5.ls()) { + Ok(f) => f, + Err(e) => { + return HeapReport { + problems: vec![Error::at(heap, format!("fractal heap header: {e}"))], + ..Default::default() + }; + } + }; + // Flags: signature(4) version(1) heap ID length(2) filter length(2) flags(1). + let flags = data.get(off + 9).copied().unwrap_or(0); + let mut w = Walk { + data, + heap, + checksum_dblocks: flags & FLAG_CHECKSUM_DBLOCKS != 0 && fh.filter_pipeline.is_none(), + boff_bytes: usize::from(fh.max_heap_size).div_ceil(8), + os: usize::from(h5.os()), + ls: usize::from(h5.ls()), + fh, + seen: HashSet::new(), + r: HeapReport::default(), + }; + if w.fh.table_width == 0 || w.fh.starting_block_size == 0 || !w.fh.table_width.is_power_of_two() + { + w.problem(heap, "fractal heap header: invalid doubling table geometry"); + return w.r; + } + let root = w.fh.root_block_address; + if !undefined(root, w.os) { + if w.fh.current_rows_in_root_indirect_block == 0 { + let size = w.fh.starting_block_size; + w.direct(root, size, 0); + } else { + let rows = w.fh.current_rows_in_root_indirect_block; + w.indirect(root, rows, 0, 0); + } + } + w.r +} + +impl Walk<'_> { + fn problem(&mut self, addr: u64, msg: impl Into) { + self.r.problems.push(Error::at(addr, msg)); + } + + fn row_size(&self, row: usize) -> Option { + let s = self.fh.starting_block_size; + if row <= 1 { + Some(s) + } else { + let sh = u32::try_from(row - 1).ok()?; + s.checked_mul(1u64.checked_shl(sh)?) + } + } + + fn max_direct_rows(&self) -> usize { + let ratio = (self.fh.max_direct_block_size / self.fh.starting_block_size).max(1); + log2(ratio) as usize + 2 + } + + fn rows_for_size(&self, size: u64) -> u16 { + let first = log2(self.fh.starting_block_size) + log2(u64::from(self.fh.table_width)); + (log2(size).saturating_sub(first) + 1) as u16 + } + + /// Common block prefix: signature, version, heap header address and + /// block offset. Returns the position after it, or `None` after + /// recording a problem. + fn prefix(&mut self, addr: u64, sig: &[u8; 4], what: &str, heap_offset: u64) -> Option { + if !self.seen.insert(addr) { + self.problem( + addr, + format!("fractal heap {what} block reached twice (cycle)"), + ); + return None; + } + if self.seen.len() > MAX_BLOCKS { + self.problem(self.heap, "fractal heap has too many blocks; stopped"); + return None; + } + let Ok(start) = usize::try_from(addr) else { + self.problem( + addr, + format!("fractal heap {what} block address out of range"), + ); + return None; + }; + let hdr_len = 5 + self.os + self.boff_bytes; + let Some(b) = start + .checked_add(hdr_len) + .and_then(|e| self.data.get(start..e)) + else { + self.problem( + addr, + format!("fractal heap {what} block lies past the end of the file"), + ); + return None; + }; + if &b[..4] != sig { + self.problem(addr, format!("fractal heap {what} block: bad signature")); + return None; + } + if b[4] != 0 { + self.problem(addr, format!("fractal heap {what} block: version {}", b[4])); + return None; + } + let back = le(&b[5..5 + self.os]); + if back != self.heap { + self.problem( + addr, + format!( + "fractal heap {what} block points at heap header {back:#x}, not {:#x}", + self.heap + ), + ); + } + let boff = le(&b[5 + self.os..hdr_len]); + if boff != heap_offset { + self.problem( + addr, + format!("fractal heap {what} block has heap offset {boff}, expected {heap_offset}"), + ); + } + Some(start + hdr_len) + } + + fn direct(&mut self, addr: u64, size: u64, heap_offset: u64) { + let Some(pos) = self.prefix(addr, b"FHDB", "direct", heap_offset) else { + return; + }; + self.r.direct_blocks += 1; + if self.fh.filter_pipeline.is_some() { + return; // stored filtered: its size on disk is not the block size + } + let start = pos - (5 + self.os + self.boff_bytes); + let Some(end) = usize::try_from(size) + .ok() + .and_then(|s| start.checked_add(s)) + else { + self.problem(addr, "fractal heap direct block size out of range"); + return; + }; + let Some(block) = self.data.get(start..end) else { + self.problem( + addr, + "fractal heap direct block extends past the end of the file", + ); + return; + }; + if self.checksum_dblocks { + let Some(stored) = block.get(pos - start..pos - start + 4) else { + self.problem(addr, "fractal heap direct block too small for its checksum"); + return; + }; + let stored = u32::from_le_bytes([stored[0], stored[1], stored[2], stored[3]]); + let mut copy = block.to_vec(); + copy[pos - start..pos - start + 4].fill(0); + let computed = jenkins_lookup3(©); + self.r.checksums += 1; + if computed != stored { + self.problem( + addr, + format!( + "fractal heap direct block: checksum mismatch: stored {stored:#010x}, computed {computed:#010x}" + ), + ); + } + } + } + + fn indirect(&mut self, addr: u64, nrows: u16, heap_offset: u64, depth: u32) { + if depth > MAX_DEPTH { + self.problem(addr, "fractal heap indirect blocks nested too deeply"); + return; + } + let Some(mut pos) = self.prefix(addr, b"FHIB", "indirect", heap_offset) else { + return; + }; + self.r.indirect_blocks += 1; + let start = pos - (5 + self.os + self.boff_bytes); + let width = usize::from(self.fh.table_width); + let filtered = self.fh.filter_pipeline.is_some(); + let direct_rows = self.max_direct_rows(); + let mut children: Vec<(u64, bool, u64, u64)> = Vec::new(); // addr, direct, size/rows, offset + let mut off = heap_offset; + for row in 0..usize::from(nrows) { + let Some(rs) = self.row_size(row) else { + self.problem(addr, "fractal heap row size overflows"); + return; + }; + let direct = row < direct_rows; + for _ in 0..width { + let Some(b) = self.data.get(pos..pos + self.os) else { + self.problem( + addr, + "fractal heap indirect block extends past the end of the file", + ); + return; + }; + let child = le(b); + pos += self.os; + if direct && filtered { + pos += self.ls + 4; + } + if !undefined(child, self.os) { + children.push((child, direct, rs, off)); + } + off = off.saturating_add(rs); + } + } + let Some(stored) = self.data.get(pos..pos + 4) else { + self.problem( + addr, + "fractal heap indirect block extends past the end of the file", + ); + return; + }; + let stored = u32::from_le_bytes([stored[0], stored[1], stored[2], stored[3]]); + let computed = jenkins_lookup3(&self.data[start..pos]); + self.r.checksums += 1; + if computed != stored { + self.problem( + addr, + format!( + "fractal heap indirect block: checksum mismatch: stored {stored:#010x}, computed {computed:#010x}" + ), + ); + return; // its child pointers cannot be trusted + } + for (child, direct, size, off) in children { + if direct { + self.direct(child, size, off); + } else { + let rows = self.rows_for_size(size); + self.indirect(child, rows, off, depth + 1); + } + } + } +} diff --git a/crates/clawhdf5-tools/src/info.rs b/crates/clawhdf5-tools/src/info.rs new file mode 100644 index 0000000..606457b --- /dev/null +++ b/crates/clawhdf5-tools/src/info.rs @@ -0,0 +1,245 @@ +//! Dataset facts shared by `ls`, `dump`, `stat` and `check`: shape text, +//! layout, filters and storage. + +use clawhdf5_format::chunked_read::{ChunkInfo, list_chunks}; +use clawhdf5_format::data_layout::DataLayout; +use clawhdf5_format::dataspace::{Dataspace, DataspaceType}; +use clawhdf5_format::datatype::Datatype; +use clawhdf5_format::filter_pipeline::{FilterDescription, FilterPipeline}; +use clawhdf5_format::message_type::MessageType; +use clawhdf5_format::object_header::ObjectHeader; + +use crate::h5::{Error, H5, Result}; + +/// h5ls's `{10/Inf, 20}` shape text. `always_max` prints `cur/max` for every +/// dimension (h5ls -v). +pub fn shape_text(ds: &Dataspace, always_max: bool) -> String { + match ds.space_type { + DataspaceType::Null => "{NULL}".into(), + DataspaceType::Scalar => "{SCALAR}".into(), + DataspaceType::Simple => { + let dims: Vec = ds + .dimensions + .iter() + .enumerate() + .map(|(i, &d)| { + let m = ds.max_dimensions.as_ref().and_then(|m| m.get(i).copied()); + match m { + Some(u64::MAX) => format!("{d}/Inf"), + Some(m) if m != d || always_max => format!("{d}/{m}"), + None if always_max => format!("{d}/{d}"), + _ => d.to_string(), + } + }) + .collect(); + format!("{{{}}}", dims.join(", ")) + } + } +} + +/// h5dump's `SIMPLE { ( 3, 4 ) / ( 3, H5S_UNLIMITED ) }`. +pub fn dataspace_ddl(ds: &Dataspace) -> String { + match ds.space_type { + DataspaceType::Null => "NULL".into(), + DataspaceType::Scalar => "SCALAR".into(), + DataspaceType::Simple => { + let cur: Vec = ds.dimensions.iter().map(|d| d.to_string()).collect(); + let max: Vec = ds + .dimensions + .iter() + .enumerate() + .map( + |(i, &d)| match ds.max_dimensions.as_ref().and_then(|m| m.get(i).copied()) { + Some(u64::MAX) => "H5S_UNLIMITED".into(), + Some(m) => m.to_string(), + None => d.to_string(), + }, + ) + .collect(); + format!( + "SIMPLE {{ ( {} ) / ( {} ) }}", + cur.join(", "), + max.join(", ") + ) + } + } +} + +pub fn filter_name(f: &FilterDescription) -> String { + let known = match f.filter_id { + 1 => "deflate", + 2 => "shuffle", + 3 => "fletcher32", + 4 => "szip", + 5 => "nbit", + 6 => "scaleoffset", + 307 => "bzip2", + 32000 => "lzf", + 32001 => "blosc", + 32004 => "lz4", + 32008 => "bitshuffle", + 32013 => "zfp", + 32015 => "zstd", + 32026 => "blosc2", + _ => "", + }; + if !known.is_empty() { + return known.into(); + } + match &f.name { + Some(n) if !n.is_empty() => n.clone(), + _ => "user-defined".into(), + } +} + +/// `deflate-1 OPT {4}` as h5ls prints a filter. +pub fn filter_text(f: &FilterDescription) -> String { + let opt = if f.flags & 1 != 0 { " OPT" } else { "" }; + let cd = if f.client_data.is_empty() { + String::new() + } else { + format!( + " {{{}}}", + f.client_data + .iter() + .map(|c| c.to_string()) + .collect::>() + .join(", ") + ) + }; + format!("{}-{}{opt}{cd}", filter_name(f), f.filter_id) +} + +pub fn layout_name(l: &DataLayout) -> &'static str { + match l { + DataLayout::Compact { .. } => "compact", + DataLayout::Contiguous { .. } => "contiguous", + DataLayout::Chunked { .. } => "chunked", + DataLayout::Virtual { .. } => "virtual", + } +} + +/// Chunk index kind of a chunked layout. +pub fn chunk_index_name(l: &DataLayout) -> &'static str { + match l { + DataLayout::Chunked { + version, + chunk_index_type, + .. + } => { + if *version < 4 { + return "v1 B-tree"; + } + match chunk_index_type { + Some(1) => "single chunk", + Some(2) => "implicit", + Some(3) => "fixed array", + Some(4) => "extensible array", + Some(5) => "v2 B-tree", + _ => "unknown", + } + } + _ => "", + } +} + +/// Everything about one dataset that can be learned without reading its +/// values. +pub struct DsInfo { + pub dt: Result, + pub ds: Result, + pub layout: Result, + pub filters: Result>, + pub external: bool, +} + +impl DsInfo { + pub fn read(h5: &H5, path: &str, h: &ObjectHeader) -> DsInfo { + DsInfo { + dt: h5.datatype(h), + ds: h5.resolved_dataspace(path, h), + layout: h5.layout(h), + filters: h5.filters(h), + external: h + .messages + .iter() + .any(|m| m.msg_type == MessageType::ExternalDataFiles), + } + } + + pub fn logical_bytes(&self) -> Option { + let (Ok(dt), Ok(ds)) = (&self.dt, &self.ds) else { + return None; + }; + crate::h5::byte_len(ds, dt).ok() + } +} + +/// Every allocated chunk (empty when none are). Errors only for a corrupt +/// chunk index. +pub fn chunks( + h5: &H5, + layout: &DataLayout, + ds: &Dataspace, + dt: &Datatype, +) -> Result> { + let DataLayout::Chunked { btree_address, .. } = layout else { + return Ok(Vec::new()); + }; + let Some(addr) = *btree_address else { + return Ok(Vec::new()); + }; + list_chunks( + h5.data(), + layout, + ds, + dt.type_size() as usize, + h5.os(), + h5.ls(), + ) + .map(|(c, _)| c) + .map_err(|e| { + Error::at( + addr, + format!("chunk index ({}): {e}", chunk_index_name(layout)), + ) + }) +} + +/// Bytes of raw data the dataset has allocated in the file (what libhdf5's +/// `H5Dget_storage_size` reports). +pub fn allocated_bytes(h5: &H5, info: &DsInfo) -> Result { + let layout = info.layout.as_ref().map_err(Clone::clone)?; + Ok(match layout { + DataLayout::Compact { data } => data.len() as u64, + DataLayout::Contiguous { address, size } => { + if address.is_some() { + *size + } else { + 0 + } + } + DataLayout::Chunked { .. } => { + let dt = info.dt.as_ref().map_err(Clone::clone)?; + let ds = info.ds.as_ref().map_err(Clone::clone)?; + chunks(h5, layout, ds, dt)? + .iter() + .map(|c| u64::from(c.chunk_size)) + .sum() + } + DataLayout::Virtual { .. } => 0, + }) +} + +/// Number of hard links to an object, as its header records it. +pub fn link_count(h: &ObjectHeader) -> u64 { + if let Some(rc) = h.reference_count { + return u64::from(rc); + } + h.messages + .iter() + .find(|m| m.msg_type == MessageType::ObjectReferenceCount) + .and_then(|m| m.data.get(1..5)) + .map(|b| u64::from(u32::from_le_bytes([b[0], b[1], b[2], b[3]]))) + .unwrap_or(1) +} diff --git a/crates/clawhdf5-tools/src/lib.rs b/crates/clawhdf5-tools/src/lib.rs new file mode 100644 index 0000000..693ea9a --- /dev/null +++ b/crates/clawhdf5-tools/src/lib.rs @@ -0,0 +1,63 @@ +//! `h5rs`: HDF5 command-line tools built on clawhdf5 alone, no libhdf5. +//! +//! The binary's subcommands (`ls`, `dump`, `stat`, `diff`, `check`) live in +//! the modules below; [`run`] dispatches to them. See the crate README for +//! the command reference. + +pub mod check; +pub mod cli; +pub mod diff; +pub mod dtype; +pub mod dump; +pub mod h5; +pub mod heap_blocks; +pub mod info; +pub mod ls; +pub mod stat; +pub mod value; + +use cli::{Args, Out}; + +pub const USAGE: &str = "\ +h5rs: HDF5 tools in pure Rust (clawhdf5, no libhdf5) + +usage: h5rs [options] ... + +commands: + ls list objects (like h5ls) + dump print a file's structure and values as DDL or JSON (like h5dump) + stat object, layout, filter and storage statistics (like h5stat) + diff compare two files (like h5diff) + check validate a file's structure and checksums + +Run `h5rs --help` for a command's options."; + +/// Run `h5rs` with `argv` (without the program name). Returns the exit +/// status. +pub fn run(argv: Vec, out: &mut Out) -> std::io::Result { + let mut it = argv.into_iter(); + let Some(cmd) = it.next() else { + writeln!(out.e, "{USAGE}")?; + return Ok(2); + }; + let rest: Vec = it.collect(); + match cmd.as_str() { + "ls" => ls::run(&mut Args::new("ls", rest), out), + "dump" => dump::run(&mut Args::new("dump", rest), out), + "stat" => stat::run(&mut Args::new("stat", rest), out), + "diff" => diff::run(&mut Args::new("diff", rest), out), + "check" => check::run(&mut Args::new("check", rest), out), + "-h" | "--help" | "help" => { + writeln!(out.o, "{USAGE}")?; + Ok(0) + } + "-V" | "--version" => { + writeln!(out.o, "h5rs {}", env!("CARGO_PKG_VERSION"))?; + Ok(0) + } + other => { + writeln!(out.e, "h5rs: unknown command {other:?}\n\n{USAGE}")?; + Ok(2) + } + } +} diff --git a/crates/clawhdf5-tools/src/ls.rs b/crates/clawhdf5-tools/src/ls.rs new file mode 100644 index 0000000..ef966f6 --- /dev/null +++ b/crates/clawhdf5-tools/src/ls.rs @@ -0,0 +1,385 @@ +//! `h5rs ls`: list a file's objects like h5ls. + +use std::collections::HashMap; + +use clawhdf5_format::attribute::AttributeMessage; +use clawhdf5_format::dataspace::DataspaceType; +use clawhdf5_format::object_header::ObjectHeader; + +use crate::cli::{Args, Out}; +use crate::h5::{H5, Kind, Link, LinkKind, split_file_arg}; +use crate::info::{self, DsInfo}; +use crate::value::{self, Decoder}; + +pub const USAGE: &str = "\ +usage: h5rs ls [-r] [-v] [--max-bytes N] FILE[/OBJECT] + +List the objects in FILE (or in the group OBJECT), one per line, like h5ls: +name, kind, shape and, for datasets, the datatype. + + -r, --recursive list every object below, with full paths + -v, --verbose also print address, link count, layout, chunking, + storage, filters, datatype and attributes + --max-bytes N largest attribute value decoded for -v (default 1 GiB) + +Exit status: 0 listed, 1 some object could not be described, 2 error."; + +struct Ls<'a> { + h5: &'a H5, + verbose: bool, + kinds: HashMap, + problems: usize, +} + +pub fn run(args: &mut Args, out: &mut Out) -> std::io::Result { + let mut recursive = false; + let mut verbose = false; + let mut max_bytes = None; + let mut target = None; + while let Some(a) = args.next() { + match a.as_str() { + "-r" | "--recursive" => recursive = true, + "-v" | "--verbose" => verbose = true, + "-rv" | "-vr" => { + recursive = true; + verbose = true; + } + "--max-bytes" => match args.number() { + Some(n) => max_bytes = Some(n), + None => return args.usage_error(out, "--max-bytes needs a number", USAGE), + }, + "-h" | "--help" => { + writeln!(out.o, "{USAGE}")?; + return Ok(0); + } + s if s.starts_with('-') && s.len() > 1 => { + return args.usage_error(out, &format!("unknown option {a}"), USAGE); + } + _ if target.is_none() => target = Some(a), + _ => return args.usage_error(out, &format!("unexpected argument {a}"), USAGE), + } + } + let Some(target) = target else { + return args.usage_error(out, "missing FILE", USAGE); + }; + let (file, obj) = split_file_arg(&target); + let mut h5 = match H5::open(std::path::Path::new(&file)) { + Ok(h) => h, + Err(e) => { + writeln!(out.e, "h5rs ls: {e}")?; + return Ok(2); + } + }; + if let Some(m) = max_bytes { + h5.max_bytes = m; + } + let mut ls = Ls { + h5: &h5, + verbose, + kinds: HashMap::new(), + problems: 0, + }; + let obj_path = obj.unwrap_or_else(|| "/".into()); + let addr = match h5.resolve(&obj_path) { + Ok(a) => a, + Err(e) => { + writeln!(out.e, "h5rs ls: {obj_path}: not found: {e}")?; + return Ok(2); + } + }; + let header = match h5.header(addr) { + Ok(h) => h, + Err(e) => { + writeln!(out.e, "h5rs ls: {obj_path}: {e}")?; + return Ok(2); + } + }; + let is_root = obj_path.trim_matches('/').is_empty(); + if Kind::of(&header) != Kind::Group && !is_root { + // h5ls names a single object by its base name, or with -r by its + // path as given (without the leading slash). + let name = if recursive { + obj_path.trim_start_matches('/').to_string() + } else { + obj_path.rsplit('/').next().unwrap_or(&obj_path).to_string() + }; + ls.object_line(out, &name, &obj_path, addr, &Ok(header))?; + } else if recursive { + let start = if is_root { "/" } else { "" }; + let mut io_err = None; + let res = h5.walk_from(addr, start, &mut |item| { + if io_err.is_some() || (item.link.is_none() && !is_root) { + return; // listing a group's contents, not the group + } + let path = if item.path.is_empty() { "/" } else { item.path }; + let full = if is_root { + path.to_string() + } else { + format!("{}{path}", obj_path.trim_end_matches('/')) + }; + if let Err(e) = ls.item( + out, + path, + &full, + item.link, + item.addr, + item.first_path, + item.header, + ) { + io_err = Some(e); + } + }); + if let Some(e) = io_err { + return Err(e); + } + if let Err(e) = res { + writeln!(out.e, "h5rs ls: {e}")?; + ls.problems += 1; + } + } else { + match h5.links(&header) { + Ok(links) => { + for l in &links { + let (a, hdr) = match l.kind { + LinkKind::Hard(a) => (Some(a), Some(h5.header(a))), + _ => (None, None), + }; + let full = format!("{}/{}", obj_path.trim_end_matches('/'), l.name); + ls.item(out, &l.name, &full, Some(l), a, None, hdr.as_ref())?; + } + } + Err(e) => { + writeln!(out.e, "h5rs ls: {obj_path}: cannot list group: {e}")?; + ls.problems += 1; + } + } + } + Ok(if ls.problems > 0 { 1 } else { 0 }) +} + +impl Ls<'_> { + #[allow(clippy::too_many_arguments)] + fn item( + &mut self, + out: &mut Out, + name: &str, + full: &str, + link: Option<&Link>, + addr: Option, + first: Option<&str>, + header: Option<&crate::h5::Result>, + ) -> std::io::Result<()> { + if let Some(l) = link { + match &l.kind { + LinkKind::Soft(t) => return writeln!(out.o, "{name:<24} Soft Link {{{t}}}"), + LinkKind::External { file, path } => { + return writeln!(out.o, "{name:<24} External Link {{{file}/{path}}}"); + } + LinkKind::UserDefined(t) => { + return writeln!(out.o, "{name:<24} User-defined link (type {t})"); + } + LinkKind::Hard(_) => {} + } + } + let Some(addr) = addr else { return Ok(()) }; + if let Some(first) = first { + let k = self.kinds.get(&addr).copied().unwrap_or(Kind::Unknown); + return writeln!(out.o, "{name:<24} {}, same as {first}", k.name()); + } + match header { + Some(h) => self.object_line(out, name, full, addr, h), + None => Ok(()), + } + } + + fn object_line( + &mut self, + out: &mut Out, + name: &str, + full: &str, + addr: u64, + header: &crate::h5::Result, + ) -> std::io::Result<()> { + let h = match header { + Ok(h) => h, + Err(e) => { + self.problems += 1; + writeln!(out.o, "{name:<24} ** error **")?; + return writeln!(out.e, "h5rs ls: {name}: {e}"); + } + }; + let kind = Kind::of(h); + self.kinds.insert(addr, kind); + match kind { + Kind::Dataset => { + let info = DsInfo::read(self.h5, full, h); + let shape = match &info.ds { + Ok(ds) => info::shape_text(ds, self.verbose), + Err(_) => "{?}".into(), + }; + if self.verbose { + writeln!(out.o, "{name:<24} Dataset {shape}")?; + } else { + let t = match &info.dt { + Ok(dt) => crate::dtype::short(dt), + Err(_) => "?".into(), + }; + writeln!(out.o, "{name:<24} Dataset {shape} {t}")?; + } + for e in [ + info.dt.as_ref().err(), + info.ds.as_ref().err(), + info.layout.as_ref().err(), + info.filters.as_ref().err(), + ] + .into_iter() + .flatten() + { + self.problems += 1; + writeln!(out.e, "h5rs ls: {name}: {e}")?; + } + if self.verbose { + self.dataset_details(out, addr, h, &info)?; + } + } + k => { + writeln!(out.o, "{name:<24} {}", k.name())?; + if self.verbose { + writeln!(out.o, " Address: {addr}")?; + writeln!(out.o, " Links: {}", info::link_count(h))?; + if k == Kind::Datatype + && let Ok(dt) = self.h5.datatype(h) + { + writeln!(out.o, " Type: {}", crate::dtype::long(&dt))?; + } + self.attributes(out, name, h)?; + } + } + } + Ok(()) + } + + fn dataset_details( + &mut self, + out: &mut Out, + addr: u64, + h: &ObjectHeader, + info: &DsInfo, + ) -> std::io::Result<()> { + writeln!(out.o, " Address: {addr}")?; + writeln!(out.o, " Links: {}", info::link_count(h))?; + if let Ok(l) = &info.layout { + let idx = info::chunk_index_name(l); + if idx.is_empty() { + writeln!(out.o, " Layout: {}", info::layout_name(l))?; + } else { + writeln!( + out.o, + " Layout: {} ({idx} index)", + info::layout_name(l) + )?; + } + if let clawhdf5_format::data_layout::DataLayout::Chunked { + chunk_dimensions, .. + } = l + { + let rank = chunk_dimensions.len().saturating_sub(1); + let dims = &chunk_dimensions[..rank]; + let esize = info + .dt + .as_ref() + .map(|d| u64::from(d.type_size())) + .unwrap_or(0); + let bytes = dims + .iter() + .try_fold(esize, |a, &d| a.checked_mul(u64::from(d))) + .map(|b| b.to_string()) + .unwrap_or_else(|| "?".into()); + writeln!( + out.o, + " Chunks: {{{}}} {bytes} bytes", + dims.iter() + .map(|d| d.to_string()) + .collect::>() + .join(", ") + )?; + } + } + if info.external { + writeln!(out.o, " External: raw data in external files")?; + } + let logical = info.logical_bytes(); + match (logical, info::allocated_bytes(self.h5, info)) { + (Some(l), Ok(a)) => { + if a > 0 { + writeln!( + out.o, + " Storage: {l} logical bytes, {a} allocated bytes, {:.2}% utilization", + l as f64 * 100.0 / a as f64 + )?; + } else { + writeln!(out.o, " Storage: {l} logical bytes, 0 allocated bytes")?; + } + } + (_, Err(e)) => { + self.problems += 1; + writeln!(out.e, "h5rs ls: {e}")?; + } + _ => {} + } + if let Ok(Some(p)) = &info.filters { + for (i, f) in p.filters.iter().enumerate() { + writeln!(out.o, " Filter-{i}: {}", info::filter_text(f))?; + } + } + if let Ok(dt) = &info.dt { + writeln!(out.o, " Type: {}", crate::dtype::long(dt))?; + } + self.attributes(out, "", h) + } + + fn attributes(&mut self, out: &mut Out, name: &str, h: &ObjectHeader) -> std::io::Result<()> { + let (attrs, errs) = match self.h5.attributes(h) { + Ok(x) => x, + Err(e) => { + self.problems += 1; + return writeln!(out.e, "h5rs ls: {name}: {e}"); + } + }; + for e in errs { + self.problems += 1; + writeln!(out.e, "h5rs ls: {name}: attribute: {e}")?; + } + for a in &attrs { + self.attribute(out, a)?; + } + Ok(()) + } + + fn attribute(&mut self, out: &mut Out, a: &AttributeMessage) -> std::io::Result<()> { + let shape = match a.dataspace.space_type { + DataspaceType::Scalar => "scalar".to_string(), + DataspaceType::Null => "null".to_string(), + DataspaceType::Simple => info::shape_text(&a.dataspace, false), + }; + writeln!(out.o, " Attribute: {} {shape}", a.name)?; + writeln!( + out.o, + " Type: {}", + crate::dtype::long(&a.datatype) + )?; + let n = crate::h5::num_elements(&a.dataspace).unwrap_or(0); + if n == 0 { + return Ok(()); + } + let dec = Decoder::new(self.h5); + let shown = n.min(8) as usize; + let mut vals = Vec::with_capacity(shown); + for i in 0..shown { + let v = dec.element(&a.datatype, &a.raw_data, i); + vals.push(value::text(&v, &|_| None)); + } + let more = if n > 8 { ", ..." } else { "" }; + writeln!(out.o, " Data: {}{more}", vals.join(", ")) + } +} diff --git a/crates/clawhdf5-tools/src/main.rs b/crates/clawhdf5-tools/src/main.rs new file mode 100644 index 0000000..4e6d431 --- /dev/null +++ b/crates/clawhdf5-tools/src/main.rs @@ -0,0 +1,58 @@ +//! The `h5rs` binary. A panic anywhere is a bug: it is caught, reported as +//! an internal error and turned into exit status 3, never a crash. + +use std::io::Write; +use std::panic::{self, AssertUnwindSafe}; + +use clawhdf5_tools::cli::Out; + +/// Exit status for an internal error (a caught panic). +const INTERNAL_ERROR: i32 = 3; + +fn main() { + panic::set_hook(Box::new(|info| { + let msg = if let Some(s) = info.payload().downcast_ref::<&str>() { + (*s).to_string() + } else if let Some(s) = info.payload().downcast_ref::() { + s.clone() + } else { + "unknown panic".into() + }; + let at = info + .location() + .map(|l| format!(" at {}:{}", l.file(), l.line())) + .unwrap_or_default(); + eprintln!("h5rs: internal error (this is a bug; please report it): {msg}{at}"); + })); + let argv: Vec = std::env::args().skip(1).collect(); + let stdout = std::io::stdout(); + let mut o = std::io::BufWriter::new(stdout.lock()); + let mut e = std::io::stderr(); + let result = panic::catch_unwind(AssertUnwindSafe(|| { + let mut out = Out { + o: &mut o, + e: &mut e, + }; + clawhdf5_tools::run(argv, &mut out) + })); + let code = match result { + Ok(Ok(code)) => match o.flush() { + Ok(()) => code, + Err(err) if err.kind() == std::io::ErrorKind::BrokenPipe => code, + Err(err) => { + eprintln!("h5rs: {err}"); + 2 + } + }, + Ok(Err(err)) if err.kind() == std::io::ErrorKind::BrokenPipe => 0, + Ok(Err(err)) => { + eprintln!("h5rs: {err}"); + 2 + } + Err(_) => { + let _ = o.flush(); + INTERNAL_ERROR + } + }; + std::process::exit(code); +} diff --git a/crates/clawhdf5-tools/src/stat.rs b/crates/clawhdf5-tools/src/stat.rs new file mode 100644 index 0000000..c7e44da --- /dev/null +++ b/crates/clawhdf5-tools/src/stat.rs @@ -0,0 +1,305 @@ +//! `h5rs stat`: object, layout, filter and storage statistics, laid out like +//! h5stat's report. + +use std::collections::BTreeMap; + +use clawhdf5_format::data_layout::DataLayout; +use clawhdf5_format::dataspace::DataspaceType; + +use crate::cli::{Args, Out}; +use crate::h5::{H5, Kind, LinkKind}; +use crate::info::{self, DsInfo}; + +pub const USAGE: &str = "\ +usage: h5rs stat FILE + +Print statistics for FILE, in the layout of h5stat's report: object counts, +links, dataset ranks, layouts, filters, attribute counts, raw-data size and +a file-space summary. Metadata space is not broken down by structure (h5stat +does); it is reported as one figure together with free space. + +Exit status: 0 printed, 1 some object could not be read (counted as +\"unreadable\"), 2 error."; + +#[derive(Default)] +struct Stats { + groups: u64, + datasets: u64, + datatypes: u64, + other: u64, + unreadable: u64, + links: u64, + max_links_to_object: u64, + max_objects_in_group: u64, + group_sizes: BTreeMap, + ranks: BTreeMap, + max_1d: u64, + layout: [u64; 4], + external: u64, + no_filter: u64, + filter_counts: BTreeMap<&'static str, u64>, + raw_data: u64, + raw_errors: u64, + attr_objects: u64, + max_attrs: u64, + attr_counts: BTreeMap, +} + +const FILTER_KINDS: [&str; 8] = [ + "GZIP", + "SHUFFLE", + "FLETCHER32", + "SZIP", + "NBIT", + "SCALEOFFSET", + "USER-DEFINED", + "", +]; + +fn filter_kind(id: u16) -> &'static str { + match id { + 1 => "GZIP", + 2 => "SHUFFLE", + 3 => "FLETCHER32", + 4 => "SZIP", + 5 => "NBIT", + 6 => "SCALEOFFSET", + _ => "USER-DEFINED", + } +} + +pub fn run(args: &mut Args, out: &mut Out) -> std::io::Result { + let mut file = None; + while let Some(a) = args.next() { + match a.as_str() { + "-h" | "--help" => { + writeln!(out.o, "{USAGE}")?; + return Ok(0); + } + s if s.starts_with('-') && s.len() > 1 => { + return args.usage_error(out, &format!("unknown option {a}"), USAGE); + } + _ if file.is_none() => file = Some(a), + _ => return args.usage_error(out, &format!("unexpected argument {a}"), USAGE), + } + } + let Some(file) = file else { + return args.usage_error(out, "missing FILE", USAGE); + }; + let h5 = match H5::open(std::path::Path::new(&file)) { + Ok(h) => h, + Err(e) => { + writeln!(out.e, "h5rs stat: {e}")?; + return Ok(2); + } + }; + let mut s = Stats::default(); + let mut errors: Vec = Vec::new(); + let walk = h5.walk(|it| { + if let Some(l) = it.link + && !matches!(l.kind, LinkKind::Hard(_)) + { + s.links += 1; + return; + } + if it.first_path.is_some() { + return; + } + let Some(Ok(h)) = it.header else { + if let Some(Err(e)) = it.header { + s.unreadable += 1; + errors.push(format!("{}: {e}", it.path)); + } + return; + }; + s.max_links_to_object = s.max_links_to_object.max(info::link_count(h)); + match h5.attributes(h) { + Ok((a, errs)) => { + let n = a.len() + errs.len(); + if n > 0 { + s.attr_objects += 1; + s.max_attrs = s.max_attrs.max(n as u64); + *s.attr_counts.entry(n).or_default() += 1; + } + for e in errs { + errors.push(format!("{}: attribute: {e}", it.path)); + } + } + Err(e) => errors.push(format!("{}: {e}", it.path)), + } + match Kind::of(h) { + Kind::Group => { + s.groups += 1; + match h5.links(h) { + Ok(l) => { + s.max_objects_in_group = s.max_objects_in_group.max(l.len() as u64); + *s.group_sizes.entry(l.len()).or_default() += 1; + } + Err(e) => { + s.unreadable += 1; + errors.push(format!("{}: {e}", it.path)); + } + } + } + Kind::Datatype => s.datatypes += 1, + Kind::Unknown => { + // The root group of a file with no links at all has no + // group messages; count it as a group. + if it.link.is_none() { + s.groups += 1; + *s.group_sizes.entry(0).or_default() += 1; + } else { + s.other += 1; + } + } + Kind::Dataset => { + s.datasets += 1; + let info = DsInfo::read(&h5, it.path, h); + if let Ok(ds) = &info.ds { + let rank = match ds.space_type { + DataspaceType::Simple => ds.dimensions.len(), + _ => 0, + }; + *s.ranks.entry(rank).or_default() += 1; + if rank == 1 { + s.max_1d = s.max_1d.max(ds.dimensions[0]); + } + } + match &info.layout { + Ok(l) => { + let i = match l { + DataLayout::Compact { .. } => 0, + DataLayout::Contiguous { .. } => 1, + DataLayout::Chunked { .. } => 2, + DataLayout::Virtual { .. } => 3, + }; + s.layout[i] += 1; + } + Err(e) => errors.push(format!("{}: {e}", it.path)), + } + if info.external { + s.external += 1; + } + match &info.filters { + Ok(Some(p)) if !p.filters.is_empty() => { + let mut kinds: Vec<&'static str> = + p.filters.iter().map(|f| filter_kind(f.filter_id)).collect(); + kinds.sort_unstable(); + kinds.dedup(); + for k in kinds { + *s.filter_counts.entry(k).or_default() += 1; + } + } + Ok(_) => s.no_filter += 1, + Err(e) => errors.push(format!("{}: {e}", it.path)), + } + match info::allocated_bytes(&h5, &info) { + Ok(b) => s.raw_data = s.raw_data.saturating_add(b), + Err(e) => { + s.raw_errors += 1; + errors.push(format!("{}: storage size: {e}", it.path)); + } + } + } + } + }); + if let Err(e) = walk { + errors.push(e.to_string()); + } + report(&h5, &file, &s, out)?; + for e in &errors { + writeln!(out.e, "h5rs stat: {e}")?; + } + Ok(if errors.is_empty() { 0 } else { 1 }) +} + +fn report(h5: &H5, file: &str, s: &Stats, out: &mut Out) -> std::io::Result<()> { + let o = &mut out.o; + writeln!(o, "Filename: {file}")?; + writeln!(o, "File information")?; + writeln!(o, "\t# of unique groups: {}", s.groups)?; + writeln!(o, "\t# of unique datasets: {}", s.datasets)?; + writeln!(o, "\t# of unique named datatypes: {}", s.datatypes)?; + writeln!(o, "\t# of unique links: {}", s.links)?; + writeln!(o, "\t# of unique other: {}", s.other)?; + if s.unreadable > 0 { + writeln!(o, "\t# of unreadable objects: {}", s.unreadable)?; + } + writeln!(o, "\tMax. # of links to object: {}", s.max_links_to_object)?; + writeln!( + o, + "\tMax. # of objects in group: {}", + s.max_objects_in_group + )?; + let sb = h5.sb(); + writeln!(o, "Superblock:")?; + writeln!(o, "\tVersion: {}", sb.version)?; + writeln!(o, "\tSize of offsets: {} bytes", sb.offset_size)?; + writeln!(o, "\tSize of lengths: {} bytes", sb.length_size)?; + writeln!(o, "\tUser block: {} bytes", h5.file.user_block_size())?; + writeln!(o, "Group bins:")?; + for (n, c) in &s.group_sizes { + writeln!(o, "\t# of groups with {n} link(s): {c}")?; + } + writeln!(o, "\tTotal # of groups: {}", s.groups)?; + writeln!(o, "Dataset dimension information:")?; + writeln!( + o, + "\tMax. rank of datasets: {}", + s.ranks.keys().next_back().copied().unwrap_or(0) + )?; + writeln!(o, "\tDataset ranks:")?; + for (r, c) in &s.ranks { + writeln!(o, "\t\t# of dataset with rank {r}: {c}")?; + } + writeln!(o, "1-D Dataset information:")?; + writeln!(o, "\tMax. dimension size of 1-D datasets: {}", s.max_1d)?; + writeln!(o, "Dataset storage information:")?; + writeln!(o, "\tTotal raw data size: {}", s.raw_data)?; + if s.raw_errors > 0 { + writeln!( + o, + "\tDatasets whose storage size could not be read: {}", + s.raw_errors + )?; + } + writeln!(o, "Dataset layout information:")?; + for (i, n) in ["COMPACT", "CONTIG", "CHUNKED", "VIRTUAL"] + .iter() + .enumerate() + { + writeln!(o, "\tDataset layout counts[{n}]: {}", s.layout[i])?; + } + writeln!(o, "\tDatasets with external raw data: {}", s.external)?; + writeln!(o, "Dataset filters information:")?; + writeln!(o, "\tNumber of datasets with:")?; + writeln!(o, "\t\tNO filter: {}", s.no_filter)?; + for k in FILTER_KINDS.iter().filter(|k| !k.is_empty()) { + writeln!( + o, + "\t\t{k} filter: {}", + s.filter_counts.get(k).copied().unwrap_or(0) + )?; + } + writeln!(o, "Attribute information:")?; + for (n, c) in &s.attr_counts { + writeln!(o, "\t# of objects with {n} attribute(s): {c}")?; + } + writeln!( + o, + "\tTotal # of objects with attributes: {}", + s.attr_objects + )?; + writeln!(o, "\tMax. # of attributes to objects: {}", s.max_attrs)?; + let total = std::fs::metadata(&h5.path).map(|m| m.len()).unwrap_or(0); + let ub = h5.file.user_block_size(); + writeln!(o, "Summary of file space information:")?; + writeln!(o, " User block: {ub} bytes")?; + writeln!(o, " Raw data: {} bytes", s.raw_data)?; + writeln!( + o, + " Metadata and free space: {} bytes", + total.saturating_sub(s.raw_data).saturating_sub(ub) + )?; + writeln!(o, "Total space: {total} bytes") +} diff --git a/crates/clawhdf5-tools/src/value.rs b/crates/clawhdf5-tools/src/value.rs new file mode 100644 index 0000000..8160574 --- /dev/null +++ b/crates/clawhdf5-tools/src/value.rs @@ -0,0 +1,483 @@ +//! Decoding one element of any datatype into a [`Value`], and printing it. +//! +//! Decoding never panics: a short buffer, an unknown byte order or a +//! dangling heap reference becomes [`Value::Error`]. + +use clawhdf5_format::datatype::{Datatype, DatatypeByteOrder, ReferenceType, StringPadding}; +use serde_json::Value as J; + +use crate::dtype; +use crate::h5::H5; + +#[derive(Debug, Clone, PartialEq)] +pub enum Value { + Int(i128), + /// A float and the width (in bits) it was stored with, so it prints at + /// its own precision. + Float(f64, u8), + Str(String), + /// Opaque, bitfield, time and oversized integers. + Bytes(Vec), + /// An enum member (name, when the value matches one) and its value. + Enum(Option, i128), + Compound(Vec<(String, Value)>), + Array(Vec), + /// A variable-length sequence. + Seq(Vec), + /// A reference: the referenced object's address (`None` = null). + Ref(Option), + /// A region or attribute reference, kept as its bytes. + OtherRef(Vec), + Error(String), +} + +pub fn hex(b: &[u8]) -> String { + let mut s = String::from("0x"); + for x in b { + s.push_str(&format!("{x:02x}")); + } + s +} + +/// Element bytes as an unsigned integer (at most 16 bytes). +fn bits(b: &[u8], order: &DatatypeByteOrder) -> Option { + if b.len() > 16 { + return None; + } + let mut v = 0u128; + match order { + DatatypeByteOrder::LittleEndian => { + for (i, x) in b.iter().enumerate() { + v |= u128::from(*x) << (8 * i); + } + } + DatatypeByteOrder::BigEndian => { + for x in b { + v = (v << 8) | u128::from(*x); + } + } + DatatypeByteOrder::Vax => return None, + } + Some(v) +} + +/// The value of an integer (fixed-point) element, honouring its bit offset +/// and precision. `None` when it cannot be represented (over 16 bytes) or +/// `dt` is not an integer. +pub fn decode_int(dt: &Datatype, b: &[u8]) -> Option { + let Datatype::FixedPoint { + size, + byte_order, + signed, + bit_offset, + bit_precision, + } = dt + else { + return None; + }; + let size = usize::try_from(*size).ok()?; + let v = bits(b.get(..size)?, byte_order)?; + let off = u32::from(*bit_offset); + let prec = u32::from(*bit_precision).min(128); + if off >= 128 || prec == 0 { + return Some(0); + } + let mut x = v >> off; + if prec < 128 { + x &= (1u128 << prec) - 1; + } + if *signed && prec < 128 && (x >> (prec - 1)) & 1 == 1 { + x |= !0u128 << prec; + return Some(x as i128); + } + if !*signed && prec == 128 && x > i128::MAX as u128 { + return None; + } + Some(x as i128) +} + +fn decode_float(dt: &Datatype, b: &[u8]) -> Value { + let Datatype::FloatingPoint { + size, byte_order, .. + } = dt + else { + return Value::Error("not a float".into()); + }; + let Ok(n) = usize::try_from(*size) else { + return Value::Error("float size".into()); + }; + let Some(b) = b.get(..n) else { + return Value::Error("short element".into()); + }; + if dtype::is_ieee(dt) + && let Some(v) = bits(b, byte_order) + { + return match n { + 2 => Value::Float( + f64::from(clawhdf5_format::float16::f16_bits_to_f32(v as u16)), + 16, + ), + 4 => Value::Float(f64::from(f32::from_bits(v as u32)), 32), + _ => Value::Float(f64::from_bits(v as u64), 64), + }; + } + // Non-IEEE layouts (N-Bit floats, VAX order): the library converts. + match clawhdf5_format::data_read::read_as_f64(b, dt) { + Ok(v) if v.len() == 1 => Value::Float(v[0], (n * 8).min(64) as u8), + Ok(_) => Value::Error("float conversion".into()), + Err(e) => Value::Error(e.to_string()), + } +} + +fn trim_string(b: &[u8], pad: Option<&StringPadding>) -> String { + let cut = b.iter().position(|&c| c == 0).unwrap_or(b.len()); + let mut s = &b[..cut]; + if matches!(pad, Some(StringPadding::SpacePad)) { + while let [rest @ .., b' '] = s { + s = rest; + } + } + String::from_utf8_lossy(s).into_owned() +} + +/// Little-endian unsigned integer of `b` (up to 8 bytes). +fn le(b: &[u8]) -> u64 { + b.iter() + .take(8) + .enumerate() + .fold(0u64, |a, (i, &x)| a | (u64::from(x) << (8 * i))) +} + +/// Decodes elements of one file. +pub struct Decoder<'a> { + pub h5: &'a H5, +} + +impl<'a> Decoder<'a> { + pub fn new(h5: &'a H5) -> Self { + Self { h5 } + } + + /// Decode element `i` of `raw`, an array of `dt` elements. + pub fn element(&self, dt: &Datatype, raw: &[u8], i: usize) -> Value { + let size = dt.type_size() as usize; + match i + .checked_mul(size) + .and_then(|s| raw.get(s..s.checked_add(size)?)) + { + Some(b) => self.decode(dt, b, 0), + None => Value::Error("element out of range".into()), + } + } + + pub fn decode(&self, dt: &Datatype, b: &[u8], depth: u32) -> Value { + if depth > 32 { + return Value::Error("datatype nesting too deep".into()); + } + let size = dt.type_size() as usize; + let Some(b) = b.get(..size) else { + return Value::Error("short element".into()); + }; + match dt { + Datatype::FixedPoint { .. } => match decode_int(dt, b) { + Some(v) => Value::Int(v), + None => Value::Bytes(b.to_vec()), + }, + Datatype::FloatingPoint { .. } => decode_float(dt, b), + Datatype::Time { .. } | Datatype::BitField { .. } | Datatype::Opaque { .. } => { + Value::Bytes(b.to_vec()) + } + Datatype::String { padding, .. } => Value::Str(trim_string(b, Some(padding))), + Datatype::Compound { members, .. } => { + let mut out = Vec::with_capacity(members.len()); + for m in members { + let off = usize::try_from(m.byte_offset).unwrap_or(usize::MAX); + let v = match b.get(off..) { + Some(mb) => self.decode(&m.datatype, mb, depth + 1), + None => Value::Error("member out of bounds".into()), + }; + out.push((m.name.clone(), v)); + } + Value::Compound(out) + } + Datatype::Reference { ref_type, .. } => match ref_type { + ReferenceType::Object | ReferenceType::Object2 => { + match clawhdf5_format::data_read::read_object_references(b, dt, self.h5.os()) { + Ok(r) if r.len() == 1 => { + let a = r[0].address; + let undef = a == u64::MAX + || (self.h5.os() < 8 && a == (1u64 << (8 * self.h5.os())) - 1); + Value::Ref(if undef || a == 0 { None } else { Some(a) }) + } + Ok(_) => Value::Error("reference".into()), + Err(e) => Value::Error(e.to_string()), + } + } + _ => Value::OtherRef(b.to_vec()), + }, + Datatype::Enumeration { + base_type, members, .. + } => { + let Some(v) = decode_int(base_type, b) else { + return Value::Bytes(b.to_vec()); + }; + let bs = base_type.type_size() as usize; + let name = members + .iter() + .find(|m| m.value.get(..bs) == b.get(..bs)) + .map(|m| m.name.clone()); + Value::Enum(name, v) + } + Datatype::Array { + base_type, + dimensions, + } => { + let n = dimensions + .iter() + .try_fold(1usize, |a, &d| a.checked_mul(d as usize)); + let bs = base_type.type_size() as usize; + let Some(n) = n.filter(|n| n.checked_mul(bs).is_some_and(|t| t <= b.len())) else { + return Value::Error("array larger than its element".into()); + }; + let mut out = Vec::with_capacity(n); + for k in 0..n { + out.push(self.decode(base_type, &b[k * bs..], depth + 1)); + } + Value::Array(out) + } + Datatype::VariableLength { + is_string, + padding, + base_type, + .. + } => self.decode_vlen(*is_string, padding.as_ref(), base_type, b, depth), + } + } + + fn decode_vlen( + &self, + is_string: bool, + padding: Option<&StringPadding>, + base: &Datatype, + b: &[u8], + depth: u32, + ) -> Value { + let os = usize::from(self.h5.os()); + let (Some(lenb), Some(addrb), Some(idxb)) = + (b.get(..4), b.get(4..4 + os), b.get(4 + os..8 + os)) + else { + return Value::Error("short VL element".into()); + }; + let len = le(lenb) as usize; + let addr = le(addrb); + let idx = le(idxb) as u32; + let undef = if os >= 8 { + u64::MAX + } else { + (1u64 << (8 * os)) - 1 + }; + let obj = if len == 0 || addr == 0 || addr == undef { + Vec::new() + } else { + match self.h5.heap_object(addr, idx) { + Ok(o) => o, + Err(e) => return Value::Error(e.to_string()), + } + }; + if is_string { + let l = len.min(obj.len()); + return Value::Str(trim_string(&obj[..l], padding)); + } + let bs = base.type_size() as usize; + if bs == 0 { + return Value::Error("VL base type of size 0".into()); + } + match len.checked_mul(bs) { + Some(need) if need <= obj.len() => {} + _ => return Value::Error("VL sequence longer than its heap object".into()), + } + let mut out = Vec::with_capacity(len); + for k in 0..len { + out.push(self.decode(base, &obj[k * bs..], depth + 1)); + } + Value::Seq(out) + } +} + +/// Format a float like C's `%g` would at full round-trip precision: plain +/// digits for moderate magnitudes, an exponent otherwise. +pub fn fmt_float(v: f64, width: u8) -> String { + if v.is_nan() { + return "NaN".into(); + } + if v.is_infinite() { + return if v > 0.0 { "Inf".into() } else { "-Inf".into() }; + } + let a = v.abs(); + let plain = a == 0.0 || (1e-5..1e16).contains(&a); + match (width, plain) { + (16 | 32, true) => format!("{}", v as f32), + (16 | 32, false) => format!("{:e}", v as f32), + (_, true) => format!("{v}"), + (_, false) => format!("{v:e}"), + } +} + +fn escape(s: &str) -> String { + let mut o = String::with_capacity(s.len() + 2); + for c in s.chars() { + match c { + '"' => o.push_str("\\\""), + '\\' => o.push_str("\\\\"), + '\n' => o.push_str("\\n"), + '\r' => o.push_str("\\r"), + '\t' => o.push_str("\\t"), + c if (c as u32) < 0x20 => o.push_str(&format!("\\{:03o}", c as u32)), + c => o.push(c), + } + } + o +} + +/// A fixed-length string element's bytes quoted as h5dump prints a +/// null-padded string: every byte, NULs as `\000`. +pub fn quote_bytes(b: &[u8]) -> String { + format!("\"{}\"", escape(&String::from_utf8_lossy(b))) +} + +/// Text form, as in an h5dump DATA block. +pub fn text(v: &Value, h5paths: &dyn Fn(u64) -> Option) -> String { + match v { + Value::Int(i) => i.to_string(), + Value::Float(f, w) => fmt_float(*f, *w), + Value::Str(s) => format!("\"{}\"", escape(s)), + Value::Bytes(b) => hex(b), + Value::Enum(Some(n), _) => n.clone(), + Value::Enum(None, i) => i.to_string(), + Value::Compound(ms) => format!( + "{{ {} }}", + ms.iter() + .map(|(_, v)| text(v, h5paths)) + .collect::>() + .join(", ") + ), + Value::Array(vs) => format!( + "[ {} ]", + vs.iter() + .map(|v| text(v, h5paths)) + .collect::>() + .join(", ") + ), + Value::Seq(vs) => format!( + "({})", + vs.iter() + .map(|v| text(v, h5paths)) + .collect::>() + .join(", ") + ), + Value::Ref(None) => "NULL".into(), + Value::Ref(Some(a)) => match h5paths(*a) { + Some(p) => format!("\"{p}\""), + None => format!("{a:#x}"), + }, + Value::OtherRef(b) => hex(b), + Value::Error(e) => format!(""), + } +} + +/// hdf5-json value form. +pub fn to_json(v: &Value, h5paths: &dyn Fn(u64) -> Option) -> J { + match v { + Value::Int(i) => { + if let Ok(x) = i64::try_from(*i) { + J::from(x) + } else if let Ok(x) = u64::try_from(*i) { + J::from(x) + } else { + J::from(i.to_string()) + } + } + Value::Float(f, _) => { + if f.is_finite() { + serde_json::Number::from_f64(*f) + .map(J::Number) + .unwrap_or(J::Null) + } else if f.is_nan() { + J::from("NaN") + } else if *f > 0.0 { + J::from("Infinity") + } else { + J::from("-Infinity") + } + } + Value::Str(s) => J::from(s.as_str()), + Value::Bytes(b) | Value::OtherRef(b) => J::from(hex(b)), + Value::Enum(_, i) => to_json(&Value::Int(*i), h5paths), + Value::Compound(ms) => J::Array(ms.iter().map(|(_, v)| to_json(v, h5paths)).collect()), + Value::Array(vs) | Value::Seq(vs) => { + J::Array(vs.iter().map(|v| to_json(v, h5paths)).collect()) + } + Value::Ref(None) => J::Null, + Value::Ref(Some(a)) => J::from(h5paths(*a).unwrap_or_else(|| format!("{a:#x}"))), + Value::Error(e) => serde_json::json!({ "error": e }), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn int(size: u32, signed: bool, order: DatatypeByteOrder, off: u16, prec: u16) -> Datatype { + Datatype::FixedPoint { + size, + byte_order: order, + signed, + bit_offset: off, + bit_precision: prec, + } + } + + #[test] + fn integers_decode_with_order_offset_and_sign() { + let le = DatatypeByteOrder::LittleEndian; + let be = DatatypeByteOrder::BigEndian; + assert_eq!( + decode_int(&int(2, true, le.clone(), 0, 16), &[0xff, 0xff]), + Some(-1) + ); + assert_eq!( + decode_int(&int(2, false, le, 0, 16), &[0xff, 0xff]), + Some(65535) + ); + assert_eq!( + decode_int(&int(2, true, be.clone(), 0, 16), &[0x80, 0x00]), + Some(-32768) + ); + // 17-bit signed field at offset 4: -5 + let stored = (((-5i32) as u32) & 0x1_FFFF) << 4; + assert_eq!( + decode_int(&int(4, true, be, 4, 17), &stored.to_be_bytes()), + Some(-5) + ); + assert_eq!( + decode_int(&int(4, true, DatatypeByteOrder::Vax, 0, 32), &[0; 4]), + None + ); + assert_eq!( + decode_int( + &int(4, true, DatatypeByteOrder::LittleEndian, 0, 32), + &[0; 2] + ), + None + ); + } + + #[test] + fn floats_print_at_their_own_precision() { + assert_eq!(fmt_float(f64::from(0.1f32), 32), "0.1"); + assert_eq!(fmt_float(0.1, 64), "0.1"); + assert_eq!(fmt_float(1e20, 64), "1e20"); + assert_eq!(fmt_float(2.0, 64), "2"); + assert_eq!(fmt_float(f64::NAN, 64), "NaN"); + } +} diff --git a/crates/clawhdf5-tools/tests/gen_files.py b/crates/clawhdf5-tools/tests/gen_files.py new file mode 100644 index 0000000..8bc4e1e --- /dev/null +++ b/crates/clawhdf5-tools/tests/gen_files.py @@ -0,0 +1,164 @@ +"""Write the HDF5 files the h5rs interop tests run on. + +usage: gen_files.py OUTDIR + +Writes OUTDIR/{earliest,latest}.h5 (the same content with the oldest and the +newest file-format structures: symbol tables and v1 B-trees vs. v2 object +headers, fractal heaps, v2 B-trees and the chunk indexes of HDF5 1.10+), the +pairs the diff tests compare, and a file with a user block. Prints one JSON +object with the values h5py reads back, for the dump tests. +""" + +import json +import os +import sys + +import h5py +import numpy as np + +out = sys.argv[1] + + +def content(f, dense): + f.attrs["title"] = "h5rs test" + f.attrs["version"] = np.int64(3) + f.attrs["scale"] = np.array([0.5, 1.5], dtype="f4") + f["contig"] = np.arange(12, dtype="f8").reshape(3, 4) + dcpl = h5py.h5p.create(h5py.h5p.DATASET_CREATE) + dcpl.set_layout(h5py.h5d.COMPACT) + space = h5py.h5s.create_simple((4,)) + h5py.h5d.create(f.id, b"compact", h5py.h5t.STD_I16LE, space, dcpl).write( + h5py.h5s.ALL, h5py.h5s.ALL, np.arange(4, dtype="f4")]) + f.create_dataset( + "enum", data=np.array([0, 1, 1], dtype="u1"), + dtype=h5py.enum_dtype({"RED": 0, "GREEN": 1}, basetype="u1"), + ) + f["be"] = np.arange(5, dtype=">i4") + f["arr"] = np.array([([1, 2, 3],), ([4, 5, 6],)], dtype=[("v", "3i4")]) + f["named_t"] = np.dtype("i8") + f["soft"] = h5py.SoftLink("/contig") + f["dangling"] = h5py.SoftLink("/nowhere") + f["external"] = h5py.ExternalLink("other.h5", "/x") + f["hard2"] = g["sub"] + many = f.create_group("many") + for i in range(12 if dense else 4): + many[f"d{i:02}"] = np.int32(i) + many.attrs[f"a{i:02}"] = i + + +values = {} +# "latest" under HDF5 2.0 writes datatype messages that libhdf5 1.14 tools +# cannot read, so the newest format is taken as 1.14's. +for libver in ("earliest", "latest"): + path = os.path.join(out, f"{libver}.h5") + bounds = ("earliest", "v114") if libver == "earliest" else ("v114", "v114") + with h5py.File(path, "w", libver=bounds) as f: + content(f, True) + with h5py.File(path, "r") as f: + vals = {} + + def grab(name, obj): + if isinstance(obj, h5py.Dataset) and obj.dtype.kind in "iuf" and obj.shape is not None: + vals["/" + name] = obj[()].tolist() + + f.visititems(grab) + values[libver] = vals + +# diff pairs +def small(path, data=None, extra=False, attr=False, shape=(3, 4), dtype="f8"): + with h5py.File(os.path.join(out, path), "w") as f: + d = np.arange(12, dtype=dtype).reshape(shape) if data is None else data + f["d"] = d + f["g/x"] = np.arange(3) + if extra: + f["only_here"] = 1 + if attr: + f["d"].attrs["u"] = 1 + + +base = np.arange(12, dtype="f8").reshape(3, 4) +small("base.h5") +small("same.h5") +changed = base.copy() +changed[0, 2] += 0.001 +changed[2, 3] += 0.001 +small("changed.h5", data=changed) +small("extra.h5", extra=True) +small("attr.h5", attr=True) +small("reshaped.h5", shape=(4, 3)) +small("int.h5", dtype="i4") + +# One object under two names (a hard link) against two separate copies. +with h5py.File(os.path.join(out, "hardlinked.h5"), "w") as f: + f["x"] = np.arange(5) + f["y"] = f["x"] + g = f.create_group("g") + g["d"] = np.arange(3) + g.create_group("s")["e"] = np.arange(2) + f["h"] = g +for name, last in (("copied.h5", 1), ("copied_changed.h5", 9)): + with h5py.File(os.path.join(out, name), "w") as f: + f["x"] = np.arange(5) + f["y"] = np.arange(5) + for gname in ("g", "h"): + g = f.create_group(gname) + g["d"] = np.arange(3) + g.create_group("s")["e"] = np.array([0, last if gname == "h" else 1]) + +# 64-bit integers one apart, beyond f64's 2^53 integer precision. +for name, d in (("big1.h5", 0), ("big2.h5", 1)): + with h5py.File(os.path.join(out, name), "w") as f: + f["i"] = np.array([2**60 + d, -(2**62) - d], dtype="i8") + f["u"] = np.array([2**64 - 1 - d], dtype="u8") + +# Soft links: the same link targets, whose target objects differ. +for name, v in (("soft1.h5", 0), ("soft2.h5", 1)): + with h5py.File(os.path.join(out, name), "w") as f: + f["z"] = np.arange(4) + v + f.create_group("g")["s"] = h5py.SoftLink("/z") + grp = f.create_group("grp") + grp["d"] = np.arange(3) + v + f["lnk"] = h5py.SoftLink("/grp") + f["dang"] = h5py.SoftLink("/nowhere") +# Soft links: different link targets, whose target objects are equal. +for name, t in (("target1.h5", "/a"), ("target2.h5", "/b")): + with h5py.File(os.path.join(out, name), "w") as f: + f["a"] = np.arange(4) + f["b"] = np.arange(4) + f["s"] = h5py.SoftLink(t) + f["rel"] = h5py.SoftLink("a") + +# Null-padded fixed strings with NULs, at the top level and nested in a +# compound and in an array member. +with h5py.File(os.path.join(out, "nulstrings.h5"), "w") as f: + f["top"] = np.array([b"", b"ab", b"a\x00b"], dtype="S3") + f["cmp"] = np.array( + [(b"", 1), (b"ab", 2), (b"a\x00b", 3)], dtype=[("s", "S3"), ("i", "i4")] + ) + f["arr"] = np.array([([b"", b"x"],)], dtype=[("v", "(2,)S2")]) + +with h5py.File(os.path.join(out, "userblock.h5"), "w", userblock_size=1024) as f: + f["d"] = np.arange(10) + +print(json.dumps(values)) diff --git a/crates/clawhdf5-tools/tests/h5rs_interop.rs b/crates/clawhdf5-tools/tests/h5rs_interop.rs new file mode 100644 index 0000000..1669beb --- /dev/null +++ b/crates/clawhdf5-tools/tests/h5rs_interop.rs @@ -0,0 +1,772 @@ +//! `h5rs` against libhdf5's own tools, on files h5py writes +//! (`tests/gen_files.py`): `ls` against h5ls, `stat` against h5stat, `dump` +//! against h5dump, `diff` exit codes against h5diff, `dump --json` values +//! against h5py, and `check` on valid and deliberately corrupted files. +//! +//! Needs python3 with h5py (`CLAWHDF5_PYTHON`) and, for the comparisons, +//! the libhdf5 command-line tools (h5ls, h5stat, h5dump, h5diff) on `PATH`. +//! Each test skips when what it needs is missing, unless +//! `CLAWHDF5_REQUIRE_INTEROP=1`, which turns a skip into a failure. + +use std::collections::BTreeMap; +use std::path::{Path, PathBuf}; +use std::process::{Command, Output}; + +use clawhdf5_format::checksum::jenkins_lookup3; + +fn python() -> String { + std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string()) +} + +fn interop_required() -> bool { + std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1") +} + +fn python_available() -> bool { + Command::new(python()) + .args(["-c", "import h5py, numpy"]) + .output() + .map(|o| o.status.success()) + .unwrap_or(false) +} + +fn tool_available(name: &str) -> bool { + Command::new(name) + .arg("--version") + .output() + .map(|o| o.status.success()) + .unwrap_or(false) +} + +/// Skip (return true) when `what` is unavailable, or fail when interop is +/// required. +fn missing(ok: bool, what: &str) -> bool { + if ok { + return false; + } + assert!( + !interop_required(), + "CLAWHDF5_REQUIRE_INTEROP=1 but {what} is not available" + ); + eprintln!("SKIP: {what} not available"); + true +} + +fn h5rs(args: &[&str]) -> Output { + Command::new(env!("CARGO_BIN_EXE_h5rs")) + .args(args) + .output() + .expect("run h5rs") +} + +fn run(tool: &str, args: &[&str]) -> Output { + Command::new(tool).args(args).output().expect("run tool") +} + +fn stdout(o: &Output) -> String { + String::from_utf8_lossy(&o.stdout).into_owned() +} + +/// The generated files and what h5py read from them. +struct Files { + dir: tempfile::TempDir, + values: serde_json::Value, +} + +impl Files { + fn path(&self, name: &str) -> PathBuf { + self.dir.path().join(name) + } + + fn p(&self, name: &str) -> String { + self.path(name).to_string_lossy().into_owned() + } +} + +fn generate() -> Option { + if missing(python_available(), "python3 with h5py") { + return None; + } + let dir = tempfile::tempdir().unwrap(); + let script = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/gen_files.py"); + let out = Command::new(python()) + .arg(&script) + .arg(dir.path()) + .output() + .expect("run gen_files.py"); + assert!( + out.status.success(), + "gen_files.py failed:\n{}", + String::from_utf8_lossy(&out.stderr) + ); + let values = serde_json::from_slice(&out.stdout).expect("gen_files.py output"); + Some(Files { dir, values }) +} + +const LIBVERS: [&str; 2] = ["earliest.h5", "latest.h5"]; + +// --------------------------------------------------------------------------- +// ls +// --------------------------------------------------------------------------- + +/// Every h5ls line must be the start of the matching h5rs line (h5rs adds +/// the datatype after a dataset's shape). +fn assert_ls_like(ours: &str, reference: &str, what: &str) { + let ours: Vec<&str> = ours.lines().collect(); + let refs: Vec<&str> = reference.lines().collect(); + assert_eq!( + ours.len(), + refs.len(), + "{what}: line count\nh5rs:\n{}\nh5ls:\n{}", + ours.join("\n"), + refs.join("\n") + ); + for (o, r) in ours.iter().zip(&refs) { + let r = r.trim_end(); + assert!( + o.starts_with(r) && (o.len() == r.len() || r.contains("Dataset {")), + "{what}:\n h5rs: {o}\n h5ls: {r}" + ); + } +} + +#[test] +fn ls_matches_h5ls() { + let Some(f) = generate() else { return }; + if missing(tool_available("h5ls"), "h5ls") { + return; + } + for name in LIBVERS.iter().chain(&["userblock.h5"]) { + let p = f.p(name); + for (args, what) in [(vec!["-r"], "recursive"), (vec![], "root")] { + let mut a: Vec<&str> = args.clone(); + a.push(&p); + let ours = h5rs(&[&["ls"], a.as_slice()].concat()); + assert!(ours.status.success(), "{name} {what}: {:?}", ours); + let reference = run("h5ls", &a); + assert_ls_like( + &stdout(&ours), + &stdout(&reference), + &format!("{name} {what}"), + ); + } + } + // A group, and a dataset, named after the file. + for obj in ["/grp", "/grp/sub", "/contig", "/grp/ext2"] { + let arg = format!("{}{obj}", f.p("latest.h5")); + let ours = h5rs(&["ls", &arg]); + let reference = run("h5ls", &[&arg]); + assert_ls_like(&stdout(&ours), &stdout(&reference), &arg); + let ours = h5rs(&["ls", "-r", &arg]); + let reference = run("h5ls", &["-r", &arg]); + assert_ls_like(&stdout(&ours), &stdout(&reference), &format!("-r {arg}")); + } +} + +#[test] +fn ls_verbose_describes_layout_filters_and_attributes() { + let Some(f) = generate() else { return }; + let o = h5rs(&["ls", "-r", "-v", &f.p("latest.h5")]); + assert!(o.status.success()); + let s = stdout(&o); + for want in [ + "Layout: chunked (extensible array index)", + "Layout: chunked (v2 B-tree index)", + "Layout: chunked (fixed array index)", + "Layout: chunked (single chunk index)", + "Filter-0: shuffle-2", + "Filter-1: deflate-1", + "Filter-2: fletcher32-3", + "Chunks: {100} 400 bytes", + "Attribute: units scalar", + "Data: \"m\"", + ] { + assert!(s.contains(want), "missing {want:?} in\n{s}"); + } + let o = h5rs(&["ls", "-v", &format!("{}/grp", f.p("earliest.h5"))]); + assert!(stdout(&o).contains("Layout: chunked (v1 B-tree index)")); +} + +// --------------------------------------------------------------------------- +// stat +// --------------------------------------------------------------------------- + +/// `key: value` lines of an h5stat-style report. +fn stat_facts(s: &str) -> BTreeMap { + s.lines() + .filter_map(|l| { + let (k, v) = l.rsplit_once(':')?; + Some((k.trim().to_string(), v.trim().to_string())) + }) + .collect() +} + +#[test] +fn stat_matches_h5stat() { + let Some(f) = generate() else { return }; + if missing(tool_available("h5stat"), "h5stat") { + return; + } + let keys = [ + "# of unique groups", + "# of unique datasets", + "# of unique named datatypes", + "# of unique links", + "# of unique other", + "Max. # of links to object", + "Max. # of objects in group", + "Max. rank of datasets", + "Max. dimension size of 1-D datasets", + "Total raw data size", + "Dataset layout counts[COMPACT]", + "Dataset layout counts[CONTIG]", + "Dataset layout counts[CHUNKED]", + "Dataset layout counts[VIRTUAL]", + "NO filter", + "GZIP filter", + "SHUFFLE filter", + "FLETCHER32 filter", + "SZIP filter", + "NBIT filter", + "SCALEOFFSET filter", + "USER-DEFINED filter", + "Max. # of attributes to objects", + "Total space", + ]; + for name in LIBVERS.iter().chain(&["userblock.h5", "base.h5"]) { + let p = f.p(name); + let ours = h5rs(&["stat", &p]); + assert!(ours.status.success(), "{name}: {ours:?}"); + let reference = run("h5stat", &[&p]); + assert!(reference.status.success(), "h5stat {name} failed"); + let (o, r) = (stat_facts(&stdout(&ours)), stat_facts(&stdout(&reference))); + for k in keys { + // h5stat leaves out the attribute summary when nothing has one. + let Some(want) = r.get(k) else { continue }; + assert_eq!(o.get(k), Some(want), "{name}: {k}"); + } + } +} + +// --------------------------------------------------------------------------- +// dump +// --------------------------------------------------------------------------- + +#[test] +fn dump_matches_h5dump() { + let Some(f) = generate() else { return }; + if missing(tool_available("h5dump"), "h5dump") { + return; + } + for name in LIBVERS.iter().chain(&["userblock.h5", "changed.h5"]) { + let p = f.p(name); + for args in [vec![], vec!["-A"]] { + let mut a = args.clone(); + a.push(p.as_str()); + let ours = h5rs(&[&["dump"], a.as_slice()].concat()); + assert!(ours.status.success(), "{name} {args:?}: {ours:?}"); + let reference = run("h5dump", &a); + // h5dump names the file as given; h5rs by its file name. + let r = stdout(&reference).replacen(&p, name, 1); + assert_eq!(stdout(&ours), r, "{name} {args:?}"); + } + } + let p = f.p("latest.h5"); + let ours = h5rs(&["dump", "-d", "/grp/ext2", &p]); + let reference = run("h5dump", &["-d", "/grp/ext2", &p]); + assert_eq!( + stdout(&ours), + stdout(&reference).replacen(&p, "latest.h5", 1) + ); +} + +/// A null-padded fixed string prints every byte (NULs as `\000`), as +/// h5dump prints it, also inside a compound and an array member. +#[test] +fn dump_shows_nul_padding_in_nested_strings() { + let Some(f) = generate() else { return }; + let p = f.p("nulstrings.h5"); + let ours = stdout(&h5rs(&["dump", &p])); + for want in [ + r#"(0): "\000\000\000", "ab\000", "a\000b""#, + r#""\000\000\000","#, + r#""a\000b","#, + r#"[ "\000\000", "x\000" ]"#, + ] { + assert!(ours.contains(want), "no {want} in\n{ours}"); + } + if missing(tool_available("h5dump"), "h5dump") { + return; + } + let reference = run("h5dump", &[&p]); + assert_eq!(ours, stdout(&reference).replacen(&p, "nulstrings.h5", 1)); +} + +#[test] +fn dump_json_values_match_h5py() { + let Some(f) = generate() else { return }; + for name in LIBVERS { + let o = h5rs(&["dump", "--json", &f.p(name)]); + assert!(o.status.success(), "{name}: {o:?}"); + let doc: serde_json::Value = serde_json::from_slice(&o.stdout).unwrap(); + assert_eq!(doc["apiVersion"], "1.1.1"); + let root = doc["root"].as_str().unwrap(); + assert!(doc["groups"][root].is_object()); + let mut by_path = BTreeMap::new(); + for (_, d) in doc["datasets"].as_object().unwrap() { + for a in d["alias"].as_array().unwrap() { + by_path.insert(a.as_str().unwrap().to_string(), d.clone()); + } + } + let want = f.values[name.trim_end_matches(".h5")].as_object().unwrap(); + assert!(want.len() > 10); + for (path, v) in want { + let d = by_path + .get(path) + .unwrap_or_else(|| panic!("{name}: no {path}")); + let got = &d["value"]; + // JSON numbers: compare as f64 (h5py's ints and floats both + // round-trip exactly through JSON here). + assert_eq!(flatten(got), flatten(v), "{name}: {path}"); + } + let links = doc["groups"][root]["links"].as_array().unwrap(); + let soft = links.iter().find(|l| l["title"] == "soft").unwrap(); + assert_eq!(soft["class"], "H5L_TYPE_SOFT"); + assert_eq!(soft["h5path"], "/contig"); + let ext = links.iter().find(|l| l["title"] == "external").unwrap(); + assert_eq!(ext["class"], "H5L_TYPE_EXTERNAL"); + assert_eq!(ext["file"], "other.h5"); + } +} + +fn flatten(v: &serde_json::Value) -> Vec { + match v { + serde_json::Value::Array(a) => a.iter().flat_map(flatten).collect(), + serde_json::Value::Number(n) => vec![n.as_f64().unwrap()], + other => panic!("not numeric: {other}"), + } +} + +// --------------------------------------------------------------------------- +// diff +// --------------------------------------------------------------------------- + +fn code(o: &Output) -> i32 { + o.status.code().expect("exit code") +} + +#[test] +fn diff_exit_codes_match_h5diff() { + let Some(f) = generate() else { return }; + if missing(tool_available("h5diff"), "h5diff") { + return; + } + let (b, same, changed, extra, attr) = ( + f.p("base.h5"), + f.p("same.h5"), + f.p("changed.h5"), + f.p("extra.h5"), + f.p("attr.h5"), + ); + let missing_file = f.p("does-not-exist.h5"); + let cases: Vec<(Vec<&str>, i32)> = vec![ + (vec![&b, &same], 0), + (vec![&b, &b], 0), + (vec![&b, &changed], 1), + (vec![&b, &changed, "/d"], 1), + (vec![&b, &changed, "/g"], 0), + (vec!["-d", "0.01", &b, &changed], 0), + (vec!["-d", "0.0001", &b, &changed], 1), + (vec!["-p", "0.01", &b, &changed], 0), + (vec!["-p", "1e-9", &b, &changed], 1), + (vec![&b, &extra], 1), + (vec![&b, &attr], 1), + (vec![&b, &attr, "/g"], 0), + (vec![&b, &same, "/nope"], 2), + (vec![&b, &missing_file], 2), + (vec![&b, &same, "/d", "/d"], 0), + ]; + for (args, want) in cases { + let theirs = code(&run("h5diff", &args)); + assert_eq!(theirs, want, "h5diff {args:?} (test expectation)"); + let o = h5rs(&[&["diff"], args.as_slice()].concat()); + assert_eq!(code(&o), want, "h5rs diff {args:?}: {}", stdout(&o)); + let q = h5rs(&[&["diff", "-q"], args.as_slice()].concat()); + assert_eq!(code(&q), want, "h5rs diff -q {args:?}"); + assert!(q.stdout.is_empty(), "-q printed output"); + } + // The report lists the differing positions. + let o = h5rs(&["diff", "-r", &b, &changed, "/d"]); + let s = stdout(&o); + assert!(s.contains("[ 0 2 ]") && s.contains("[ 2 3 ]"), "{s}"); + assert!(s.contains("2 difference(s) found"), "{s}"); +} + +/// The option names are h5diff's: -c is --compare (a flag), the count is +/// -n/--count, and the --name=value forms are accepted. +#[test] +fn diff_options_are_named_like_h5diff() { + let Some(f) = generate() else { return }; + let (b, changed) = (f.p("base.h5"), f.p("changed.h5")); + let cases: Vec<(Vec<&str>, i32)> = vec![ + // "2" is then the first file name, which does not exist. + (vec!["-r", "-c", "2", &b, &changed], 2), + (vec!["-c", &b, &changed], 1), + (vec!["--compare", &b, &b], 0), + (vec!["-r", "-n", "1", &b, &changed], 1), + (vec!["-r", "--count=1", &b, &changed], 1), + (vec!["--delta=0.01", &b, &changed], 0), + (vec!["--relative=1e-9", &b, &changed], 1), + ]; + let have_h5diff = tool_available("h5diff"); + for (args, want) in cases { + if have_h5diff { + assert_eq!(code(&run("h5diff", &args)), want, "h5diff {args:?}"); + } + let o = h5rs(&[&["diff"], args.as_slice()].concat()); + assert_eq!(code(&o), want, "h5rs diff {args:?}: {}", stdout(&o)); + } + for count in [vec!["-n", "1"], vec!["--count=1"]] { + let o = h5rs(&[&["diff", "-r"], count.as_slice(), &[&b, &changed, "/d"]].concat()); + let s = stdout(&o); + assert!( + s.contains("[ 0 2 ]") && !s.contains("[ 2 3 ]"), + "{count:?}: {s}" + ); + assert!(s.contains("2 difference(s) found"), "{s}"); + } + assert_eq!(code(&h5rs(&["diff", "--bogus=1", &b, &b])), 2); +} + +/// Tolerances on 64-bit integers one apart, far beyond f64's integer +/// precision: compared exactly, as h5diff does. +#[test] +fn diff_tolerances_compare_large_integers_exactly() { + let Some(f) = generate() else { return }; + let (a, b) = (f.p("big1.h5"), f.p("big2.h5")); + let cases: Vec<(Vec<&str>, i32)> = vec![ + (vec![&a, &b], 1), + (vec!["-d", "0", &a, &b], 1), + (vec!["-d", "0.5", &a, &b], 1), + (vec!["-d", "1", &a, &b], 0), + (vec!["-d", "0", &a, &b, "/u"], 1), + (vec!["-p", "0", &a, &b], 1), + (vec!["-p", "1e-18", &a, &b, "/i"], 1), + (vec!["-p", "1e-15", &a, &b], 0), + ]; + let have_h5diff = tool_available("h5diff"); + for (args, want) in cases { + if have_h5diff { + assert_eq!(code(&run("h5diff", &args)), want, "h5diff {args:?}"); + } + let o = h5rs(&[&["diff"], args.as_slice()].concat()); + assert_eq!(code(&o), want, "h5rs diff {args:?}: {}", stdout(&o)); + } + // The reported difference is exact too. + let s = stdout(&h5rs(&["diff", "-r", "-d", "0", &a, &b, "/u"])); + assert!( + s.contains("18446744073709551615 18446744073709551614 1\n"), + "{s}" + ); +} + +/// A soft link is compared as a link (its target path), as h5diff does, +/// unless `--follow-symlinks`, which compares (and walks into) what it +/// leads to; two dangling links are then the same. +#[test] +fn diff_soft_links_like_h5diff() { + let Some(f) = generate() else { return }; + let (s1, s2, t1, t2) = ( + f.p("soft1.h5"), + f.p("soft2.h5"), + f.p("target1.h5"), + f.p("target2.h5"), + ); + let fl = "--follow-symlinks"; + let cases: Vec<(Vec<&str>, i32)> = vec![ + (vec![&s1, &s2, "/g/s"], 0), + (vec![&s1, &s2, "/g"], 0), + (vec![&s1, &s2, "/lnk"], 0), + (vec![&s1, &s2, "/dang"], 0), + (vec![&s1, &s2, "/lnk/d"], 1), + (vec![&t1, &t2, "/s"], 1), + (vec![&t1, &t2, "/s", "/rel"], 1), + (vec![fl, &s1, &s2, "/g/s"], 1), + (vec![fl, &s1, &s2, "/g"], 1), + (vec![fl, &s1, &s2, "/lnk"], 1), + (vec![fl, &s1, &s2, "/dang"], 0), + (vec![fl, &s1, &s1], 0), + (vec![fl, &t1, &t2, "/s"], 0), + (vec![fl, &t1, &t2, "/s", "/rel"], 0), + ]; + let have_h5diff = tool_available("h5diff"); + for (args, want) in cases { + if have_h5diff { + assert_eq!(code(&run("h5diff", &args)), want, "h5diff {args:?}"); + } + let o = h5rs(&[&["diff"], args.as_slice()].concat()); + assert_eq!(code(&o), want, "h5rs diff {args:?}: {}", stdout(&o)); + } +} + +/// An object hard-linked under two names is compared under both, with +/// everything below it, so a file that shares one object between two names +/// equals a file that stores two identical copies. +#[test] +fn diff_compares_every_hard_link_path() { + let Some(f) = generate() else { return }; + let (linked, copied, changed) = ( + f.p("hardlinked.h5"), + f.p("copied.h5"), + f.p("copied_changed.h5"), + ); + // A hard-linked dataset: h5diff agrees (exit 0). + for (a, b) in [(&linked, &copied), (&copied, &linked)] { + let o = h5rs(&["diff", a, b, "/y"]); + assert_eq!(code(&o), 0, "{a} {b} /y: {}", stdout(&o)); + if tool_available("h5diff") { + assert_eq!(code(&run("h5diff", &[a, b, "/y"])), 0, "h5diff /y"); + } + // The whole file, including a hard-linked group's members (h5diff + // does not descend a group's second name, so it reports those + // members as present in one file only and exits 1). + let o = h5rs(&["diff", a, b]); + assert_eq!(code(&o), 0, "{a} {b}: {}", stdout(&o)); + assert!(!stdout(&o).contains("exists only"), "{}", stdout(&o)); + } + // A value under the second name is compared, not skipped. + let o = h5rs(&["diff", "-r", &linked, &changed]); + assert_eq!(code(&o), 1, "{}", stdout(&o)); + assert!( + stdout(&o).contains("dataset: and "), + "{}", + stdout(&o) + ); + assert!(!stdout(&o).contains(""), "{}", stdout(&o)); +} + +/// Where h5rs deliberately differs from h5diff: objects that cannot be +/// compared (another shape, another datatype class) are a difference. +#[test] +fn diff_counts_incomparable_objects_as_different() { + let Some(f) = generate() else { return }; + let b = f.p("base.h5"); + for other in ["reshaped.h5", "int.h5"] { + let o = h5rs(&["diff", &b, &f.p(other)]); + assert_eq!(code(&o), 1, "{other}: {}", stdout(&o)); + assert!(stdout(&o).contains("Not comparable"), "{other}"); + } +} + +// --------------------------------------------------------------------------- +// check +// --------------------------------------------------------------------------- + +#[test] +fn check_accepts_valid_files() { + let Some(f) = generate() else { return }; + for name in LIBVERS.iter().chain(&[ + "userblock.h5", + "base.h5", + "changed.h5", + "extra.h5", + "attr.h5", + ]) { + let o = h5rs(&["check", "--data", &f.p(name)]); + assert_eq!(code(&o), 0, "{name}:\n{}", stdout(&o)); + assert!(stdout(&o).contains("no problems found")); + } + // The newest format's checksummed structures were all verified. + let s = stdout(&h5rs(&["check", &f.p("latest.h5")])); + assert!(s.contains("superblock 1, v2 object headers 33,"), "{s}"); + assert!(s.contains("chunk indexes 5"), "{s}"); +} + +/// Signatures of checksummed structures, and where each one's checksum +/// is: found by trying every end offset (the checksum is the Jenkins +/// lookup3 hash of the bytes before it). A fractal heap direct block's +/// checksum is instead stored near its start, computed over the whole block +/// with the field zeroed. +fn checksummed_structures(data: &[u8]) -> Vec<(&'static str, usize, usize)> { + const SIGS: [&str; 13] = [ + "OHDR", "OCHK", "BTHD", "BTIN", "BTLF", "FRHP", "FHIB", "EAHD", "EAIB", "EASB", "EADB", + "FAHD", "FADB", + ]; + let mut found = Vec::new(); + for start in 0..data.len().saturating_sub(4) { + let sig = &data[start..start + 4]; + if let Some(name) = SIGS.iter().find(|s| s.as_bytes() == sig) { + let limit = data.len().min(start + 65536); + for end in start + 8..limit.saturating_sub(4) { + let stored = u32::from_le_bytes(data[end..end + 4].try_into().unwrap()); + if jenkins_lookup3(&data[start..end]) == stored { + found.push((*name, start, end)); + break; + } + } + } else if sig == b"FHDB" { + // version(1) heap address(8) block offset(1..8) checksum(4) + 'found: for boff in 1..=8usize { + let at = start + 5 + 8 + boff; + let Some(stored) = data.get(at..at + 4) else { + break; + }; + let stored = u32::from_le_bytes(stored.try_into().unwrap()); + for size in [512usize, 1024, 2048, 4096, 8192, 16384, 32768, 65536] { + let Some(block) = data.get(start..start + size) else { + break; + }; + let mut b = block.to_vec(); + b[at - start..at - start + 4].fill(0); + if jenkins_lookup3(&b) == stored { + found.push(("FHDB", start, at)); + break 'found; + } + } + } + } + } + found +} + +#[test] +fn check_flags_every_corrupted_checksum() { + let Some(f) = generate() else { return }; + let path = f.path("latest.h5"); + let data = std::fs::read(&path).unwrap(); + let structures = checksummed_structures(&data); + let kinds: std::collections::BTreeSet<&str> = structures.iter().map(|s| s.0).collect(); + for want in [ + "OHDR", "OCHK", "BTHD", "BTLF", "FRHP", "FHDB", "EAHD", "EAIB", "EADB", "FAHD", "FADB", + ] { + assert!(kinds.contains(want), "test file has no {want}: {kinds:?}"); + } + let bad = f.path("bad.h5"); + for (name, start, at) in structures { + let mut d = data.clone(); + d[at] ^= 0x01; + std::fs::write(&bad, &d).unwrap(); + let o = h5rs(&["check", &bad.to_string_lossy()]); + let s = stdout(&o); + assert_eq!( + code(&o), + 1, + "{name} at {start:#x}: checksum flip not flagged\n{s}" + ); + assert!( + s.lines() + .any(|l| l.starts_with("problem: 0x") && l.contains("checksum")), + "{name} at {start:#x}: no checksum problem reported\n{s}" + ); + } +} + +/// `check --data` follows variable-length elements into the global heap: +/// a damaged collection is reported at its address, for the dataset and +/// the attribute that point into it. Without --data it is not read. +#[test] +fn check_data_follows_vl_data_into_the_global_heap() { + let Some(f) = generate() else { return }; + for name in LIBVERS { + let mut data = std::fs::read(f.path(name)).unwrap(); + let gcols: Vec = (0..data.len() - 4) + .filter(|&i| &data[i..i + 4] == b"GCOL") + .collect(); + assert!(!gcols.is_empty(), "{name}: no global heap"); + // "GCOL" version(1) reserved(3) size(8), then heap object 1: + // index(2) refcount(2) reserved(4) size(8) — claim 4 GiB. + let at = gcols[0]; + data[at + 24..at + 28].fill(0xff); + let bad = f.path(&format!("bad-gcol-{name}")); + std::fs::write(&bad, &data).unwrap(); + let bad = bad.to_string_lossy().into_owned(); + let o = h5rs(&["check", "--data", &bad]); + let s = stdout(&o); + assert_eq!(code(&o), 1, "{name}: {s}"); + let want = format!("problem: {at:#x} /vlstr: variable-length data: global heap"); + assert!(s.contains(&want), "{name}: no {want:?} in\n{s}"); + // h5dump refuses the file too. + if tool_available("h5dump") { + assert_ne!(code(&run("h5dump", &[&bad])), 0, "{name}: h5dump read it"); + } + assert_eq!(code(&h5rs(&["check", &bad])), 0, "{name}: without --data"); + let ok = h5rs(&["check", "--data", &f.p(name)]); + assert!( + stdout(&ok).contains(&format!("global heap collections read: {}", gcols.len())), + "{name}: {}", + stdout(&ok) + ); + } +} + +#[test] +fn check_flags_a_truncated_file() { + let Some(f) = generate() else { return }; + let data = std::fs::read(f.path("latest.h5")).unwrap(); + let bad = f.path("truncated.h5"); + std::fs::write(&bad, &data[..data.len() * 3 / 4]).unwrap(); + let o = h5rs(&["check", &bad.to_string_lossy()]); + assert_eq!(code(&o), 1); + assert!(stdout(&o).contains("file is truncated"), "{}", stdout(&o)); +} + +/// A v1 B-tree chunk index has no checksum; its keys are checked against +/// the dataset: a chunk offset that is not a multiple of the chunk size. +#[test] +fn check_flags_a_misaligned_chunk() { + let Some(f) = generate() else { return }; + let path = f.path("earliest.h5"); + let mut data = std::fs::read(&path).unwrap(); + // The leaf chunk B-tree of /grp/gz (10 deflated chunks of 100 i32, so + // each stored in under 400 bytes): "TREE" type(1)=1 level(1)=0 + // entries(2) left(8) right(8), then key, child, key, child, ... where a + // rank-1 chunk key is size(4) mask(4) offsets(2 x 8) and a child is 8. + // Key 1 starts at 24 + 24 + 8 = 56; its dimension-0 offset (100) at 64. + let mut patched = false; + for start in 0..data.len() - 72 { + if &data[start..start + 4] == b"TREE" && data[start + 4] == 1 && data[start + 5] == 0 { + let entries = u16::from_le_bytes(data[start + 6..start + 8].try_into().unwrap()); + let size0 = u32::from_le_bytes(data[start + 24..start + 28].try_into().unwrap()); + let off1 = u64::from_le_bytes(data[start + 64..start + 72].try_into().unwrap()); + if entries == 10 && size0 < 400 && off1 == 100 { + data[start + 64..start + 72].copy_from_slice(&103u64.to_le_bytes()); + patched = true; + break; + } + } + } + assert!(patched, "no chunk B-tree found for /grp/gz"); + let bad = f.path("misaligned.h5"); + std::fs::write(&bad, &data).unwrap(); + let o = h5rs(&["check", &bad.to_string_lossy()]); + let s = stdout(&o); + assert_eq!(code(&o), 1, "{s}"); + assert!( + s.contains("/grp/gz: chunk at [103, 0] offset 103 in dimension 0 is not a multiple of the chunk size 100"), + "{s}" + ); +} + +#[test] +fn every_subcommand_rejects_a_non_hdf5_file_cleanly() { + let dir = tempfile::tempdir().unwrap(); + let p = dir.path().join("junk.h5"); + std::fs::write(&p, b"this is not an HDF5 file at all").unwrap(); + let p = p.to_string_lossy().into_owned(); + for args in [ + vec!["ls", p.as_str()], + vec!["dump", p.as_str()], + vec!["stat", p.as_str()], + vec!["diff", p.as_str(), p.as_str()], + ] { + let o = h5rs(&args); + assert_eq!(code(&o), 2, "{args:?}"); + assert!(String::from_utf8_lossy(&o.stderr).contains("not an HDF5 file")); + } + let o = h5rs(&["check", &p]); + assert_eq!(code(&o), 1); + assert!(stdout(&o).contains("no HDF5 signature")); + assert_eq!(code(&h5rs(&["nonsense"])), 2); + assert_eq!(code(&h5rs(&["ls"])), 2); + assert_eq!(code(&h5rs(&["--help"])), 0); +} diff --git a/docs/known-issues.md b/docs/known-issues.md index 2e9227e..13ebc32 100644 --- a/docs/known-issues.md +++ b/docs/known-issues.md @@ -185,6 +185,8 @@ fill-value item that did is fixed). its element count times its base size. - (`cve-2024-32616` `/group1/dset3` and `cve-2025-2309`'s `Comp_OBJREF` attribute are h5py/numpy type-mapping failures, not libhdf5 refusals.) + - `h5rs check` validates with the library's parsers, so it inherits what + they accept: H5RS_CHECK_CVE_COUNTS. - **Writer:** - Nested groups beyond one level: path-like names are now refused, not created. diff --git a/scripts/ci-test.sh b/scripts/ci-test.sh index dedb2b4..df9ecdd 100755 --- a/scripts/ci-test.sh +++ b/scripts/ci-test.sh @@ -111,7 +111,8 @@ run_step "cargo clippy (fast-deflate / zlib-ng)" cargo clippy \ no_c_in_default_build() { local crate found=0 for crate in clawhdf5-format clawhdf5-io clawhdf5-filters clawhdf5 \ - clawhdf5-agent clawhdf5-ann clawhdf5-accel clawhdf5-netcdf4 clawhdf5-cli; do + clawhdf5-agent clawhdf5-ann clawhdf5-accel clawhdf5-netcdf4 clawhdf5-cli \ + clawhdf5-tools; do local c_deps c_deps=$(cargo tree -q -p "$crate" -e normal,build --prefix none \ | grep -E '^([a-z0-9_-]+-sys|cc|cmake) v' | sort -u) diff --git a/scripts/h5rs-check-ok-files.sh b/scripts/h5rs-check-ok-files.sh new file mode 100755 index 0000000..eee3fa4 --- /dev/null +++ b/scripts/h5rs-check-ok-files.sh @@ -0,0 +1,79 @@ +#!/usr/bin/env bash +# scripts/h5rs-check-ok-files.sh — run `h5rs check` over the files the +# conformance sweep reads correctly and completely: the `ok_files` of +# conformance/baseline.json (clawhdf5 and h5py agree on every object) whose +# probe results record no error at all (some ok files are deliberately +# corrupt test files that both readers refuse in the same way; those are +# left out). A validator that flags a file libhdf5 and clawhdf5 both read +# in full has a false positive — unless the file really is damaged in a way +# readers tolerate, which the report must then show. +# +# Usage: scripts/h5rs-check-ok-files.sh [--data] [SAMPLE] +# --data pass --data to check (decode every chunk) +# SAMPLE check only every SAMPLE-th file (default 1: all of them) +# +# Environment: H5RS (binary; default builds release), CONFORMANCE_CACHE +# (corpus cache, default conformance/.cache), CONFORMANCE_OUT (the sweep's +# results, default $CONFORMANCE_CACHE/results; run conformance/run.sh first), +# TMO (timeout, default 60). +# +# Exit status: 0 = no file flagged; 1 = some file flagged or failed (listed); +# 2 = setup error. +set -uo pipefail +HERE="$(cd "$(dirname "$0")" && pwd)" +ROOT="$(cd "$HERE/.." && pwd)" +export PATH="$HOME/.cargo/bin:$PATH" +DATA=() +if [ "${1:-}" = "--data" ]; then DATA=(--data); shift; fi +SAMPLE="${1:-1}" +CACHE="${CONFORMANCE_CACHE:-$ROOT/conformance/.cache}" +C="$CACHE/corpus" +R="${CONFORMANCE_OUT:-$CACHE/results}" +[ -d "$C" ] || { echo "error: no corpus in $C (run conformance/fetch-corpus.sh)" >&2; exit 2; } +[ -d "$R/runs" ] || { echo "error: no sweep results in $R (run conformance/run.sh)" >&2; exit 2; } +if [ -z "${H5RS:-}" ]; then + cargo build -q --release -p clawhdf5-tools --manifest-path "$ROOT/Cargo.toml" || exit 2 + H5RS="${CARGO_TARGET_DIR:-$ROOT/target}/release/h5rs" +fi +TMO="${TMO:-60}" + +mapfile -t FILES < <(python3 -c ' +import json, os, sys +b = json.load(open(sys.argv[1])) +runs, step = sys.argv[3], int(sys.argv[2]) +def clean(f): + """Both readers read every object, attribute and group listing.""" + for side in ("ours.json", "ref.json"): + try: + d = json.load(open(os.path.join(runs, f.replace("/", "__"), side))) + except (OSError, ValueError): + return False + if "open_error" in d or d.get("truncated"): + return False + for o in d.get("objects", []): + if "error" in o or "attrs_error" in o or "list_error" in o: + return False + if any(isinstance(a, dict) and "error" in a for a in o.get("attrs", {}).values()): + return False + return True +files = [f for f in b["ok_files"] if clean(f)] +total = len(b["ok_files"]) +print(f"{len(files)} of {total} ok files are read in full by both readers", file=sys.stderr) +for i, f in enumerate(files): + if i % step == 0: + print(f) +' "$ROOT/conformance/baseline.json" "$SAMPLE" "$R/runs") + +flagged=0 checked=0 +for f in "${FILES[@]}"; do + checked=$((checked + 1)) + out=$(timeout "$TMO" "$H5RS" check -q "${DATA[@]}" "$C/$f" 2>&1) + rc=$? + if [ $rc -ne 0 ]; then + flagged=$((flagged + 1)) + echo "== rc=$rc $f" + echo "$out" | head -5 | sed 's/^/ /' + fi +done +echo "== checked $checked ok files: $flagged flagged" +[ $flagged -eq 0 ] diff --git a/scripts/h5rs-fuzz.sh b/scripts/h5rs-fuzz.sh new file mode 100755 index 0000000..0edf97e --- /dev/null +++ b/scripts/h5rs-fuzz.sh @@ -0,0 +1,115 @@ +#!/usr/bin/env bash +# scripts/h5rs-fuzz.sh — run every `h5rs` subcommand over every file in a +# corpus of hostile HDF5 files, each under a timeout and a memory limit, and +# fail on any panic, crash or hang. +# +# Usage: scripts/h5rs-fuzz.sh [CORPUS_DIR ...] +# default corpus: conformance/.cache/corpus/cve_hdf5 (fetch it with +# conformance/fetch-corpus.sh; the HDF Group's CVE reproducers) +# +# Environment: +# H5RS h5rs binary to test (default: a debug build, for overflow checks) +# TMO per-run timeout in seconds (default 60) +# MEM_KB per-run address-space limit in KiB (default 4 GiB) +# JOBS files in parallel (default nproc) +# MUTATE also run on N byte-flipped copies of each file (default 0) +# MAX_BYTES --max-bytes for the commands that read values (default 16 MiB) +# +# A run may exit 0 (fine), 1 (problems found / could not read something) or +# 2 (error). Anything else fails the sweep: 3 is a caught panic (h5rs prints +# "internal error"), 124 a timeout, 128+N a signal (crash, abort, OOM kill). +# +# Exit status: 0 = every run ended cleanly; 1 = at least one did not (listed); +# 2 = setup error. +set -uo pipefail +HERE="$(cd "$(dirname "$0")" && pwd)" +ROOT="$(cd "$HERE/.." && pwd)" +export PATH="$HOME/.cargo/bin:$PATH" + +CORPORA=("$@") +[ ${#CORPORA[@]} -eq 0 ] && CORPORA=("$ROOT/conformance/.cache/corpus/cve_hdf5") +for c in "${CORPORA[@]}"; do + [ -d "$c" ] || { echo "error: no corpus at $c (run conformance/fetch-corpus.sh)" >&2; exit 2; } +done + +if [ -z "${H5RS:-}" ]; then + # A debug build: overflow checks turn silent wraparound on hostile sizes + # into a caught panic this sweep reports. + echo "== building h5rs (debug, with overflow checks)" + cargo build -q -p clawhdf5-tools --manifest-path "$ROOT/Cargo.toml" || exit 2 + TD="${CARGO_TARGET_DIR:-$ROOT/target}" + H5RS="$TD/debug/h5rs" +fi +[ -x "$H5RS" ] || { echo "error: $H5RS is not executable" >&2; exit 2; } +export H5RS TMO="${TMO:-60}" MEM_KB="${MEM_KB:-4194304}" MUTATE="${MUTATE:-0}" +export MAX_BYTES="${MAX_BYTES:-16777216}" +JOBS="${JOBS:-$(nproc 2>/dev/null || echo 4)}" + +WORK="$(mktemp -d)" +trap 'rm -rf "$WORK"' EXIT +export WORK + +# Every file in the corpora, whatever its extension (the reproducers often +# have none), except the corpus's own scripts and docs. +find -L "${CORPORA[@]}" -type f ! -name '*.md' ! -name '*.yml' ! -name '*.sh' \ + ! -name '*.py' ! -name 'COPYING' ! -name '*.gif' | sort > "$WORK/files.txt" +N=$(wc -l < "$WORK/files.txt") +[ "$N" -gt 0 ] || { echo "error: no files in ${CORPORA[*]}" >&2; exit 2; } +echo "== $N files, $JOBS at a time (timeout ${TMO}s, limit $((MEM_KB / 1024)) MiB, $MUTATE mutations each)" + +one() { + local f="$1" out="$WORK/fail.$$.$RANDOM" + local inputs=("$f") + if [ "$MUTATE" -gt 0 ]; then + local i + for ((i = 0; i < MUTATE; i++)); do + local m="$WORK/mut.$$.$i" + python3 - "$f" "$m" "$i" <<'PY' || continue +import random, sys +src, dst, seed = sys.argv[1], sys.argv[2], int(sys.argv[3]) +data = bytearray(open(src, "rb").read()) +if not data: + sys.exit(1) +rng = random.Random(f"{src}:{seed}") +for _ in range(rng.randint(1, 8)): + data[rng.randrange(len(data))] ^= 1 << rng.randrange(8) +open(dst, "wb").write(data) +PY + inputs+=("$m") + done + fi + local x + for x in "${inputs[@]}"; do + local cmd + # --max-bytes bounds the work per run: a valid 1 GiB dataset (the libhdf5 + # test files have some) is not what this sweep is looking for. + local mb="--max-bytes $MAX_BYTES" + for cmd in "ls -r -v $mb" "dump $mb" "dump --json $mb" "dump -p -A" "stat" "check --data $mb" "diff SELF"; do + local argv + read -r -a argv <<< "$cmd" + if [ "${argv[0]}" = diff ]; then argv=(diff --max-bytes "$MAX_BYTES" "$x"); fi + ( ulimit -v "$MEM_KB"; exec timeout -k 2 "$TMO" "$H5RS" "${argv[@]}" "$x" ) \ + >/dev/null 2>"$out.err" + local rc=$? + if [ $rc -gt 2 ]; then + { + echo "rc=$rc: h5rs ${argv[*]} $x" + [ "$x" != "$f" ] && echo " (a mutation of $f)" + grep -m3 'internal error' "$out.err" | sed 's/^/ /' + } >> "$WORK/failures.txt.$$" + fi + done + [ "$x" != "$f" ] && rm -f "$x" + done + rm -f "$out.err" +} +export -f one +xargs -a "$WORK/files.txt" -d '\n' -P "$JOBS" -I{} bash -c 'one "$1"' _ {} + +cat "$WORK"/failures.txt.* > "$WORK/failures.txt" 2>/dev/null || true +if [ -s "$WORK/failures.txt" ]; then + echo "== FAILED: $(grep -c '^rc=' "$WORK/failures.txt") run(s) panicked, crashed or hung:" + cat "$WORK/failures.txt" + exit 1 +fi +echo "== ok: every subcommand ended cleanly on all $N files"