diff --git a/.gitea/workflows/ci.yml b/.gitea/workflows/ci.yml index 1a7cdde..cfd9624 100644 --- a/.gitea/workflows/ci.yml +++ b/.gitea/workflows/ci.yml @@ -22,6 +22,9 @@ jobs: run: rustup component add rustfmt clippy - name: Install thumbv7em-none-eabihf target run: rustup target add thumbv7em-none-eabihf + - name: Install wasm32-unknown-unknown target + # ci-test.sh builds the reader and clawhdf5-wasm for the browser. + run: rustup target add wasm32-unknown-unknown - name: Install Python interop dependencies # The interop suites used to skip silently when python3/h5py were # missing, so they never ran in CI. Install them and make a missing @@ -31,12 +34,18 @@ jobs: # cmake builds libz-ng-sys for the opt-in `fast-deflate` (zlib-ng) # steps in ci-test.sh; rust:latest does not ship it. The default # build (pure-Rust zlib-rs) does not need it. - apt-get install -y --no-install-recommends python3 python3-venv cmake + # hdf5-tools: h5ls/h5stat/h5dump/h5diff, which the h5rs + # (clawhdf5-tools) interop tests compare against. + apt-get install -y --no-install-recommends python3 python3-venv cmake hdf5-tools python3 -m venv /opt/interop /opt/interop/bin/pip install --no-cache-dir h5py numpy netCDF4 xarray hdf5plugin echo "/opt/interop/bin" >> "$GITHUB_PATH" - name: Show interop library versions - run: /opt/interop/bin/python -c "import h5py, netCDF4, hdf5plugin; print('h5py', h5py.__version__, 'HDF5', h5py.version.hdf5_version, 'netCDF4', netCDF4.__version__, 'hdf5plugin', hdf5plugin.version)" + # h5dump's version too: the h5rs dump test requires its exact output + # (checked against Debian's 1.14.5 in rust:latest and 1.14.6). + run: | + /opt/interop/bin/python -c "import h5py, netCDF4, hdf5plugin; print('h5py', h5py.__version__, 'HDF5', h5py.version.hdf5_version, 'netCDF4', netCDF4.__version__, 'hdf5plugin', hdf5plugin.version)" + h5dump --version - name: Run CI script env: # Name the interpreter outright rather than relying on $GITHUB_PATH diff --git a/BENCHMARKS.md b/BENCHMARKS.md index f6e6375..c105698 100644 --- a/BENCHMARKS.md +++ b/BENCHMARKS.md @@ -482,6 +482,131 @@ The rows and columns of the uncompressed layouts are within 20% (chunked column 0.45 -> 0.49 ms, contiguous column 2.55 -> 2.61 ms). This run does not explain the slower windows. +## Concurrent reads + +### Results (2026-09-26, tank) + +Measured on tank (AMD Ryzen 7 7800X3D, 8 cores / 16 threads, 61 GiB, Linux +7.0) at commit `91644d8`, load average 1.84 when the run started (the +1-minute figure rose to 3.7 during the runs; that is mostly the benchmark's +own threads). Warm page cache. clawhdf5 2.7.0 (workspace), h5py 3.16.0 on +HDF5 2.0.0. Commands exactly as in the **Run** box below; files at their +defaults (64 datasets of 16384 x 1024 `f32`, 64 MiB each; deflate chunks +256 x 256, level 4). MB/s is decoded data, the median of the repetitions; +eff is scaling efficiency against the same tool's 1-thread row. + +Each read decoding on its calling thread (`--decode-threads 1`, like h5py): + +| layout | mode | threads | clawhdf5 MB/s (eff) | h5py threads MB/s (eff) | h5py processes MB/s (eff) | +|---|---|---:|---:|---:|---:| +| deflate | distinct | 1 | 421 (1.00) | 433 (1.00) | 421 (1.00) | +| deflate | distinct | 4 | 890 (0.53) | 428 (0.25) | 1651 (0.98) | +| deflate | distinct | 16 | 880 (0.13) | 427 (0.06) | 4424 (0.66) | +| deflate | same | 1 | 151 (1.00) | 130 (1.00) | 129 (1.00) | +| deflate | same | 4 | 490 (0.81) | 129 (0.25) | 497 (0.96) | +| deflate | same | 16 | 1244 (0.52) | 128 (0.06) | 1402 (0.68) | +| contiguous | distinct | 1 | 2495 (1.00) | 9789 (1.00) | 9169 (1.00) | +| contiguous | distinct | 16 | 8083 (0.20) | 8096 (0.05) | 12272 (0.08) | +| contiguous | same | 1 | 624 (1.00) | 5022 (1.00) | 5172 (1.00) | +| contiguous | same | 16 | 4778 (0.48) | 4411 (0.05) | 37138 (0.45) | + +With the default rayon pool decoding inside each read, deflate `distinct` +is 912 MB/s at 1 thread (2.1x h5py) and 2824 MB/s at 16 (6.6x h5py threads, +0.64x h5py processes); the other rows are within a few percent of the table +above. Full tables (2, 4, 8 threads, both decode modes) come from +`compare_concurrent_read.py` on the JSON files. + +What this shows: +- **h5py threads do not scale** (flat at about 430 MB/s on deflate, every + thread count): libhdf5's global lock. +- **clawhdf5 threads on one `File` do, for hyperslab reads of compressed + data:** 1244 MB/s at 16 threads, 9.7x h5py threads and 0.89x h5py + processes, without a process pool. +- **Where clawhdf5 is behind** (open performance bugs, see + `docs/known-issues.md`): + - *Full reads of chunked datasets stop scaling at about 4 threads* + (about 880 MB/s) while h5py processes reach 4424 MB/s. Hyperslab + reads, which bypass the `File`'s chunk cache, keep scaling, so the + cache (one mutex and one 16 MiB budget per `File`, thrashed by 64 MiB + datasets) is the suspect. + - *Contiguous reads are slow*: 2.5 GB/s for a single-threaded full read + against h5py's 9.8 GB/s (0.25x), and 0.12x for 256 x 256 hyperslabs. + Threads close the gap (about 1.0x h5py at 16), but single-thread + contiguous I/O is a real deficit. + +The question: libhdf5's threadsafe build serialises every API call under one +global mutex, and h5py holds a global lock around every call too, so threads +reading through h5py cannot decode in parallel; h5py users scale with +processes. A clawhdf5 `File` is `Send + Sync`, and nothing on the read paths +this harness uses (`read_f32`, `read_f32_selection`) takes a library-wide +lock: the one mutex is the `File`'s chunk cache (keyed per dataset), taken by +full reads of chunked datasets for each chunk's O(1) lookup and insert, never +across a decode; hyperslab reads do not use the cache. How does +decoded throughput scale with threads on one open file, against h5py threads +and h5py processes on the same files? + +Workload (`crates/clawhdf5-bench/src/bin/concurrent_read.rs`; the h5py script +mirrors it): `/deflate.h5` and `/contiguous.h5`, each with 64 `f32` +datasets of 64 MiB decoded (`[16384, 1024]`; the deflate file chunked +`256 x 256`, level 4), written by clawhdf5 on first use and reused while +`manifest.json` matches. The data is a slowly varying ramp plus 8 bits of +noise per element, every value exact in `f32`, so both harnesses check what +they read; it deflates about 3.1x (128 MiB -> 40.7 MiB for two 64 MiB +datasets). For each layout and thread count +(1, 2, 4, 8, 16; fixed total work per repetition, split among the threads): + +- `distinct`: every dataset read in full once, thread `t` taking datasets + `t, t + T, ...`; +- `same`: 1024 random `256 x 256` hyperslabs of `d00` in total, from a seeded + splitmix64 stream that both harnesses generate identically. + +Reported per row: MB/s of decoded (selected) data from the median of the +repetitions, and scaling efficiency `MB/s(T) / (T x MB/s(1))`. Each worker +times itself from a start barrier; a repetition spans the earliest start to +the latest finish. Page cache: warm by default (each file is read once before +timing); `--cold` evicts the files with `posix_fadvise(POSIX_FADV_DONTNEED)` +before every repetition (no root needed; best effort). clawhdf5 opens one +`File` per repetition, shared by all threads; h5py threads share one +`h5py.File`; h5py processes (spawned before timing) each open the file inside +the timed region. + +Decode inside a single clawhdf5 read is itself parallel in this binary +(clawhdf5-format's `parallel` feature, enabled here through clawhdf5-agent; +it is off in the facade's default features), so a 1-thread clawhdf5 full read +of the deflate file already uses the whole rayon pool. Run both +`--decode-threads 1` (each read decodes on its calling thread, like h5py — +this isolates the API's own scaling) and the default pool. + +> **Run** (from the repository root). The default files take about 5.4 GiB +> of disk (4 GiB contiguous + about 1.3 GiB deflate). Generating them is +> memory-hungry because `FileBuilder` holds a whole file in memory: peak RSS +> was 676 MB for `--datasets 2 --mib 64` (2026-09-25, tank, +> `/usr/bin/time -f %M`), about 5x one file's decoded size, so expect about +> 21 GB at the defaults (once; later runs reuse the files). Put `--dir` on a +> real disk, not tmpfs, if `--cold` is to mean anything. +> +> ```bash +> DIR=/path/on/disk/concurrent-read +> BENCH=crates/clawhdf5-bench/scripts +> PY=.venv/bin/python # h5py 3.16 / HDF5 2.0 in this repo +> cargo build --release -p clawhdf5-bench --bin concurrent_read +> B=target/release/concurrent_read +> $B --dir $DIR --json claw-pool.json # generates on first run +> $B --dir $DIR --decode-threads 1 --json claw-1.json +> $PY $BENCH/concurrent_read_h5py.py --dir $DIR --executor threads --json h5py-threads.json +> $PY $BENCH/concurrent_read_h5py.py --dir $DIR --executor processes --json h5py-procs.json +> $PY $BENCH/compare_concurrent_read.py claw-1.json h5py-threads.json h5py-procs.json +> $PY $BENCH/compare_concurrent_read.py claw-pool.json h5py-threads.json h5py-procs.json +> ``` +> +> Cold page cache: add `--cold` to every harness command. Smoke test (seconds): +> `$B --dir /tmp/cr --datasets 4 --mib 1 --threads 1,2,4 --slabs 16 --reps 1` +> and the same `--threads/--slabs/--reps` to the h5py script. + +Other flags (both harnesses): `--threads`, `--reps`, `--slab`, `--slabs`, +`--seed`, `--modes distinct,same`, `--layouts deflate,contiguous`; sizes +(`--datasets`, `--mib`) only on the Rust harness, which writes the files. + ## Search harness baseline (v2.3.0) Produced by `cargo run --release -p clawhdf5-bench --bin search_harness -- --full` diff --git a/CHANGELOG.md b/CHANGELOG.md index 1fc429d..ef4bd2f 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,57 @@ ## Unreleased +### Plugin filters (2026-09-26) +- **LZF, bitshuffle, bzip2 and Blosc read and write, in pure Rust.** Files + written by h5py with `compression="lzf"`, or with hdf5plugin's + `Bitshuffle`, `BZip2` and `Blosc`, failed with `UnsupportedFilter`. New + `clawhdf5-format`/`clawhdf5` features: `lzf` (32000, **on by default**, no + dependencies), `bitshuffle` (32008: transpose only, LZ4 and Zstandard + modes), `bzip2` (307), `blosc` (32001: Blosc 1 frames with BloscLZ, + LZ4/LZ4HC, Snappy, Zlib and Zstandard codecs and byte/bit shuffle; + BloscLZ is decoded by a port of c-blosc 1.21's decoder, and cannot be + written), and `plugin-filters` for all four. None compiles C: Zstandard is + ruzstd, bzip2 is libbz2-rs-sys. Write with `DatasetBuilder::with_lzf()`, + `with_bitshuffle(..)`, `with_bzip2(..)`, `with_blosc(..)` or + `with_plugin_filter(PluginFilter::..)`; `ChunkOptions` gains a `plugin` + field (**breaking** for code that builds `ChunkOptions` with a struct + literal and no `..Default::default()`). Tested both ways against h5py 3.16 + + hdf5plugin 7.1 over 1-3-D shapes with partial edge chunks, 1-8-byte + types in both byte orders and incompressible data + (`crates/clawhdf5/tests/plugin_filters_interop.rs`). Conformance: 573 of + 697 files ok (was 569) — h5ex_d_lzf/bshuf/bzip2/blosc. +- **Filter registry.** Filters are looked up by ID in + `clawhdf5_format::filter_registry` instead of a `match`: the built-in + table (per build), then codecs registered at run time with + `register_filter(id, codec)` — a decoding closure or a `FilterCodec` that + can also encode. Built-in IDs cannot be overridden; a registered decoder's + output is held to the chunk-size bound. Unknown IDs still fail with + `UnsupportedFilter(id)`, whose message now names known filters and the + missing feature ("unsupported filter: 32026 (Blosc2, not implemented by + clawhdf5)"). +- **Not implemented:** Blosc2 (32026) and ZFP (32013) remain a clear error. +- **Wrong data: a chunk that decodes short read as zeros** (pre-existing, every + filter). HDF5 stores every chunk at the full chunk size, so a filter + pipeline that decodes to fewer bytes means a corrupt chunk; every chunk + reader (full, cached, selection, parallel, partial) padded it with zeros. + It is now an error naming the chunk ("chunk at [16] decoded to 16 bytes, + expected 32"), via the new `filters::decompress_chunk_exact`. libhdf5 + returns the rest of such a chunk uninitialised, or fails when the filter + checks. A Blosc frame declaring no data for a non-empty chunk is an error + too. Legitimate edge chunks are unaffected (they are stored full-size, + filtered or not); conformance is unchanged at 573 of 697, with no file + changing class. +- **Crash: a hostile Blosc chunk panicked** in builds with overflow checks + (debug builds, `cargo test`, `maturin develop`): a frame size below the + 16-byte header underflowed. It is now an error. Every new decoder (LZF, + bitshuffle, bzip2, Blosc/BloscLZ) is fuzzed with random and mutated frames + in the unit tests. +- **`register_filter(32023, ..)` works with the `pcodec` feature.** 32023 is + Granular BitRound's ID; the built-in entry there only reads clawhdf5 + <= 2.7.0's pcodec chunks (filter name `"pcodec"`), so a registered codec now + handles every other chunk with that ID, and writes. It was refused as + "built in". + ### Upgrade Notes - **HDF5 correctness audit (2026-09-25).** A sweep of 686 public files (the libhdf5 test files, the HDF Group's CVE reproducers, pyfive, netcdf-c, @@ -112,6 +163,65 @@ `quantized_index = false`, or pass `create --f32-index` to the CLI, to opt out. The CLI's `--quantized-index` is still accepted but is now a no-op. +### Tools +- New crate **`clawhdf5-tools`** with the binary **`h5rs`**: HDF5 + command-line tools without libhdf5, built only on the `clawhdf5` facade + and `clawhdf5-format` (no C, so it also builds as a static musl binary). + - `h5rs ls [-r] [-v] FILE[/path]` lists objects like h5ls (its first two + columns are h5ls's text on the test files) plus the datatype; `-v` adds + address, link count, layout and chunk index, chunk size, storage, + filters, datatype and attributes. + - `h5rs dump [--json] [-A] [-p] [-d PATH] FILE` prints DDL text that is + byte-identical to h5dump 1.14.6's (and to Debian's 1.14.5, which CI + uses) on the test files (all layouts and + chunk indexes, v1/v2 groups, compound, enum, strings, links, named + types, attributes; null-padded strings show their NULs at any depth), + or JSON in the HDF Group's hdf5-json layout (schema in the crate + README). Nested compounds print inline and `long double` values as + errors (exit 1); both are listed in the README. + - `h5rs stat FILE` reports h5stat's object, link, rank, layout, filter, + attribute, raw-data and file-size figures (equal to h5stat's on the test + files); metadata space is one figure, not broken down. + - `h5rs diff [-r] [-q] [-n N] [-d D] [-p R] [--follow-symlinks] A B [OBJ1 + [OBJ2]]` (option names as h5diff's: `-c` is `--compare`, the count is + `-n`/`--count=N`) compares objects, kinds, datatypes, shapes, attributes, values and link + targets; exit status 0/1/2 as h5diff's. Soft links are compared by + target path, as h5diff's default, or with `--follow-symlinks` by the + objects they lead to (external links are never followed). Every path is + compared, including every name of a hard-linked object and the members + of a hard-linked group; with a `-d`/`-p` tolerance, integers are + compared exactly in integer arithmetic (no loss above 2^53), and a `-p` + below the f64 epsilon compares exactly, as h5diff's. Objects that cannot + be compared count as a difference (h5diff exits 0 for them), and NaN + equals NaN. + - `h5rs check [--data] FILE` is a structural validator: it walks every + object, parses every header message, verifies the checksums of every + version 2+ structure it meets (superblock, object headers and + continuation chunks, v2 B-tree nodes, fractal heap headers and — which + the library's reads do not — every direct and indirect heap block, and + extensible/fixed array chunk indexes), checks each chunk index against + its dataset (aligned, in-extent, unique, plausibly sized chunks), and + that raw data lies inside the file without overlaps. Every problem is + printed with its address; exit 1 when there are any. libhdf5's h5check + reads only the 1.8 format. On the conformance corpus it passes all 418 + files that both clawhdf5 and h5py read in full, and `check --data` flags + 134 of the 150 CVE and fuzzer files of the `cve_hdf5` corpus (tank, + 2026-09-26). `--data` also follows variable-length data into its global + heap collections and reports a damaged one at its address. It inherits + the library's tolerance, though: 9 of the 16 it passes are files h5dump + 1.14.6 rejects (see `docs/known-issues.md`, header checks). + - Values over `--max-bytes` (default 1 GiB) are reported instead of read; + a panic is caught and reported as an internal error (exit 3). + `scripts/h5rs-fuzz.sh` runs every subcommand over a corpus (default the + CVE reproducers, optionally with byte-flipped copies) with overflow + checks, a timeout and a memory limit, and fails on any panic, crash or + hang; `scripts/h5rs-check-ok-files.sh` runs `check --data` over the + fully-read conformance files. + - Because the library does not verify fractal heap block checksums when + it reads a dense group's links or dense attributes, `h5rs` verifies a + heap's blocks before reading from it and refuses a damaged one, as + libhdf5 does, instead of printing what the damaged block holds. + ### Signing - `clawhdf5-agent`: **Ed25519-signed checkpoints** — the README's "cryptographically verifiable memory", now true. With @@ -195,8 +305,22 @@ README claimed but nothing measured. - `footprint_bench` reports whether it built `float16` or `f32` stores and takes `--f32`; it had kept printing "f32" after the default changed. +- New `concurrent_read` harness, with an h5py counterpart + (`crates/clawhdf5-bench/scripts/concurrent_read_h5py.py`, threads or + processes) and `compare_concurrent_read.py`: decoded read throughput and + scaling efficiency at 1-16 threads on one open file, full reads of distinct + datasets and random hyperslabs of one dataset, deflate and contiguous, warm + or `--cold` page cache, JSON output. Not yet measured — `BENCHMARKS.md` + ("Concurrent reads") has the commands and no numbers. ### Interop +- **h5py could not open chunked datasets we wrote with a chunk dimension + from 65 536 to 16 777 215.** A version-4 layout must store its chunk + dimensions in the fewest bytes that hold the largest (3 for 70 000); + the writer rounded 3 up to 4, and HDF5 2.0.0 (h5py 3.16) refuses that + ("stored chunk dimension encoding length does not match value calculated + from chunk dimensions"). Newer libhdf5 and clawhdf5 read those files; new + files use the exact width. Test: `we_write_chunk_dimensions_in_the_fewest_bytes`. - **Conformance sweep in the repo** (`conformance/`, report in `CONFORMANCE.md`). `conformance/run.sh` fetches eight public HDF5 corpora pinned by commit (libhdf5's test files, the HDF Group's CVE reproducers, @@ -280,6 +404,32 @@ infinity; batches are all or nothing. CLI: `create --float16`. See `BENCHMARKS.md`, "float16 embedding storage". +### Browser (WebAssembly) +- **New crate `clawhdf5-wasm`:** the reader compiled to + `wasm32-unknown-unknown` with a wasm-bindgen JavaScript API — + `open(bytes)`, `list`, `info`, `attrs`, `read`, `readHyperslab` — returning + typed arrays of the stored width (`BigInt64Array` for 64-bit integers), + string arrays for strings and enums, and a thrown `Error` for types with no + typed-array form (compound, reference, opaque, VL sequences) or filters the + build lacks (Zstd, SZIP). Read-only; the file is held in memory. +- **`examples/wasm-viewer/`:** a drop-a-file HDF5/NetCDF-4 viewer page (tree, + type/shape/attributes, values paged as hyperslabs; `?file=&path=` opens a + URL). `build.sh` produces the package; `test/run.sh` checks it under Node + (251 checks against values h5py/libhdf5 read back from an h5py- and a + netCDF4-written file) and renders the page in headless Chromium. Size, + measured 2026-09-26 on tank (`gzip -9 -n`): 627,501 B of wasm, 191,639 B + gzipped, plus 21,826 B (4,487 B) of JS glue; h5wasm 0.10.3's embedded wasm + is 3,544,184 B (907,096 B) — full libhdf5, so not equal functionality. See + `examples/wasm-viewer/README.md`. +- The facade's read path already built for `wasm32-unknown-unknown` (nothing + needed gating); `ci-test.sh` now builds it (`--no-default-features`) and + lints `clawhdf5-wasm` for that target, and CI installs the target. The Node + and browser tests run in `ci-test.sh` only where `node` and `wasm-bindgen` + exist (not the CI container); CI checks the same expectations natively + (`clawhdf5-wasm`'s `h5py_interop` test). +- `Dataset::raw_datatype()` (facade) returns the full stored datatype, for + decoding `read_selection` bytes with `clawhdf5_format::data_read`. + ### Build - **Pure-Rust default.** `clawhdf5-format`, `clawhdf5-filters` and the `clawhdf5` facade default to the `zlib-rs` deflate backend; `fast-deflate` @@ -296,6 +446,84 @@ - CI keeps zlib-ng building and tested; the arm64 job no longer needs cmake. ### Correctness +- **Corrupt files libhdf5 refuses are now refused instead of read.** On the + HDF Group's CVE reproducers, 18 objects that libhdf5 (HDF5 2.0, through + h5py) refuses to open were read by clawhdf5, some as wrong data (a chunk + dimension of 0 read as all fill values; chunks read at offsets off the + chunk grid). The + parser now makes libhdf5's checks, with libhdf5's error text: + - object headers (`FormatError::InvalidObjectHeader`): every message of a + v1 chunk is read and more than the prefix's count is refused (the rest + used to be dropped); v1 message sizes must be multiples of 8 and a v1 + chunk cannot end in a gap; a message running past its chunk is an error + (it used to end the chunk quietly); contradictory message flags; a + message of a class that cannot be shared flagged shareable; a + reference-count message in a v1 header; malformed continuation, + reference-count and modification-time messages; unknown v2 header + flags. + - datatypes (`FormatError::InvalidDatatype`): size 0; integer bits outside + the type; float exponent/mantissa outside the type, empty or + overlapping; a compound with no members, a member outside the compound, + a duplicate name or overlapping members; an enum whose size differs from + its base type's or with an empty name; array rank over 32 or a zero + dimension; an opaque tag length that is not a multiple of 8; in a + version-1 (unchecksummed) header, a numeric type that leaves more than + half its bits unused (`Datatype::parse_in_header`, + `Datatype::check_unused_bits`). A v1/v2 float's class bit 6 was read as + VAX byte order; libhdf5 ignores it before version 3, and so does this. + The overlap check measures each earlier member by its stored size, as + libhdf5 does, so a variable-length member (4 + offset size + 4 bytes) + in a file with 4-byte offsets does not overlap the member after it. + - chunked layouts (`FormatError::InvalidChunkDimensions`): a zero chunk + dimension, a chunk rank that does not match the dataspace, a chunk of + 4 GiB or more indexed by a v1 B-tree (layout version 3 or earlier; + 0x80000000-sized chunks hung the reader — layout versions 4 and 5 allow + larger chunks, and HDF5 2.0 writes them), an element size in the + layout that differs from the datatype's stored size (the chunks were + laid out with the wrong element size), and v1 B-tree + chunk keys whose offsets are not multiples of the chunk dimensions, + including the keys that only bound a node + (`chunked_read::collect_chunk_info_checked`). + - truncated files (`FormatError::TruncatedFile`, `Superblock::data_end`): + a file shorter than the end of file its superblock records is refused + ("truncated file"), and nothing past that end is read. Every reader + does this: `File`, `LazyFile` and `MmapFile`, and in `clawhdf5-io` + `NativeVol` (at `open`, and on read for `from_bytes`), + `AsyncHDF5File` and `MpiVol` (the MPI path is not built in CI: it + needs an MPI installation). + - the writer: `FileWriter::finish()` / `FileBuilder::finish()` refuse a + datatype the reader would refuse (`FormatError::SerializationError`, + "datatype cannot be written: ..."), such as a compound with a repeated + field name or no fields, or an enum member with an empty name + (`CompoundTypeBuilder` and `EnumTypeBuilder` build them without + complaint). These were never valid HDF5 — h5py refuses them — and + clawhdf5 wrote them, which made files it could not read back. + + Checks newer libhdf5 releases make but HDF5 2.0 does not (bit-field + offsets, the variable-length kind, array sizes) are left out, so files + h5py opens still open. Two libhdf5 checks are skipped on purpose because + clawhdf5 up to v2.7.0 wrote files that fail them without being wrong: + the sign bit of every float at position 63, and a size-0 string type for + an empty-string attribute (new fixtures written by v2.7.0 guard this). + Conformance: 569 -> 571 ok (h5stat_err_refcount.h5, + h5clear_fsm_persist_less.h5), and 17 of the 18 CVE objects now fail as in + libhdf5 (see `docs/known-issues.md` for the one left), as do 10 files + h5py refuses as truncated. Tests: + `header_validation_interop.rs` (h5py writes, the test damages a copy, both + libraries must refuse it), `legacy_writer_files.rs`, and unit tests next + to each check. **Breaking (format crate):** `FormatError` gained + `InvalidObjectHeader`, `InvalidDatatype`, `InvalidChunkDimensions` and + `TruncatedFile`; an exhaustive `match` on it needs the new arms. +- **Chunked datasets whose chunk dimensions take 3, 5, 6 or 7 bytes did not + open.** A version-4 layout (`libver="latest"`) stores each chunk dimension + in the fewest bytes that hold the largest one, so a chunk dimension from + 65 536 to 16 777 215 (e.g. h5py `chunks=(70000,)`) takes 3 bytes; only 1, 2, + 4 and 8 were read, and the rest failed with `UnexpectedEof`. Widths 1-8 are + read now, and 0 or more than 8 is refused as libhdf5 refuses it. A width + larger than needed is accepted: HDF5 2.0.0 refuses one ("stored chunk + dimension encoding length does not match"), but libhdf5 since + HDFGroup/hdf5@e124c36 (2026-06-05) reads it, and clawhdf5 itself wrote such + layouts. - `clawhdf5-format` VDS: variable-length and reference data from a source in another file is refused. Those elements are global-heap IDs and object addresses in the source file; copied into the virtual dataset they would diff --git a/CLAUDE.md b/CLAUDE.md index 3f8d9c1..1cb7aba 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -5,13 +5,13 @@ Pure-Rust HDF5 format implementation with HNSW vector search, WAL-backed persist ## Architecture -Cargo workspace with 16 crates under `crates/` (plus `libaec-sys`, an internal FFI bindings crate for the optional `szip` feature): +Cargo workspace with 18 crates under `crates/` (plus `libaec-sys`, an internal FFI bindings crate for the optional `szip` feature): | Crate | Role | |-------|------| | `clawhdf5-format` | HDF5 binary spec parser (superblock, B-tree, heap) — also holds shared type definitions and physical constants | | `clawhdf5-io` | Read/write implementation | -| `clawhdf5-filters` | Deflate backends (zlib-rs, zlib-ng, Apple Compression); the HDF5 filter pipeline and the other codecs (LZ4, Zstd, SZIP, N-Bit, scale-offset, pcodec) live in `clawhdf5-format`. No Blosc. | +| `clawhdf5-filters` | Deflate backends (zlib-rs, zlib-ng, Apple Compression); the HDF5 filter pipeline, the filter registry (`clawhdf5_format::filter_registry`) and the other codecs (LZ4, Zstd, SZIP, N-Bit, scale-offset, pcodec, and the pure-Rust plugin filters LZF, bitshuffle, bzip2, Blosc 1) live in `clawhdf5-format`. No Blosc2 or ZFP. | | `clawhdf5-derive` | Proc-macro derive for HDF5-serializable structs | | `clawhdf5` | Main facade crate | | `clawhdf5-netcdf4` | NetCDF-4 compatibility layer | @@ -21,9 +21,11 @@ Cargo workspace with 16 crates under `crates/` (plus `libaec-sys`, an internal F | `clawhdf5-accel` | CPU SIMD acceleration path | | `clawhdf5-migrate` | SQLite → HDF5 agent-memory migration | | `clawhdf5-android` | Android JNI bindings | -| `clawhdf5-cli` | Command-line interface | +| `clawhdf5-cli` | Command-line interface (agent memory) | +| `clawhdf5-tools` | `h5rs`: pure-Rust HDF5 tools — `ls`, `dump` (DDL / hdf5-json), `stat`, `diff`, `check` (structural + checksum validator) | | `clawhdf5-napi` | Node.js native addon bindings | | `clawhdf5-py` | PyO3 Python bindings | +| `clawhdf5-wasm` | WebAssembly (wasm-bindgen) reader for the browser; demo in `examples/wasm-viewer/` | | `clawhdf5-bench` | Benchmark suite | ## Key Features @@ -149,6 +151,14 @@ Cargo workspace with 16 crates under `crates/` (plus `libaec-sys`, an internal F `MemorySource` for this bookkeeping is inferred from the caller-supplied `source_channel` string (a heuristic, not an authenticated trust boundary). - GPU-accelerated vector distance computation (`clawhdf5-gpu`, wgpu); HDF5 I/O itself is CPU-only +- Browser: `clawhdf5-wasm` (wasm-bindgen, read-only, file held in memory; + no Zstd/SZIP since they link C) and the `examples/wasm-viewer/` page. + `examples/wasm-viewer/test/run.sh` builds the package (needs the + `wasm-bindgen` CLI at the crate's exact version) and tests it under Node + and headless Chromium (a Playwright download in `~/.cache/ms-playwright` + on tank); the CI container has neither, so CI runs the native + `clawhdf5-wasm` `h5py_interop` test on the same fixture. Size numbers are + in the example's README. - Python and Node.js bindings for cross-language use - NetCDF-4 compatibility for scientific data interop @@ -188,6 +198,16 @@ cargo run -p clawhdf5-cli -- --help # create, save, search, recall, stats, flush-wal, agents-md, export, snapshot subcommands ``` +### HDF5 tools (`h5rs`, crate `clawhdf5-tools`) +```bash +cargo run -p clawhdf5-tools -- ls -r file.h5 # also dump [--json], stat, diff, check +bash scripts/h5rs-fuzz.sh # every subcommand over the CVE corpus: no panic/crash/hang +bash scripts/h5rs-check-ok-files.sh --data # check passes every fully-read conformance file +``` +Its interop tests compare against h5ls/h5stat/h5dump/h5diff (Debian +`hdf5-tools`, installed in CI); `dump` must stay byte-identical to h5dump on +the test files. + ### Python bindings ```bash cd crates/clawhdf5-py diff --git a/CONFORMANCE.md b/CONFORMANCE.md index aca1b5d..fd6fbed 100644 --- a/CONFORMANCE.md +++ b/CONFORMANCE.md @@ -13,8 +13,8 @@ fatal. This file is generated by `conformance/run.sh`; do not edit it by hand. | | | |---|---| -| date | 2026-09-26 03:46 UTC | -| clawhdf5 commit | `10d1029ead524e2fe64c2cd7f61b28067d9e449c` | +| date | 2026-09-26 06:50 UTC | +| clawhdf5 commit | `72306c601399748616bc9d061be2ebc4c1bea9e0` | | machine | `tank`: AMD Ryzen 7 7800X3D 8-Core Processor, 16 CPUs, 61 GiB, Linux 7.0.0-34-generic x86_64 | | command | `conformance/run.sh --no-fetch --update-baseline` | | rustc | rustc 1.98.1 (48a229cea 2026-09-01) | @@ -38,14 +38,14 @@ A file's class is the first that applies: | NCAS-CMS_pyfive | 33 | 32 | 0 | 1 | 0 | 0 | 0 | 0 | 0 | | cve_hdf5 | 147 | 100 | 6 | 9 | 32 | 0 | 0 | 0 | 0 | | h5py_data | 4 | 4 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | -| hdf5 | 466 | 386 | 8 | 12 | 60 | 0 | 0 | 0 | 0 | +| hdf5 | 466 | 392 | 4 | 10 | 60 | 0 | 0 | 0 | 0 | | netcdf-c | 20 | 20 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | | netcdf4-python | 18 | 18 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | | usnistgov_h5wasm | 5 | 5 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | | xarray-data | 4 | 4 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | -| **all** | **697** | **569** | **14** | **22** | **92** | **0** | **0** | **0** | **0** | +| **all** | **697** | **575** | **10** | **20** | **92** | **0** | **0** | **0** | **0** | -2 of the 22 mismatches are a known h5py bug, not ours (see *Known not-our-bug*). +2 of the 20 mismatches are a known h5py bug, not ours (see *Known not-our-bug*). Corpora (fetched by `conformance/fetch-corpus.sh` into the gitignored `conformance/.cache/`): @@ -70,9 +70,9 @@ Grouped by normalised error message. *files* counts files whose class this cause | files | objects | error | examples | |---:|---:|---|---| -| 6 | 6 | `UnsupportedFilter(N)` | `hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_blosc.h5`, `hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_blosc2.h5`, `hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_bshuf.h5` (+3 more) | | 3 | 3 | `DataSizeMismatch { expected: N, actual: N }` | `cve_hdf5/cvefiles/cve-2020-18494.h5`, `cve_hdf5/cvefiles/cve-2024-32623.h5`, `cve_hdf5/cvefiles/cve-2025-2309.h5` | | 2 | 2 | `ChunkedReadError("…")` | `cve_hdf5/cvefiles/cve-2025-2308.h5`, `hdf5/test/testfiles/bad_nbit_parms_walk.h5` | +| 2 | 2 | `UnsupportedFilter(N)` | `hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_blosc2.h5`, `hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_zfp.h5` | | 1 | 1 | `UnexpectedEof { expected: N, available: N }` | `cve_hdf5/cvefiles/cve-2019-9151.h5` | | 1 | 1 | `MissingMessage(Dataspace)` | `cve_hdf5/cvefiles/cve-2024-33874.h5` | | 1 | 1 | `InvalidObjectHeaderVersion(N)` | `hdf5/tools/test/testfiles/h5clear_mdc_image.h5` | @@ -82,9 +82,9 @@ Grouped by normalised error message. *files* counts files whose class this cause | files | objects | cause | examples | |---:|---:|---|---| | 13 | 14 | `missing-object` | `cve_hdf5/cvefiles/cve-2019-8397.h5`, `cve_hdf5/cvefiles/cve-2019-8398.h5`, `cve_hdf5/cvefiles/cve-2021-46243.h5` (+10 more) | -| 3 | 7 | `extra-attr` | `cve_hdf5/cvefiles/cve-2018-17438`, `cve_hdf5/cvefiles/cve-2018-17439`, `cve_hdf5/cvefiles/cve-2024-33874.h5` | -| 3 | 6 | `extra-object` | `cve_hdf5/cvefiles/cve-2021-46244.h5`, `hdf5/tools/test/testfiles/h5clear_fsm_persist_less.h5`, `hdf5/tools/test/testfiles/h5stat_err_refcount.h5` | +| 2 | 6 | `extra-attr` | `cve_hdf5/cvefiles/cve-2018-17438`, `cve_hdf5/cvefiles/cve-2018-17439` | | 1 | 1 | `attr-values: ours=vlen(>u8) h5py=object layout=- filters=-` | `NCAS-CMS_pyfive/tests/data/attr_datatypes.hdf5` | +| 1 | 4 | `extra-object` | `cve_hdf5/cvefiles/cve-2021-46244.h5` | | 1 | 1 | `values: ours=i2 h5py=>i2 layout=chunked filters=[6]` | `cve_hdf5/cvefiles/cve-2025-44905.h5` | | 1 | 1 | `values: ours=>f4 h5py=>f4 layout=chunked filters=[2]` | `cve_hdf5/cvefiles/cve-2025-44905.h5` | @@ -102,7 +102,7 @@ columns are. | tool | read | error | panic | crash | hang | oom | |---|---:|---:|---:|---:|---:|---:| -| clawhdf5 | 142 | 5 | 0 | 0 | 0 | 0 | +| clawhdf5 | 140 | 7 | 0 | 0 | 0 | 0 | | h5dump 1.14.6 | 16 | 129 | 0 | 2 | 0 | 0 | | h5py 3.16.0 / HDF5 2.0.0 | 115 | 31 | 0 | 1 | 0 | 0 | @@ -112,19 +112,19 @@ columns are. |---|---|---|---|---| | cvefiles/cve-2016-4330.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok | | cvefiles/cve-2016-4331.h5 | error exit | read 25 obj, 1 errors | read 25 obj, 1 errors | ok | -| cvefiles/cve-2016-4332-mtime-new.h5 | error exit | read 25 obj, 1 errors | read 25 obj | ok | -| cvefiles/cve-2016-4332-mtime.h5 | error exit | read 4 obj, 3 errors | read 4 obj | ok | -| cvefiles/cve-2016-4332-stab.h5 | error exit | open error | read 65 obj | h5py-cannot-read | +| cvefiles/cve-2016-4332-mtime-new.h5 | error exit | read 25 obj, 1 errors | read 25 obj, 1 errors | ok | +| cvefiles/cve-2016-4332-mtime.h5 | error exit | read 4 obj, 3 errors | read 4 obj, 3 errors | ok | +| cvefiles/cve-2016-4332-stab.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read | | cvefiles/cve-2016-4333.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok | | cvefiles/cve-2017-17505.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok | | cvefiles/cve-2017-17506.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok | | cvefiles/cve-2017-17507.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok | -| cvefiles/cve-2017-17508.h5 | error exit | read 2 obj, 1 errors | read 2 obj | ok | +| cvefiles/cve-2017-17508.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok | | cvefiles/cve-2017-17509.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok | | cvefiles/cve-2018-11202.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok | | cvefiles/cve-2018-11203.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok | -| cvefiles/cve-2018-11204.h5 | error exit | read 2 obj, 1 errors | read 2 obj | ok | -| cvefiles/cve-2018-11205.h5 | error exit | read 2 obj, 1 errors | read 2 obj | ok | +| cvefiles/cve-2018-11204.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok | +| cvefiles/cve-2018-11205.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok | | cvefiles/cve-2018-11206-new.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok | | cvefiles/cve-2018-11206-old.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok | | cvefiles/cve-2018-11207.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok | @@ -135,10 +135,10 @@ columns are. | cvefiles/cve-2018-13870.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok | | cvefiles/cve-2018-13871.h5 | error exit | read 2 obj | read 2 obj | ok | | cvefiles/cve-2018-13872.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok | -| cvefiles/cve-2018-13873.h5 | error exit | read 1 obj, 1 errors | read 1 obj | ok | -| cvefiles/cve-2018-13874.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read | +| cvefiles/cve-2018-13873.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok | +| cvefiles/cve-2018-13874.h5 | error exit | open error | open error | h5py-cannot-read | | cvefiles/cve-2018-13875.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok | -| cvefiles/cve-2018-13876.h5 | error exit | open error | read 2 obj, 1 errors | h5py-cannot-read | +| cvefiles/cve-2018-13876.h5 | error exit | open error | open error | h5py-cannot-read | | cvefiles/cve-2018-14031.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok | | cvefiles/cve-2018-14033.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok | | cvefiles/cve-2018-14034.h5 | error exit | read 1 obj, 2 errors | read 1 obj | ok | @@ -176,13 +176,13 @@ columns are. | cvefiles/cve-2021-45833.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok | | cvefiles/cve-2021-46242.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read | | cvefiles/cve-2021-46243.h5 | error exit | read 3 obj, 2 errors | read 2 obj, 1 errors | mismatch | -| cvefiles/cve-2021-46244.h5 | error exit | read 2 obj, 1 errors | read 6 obj, 3 errors | mismatch | +| cvefiles/cve-2021-46244.h5 | error exit | read 2 obj, 1 errors | read 6 obj, 4 errors | mismatch | | cvefiles/cve-2024-29157.h5 | error exit | read 4 obj, 7 errors | read 4 obj, 7 errors | ok | | cvefiles/cve-2024-29158.h5 | ok | read 3 obj, 1 errors | read 3 obj, 1 errors | ok | | cvefiles/cve-2024-29159.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok | | cvefiles/cve-2024-29160.h5 | error exit | read 4 obj, 1 errors | read 4 obj, 1 errors | ok | -| cvefiles/cve-2024-29161.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 2 errors | ok | -| cvefiles/cve-2024-29162.h5 | error exit | read 17 obj, 4 errors | read 17 obj, 3 errors | ok | +| cvefiles/cve-2024-29161.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok | +| cvefiles/cve-2024-29162.h5 | error exit | read 17 obj, 4 errors | read 17 obj, 4 errors | ok | | cvefiles/cve-2024-29163.h5 | error exit | read 7 obj, 1 errors | read 7 obj, 1 errors | ok | | cvefiles/cve-2024-29164.h5 | ok | read 3 obj | read 3 obj | ok | | cvefiles/cve-2024-29165.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok | @@ -197,19 +197,19 @@ columns are. | cvefiles/cve-2024-32611.h5 | ok | read 6 obj | read 6 obj | ok | | cvefiles/cve-2024-32612.h5 | ok | read 3 obj | read 3 obj | ok | | cvefiles/cve-2024-32613.h5 | error exit | read 7 obj, 1 errors | read 7 obj, 1 errors | ok | -| cvefiles/cve-2024-32614.h5 | error exit | read 25 obj, 2 errors | read 25 obj, 1 errors | ok | +| cvefiles/cve-2024-32614.h5 | error exit | read 25 obj, 2 errors | read 25 obj, 2 errors | ok | | cvefiles/cve-2024-32615.h5 | error exit | read 4 obj, 1 errors | read 4 obj, 1 errors | ok | -| cvefiles/cve-2024-32616.h5 | error exit | read 10 obj, 7 errors | read 10 obj, 5 errors | ok | +| cvefiles/cve-2024-32616.h5 | error exit | read 10 obj, 7 errors | read 10 obj, 6 errors | ok | | cvefiles/cve-2024-32617.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok | -| cvefiles/cve-2024-32618.h5 | error exit | read 4 obj, 2 errors | read 3 obj | mismatch | -| cvefiles/cve-2024-32619.h5 | error exit | read 3 obj, 2 errors | read 3 obj | ok | +| cvefiles/cve-2024-32618.h5 | error exit | read 4 obj, 2 errors | read 3 obj, 1 errors | mismatch | +| cvefiles/cve-2024-32619.h5 | error exit | read 3 obj, 2 errors | read 3 obj, 2 errors | ok | | cvefiles/cve-2024-32620.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok | | cvefiles/cve-2024-32621.h5 | ok | read 3 obj, 1 errors | read 3 obj, 1 errors | ok | | cvefiles/cve-2024-32622.h5 | ok | read 3 obj, 1 errors | read 3 obj, 1 errors | ok | | cvefiles/cve-2024-32623.h5 | ok | read 6 obj | read 6 obj, 1 errors | our-error | | cvefiles/cve-2024-32624.h5 | error exit | read 6 obj, 1 errors | read 6 obj | ok | -| cvefiles/cve-2024-33873.h5 | error exit | read 4 obj, 1 errors | read 4 obj | ok | -| cvefiles/cve-2024-33874.h5 | ok | read 6 obj, 1 errors | read 6 obj, 1 errors | our-error | +| cvefiles/cve-2024-33873.h5 | error exit | read 4 obj, 1 errors | read 4 obj, 1 errors | ok | +| cvefiles/cve-2024-33874.h5 | ok | read 6 obj, 1 errors | read 6 obj, 2 errors | our-error | | cvefiles/cve-2024-33875.h5 | ok | read 2 obj | read 2 obj | ok | | cvefiles/cve-2024-33876.h5 | ok | read 3 obj, 1 errors | read 3 obj, 1 errors | ok | | cvefiles/cve-2024-33877.h5 | error exit | read 8 obj, 1 errors | read 8 obj, 1 errors | ok | @@ -246,12 +246,12 @@ columns are. | cvefiles/cve-2025-7068.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read | | cvefiles/cve-2025-7069.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read | | cvefiles/cve-2026-26200.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok | -| cvefiles/cve-2026-34734.h5 | error exit | read 2 obj, 1 errors | read 2 obj | ok | +| cvefiles/cve-2026-34734.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok | | cvefiles/cve-2026-92627.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok | | cvefiles/unknown-1.h5 | error exit | read 11 obj, 1 errors | read 11 obj, 1 errors | ok | | fuzzerfiles/gh-4431-poc-03.h5 | error exit | read 1 obj | read 1 obj | ok | | fuzzerfiles/gh-4432-poc-05.h5 | SIGSEGV | read 1 obj, 1 errors | read 1 obj, 1 errors | ok | -| fuzzerfiles/gh-4433-poc-08.h5 | error exit | read 1 obj, 1 errors | read 1 obj | ok | +| fuzzerfiles/gh-4433-poc-08.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok | | fuzzerfiles/gh-4434-poc-09.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read | | fuzzerfiles/gh-4435-poc-10.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok | | fuzzerfiles/gh-4585.h5 | error exit | open error | open error | h5py-cannot-read | @@ -279,10 +279,8 @@ columns are. ## Objects h5py fails on but clawhdf5 reads -- 19 x `KeyError: '…'` - 19 x `OSError: Can't synchronously read data (no appropriate function for conversion path)` - 1 x `TypeError: unhandled dtype kind M (dtype('…'))` -- 1 x `OSError: Can't synchronously read data (bad coordinate offset)` - 1 x `TypeError: No NumPy equivalent for TypeTimeID exists` - 1 x `KeyError: "…"` - 1 x `ValueError: Insufficient precision in available types to represent (N, N, N, N, N)` diff --git a/Cargo.toml b/Cargo.toml index 2e1f6fe..0801b3d 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -16,6 +16,8 @@ members = [ "crates/clawhdf5-cli", "crates/clawhdf5-napi", "crates/clawhdf5-bench", + "crates/clawhdf5-tools", + "crates/clawhdf5-wasm", "crates/libaec-sys", ] resolver = "2" @@ -34,3 +36,12 @@ tempfile = "3" criterion = { version = "0.5", features = ["html_reports"] } half = "2.7" serde = { version = "1", features = ["derive"] } + +# The browser build of clawhdf5-wasm (examples/wasm-viewer/build.sh): size +# over speed, whole-program optimisation. Native profiles are unaffected. +[profile.wasm-release] +inherits = "release" +opt-level = "s" +lto = true +codegen-units = 1 +panic = "abort" diff --git a/README.md b/README.md index 5d4c707..b9e3aa1 100644 --- a/README.md +++ b/README.md @@ -587,14 +587,14 @@ let exported = backend.export_markdown("MEMORY.md")?; ## Crate Map ``` -clawhdf5 workspace (16 crates, ~86K lines of Rust in src/, ~104K with tests +clawhdf5 workspace (17 crates, ~86K lines of Rust in src/, ~104K with tests and benches; plus libaec-sys, an internal FFI bindings crate for the optional szip feature) │ ├── Core HDF5 │ ├── clawhdf5-format — Binary parser/writer (no_std-capable), shared type definitions │ ├── clawhdf5-io — I/O abstraction (file/memory readers; optional mmap, async, HSDS, MPI) -│ ├── clawhdf5-filters — Fast deflate path (zlib-ng); lz4/zstd/pcodec/szip filters live in clawhdf5-format +│ ├── clawhdf5-filters — Fast deflate path (zlib-ng); the filter registry and the lz4/zstd/pcodec/szip/LZF/bitshuffle/bzip2/Blosc filters live in clawhdf5-format │ ├── clawhdf5-derive — Proc macros │ ├── clawhdf5 — High-level API │ ├── clawhdf5-netcdf4 — NetCDF-4 support @@ -610,7 +610,8 @@ clawhdf5 workspace (16 crates, ~86K lines of Rust in src/, ~104K with tests │ ├── Bindings │ ├── clawhdf5-py — Python (PyO3) -│ └── clawhdf5-napi — Node.js (napi-rs) +│ ├── clawhdf5-napi — Node.js (napi-rs) +│ └── clawhdf5-wasm — Browser (WebAssembly, wasm-bindgen; read-only) │ └── Tooling └── clawhdf5-bench — Benchmark suite @@ -700,6 +701,22 @@ stores keep their setting. Opt out with `float16 = false` or | `system-zlib` | no | System zlib backend for deflate (C) | | `blake3_hash` | no | BLAKE3 content hashing for provenance | | `szip` | no | SZIP filter (id 4) via libaec (C, through the internal `libaec-sys` crate) | +| `lzf` | **yes** | LZF filter (id 32000), h5py's built-in `compression="lzf"`: read and write. No dependencies | +| `bitshuffle` | no | Bitshuffle filter (id 32008) with its LZ4 and Zstandard modes: read and write. Pure Rust (lz4_flex, ruzstd) | +| `bzip2` | no | bzip2 filter (id 307): read and write. Pure Rust (the `bzip2` crate's libbz2-rs-sys backend compiles no C) | +| `blosc` | no | Blosc 1 filter (id 32001): reads BloscLZ, LZ4/LZ4HC, Snappy, Zlib and Zstandard frames with byte or bit shuffle; writes LZ4, Snappy, Zlib or Zstandard (not BloscLZ). Pure Rust | +| `plugin-filters` | no | All four above | + +Blosc2 (32026) and ZFP (32013) are not implemented: reading them fails with +`UnsupportedFilter`, whose message names the filter. Any other filter can be +supplied at run time with `filter_registry::register_filter` (a decoder +closure, or a `FilterCodec` that also encodes). The facade (`clawhdf5`) +forwards `lzf`, `bitshuffle`, `bzip2`, `blosc` and `plugin-filters`. Write +with `DatasetBuilder::with_lzf()`, `with_bitshuffle(..)`, `with_bzip2(..)` +and `with_blosc(..)`; h5py + hdf5plugin read the result (tested both ways in +`crates/clawhdf5/tests/plugin_filters_interop.rs`). The pure-Rust Zstandard +encoder has one level (about zstd's level 1); no speed or ratio claims are +made for these codecs. ### `clawhdf5-ann` diff --git a/conformance/baseline.json b/conformance/baseline.json index d17769a..eb3747e 100644 --- a/conformance/baseline.json +++ b/conformance/baseline.json @@ -1,15 +1,15 @@ { "comment": "conformance/run.sh fails if the ok count drops below `ok` or a file in `ok_files` stops being ok. Regenerate with `conformance/run.sh --update-baseline` after an intended change.", - "commit": "10d1029ead524e2fe64c2cd7f61b28067d9e449c", - "date": "2026-09-26 03:46 UTC", + "commit": "72306c601399748616bc9d061be2ebc4c1bea9e0", + "date": "2026-09-26 06:50 UTC", "reference": "h5py 3.16.0 / HDF5 2.0.0", "files": 697, - "ok": 569, + "ok": 575, "counts": { "h5py-cannot-read": 92, - "mismatch": 22, - "ok": 569, - "our-error": 14 + "mismatch": 20, + "ok": 575, + "our-error": 10 }, "per_corpus": { "NCAS-CMS_pyfive": { @@ -27,9 +27,9 @@ }, "hdf5": { "h5py-cannot-read": 60, - "mismatch": 12, - "ok": 386, - "our-error": 8 + "mismatch": 10, + "ok": 392, + "our-error": 4 }, "netcdf-c": { "ok": 20 @@ -182,9 +182,13 @@ "h5py_data/vlen_string_dset_utc.h5", "h5py_data/vlen_string_s390x.h5", "hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_bitgroom.h5", + "hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_blosc.h5", + "hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_bshuf.h5", + "hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_bzip2.h5", "hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_granularbr.h5", "hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_jpeg.h5", "hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_lz4.h5", + "hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_lzf.h5", "hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_zstd.h5", "hdf5/HDF5Examples/C/H5G/16/h5ex_g_iterate.h5", "hdf5/HDF5Examples/C/H5G/16/h5ex_g_traverse.h5", @@ -276,6 +280,7 @@ "hdf5/tools/test/testfiles/file_space.h5", "hdf5/tools/test/testfiles/filter_fail.h5", "hdf5/tools/test/testfiles/h5clear_fsm_persist_equal.h5", + "hdf5/tools/test/testfiles/h5clear_fsm_persist_less.h5", "hdf5/tools/test/testfiles/h5clear_fsm_persist_noclose.h5", "hdf5/tools/test/testfiles/h5clear_fsm_persist_user_equal.h5", "hdf5/tools/test/testfiles/h5clear_fsm_persist_user_less.h5", @@ -386,6 +391,7 @@ "hdf5/tools/test/testfiles/h5repack_uint8be_ex.h5", "hdf5/tools/test/testfiles/h5stat_err_old_fill.h5", "hdf5/tools/test/testfiles/h5stat_err_old_layout.h5", + "hdf5/tools/test/testfiles/h5stat_err_refcount.h5", "hdf5/tools/test/testfiles/h5stat_filters.h5", "hdf5/tools/test/testfiles/h5stat_idx.h5", "hdf5/tools/test/testfiles/h5stat_newgrat.h5", diff --git a/conformance/probe/Cargo.lock b/conformance/probe/Cargo.lock index 0ffb1f7..964e0ab 100644 --- a/conformance/probe/Cargo.lock +++ b/conformance/probe/Cargo.lock @@ -29,6 +29,15 @@ version = "1.5.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "1fd0f2584146f6f2ef48085050886acf353beff7305ebd1ae69500e27c67f64b" +[[package]] +name = "bzip2" +version = "0.6.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f3a53fac24f34a81bc9954b5d6cfce0c21e18ec6959f44f56e8e90e4bb7c346c" +dependencies = [ + "libbz2-rs-sys", +] + [[package]] name = "cc" version = "1.5.1" @@ -52,12 +61,15 @@ name = "clawhdf5-format" version = "2.7.0" dependencies = [ "byteorder", + "bzip2", "flate2", "libaec-sys", "lz4_flex", "pco", "portable-atomic", + "ruzstd", "sha2", + "snap", "zstd", ] @@ -192,6 +204,12 @@ dependencies = [ "pkg-config", ] +[[package]] +name = "libbz2-rs-sys" +version = "0.2.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "34b357333733e8260735ba5894eb928c02ecc69c78715f01a8019e7fa7f2db4c" + [[package]] name = "libc" version = "0.2.189" @@ -286,6 +304,15 @@ dependencies = [ "rand_core", ] +[[package]] +name = "ruzstd" +version = "0.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a252f5e20f038fe7b4ea53e073e65398d652c864cc162fc77c56c2f13717b888" +dependencies = [ + "twox-hash", +] + [[package]] name = "serde" version = "1.0.229" @@ -351,6 +378,12 @@ version = "0.3.10" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "3a219298ac11a56ea9a6d2120044824d6f01aeb034955e7af7bc16858527deea" +[[package]] +name = "snap" +version = "1.1.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "199905e6153d6405f9728fe44daace35f8f837bbf830bb6e85fbd5828709a886" + [[package]] name = "syn" version = "2.0.119" diff --git a/conformance/probe/Cargo.toml b/conformance/probe/Cargo.toml index f5c68af..62470b2 100644 --- a/conformance/probe/Cargo.toml +++ b/conformance/probe/Cargo.toml @@ -12,7 +12,7 @@ description = "Walks an HDF5 file with clawhdf5-format and prints a canonical JS [workspace] [dependencies] -clawhdf5-format = { path = "../../crates/clawhdf5-format", features = ["lz4", "zstd", "szip", "pcodec"] } +clawhdf5-format = { path = "../../crates/clawhdf5-format", features = ["lz4", "zstd", "szip", "pcodec", "plugin-filters"] } serde_json = "1" sha2 = "0.10" diff --git a/conformance/probe/src/main.rs b/conformance/probe/src/main.rs index 1298cf0..87c660d 100644 --- a/conformance/probe/src/main.rs +++ b/conformance/probe/src/main.rs @@ -309,11 +309,19 @@ impl<'a> Ctx<'a> { } } + fn read_named_datatype(&self, h: &ObjectHeader) -> Result<(), String> { + let dtb = self + .payload(h, MessageType::Datatype)? + .ok_or("MissingMessage(Datatype)")?; + Datatype::parse_in_header(&dtb, h.version).map_err(e)?; + Ok(()) + } + fn read_dataset(&self, h: &ObjectHeader, rec: &mut Map) -> Result<(), String> { let dtb = self .payload(h, MessageType::Datatype)? .ok_or("MissingMessage(Datatype)")?; - let (dt, _) = Datatype::parse(&dtb).map_err(e)?; + let (dt, _) = Datatype::parse_in_header(&dtb, h.version).map_err(e)?; rec.insert("dtype".into(), Value::String(dtype_str(&dt))); let dsb = self .payload(h, MessageType::Dataspace)? @@ -716,6 +724,17 @@ fn main() { return; } }; + // libhdf5 refuses a truncated file and reads nothing past the recorded + // end of file. + let base = (data.len() - hdf5.len()) as u64; + let hdf5 = match sb.data_end(base, data.len() as u64) { + Ok(end) => &hdf5[..end as usize], + Err(err) => { + top.insert("open_error".into(), Value::String(e(err))); + println!("{}", Value::Object(top)); + return; + } + }; top.insert("superblock_version".into(), json!(sb.version)); let ctx = Ctx { data: hdf5, @@ -778,6 +797,13 @@ fn main() { { rec.insert("error".into(), Value::String(msg)); } + // Opening a committed datatype decodes it (h5py's `f[name]` fails on + // one libhdf5 cannot decode), so decode it here too. + if kind == "datatype" + && let Err(msg) = guarded(|| ctx.read_named_datatype(&h)) + { + rec.insert("error".into(), Value::String(msg)); + } if kind != "datatype" { match guarded(|| ctx.attrs(&h)) { Ok(m) => { diff --git a/crates/clawhdf5-bench/Cargo.toml b/crates/clawhdf5-bench/Cargo.toml index 5ac6350..b48afbf 100644 --- a/crates/clawhdf5-bench/Cargo.toml +++ b/crates/clawhdf5-bench/Cargo.toml @@ -34,6 +34,10 @@ path = "src/bin/consolidation_efficiency.rs" name = "ephemeral_perf" path = "src/bin/ephemeral_perf.rs" +[[bin]] +name = "concurrent_read" +path = "src/bin/concurrent_read.rs" + [[bin]] name = "mpi_io_bench" path = "src/bin/mpi_io_bench.rs" @@ -64,6 +68,10 @@ clawhdf5-io = { path = "../clawhdf5-io" } mpi = { version = "0.8", optional = true } serde = { workspace = true } serde_json = "1" +# concurrent_read: size the decode pool (--decode-threads) and evict files +# from the page cache (--cold, posix_fadvise). Both pure Rust / bindings only. +rayon = "1" +libc = "0.2" tempfile = { workspace = true } # Optional: libhdf5 C wrapper for side-by-side comparison (requires system libhdf5). # Enable with: cargo bench -p clawhdf5-bench --features libhdf5-compare diff --git a/crates/clawhdf5-bench/scripts/__pycache__/compare_concurrent_read.cpython-314.pyc b/crates/clawhdf5-bench/scripts/__pycache__/compare_concurrent_read.cpython-314.pyc new file mode 100644 index 0000000..85432af Binary files /dev/null and b/crates/clawhdf5-bench/scripts/__pycache__/compare_concurrent_read.cpython-314.pyc differ diff --git a/crates/clawhdf5-bench/scripts/__pycache__/concurrent_read_h5py.cpython-314.pyc b/crates/clawhdf5-bench/scripts/__pycache__/concurrent_read_h5py.cpython-314.pyc new file mode 100644 index 0000000..4f759f2 Binary files /dev/null and b/crates/clawhdf5-bench/scripts/__pycache__/concurrent_read_h5py.cpython-314.pyc differ diff --git a/crates/clawhdf5-bench/scripts/compare_concurrent_read.py b/crates/clawhdf5-bench/scripts/compare_concurrent_read.py new file mode 100644 index 0000000..b7c75c2 --- /dev/null +++ b/crates/clawhdf5-bench/scripts/compare_concurrent_read.py @@ -0,0 +1,70 @@ +#!/usr/bin/env python3 +"""Tabulate concurrent_read JSON results (clawhdf5, h5py threads/processes). + + python compare_concurrent_read.py clawhdf5.json h5py-threads.json h5py-procs.json + +Prints one Markdown table: for each layout, mode and thread count, every +tool's MB/s and scaling efficiency, and the first file's MB/s relative to each +of the others. Refuses to compare runs whose workload parameters differ. +""" + +import json +import sys + +COMPARED = ("datasets", "rows", "cols", "chunk", "deflate_level", "slab", "slabs", "seed") + + +def main(paths): + if len(paths) < 2: + sys.exit(__doc__) + docs = [] + for p in paths: + with open(p) as fh: + docs.append(json.load(fh)) + ref = docs[0] + for d, p in zip(docs[1:], paths[1:]): + diff = [k for k in COMPARED if d["params"].get(k) != ref["params"].get(k)] + if diff: + sys.exit(f"{p}: workload differs from {paths[0]} in {', '.join(diff)}") + if d["cache"] != ref["cache"]: + print(f"warning: {p} ran {d['cache']!r}, {paths[0]} ran {ref['cache']!r}", + file=sys.stderr) + if d.get("host") != ref.get("host"): + print(f"warning: {p} ran on {d.get('host')}, {paths[0]} on {ref.get('host')}", + file=sys.stderr) + + names = [d["tool"] for d in docs] + for d in docs: + extra = f", HDF5 {d['hdf5_version']}" if "hdf5_version" in d else "" + print(f"- {d['tool']} {d['version']}{extra}: host {d.get('host')}, " + f"{d.get('cpus')} CPUs, cache {d['cache']}, decode threads per read " + f"{d.get('decode_threads')}") + p = ref["params"] + print(f"\n{p['datasets']} datasets of {p['rows']} x {p['cols']} f32, chunks " + f"{p['chunk'][0]} x {p['chunk'][1]} (deflate {p['deflate_level']}); " + f"`same`: {p['slabs']} slabs of {p['slab']} x {p['slab']}\n") + + index = [{(r["layout"], r["mode"], r["threads"]): r for r in d["results"]} for d in docs] + keys = [(r["layout"], r["mode"], r["threads"]) for r in ref["results"]] + + head = ["layout", "mode", "threads"] + head += [f"{n} MB/s (eff)" for n in names] + head += [f"{names[0]} / {n}" for n in names[1:]] + print("| " + " | ".join(head) + " |") + print("|---|---|" + "---:|" * (len(head) - 2)) + for key in keys: + cells = [key[0], key[1], str(key[2])] + rs = [ix.get(key) for ix in index] + for r in rs: + if r is None: + cells.append("-") + else: + eff = "-" if r["efficiency"] is None else f"{r['efficiency']:.2f}" + cells.append(f"{r['mb_s']:.0f} ({eff})") + for r in rs[1:]: + cells.append("-" if r is None else f"{rs[0]['mb_s'] / r['mb_s']:.2f}x") + print("| " + " | ".join(cells) + " |") + + +if __name__ == "__main__": + main(sys.argv[1:]) diff --git a/crates/clawhdf5-bench/scripts/concurrent_read_h5py.py b/crates/clawhdf5-bench/scripts/concurrent_read_h5py.py new file mode 100644 index 0000000..883a850 --- /dev/null +++ b/crates/clawhdf5-bench/scripts/concurrent_read_h5py.py @@ -0,0 +1,265 @@ +#!/usr/bin/env python3 +"""The concurrent_read workload with h5py, on the files concurrent_read wrote. + +libhdf5 serialises every API call under one global lock, and h5py holds its +own global lock around every call as well, so h5py *threads* cannot decode in +parallel. h5py users scale with *processes* instead; ``--executor processes`` +measures that (each worker opens the file itself). + +The workload mirrors ``crates/clawhdf5-bench/src/bin/concurrent_read.rs``: + +* ``distinct``: every dataset read in full once per repetition; worker ``t`` + of ``T`` reads datasets ``t, t + T, ...``. +* ``same``: ``--slabs`` random ``--slab`` x ``--slab`` hyperslabs of ``d00`` + (slab ``j`` to worker ``j % T``), offsets from the same splitmix64 stream. + +Each worker times itself from a start barrier; a repetition spans the earliest +start to the latest finish (CLOCK_MONOTONIC, comparable across processes). +Threads share one ``h5py.File`` per repetition; process workers open the file +inside the timed region (a few ms against reads of many MiB). + +Generate the files first with the Rust harness (it writes ``manifest.json``), +then, for example:: + + python concurrent_read_h5py.py --dir DIR --executor threads --json h5py-threads.json + python concurrent_read_h5py.py --dir DIR --executor processes --json h5py-procs.json +""" + +import argparse +import json +import multiprocessing as mp +import os +import platform +import socket +import sys +import threading +import time + +import h5py +import numpy as np + +M64 = (1 << 64) - 1 + + +def splitmix64(state): + """Return (new_state, value); the same stream as the Rust harness.""" + state = (state + 0x9E3779B97F4A7C15) & M64 + z = state + z = ((z ^ (z >> 30)) * 0xBF58476D1CE4E5B9) & M64 + z = ((z ^ (z >> 27)) * 0x94D049BB133111EB) & M64 + return state, z ^ (z >> 31) + + +def value(k, i): + """Element i (row-major) of dataset k, exactly as concurrent_read writes it.""" + _, noise = splitmix64(i ^ (k << 40)) + return np.float32((((i >> 6) % 16384) + k) + (noise & 0xFF) / 256.0) + + +def slab_offsets(seed, count, rows, cols, slab): + s = seed + out = [] + for _ in range(count): + s, r = splitmix64(s) + s, c = splitmix64(s) + out.append((r % (rows - slab + 1), c % (cols - slab + 1))) + return out + + +def now(): + return time.clock_gettime(time.CLOCK_MONOTONIC) + + +def work(f, mode, t, threads, m, slabs, slab, verify): + """Worker t's share of one repetition on an open h5py.File.""" + n = m["rows"] * m["cols"] + if mode == "distinct": + for k in range(t, m["datasets"], threads): + got = f[f"d{k:02d}"][...] + assert got.size == n + if verify: + flat = got.reshape(-1) + for i in (0, n // 3, n - 1): + assert flat[i] == value(k, i), f"d{k:02d}[{i}]" + else: + ds = f["d00"] + cols = m["cols"] + for r, c in slabs[t::threads]: + got = ds[r : r + slab, c : c + slab] + assert got.shape == (slab, slab) + if verify: + assert got[0, 0] == value(0, r * cols + c) + last = (r + slab - 1) * cols + c + slab - 1 + assert got[-1, -1] == value(0, last) + + +# ----- process workers ------------------------------------------------------ + +_barrier = None + + +def _init(barrier): + global _barrier + _barrier = barrier + + +def _proc_task(task): + path, mode, t, threads, m, slabs, slab = task + _barrier.wait() + start = now() + with h5py.File(path, "r") as f: + work(f, mode, t, threads, m, slabs, slab, False) + return start, now() + + +def _noop(_): + return os.getpid() + + +def run_threads(path, mode, threads, m, slabs, slab): + spans = [None] * threads + barrier = threading.Barrier(threads) + with h5py.File(path, "r") as f: + + def body(t): + barrier.wait() + start = now() + work(f, mode, t, threads, m, slabs, slab, False) + spans[t] = (start, now()) + + ts = [threading.Thread(target=body, args=(t,)) for t in range(threads)] + for th in ts: + th.start() + for th in ts: + th.join() + return max(e for _, e in spans) - min(s for s, _ in spans) + + +def run_processes(pool, path, mode, threads, m, slabs, slab): + tasks = [(path, mode, t, threads, m, slabs, slab) for t in range(threads)] + # One task per worker: each blocks in the barrier until all T have + # started, so no worker can take a second task. + spans = pool.map(_proc_task, tasks, chunksize=1) + return max(e for _, e in spans) - min(s for s, _ in spans) + + +def warm(path): + with open(path, "rb") as fh: + while fh.read(1 << 24): + pass + + +def evict(path): + fd = os.open(path, os.O_RDONLY) + try: + os.posix_fadvise(fd, 0, 0, os.POSIX_FADV_DONTNEED) + finally: + os.close(fd) + + +def main(): + ap = argparse.ArgumentParser(description=__doc__.split("\n\n")[0]) + ap.add_argument("--dir", default="concurrent-read-data") + ap.add_argument("--executor", choices=["threads", "processes"], default="threads") + ap.add_argument("--threads", default="1,2,4,8,16") + ap.add_argument("--reps", type=int, default=3) + ap.add_argument("--slab", type=int, default=256) + ap.add_argument("--slabs", type=int, default=1024) + ap.add_argument("--seed", type=int, default=42) + ap.add_argument("--cold", action="store_true") + ap.add_argument("--modes", default="distinct,same") + ap.add_argument("--layouts", default="deflate,contiguous") + ap.add_argument("--json") + a = ap.parse_args() + + # The Rust harness pins this value (splitmix64_reference). + assert splitmix64(42)[1] == 0xBDD732262FEB6E95, "splitmix64 port is wrong" + + try: + with open(os.path.join(a.dir, "manifest.json")) as fh: + m = json.load(fh) + except FileNotFoundError: + sys.exit(f"{a.dir}/manifest.json not found: generate the files with " + "`cargo run --release -p clawhdf5-bench --bin concurrent_read -- --dir ...` first") + threads_list = [int(x) for x in a.threads.split(",")] + modes = a.modes.split(",") + layouts = a.layouts.split(",") + if a.slab < 1 or a.slab > min(m["rows"], m["cols"]): + sys.exit(f"--slab must be 1..={min(m['rows'], m['cols'])}") + files = dict(m["files"]) + slabs = slab_offsets(a.seed, a.slabs, m["rows"], m["cols"], a.slab) + dataset_bytes = m["rows"] * m["cols"] * 4 + tool = f"h5py-{a.executor}" + + ctx = mp.get_context("spawn") # never fork a process holding HDF5 state + pools = {} + if a.executor == "processes": + for t in threads_list: + pool = ctx.Pool(t, initializer=_init, initargs=(ctx.Barrier(t),)) + pool.map(_noop, range(t)) # start the workers outside the timing + pools[t] = pool + + rows = [] + print("| layout | mode | threads | MB/s | efficiency | median s |") + print("|---|---|---:|---:|---:|---:|") + try: + for layout in layouts: + path = os.path.join(a.dir, files[layout]) + if not a.cold: + warm(path) + for mode in modes: + with h5py.File(path, "r") as f: # untimed, checked pass + work(f, mode, 0, 1, m, slabs, a.slab, True) + nbytes = (dataset_bytes * m["datasets"] if mode == "distinct" + else a.slab * a.slab * 4 * a.slabs) + base = None + for t in threads_list: + times = [] + for _ in range(a.reps): + if a.cold: + evict(path) + if a.executor == "threads": + times.append(run_threads(path, mode, t, m, slabs, a.slab)) + else: + times.append(run_processes(pools[t], path, mode, t, m, slabs, a.slab)) + med = sorted(times)[len(times) // 2] + mb_s = nbytes / (1 << 20) / med + if t == 1: + base = mb_s + eff = mb_s / (t * base) if base else None + print(f"| {layout} | {mode} | {t} | {mb_s:.0f} | " + f"{'-' if eff is None else f'{eff:.2f}'} | {med:.4f} |") + rows.append({ + "layout": layout, "mode": mode, "threads": t, "bytes": nbytes, + "times_s": times, "median_s": med, "mb_s": mb_s, "efficiency": eff, + }) + finally: + for pool in pools.values(): + pool.terminate() + + if a.json: + doc = { + "tool": tool, + "version": h5py.__version__, + "hdf5_version": h5py.version.hdf5_version, + "python": platform.python_version(), + "host": socket.gethostname(), + "cpus": os.cpu_count(), + "unix_time": int(time.time()), + "cache": ("cold (posix_fadvise DONTNEED before each repetition)" + if a.cold else "warm"), + "decode_threads": 1, + "params": { + "datasets": m["datasets"], "rows": m["rows"], "cols": m["cols"], + "chunk": m["chunk"], "deflate_level": m["deflate_level"], + "mib": dataset_bytes // (1 << 20), "slab": a.slab, "slabs": a.slabs, + "seed": a.seed, "reps": a.reps, "dir": a.dir, + }, + "results": rows, + } + with open(a.json, "w") as fh: + json.dump(doc, fh, indent=2) + + +if __name__ == "__main__": + main() diff --git a/crates/clawhdf5-bench/src/bin/concurrent_read.rs b/crates/clawhdf5-bench/src/bin/concurrent_read.rs new file mode 100644 index 0000000..a87ac26 --- /dev/null +++ b/crates/clawhdf5-bench/src/bin/concurrent_read.rs @@ -0,0 +1,523 @@ +//! Concurrent-read harness: how does decoded read throughput scale with the +//! number of threads reading one open file? +//! +//! libhdf5 (threadsafe build) serialises every API call under one global +//! mutex, and h5py holds it too, so threads cannot decode in parallel there. +//! A clawhdf5 [`File`] is `Send + Sync`; this harness measures what that buys. +//! `crates/clawhdf5-bench/scripts/concurrent_read_h5py.py` runs the same +//! workload on the same files with h5py (threads, and processes), and +//! `compare_concurrent_read.py` tabulates the JSON both write. +//! +//! Files (generated on first use, reused while `manifest.json` matches): +//! +//! * `/deflate.h5`: `--datasets` datasets `d00`, `d01`, ... of `f32`, +//! `--mib` MiB decoded each, shape `[mib * 256, 1024]`, chunks `256 x 256`, +//! deflate level 4. +//! * `/contiguous.h5`: the same datasets, contiguous. +//! +//! Modes, for each layout and each thread count `T` (strong scaling: the total +//! work per repetition is fixed, split among the threads): +//! +//! * `distinct`: every dataset is read in full once; thread `t` reads datasets +//! `t, t + T, t + 2T, ...`. +//! * `same`: all threads read `d00`, `--slabs` random `--slab` x `--slab` +//! hyperslabs in total (slab `j` goes to thread `j % T`). The offsets come +//! from a splitmix64 stream seeded with `--seed`, identical in the h5py +//! script. +//! +//! One `File` per layout per repetition is shared by all threads (opened +//! fresh each repetition, so no chunk cache carries over). Page cache: +//! `warm` (default) reads every file once before timing; `--cold` evicts the +//! files from the page cache with `posix_fadvise(POSIX_FADV_DONTNEED)` before +//! every repetition (no root needed; it only evicts clean, unmapped pages, so +//! it is best effort — the JSON says which was used). +//! +//! Decode inside one read is itself parallel when clawhdf5-format's `parallel` +//! feature is on (it is in this binary, via clawhdf5-agent). `--decode-threads +//! N` sizes that rayon pool; `--decode-threads 1` measures the API's own +//! thread scaling, comparable with h5py where each call decodes on the +//! calling thread. +//! +//! ```text +//! cargo run --release -p clawhdf5-bench --bin concurrent_read -- \ +//! --dir /data/concurrent-read --json clawhdf5.json +//! cargo run --release -p clawhdf5-bench --bin concurrent_read -- \ +//! --dir /tmp/cr --datasets 4 --mib 1 --threads 1,2 --slabs 16 --reps 1 # smoke +//! ``` + +use std::path::{Path, PathBuf}; +use std::sync::Barrier; +use std::time::Instant; + +use clawhdf5::{File, FileBuilder, Selection}; +use serde::{Deserialize, Serialize}; + +const COLS: u64 = 1024; +const ROWS_PER_MIB: u64 = 256; // 256 rows x 1024 cols x 4 bytes = 1 MiB +const CHUNK: u64 = 256; +const DEFLATE_LEVEL: u32 = 4; +const LAYOUTS: [&str; 2] = ["deflate", "contiguous"]; +const MANIFEST_VERSION: u32 = 1; + +/// splitmix64 — shared with the h5py script, which must produce the same +/// stream (both the data and the hyperslab offsets depend on it). +fn splitmix64(state: &mut u64) -> u64 { + *state = state.wrapping_add(0x9E37_79B9_7F4A_7C15); + let mut z = *state; + z = (z ^ (z >> 30)).wrapping_mul(0xBF58_476D_1CE4_E5B9); + z = (z ^ (z >> 27)).wrapping_mul(0x94D0_49BB_1331_11EB); + z ^ (z >> 31) +} + +/// Element `i` (row-major) of dataset `k`: a slowly varying integer part plus +/// 8 bits of noise, so deflate has real work to do (about 3.1x) and every value +/// is exact in `f32` (< 2^15 with 8 fraction bits), which lets both harnesses +/// check what they read against this formula. +fn value(k: u64, i: u64) -> f32 { + let mut s = i ^ (k << 40); + let noise = splitmix64(&mut s) & 0xff; + (((i >> 6) % 16384) + k) as f32 + noise as f32 / 256.0 +} + +#[derive(Serialize, Deserialize, PartialEq, Debug, Clone)] +struct Manifest { + version: u32, + datasets: u64, + rows: u64, + cols: u64, + chunk: [u64; 2], + deflate_level: u32, + files: Vec<(String, String)>, // (layout, file name) + writer: String, +} + +fn manifest_for(datasets: u64, mib: u64) -> Manifest { + Manifest { + version: MANIFEST_VERSION, + datasets, + rows: mib * ROWS_PER_MIB, + cols: COLS, + chunk: [CHUNK, CHUNK], + deflate_level: DEFLATE_LEVEL, + files: LAYOUTS + .iter() + .map(|l| (l.to_string(), format!("{l}.h5"))) + .collect(), + writer: format!("clawhdf5 {}", env!("CARGO_PKG_VERSION")), + } +} + +fn dataset_values(k: u64, n: u64) -> Vec { + (0..n).map(|i| value(k, i)).collect() +} + +/// Write the files unless `dir` already holds ones matching `want`. +fn ensure_files(dir: &Path, want: &Manifest) -> std::io::Result { + let manifest_path = dir.join("manifest.json"); + if let Ok(text) = std::fs::read_to_string(&manifest_path) + && let Ok(have) = serde_json::from_str::(&text) + && have.version == want.version + && have.datasets == want.datasets + && have.rows == want.rows + && have.cols == want.cols + && have.chunk == want.chunk + && have.deflate_level == want.deflate_level + && have.files == want.files + && want.files.iter().all(|(_, f)| dir.join(f).exists()) + { + return Ok(false); + } + std::fs::create_dir_all(dir)?; + // A stale manifest must not survive a half-written regeneration. + let _ = std::fs::remove_file(&manifest_path); + let n = want.rows * want.cols; + for (layout, file) in &want.files { + // One layout at a time keeps the peak memory to about twice one + // file's decoded size. + let mut b = FileBuilder::new(); + for k in 0..want.datasets { + let ds = b.create_dataset(&format!("d{k:02}")); + ds.with_f32_data(&dataset_values(k, n)) + .with_shape(&[want.rows, want.cols]); + if layout == "deflate" { + ds.with_chunks(&[CHUNK.min(want.rows), CHUNK]) + .with_deflate(DEFLATE_LEVEL); + } + } + b.write(dir.join(file)).map_err(std::io::Error::other)?; + } + std::fs::write( + &manifest_path, + serde_json::to_string_pretty(want).map_err(std::io::Error::other)?, + )?; + Ok(true) +} + +fn slab_offsets(seed: u64, count: usize, rows: u64, cols: u64, slab: u64) -> Vec<(u64, u64)> { + let mut s = seed; + (0..count) + .map(|_| { + let r = splitmix64(&mut s) % (rows - slab + 1); + let c = splitmix64(&mut s) % (cols - slab + 1); + (r, c) + }) + .collect() +} + +/// Warm the page cache by reading every byte of `path`. +fn warm(path: &Path) -> std::io::Result<()> { + let mut f = std::fs::File::open(path)?; + std::io::copy(&mut f, &mut std::io::sink())?; + Ok(()) +} + +/// Ask the kernel to drop `path`'s pages from the page cache. +fn evict(path: &Path) -> std::io::Result<()> { + use std::os::fd::AsRawFd; + let f = std::fs::File::open(path)?; + // SAFETY: plain syscall on a valid, open file descriptor. + let rc = unsafe { libc::posix_fadvise(f.as_raw_fd(), 0, 0, libc::POSIX_FADV_DONTNEED) }; + if rc != 0 { + return Err(std::io::Error::from_raw_os_error(rc)); + } + Ok(()) +} + +#[derive(Serialize)] +struct Row { + layout: String, + mode: String, + threads: usize, + /// Decoded (selected) bytes read per repetition. + bytes: u64, + times_s: Vec, + median_s: f64, + mb_s: f64, + /// `mb_s / (threads * mb_s at threads = 1)`; null without a 1-thread row. + efficiency: Option, +} + +struct Args { + dir: PathBuf, + datasets: u64, + mib: u64, + threads: Vec, + reps: usize, + slab: u64, + slabs: usize, + seed: u64, + cold: bool, + decode_threads: usize, + modes: Vec, + layouts: Vec, + json: Option, +} + +const USAGE: &str = "\ +usage: concurrent_read [--dir DIR] [--datasets N] [--mib N] [--threads 1,2,4,8,16] + [--reps N] [--slab N] [--slabs N] [--seed N] [--cold] + [--decode-threads N] [--modes distinct,same] + [--layouts deflate,contiguous] [--json FILE]"; + +fn parse_list(s: &str) -> Result, String> { + s.split(',') + .map(|x| x.trim().parse().map_err(|_| format!("bad list item {x:?}"))) + .collect() +} + +fn parse_args() -> Result { + let mut a = Args { + dir: PathBuf::from("concurrent-read-data"), + datasets: 64, + mib: 64, + threads: vec![1, 2, 4, 8, 16], + reps: 3, + slab: 256, + slabs: 1024, + seed: 42, + cold: false, + decode_threads: 0, + modes: vec!["distinct".into(), "same".into()], + layouts: LAYOUTS.iter().map(|s| s.to_string()).collect(), + json: None, + }; + let mut it = std::env::args().skip(1); + while let Some(flag) = it.next() { + if flag == "--cold" { + a.cold = true; + continue; + } + if flag == "-h" || flag == "--help" { + return Err(USAGE.into()); + } + let v = it.next().ok_or(format!("{flag} needs a value\n{USAGE}"))?; + let num = |v: &str| { + v.parse::() + .map_err(|_| format!("{flag}: bad number {v:?}")) + }; + match flag.as_str() { + "--dir" => a.dir = v.into(), + "--datasets" => a.datasets = num(&v)?, + "--mib" => a.mib = num(&v)?, + "--threads" => a.threads = parse_list(&v)?, + "--reps" => a.reps = num(&v)? as usize, + "--slab" => a.slab = num(&v)?, + "--slabs" => a.slabs = num(&v)? as usize, + "--seed" => a.seed = num(&v)?, + "--decode-threads" => a.decode_threads = num(&v)? as usize, + "--modes" => a.modes = parse_list(&v)?, + "--layouts" => a.layouts = parse_list(&v)?, + "--json" => a.json = Some(v.into()), + _ => return Err(format!("unknown flag {flag}\n{USAGE}")), + } + } + if a.datasets == 0 || a.datasets > 100 { + return Err("--datasets must be 1..=100".into()); + } + if a.mib == 0 || a.reps == 0 || a.slabs == 0 || a.threads.contains(&0) { + return Err("--mib, --reps, --slabs and every --threads value must be > 0".into()); + } + if a.slab == 0 || a.slab > COLS || a.slab > a.mib * ROWS_PER_MIB { + return Err(format!( + "--slab must be 1..={}", + COLS.min(a.mib * ROWS_PER_MIB) + )); + } + for m in &a.modes { + if m != "distinct" && m != "same" { + return Err(format!("unknown mode {m:?}")); + } + } + for l in &a.layouts { + if !LAYOUTS.contains(&l.as_str()) { + return Err(format!("unknown layout {l:?}")); + } + } + Ok(a) +} + +/// One timed repetition: `T` threads on one shared `File`. Returns seconds. +fn run_once( + path: &Path, + mode: &str, + threads: usize, + m: &Manifest, + slabs: &[(u64, u64)], + slab: u64, + verify: bool, +) -> f64 { + let file = File::open(path).expect("open"); + let barrier = Barrier::new(threads + 1); // + the spawning thread + let n = m.rows * m.cols; + // Each thread times itself from the barrier; the repetition spans the + // earliest start to the latest finish (timing on the spawning thread + // instead undercounts whenever it is scheduled after the workers ran). + let spans: Vec<(Instant, Instant)> = std::thread::scope(|s| { + let handles: Vec<_> = (0..threads) + .map(|t| { + let (file, barrier) = (&file, &barrier); + s.spawn(move || { + barrier.wait(); + let start = Instant::now(); + match mode { + "distinct" => { + for k in (t as u64..m.datasets).step_by(threads) { + let got = file.dataset(&format!("d{k:02}")).unwrap().read_f32(); + let got = got.unwrap(); + assert_eq!(got.len() as u64, n); + if verify { + for i in [0, n / 3, n - 1] { + assert_eq!(got[i as usize], value(k, i), "d{k:02}[{i}]"); + } + } + std::hint::black_box(got); + } + } + _ => { + let ds = file.dataset("d00").unwrap(); + for &(r, c) in slabs.iter().skip(t).step_by(threads) { + let sel = Selection::Hyperslab { + start: vec![r, c], + stride: vec![1, 1], + count: vec![slab, slab], + block: vec![1, 1], + }; + let got = ds.read_f32_selection(&sel).unwrap(); + assert_eq!(got.len() as u64, slab * slab); + if verify { + let last = (r + slab - 1) * m.cols + c + slab - 1; + assert_eq!(got[0], value(0, r * m.cols + c)); + assert_eq!(*got.last().unwrap(), value(0, last)); + } + std::hint::black_box(got); + } + } + } + (start, Instant::now()) + }) + }) + .collect(); + barrier.wait(); + handles.into_iter().map(|h| h.join().unwrap()).collect() + }); + let start = spans.iter().map(|s| s.0).min().unwrap(); + let end = spans.iter().map(|s| s.1).max().unwrap(); + (end - start).as_secs_f64() +} + +fn median(v: &[f64]) -> f64 { + let mut s = v.to_vec(); + s.sort_by(f64::total_cmp); + s[s.len() / 2] +} + +fn hostname() -> String { + std::fs::read_to_string("/proc/sys/kernel/hostname") + .map(|s| s.trim().to_string()) + .unwrap_or_else(|_| "unknown".into()) +} + +fn main() { + let args = match parse_args() { + Ok(a) => a, + Err(e) => { + eprintln!("{e}"); + std::process::exit(2); + } + }; + if cfg!(debug_assertions) { + eprintln!("warning: debug build — numbers are meaningless. Use --release."); + } + if args.decode_threads > 0 { + rayon::ThreadPoolBuilder::new() + .num_threads(args.decode_threads) + .build_global() + .expect("configure rayon pool"); + } + + let manifest = manifest_for(args.datasets, args.mib); + let t = Instant::now(); + match ensure_files(&args.dir, &manifest) { + Ok(true) => eprintln!( + "generated {} in {:.1} s", + args.dir.display(), + t.elapsed().as_secs_f64() + ), + Ok(false) => eprintln!("reusing {}", args.dir.display()), + Err(e) => { + eprintln!("cannot write test files in {}: {e}", args.dir.display()); + std::process::exit(1); + } + } + let path_of = |layout: &str| args.dir.join(format!("{layout}.h5")); + let slabs = slab_offsets( + args.seed, + args.slabs, + manifest.rows, + manifest.cols, + args.slab, + ); + let dataset_bytes = manifest.rows * manifest.cols * 4; + + let mut rows: Vec = Vec::new(); + println!("| layout | mode | threads | MB/s | efficiency | median s |"); + println!("|---|---|---:|---:|---:|---:|"); + for layout in &args.layouts { + let path = path_of(layout); + // Untimed pass: page cache warm (unless --cold), results checked. + if !args.cold { + warm(&path).expect("warm page cache"); + } + for mode in &args.modes { + run_once(&path, mode, 1, &manifest, &slabs, args.slab, true); + let bytes = match mode.as_str() { + "distinct" => dataset_bytes * manifest.datasets, + _ => args.slab * args.slab * 4 * args.slabs as u64, + }; + let mut base: Option = None; + for &threads in &args.threads { + let times: Vec = (0..args.reps) + .map(|_| { + if args.cold { + evict(&path).expect("posix_fadvise"); + } + run_once(&path, mode, threads, &manifest, &slabs, args.slab, false) + }) + .collect(); + let med = median(×); + let mb_s = bytes as f64 / (1 << 20) as f64 / med; + if threads == 1 { + base = Some(mb_s); + } + let efficiency = base.map(|b| mb_s / (threads as f64 * b)); + println!( + "| {layout} | {mode} | {threads} | {mb_s:.0} | {} | {med:.4} |", + efficiency.map_or("-".into(), |e| format!("{e:.2}")) + ); + rows.push(Row { + layout: layout.clone(), + mode: mode.clone(), + threads, + bytes, + times_s: times, + median_s: med, + mb_s, + efficiency, + }); + } + } + } + + if let Some(out) = &args.json { + let doc = serde_json::json!({ + "tool": "clawhdf5", + "version": env!("CARGO_PKG_VERSION"), + "host": hostname(), + "cpus": std::thread::available_parallelism().map_or(0, |n| n.get()), + "unix_time": std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map_or(0, |d| d.as_secs()), + "cache": if args.cold { "cold (posix_fadvise DONTNEED before each repetition)" } else { "warm" }, + "decode_threads": rayon::current_num_threads(), + "params": { + "datasets": manifest.datasets, + "mib": args.mib, + "rows": manifest.rows, + "cols": manifest.cols, + "chunk": manifest.chunk, + "deflate_level": manifest.deflate_level, + "slab": args.slab, + "slabs": args.slabs, + "seed": args.seed, + "reps": args.reps, + "dir": args.dir, + }, + "results": rows, + }); + std::fs::write(out, serde_json::to_string_pretty(&doc).unwrap()).expect("write json"); + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn values_are_exact_in_f32() { + for k in [0, 7, 63] { + for i in [0u64, 1, 4095, 1 << 20, (1 << 24) - 1] { + let v = value(k, i); + assert_eq!(v, (v as f64) as f32); + assert!(v < 32768.0); + assert_eq!((v * 256.0).fract(), 0.0); + } + } + } + + /// The h5py script hard-codes this vector to check its splitmix64 port. + #[test] + fn splitmix64_reference() { + let mut s = 42; + assert_eq!(splitmix64(&mut s), 0xBDD7_3226_2FEB_6E95); + } +} diff --git a/crates/clawhdf5-bench/tests/concurrent_read_smoke.rs b/crates/clawhdf5-bench/tests/concurrent_read_smoke.rs new file mode 100644 index 0000000..a8ab5c7 --- /dev/null +++ b/crates/clawhdf5-bench/tests/concurrent_read_smoke.rs @@ -0,0 +1,148 @@ +//! Keeps the concurrent-read harnesses working: runs `concurrent_read`, the +//! h5py script (threads and processes) and the comparison script end to end +//! on tiny files. h5py reading the files also checks, element by element at +//! spot positions, that both harnesses generate the same data and slabs. +//! +//! The h5py half is skipped when python3 with h5py is unavailable, unless +//! `CLAWHDF5_REQUIRE_INTEROP=1`; `CLAWHDF5_PYTHON` picks the interpreter. + +use std::path::{Path, PathBuf}; +use std::process::Command; + +fn python() -> String { + std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string()) +} + +fn interop_required() -> bool { + std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1") +} + +fn python_available() -> bool { + Command::new(python()) + .args(["-c", "import h5py, numpy"]) + .output() + .map(|o| o.status.success()) + .unwrap_or(false) +} + +fn scripts() -> PathBuf { + Path::new(env!("CARGO_MANIFEST_DIR")).join("scripts") +} + +fn run(cmd: &mut Command) -> String { + let out = cmd.output().expect("spawn"); + assert!( + out.status.success(), + "{cmd:?} failed\nSTDOUT:\n{}\nSTDERR:\n{}", + String::from_utf8_lossy(&out.stdout), + String::from_utf8_lossy(&out.stderr) + ); + String::from_utf8_lossy(&out.stdout).into_owned() +} + +const SMALL: [&str; 8] = [ + "--threads", + "1,2", + "--slabs", + "8", + "--reps", + "1", + "--slab", + "64", +]; + +fn results(path: &Path) -> serde_json::Value { + serde_json::from_str(&std::fs::read_to_string(path).unwrap()).unwrap() +} + +#[test] +fn harnesses_run_end_to_end_on_tiny_files() { + let dir = tempfile::TempDir::new().unwrap(); + let data = dir.path().join("data"); + let claw = dir.path().join("claw.json"); + + let bin = env!("CARGO_BIN_EXE_concurrent_read"); + run(Command::new(bin) + .arg("--dir") + .arg(&data) + .args(["--datasets", "3", "--mib", "1"]) + .args(SMALL) + .arg("--json") + .arg(&claw)); + // Second run reuses the files (and exercises --cold). + let out = Command::new(bin) + .arg("--dir") + .arg(&data) + .args(["--datasets", "3", "--mib", "1", "--cold"]) + .args(SMALL) + .output() + .unwrap(); + assert!(out.status.success()); + assert!(String::from_utf8_lossy(&out.stderr).contains("reusing")); + + let doc = results(&claw); + assert_eq!(doc["tool"], "clawhdf5"); + // 2 layouts x 2 modes x 2 thread counts. + assert_eq!(doc["results"].as_array().unwrap().len(), 8); + for r in doc["results"].as_array().unwrap() { + assert!(r["mb_s"].as_f64().unwrap() > 0.0, "{r}"); + } + + if !python_available() { + assert!( + !interop_required(), + "CLAWHDF5_REQUIRE_INTEROP=1 but {} has no h5py", + python() + ); + eprintln!("skipping the h5py half: no h5py in {}", python()); + return; + } + let mut jsons = vec![claw]; + for executor in ["threads", "processes"] { + let out = dir.path().join(format!("h5py-{executor}.json")); + run(Command::new(python()) + .arg(scripts().join("concurrent_read_h5py.py")) + .arg("--dir") + .arg(&data) + .args(["--executor", executor]) + .args(SMALL) + .arg("--json") + .arg(&out)); + let doc = results(&out); + assert_eq!(doc["tool"], format!("h5py-{executor}")); + assert_eq!(doc["results"].as_array().unwrap().len(), 8); + jsons.push(out); + } + let table = run(Command::new(python()) + .arg(scripts().join("compare_concurrent_read.py")) + .args(&jsons)); + assert!(table.contains("| deflate | same | 2 |"), "{table}"); + assert!(table.contains("clawhdf5 / h5py-processes"), "{table}"); + + // A different workload must not be compared. + let other = dir.path().join("other.json"); + run(Command::new(python()) + .arg(scripts().join("concurrent_read_h5py.py")) + .arg("--dir") + .arg(&data) + .args([ + "--threads", + "1", + "--slabs", + "4", + "--reps", + "1", + "--slab", + "64", + ]) + .arg("--json") + .arg(&other)); + let out = Command::new(python()) + .arg(scripts().join("compare_concurrent_read.py")) + .arg(&jsons[0]) + .arg(&other) + .output() + .unwrap(); + assert!(!out.status.success()); + assert!(String::from_utf8_lossy(&out.stderr).contains("slabs")); +} diff --git a/crates/clawhdf5-format/Cargo.toml b/crates/clawhdf5-format/Cargo.toml index a88b734..75fd233 100644 --- a/crates/clawhdf5-format/Cargo.toml +++ b/crates/clawhdf5-format/Cargo.toml @@ -22,6 +22,13 @@ zstd = { version = "0.13", optional = true } blake3 = { version = "1", optional = true } libaec-sys = { path = "../libaec-sys", version = "0.1", optional = true } pco = { version = "1.0", optional = true } +# Pure-Rust Zstandard, for the plugin filters that embed zstd (bitshuffle, +# blosc). The `zstd` feature (filter 32015) links libzstd instead. +ruzstd = { version = "0.9", optional = true } +# bzip2 with its default backend, libbz2-rs-sys: a pure-Rust port of +# libbzip2 (no C is compiled, despite the -sys name). +bzip2 = { version = "0.6", optional = true } +snap = { version = "1", optional = true } [dev-dependencies] half = { workspace = true } @@ -37,7 +44,7 @@ harness = false # Deflate backend: `zlib-rs` (pure Rust) by default. `fast-deflate` selects # zlib-ng instead (C, built with cmake); flate2 prefers a C zlib whenever one # is enabled, so turning it on anywhere in the build overrides the default. -default = ["std", "checksum", "deflate", "provenance", "zlib-rs", "system-zlib-decompress"] +default = ["std", "checksum", "deflate", "provenance", "zlib-rs", "system-zlib-decompress", "lzf"] std = [] checksum = [] deflate = ["flate2"] @@ -56,6 +63,17 @@ zstd = ["dep:zstd"] blake3_hash = ["blake3"] szip = ["libaec-sys"] pcodec = ["dep:pco"] +# Plugin filters, pure Rust. LZF (32000) is h5py's built-in compression; it +# has no dependencies, so it is on by default. +lzf = [] +# Bitshuffle (32008), with its LZ4 and Zstandard modes. +bitshuffle = ["lz4_flex", "ruzstd"] +# bzip2 (307). +bzip2 = ["dep:bzip2", "std"] +# Blosc 1 (32001) with its BloscLZ, LZ4, Snappy, Zlib and Zstandard codecs. +blosc = ["lz4_flex", "ruzstd", "snap", "deflate", "std"] +# Every plugin filter above. +plugin-filters = ["lzf", "bitshuffle", "bzip2", "blosc"] [[bench]] name = "parallel_decompress_bench" diff --git a/crates/clawhdf5-format/src/attribute.rs b/crates/clawhdf5-format/src/attribute.rs index c844c3c..11bde3e 100644 --- a/crates/clawhdf5-format/src/attribute.rs +++ b/crates/clawhdf5-format/src/attribute.rs @@ -362,6 +362,18 @@ fn extract_name(bytes: &[u8]) -> String { String::from_utf8_lossy(&bytes[..end]).into_owned() } +/// An attribute's datatype gets libhdf5's extra check for a header without +/// a checksum (see [`Datatype::check_unused_bits`]). +fn check_in_header( + attr: AttributeMessage, + header: &ObjectHeader, +) -> Result { + if header.version == 1 { + attr.datatype.check_unused_bits()?; + } + Ok(attr) +} + /// Extract all attribute messages from an object header. pub fn extract_attributes( header: &ObjectHeader, @@ -371,7 +383,7 @@ pub fn extract_attributes( for msg in &header.messages { if msg.msg_type == MessageType::Attribute { let attr = AttributeMessage::parse(&msg.data, length_size)?; - attrs.push(attr); + attrs.push(check_in_header(attr, header)?); } } Ok(attrs) @@ -465,6 +477,7 @@ fn extract_attributes_with( } else { AttributeMessage::parse_in_file(&msg.data, file_data, offset_size, length_size) }; + let attr = attr.and_then(|a| check_in_header(a, header)); match attr { Ok(attr) => attrs.push(attr), Err(e) => on_error(e)?, @@ -573,7 +586,8 @@ mod tests { /// Build an f64 LE datatype message. fn build_f64_dt() -> Vec { - let mut buf = build_dt_header(1, 1, [0x00, 0x00, 0x02], 8); + // Sign bit 63 (bits 8-15 of the class bits). + let mut buf = build_dt_header(1, 1, [0x20, 63, 0x00], 8); let mut props = [0u8; 12]; props[2..4].copy_from_slice(&64u16.to_le_bytes()); // bit_precision props[4] = 52; // exp_location diff --git a/crates/clawhdf5-format/src/chunked_read.rs b/crates/clawhdf5-format/src/chunked_read.rs index 3a3dc05..cbbaa69 100644 --- a/crates/clawhdf5-format/src/chunked_read.rs +++ b/crates/clawhdf5-format/src/chunked_read.rs @@ -15,7 +15,7 @@ use crate::datatype::Datatype; use crate::error::FormatError; use crate::extensible_array::{ExtensibleArrayHeader, read_extensible_array_chunks}; use crate::filter_pipeline::FilterPipeline; -use crate::filters::{all_filters_skipped, decompress_chunk_masked}; +use crate::filters::{all_filters_skipped, decompress_chunk_exact}; use crate::fixed_array::{FixedArrayHeader, read_fixed_array_chunks}; #[cfg(feature = "std")] use std::sync::Arc; @@ -65,12 +65,13 @@ fn decompress_all_chunks( let raw_chunk = &file_data[c_addr..c_addr + size]; let decompressed = if let Some(pl) = pipeline { - decompress_chunk_masked( + decompress_chunk_exact( raw_chunk, pl, chunk_total_bytes, element_size, chunk_info.filter_mask, + &chunk_info.offsets, )? } else { raw_chunk.to_vec() @@ -148,6 +149,96 @@ pub(crate) fn checked_byte_len(elements: u64, elem_size: usize) -> Result Result<(usize, Vec), FormatError> { + let rank = chunk_dimensions.len().checked_sub(1).ok_or_else(|| { + FormatError::InvalidChunkDimensions("chunked layout has no dimensions".into()) + })?; + if dataspace.dimensions.len() != rank { + return Err(FormatError::InvalidChunkDimensions(format!( + "dimensionality of chunks doesn't match the dataspace (chunk rank {rank}, \ + dataspace rank {})", + dataspace.dimensions.len() + ))); + } + let spatial = &chunk_dimensions[..rank]; + if let Some(d) = spatial.iter().position(|&c| c == 0) { + return Err(FormatError::InvalidChunkDimensions(format!( + "chunk size must be > 0, dim = {d}" + ))); + } + let bytes = spatial + .iter() + .fold(elem_size as u128, |acc, &c| acc * u128::from(c)); + if layout_version < 4 && bytes > u128::from(u32::MAX) { + return Err(FormatError::InvalidChunkDimensions(format!( + "chunk size must be < 4GB with v1 b-tree index (chunk {spatial:?} of {elem_size}-byte elements)" + ))); + } + Ok((rank, spatial.iter().map(|&c| c as usize).collect())) +} + +/// The size of one element of `dt` as stored in the file: a +/// variable-length element is its length (4), a global heap address +/// (`offset_size`) and an index (4), not the 16 of [`Datatype::type_size`]. +fn stored_element_size(dt: &Datatype, offset_size: u8) -> u64 { + match dt { + Datatype::VariableLength { .. } => 8 + u64::from(offset_size), + Datatype::Array { + base_type, + dimensions, + } => dimensions + .iter() + .fold(stored_element_size(base_type, offset_size), |acc, &d| { + acc.saturating_mul(u64::from(d)) + }), + _ => u64::from(dt.type_size()), + } +} + +/// A chunked layout records the element size as its last dimension, and +/// libhdf5 refuses a dataset whose datatype has another size +/// (`H5D__chunk_set_sizes`: "stored datatype size in chunk layout does not +/// match datatype description"). Reading it anyway laid the chunks out with +/// the wrong element size. +pub(crate) fn check_chunk_element_size( + layout: &DataLayout, + datatype: &Datatype, + offset_size: u8, +) -> Result<(), FormatError> { + let DataLayout::Chunked { + chunk_dimensions, .. + } = layout + else { + return Ok(()); + }; + let Some(&stored) = chunk_dimensions.last() else { + return Ok(()); + }; + let expected = stored_element_size(datatype, offset_size); + if u64::from(stored) != expected { + return Err(FormatError::InvalidChunkDimensions(format!( + "stored datatype size in chunk layout does not match datatype description \ + (layout {stored} bytes, datatype {expected})" + ))); + } + Ok(()) +} + /// Product of chunk dimensions times the element size, overflow-checked. pub(crate) fn checked_chunk_byte_len( chunk_dims: &[usize], @@ -222,7 +313,76 @@ pub fn collect_chunk_info( offset_size: u8, length_size: u8, ) -> Result, FormatError> { - collect_chunk_info_inner(file_data, btree_address, ndims, offset_size, length_size, 0) + collect_chunk_info_inner( + file_data, + btree_address, + ndims, + None, + offset_size, + length_size, + 0, + ) +} + +/// [`collect_chunk_info`] for a layout with these `chunk_dimensions` (the +/// layout message's list, element size last), checking every key of the +/// B-tree as libhdf5 does (`H5D__btree_decode_key`): each coordinate offset +/// must be a multiple of its chunk dimension. That includes the keys that +/// only bound a node (internal-node keys and each node's final key), which +/// is where a corrupt chunk dimension shows when the chunks themselves all +/// start at offset 0 in that dimension (`cve-2018-11205`). A key that fails +/// ("bad coordinate offset") means a corrupt index or chunk dimension; the +/// chunks were read at the wrong place, or the dataset read as fill values. +pub fn collect_chunk_info_checked( + file_data: &[u8], + btree_address: u64, + chunk_dimensions: &[u32], + offset_size: u8, + length_size: u8, +) -> Result, FormatError> { + collect_chunk_info_inner( + file_data, + btree_address, + chunk_dimensions.len(), + Some(chunk_dimensions), + offset_size, + length_size, + 0, + ) +} + +/// Check one v1 B-tree chunk key's offsets (see +/// [`collect_chunk_info_checked`]). +fn check_key_offsets(offsets: &[u64], chunk_dimensions: &[u32]) -> Result<(), FormatError> { + for (&offset, &dim) in offsets.iter().zip(chunk_dimensions) { + if dim == 0 || offset % u64::from(dim) != 0 { + return Err(FormatError::ChunkedReadError(format!( + "bad coordinate offset {offsets:?} for chunk dimensions {chunk_dimensions:?}" + ))); + } + } + Ok(()) +} + +/// Read the `ndims` 8-byte offsets of the chunk key at `pos` (after its +/// chunk size and filter mask) and check them when `chunk_dimensions` is +/// given. +fn read_key_offsets( + file_data: &[u8], + pos: usize, + ndims: usize, + chunk_dimensions: Option<&[u32]>, +) -> Result, FormatError> { + let mut offsets = Vec::with_capacity(ndims); + let mut kp = pos + 8; + for _ in 0..ndims { + offsets.push(read_offset(file_data, kp, CHUNK_KEY_OFFSET_SIZE)?); + kp += CHUNK_KEY_OFFSET_SIZE as usize; + } + if let Some(dims) = chunk_dimensions { + check_key_offsets(&offsets, dims)?; + } + Ok(offsets) } /// Width of each chunk offset in a v1 chunk B-tree key, independent of the @@ -237,6 +397,7 @@ fn collect_chunk_info_inner( file_data: &[u8], btree_address: u64, ndims: usize, + chunk_dimensions: Option<&[u32]>, offset_size: u8, _length_size: u8, depth: usize, @@ -296,12 +457,7 @@ fn collect_chunk_info_inner( file_data[pos + 6], file_data[pos + 7], ]); - let mut offsets = Vec::with_capacity(ndims); - let mut kp = pos + 8; - for _ in 0..ndims { - offsets.push(read_offset(file_data, kp, CHUNK_KEY_OFFSET_SIZE)?); - kp += CHUNK_KEY_OFFSET_SIZE as usize; - } + let offsets = read_key_offsets(file_data, pos, ndims, chunk_dimensions)?; pos += key_size; // Parse child address @@ -315,7 +471,8 @@ fn collect_chunk_info_inner( address, }); } - // Skip final key + // The final key only bounds the node; libhdf5 still checks it. + read_key_offsets(file_data, pos, ndims, chunk_dimensions)?; Ok(chunks) } else { // Internal node: recurse into children @@ -324,11 +481,13 @@ fn collect_chunk_info_inner( let mut child_addrs = Vec::with_capacity(entries_used); for _ in 0..entries_used { - pos += key_size; // skip key + read_key_offsets(file_data, pos, ndims, chunk_dimensions)?; + pos += key_size; let child_addr = read_offset(file_data, pos, offset_size)?; child_addrs.push(child_addr); pos += os; } + read_key_offsets(file_data, pos, ndims, chunk_dimensions)?; let mut all_chunks = Vec::new(); for child_addr in child_addrs { @@ -336,6 +495,7 @@ fn collect_chunk_info_inner( file_data, child_addr, ndims, + chunk_dimensions, offset_size, _length_size, depth + 1, @@ -549,30 +709,13 @@ pub fn list_chunks( .ok_or_else(|| FormatError::ChunkedReadError("no address for chunked layout".into()))?; // Both v3 and v4 include element size as last dim (rank+1) - let ndims = chunk_dimensions.len(); - let rank = ndims - .checked_sub(1) - .ok_or_else(|| FormatError::ChunkedReadError("chunked layout has no dimensions".into()))?; - let chunk_dims: Vec = chunk_dimensions[..rank] - .iter() - .map(|&d| d as usize) - .collect(); - + let (rank, chunk_dims) = chunk_geometry(chunk_dimensions, version, dataspace, elem_size)?; let ds_dims: Vec = dataspace.dimensions.iter().map(|&d| d as usize).collect(); - if ds_dims.len() != rank { - return Err(FormatError::ChunkedReadError(format!( - "rank mismatch: dataspace has {} dims, layout has {} chunk dims (rank={})", - ds_dims.len(), - chunk_dimensions.len(), - rank - ))); - } // Collect chunks based on version and index type let mut chunks = match (version, chunk_index_type) { (3, _) => { - let ndims = chunk_dimensions.len(); // rank+1 - collect_chunk_info(file_data, addr, ndims, offset_size, length_size)? + collect_chunk_info_checked(file_data, addr, chunk_dimensions, offset_size, length_size)? } (4, Some(1)) => { // Single chunk — one chunk covering the entire dataset @@ -679,6 +822,7 @@ pub fn read_chunked_data( offset_size: u8, length_size: u8, ) -> Result, FormatError> { + check_chunk_element_size(layout, datatype, offset_size)?; let elem_size = datatype.type_size() as usize; let (chunks, chunk_dims) = list_chunks( file_data, @@ -802,12 +946,13 @@ pub fn read_chunked_data_cached( length_size: u8, cache: &ChunkCache, ) -> Result, FormatError> { - let (chunk_dimensions, addr_opt) = match layout { + let (chunk_dimensions, version, addr_opt) = match layout { DataLayout::Chunked { chunk_dimensions, + version, btree_address, .. - } => (chunk_dimensions, *btree_address), + } => (chunk_dimensions, *version, *btree_address), _ => { return Err(FormatError::ChunkedReadError( "expected chunked layout".into(), @@ -818,25 +963,10 @@ pub fn read_chunked_data_cached( let addr = addr_opt .ok_or_else(|| FormatError::ChunkedReadError("no address for chunked layout".into()))?; + check_chunk_element_size(layout, datatype, offset_size)?; let elem_size = datatype.type_size() as usize; - let ndims = chunk_dimensions.len(); - let rank = ndims - .checked_sub(1) - .ok_or_else(|| FormatError::ChunkedReadError("chunked layout has no dimensions".into()))?; - let chunk_dims: Vec = chunk_dimensions[..rank] - .iter() - .map(|&d| d as usize) - .collect(); - + let (rank, chunk_dims) = chunk_geometry(chunk_dimensions, version, dataspace, elem_size)?; let ds_dims: Vec = dataspace.dimensions.iter().map(|&d| d as usize).collect(); - if ds_dims.len() != rank { - return Err(FormatError::ChunkedReadError(format!( - "rank mismatch: dataspace has {} dims, layout has {} chunk dims (rank={})", - ds_dims.len(), - chunk_dimensions.len(), - rank - ))); - } // The per-file cache is shared across datasets (and threads); every // lookup is keyed by this dataset's chunk-index address, so another @@ -932,12 +1062,13 @@ pub fn read_chunked_data_cached( let cache_them = total_bytes <= cache.max_bytes(); if let Some(pl) = pipeline { let decode = |c: &&ChunkInfo| -> Result, FormatError> { - decompress_chunk_masked( + decompress_chunk_exact( raw_bytes(c)?, pl, chunk_total_bytes, elem_size as u32, c.filter_mask, + &c.offsets, ) }; for batch in misses.chunks(DECODE_BATCH) { @@ -1123,12 +1254,13 @@ pub fn read_chunked_data_sweep( cache: &ChunkCache, sweep: &mut SweepContext, ) -> Result, FormatError> { - let (chunk_dimensions, addr_opt) = match layout { + let (chunk_dimensions, version, addr_opt) = match layout { DataLayout::Chunked { chunk_dimensions, + version, btree_address, .. - } => (chunk_dimensions, *btree_address), + } => (chunk_dimensions, *version, *btree_address), _ => { return Err(FormatError::ChunkedReadError( "expected chunked layout".into(), @@ -1139,25 +1271,10 @@ pub fn read_chunked_data_sweep( let addr = addr_opt .ok_or_else(|| FormatError::ChunkedReadError("no address for chunked layout".into()))?; + check_chunk_element_size(layout, datatype, offset_size)?; let elem_size = datatype.type_size() as usize; - let ndims = chunk_dimensions.len(); - let rank = ndims - .checked_sub(1) - .ok_or_else(|| FormatError::ChunkedReadError("chunked layout has no dimensions".into()))?; - let chunk_dims: Vec = chunk_dimensions[..rank] - .iter() - .map(|&d| d as usize) - .collect(); - + let (rank, chunk_dims) = chunk_geometry(chunk_dimensions, version, dataspace, elem_size)?; let ds_dims: Vec = dataspace.dimensions.iter().map(|&d| d as usize).collect(); - if ds_dims.len() != rank { - return Err(FormatError::ChunkedReadError(format!( - "rank mismatch: dataspace has {} dims, layout has {} chunk dims (rank={})", - ds_dims.len(), - chunk_dimensions.len(), - rank - ))); - } // The per-file cache is shared across datasets (and threads); every // lookup is keyed by this dataset's chunk-index address, so another @@ -1217,12 +1334,13 @@ pub fn read_chunked_data_sweep( ensure_len(file_data, c_addr, size)?; let raw_chunk = &file_data[c_addr..c_addr + size]; let dec = if let Some(pl) = pipeline { - decompress_chunk_masked( + decompress_chunk_exact( raw_chunk, pl, chunk_total_bytes, elem_size as u32, chunk_info.filter_mask, + &coord, )? } else { raw_chunk.to_vec() @@ -1275,12 +1393,13 @@ pub fn read_chunked_data_indexed( length_size: u8, cache: &ChunkCache, ) -> Result, FormatError> { - let (chunk_dimensions, addr_opt) = match layout { + let (chunk_dimensions, version, addr_opt) = match layout { DataLayout::Chunked { chunk_dimensions, + version, btree_address, .. - } => (chunk_dimensions, *btree_address), + } => (chunk_dimensions, *version, *btree_address), _ => { return Err(FormatError::ChunkedReadError( "expected chunked layout".into(), @@ -1291,25 +1410,10 @@ pub fn read_chunked_data_indexed( let addr = addr_opt .ok_or_else(|| FormatError::ChunkedReadError("no address for chunked layout".into()))?; + check_chunk_element_size(layout, datatype, offset_size)?; let elem_size = datatype.type_size() as usize; - let ndims = chunk_dimensions.len(); - let rank = ndims - .checked_sub(1) - .ok_or_else(|| FormatError::ChunkedReadError("chunked layout has no dimensions".into()))?; - let chunk_dims: Vec = chunk_dimensions[..rank] - .iter() - .map(|&d| d as usize) - .collect(); - + let (rank, chunk_dims) = chunk_geometry(chunk_dimensions, version, dataspace, elem_size)?; let ds_dims: Vec = dataspace.dimensions.iter().map(|&d| d as usize).collect(); - if ds_dims.len() != rank { - return Err(FormatError::ChunkedReadError(format!( - "rank mismatch: dataspace has {} dims, layout has {} chunk dims (rank={})", - ds_dims.len(), - chunk_dimensions.len(), - rank - ))); - } // Chunk index and assembly plan for this dataset, built on first access // and kept per dataset (keyed by chunk-index address) in the shared cache. @@ -1346,12 +1450,13 @@ pub fn read_chunked_data_indexed( ensure_len(file_data, c_addr, size)?; let raw_chunk = &file_data[c_addr..c_addr + size]; let decompressed = if let Some(pl) = pipeline { - decompress_chunk_masked( + decompress_chunk_exact( raw_chunk, pl, chunk_total_bytes, elem_size as u32, *filter_mask, + coord, )? } else { raw_chunk.to_vec() @@ -1620,11 +1725,12 @@ mod tests { write_offset(&mut buf, chunk.address, offset_size); } - // Final key (dummy) + // Final key (its offsets must be on the chunk grid, as libhdf5 + // checks; 0 always is) buf.extend_from_slice(&0u32.to_le_bytes()); // chunk_size buf.extend_from_slice(&0u32.to_le_bytes()); // filter_mask for _ in 0..ndims { - write_offset(&mut buf, u64::MAX, 8); + write_offset(&mut buf, 0, 8); } buf @@ -1632,6 +1738,48 @@ mod tests { // --- ChunkInfo collection tests --- + #[test] + fn checked_collection_refuses_keys_off_the_chunk_grid() { + let chunk = |offsets: Vec, address| ChunkInfo { + chunk_size: 80, + filter_mask: 0, + offsets, + address, + }; + let good = + build_chunk_btree_leaf(&[chunk(vec![0, 0], 0x100), chunk(vec![10, 0], 0x200)], 2, 8); + assert_eq!( + collect_chunk_info_checked(&good, 0, &[10, 8], 8, 8) + .unwrap() + .len(), + 2 + ); + // A chunk key off the grid. + let bad = + build_chunk_btree_leaf(&[chunk(vec![0, 0], 0x100), chunk(vec![7, 0], 0x200)], 2, 8); + assert!(collect_chunk_info(&bad, 0, 2, 8, 8).is_ok()); + assert!(matches!( + collect_chunk_info_checked(&bad, 0, &[10, 8], 8, 8), + Err(FormatError::ChunkedReadError(m)) if m.starts_with("bad coordinate offset") + )); + // cve-2018-11205: the chunks all start at 0 in dimension 1, and only + // the node's final key shows the chunk dimension is wrong. + let mut two_d = build_chunk_btree_leaf( + &[chunk(vec![0, 0, 0], 0x100), chunk(vec![10, 0, 0], 0x200)], + 3, + 8, + ); + // Final key: (20, 20, 0), the end of a 20 x 20 dataset. + let final_key = two_d.len() - 24; + two_d[final_key..final_key + 8].copy_from_slice(&20u64.to_le_bytes()); + two_d[final_key + 8..final_key + 16].copy_from_slice(&20u64.to_le_bytes()); + assert!(collect_chunk_info_checked(&two_d, 0, &[10, 20, 4], 8, 8).is_ok()); + assert!(matches!( + collect_chunk_info_checked(&two_d, 0, &[10, 32788, 4], 8, 8), + Err(FormatError::ChunkedReadError(m)) if m.starts_with("bad coordinate offset [20, 20, 0]") + )); + } + #[test] fn collect_two_chunks_from_leaf() { let ndims = 2; // rank+1 for 1D dataset @@ -1750,6 +1898,51 @@ mod tests { use crate::dataspace::{Dataspace, DataspaceType}; use crate::datatype::{Datatype, DatatypeByteOrder}; + #[test] + fn chunk_geometry_matches_libhdf5_open_checks() { + let space = |dims: &[u64]| Dataspace { + space_type: DataspaceType::Simple, + rank: dims.len() as u8, + dimensions: dims.to_vec(), + max_dimensions: None, + }; + for v in [3, 4] { + assert_eq!( + chunk_geometry(&[4, 5, 8], v, &space(&[10, 10]), 8).unwrap(), + (2, vec![4, 5]) + ); + // Rank mismatch. + assert!(matches!( + chunk_geometry(&[4, 8], v, &space(&[10, 10]), 8), + Err(FormatError::InvalidChunkDimensions(m)) if m.contains("doesn't match") + )); + // Zero dimension (a layout built in memory, bypassing the parser). + assert!(matches!( + chunk_geometry(&[4, 0, 8], v, &space(&[10, 10]), 8), + Err(FormatError::InvalidChunkDimensions(m)) if m.contains("must be > 0") + )); + assert!(chunk_geometry(&[0xFFFF_FFFF, 1], v, &space(&[10]), 1).is_ok()); + } + // With a v1 B-tree index (layout version 3) the largest chunk is + // 4 GiB - 1 bytes: 0x80000000 x 4-byte elements (8 GiB) is refused. + // These dims used to hang the reader. + assert!(matches!( + chunk_geometry(&[0x8000_0000, 4], 3, &space(&[10]), 4), + Err(FormatError::InvalidChunkDimensions(m)) if m.contains("4GB with v1 b-tree") + )); + assert!(matches!( + chunk_geometry(&[0xFFFF_FFFF, 0xFFFF_FFFF, 1], 3, &space(&[10, 10]), 1), + Err(FormatError::InvalidChunkDimensions(m)) if m.contains("4GB with v1 b-tree") + )); + // The other chunk indexes (layout version 4, and 5, which is read as + // 4) allow chunks of 4 GiB and more; HDF5 2.0 writes them. + assert_eq!( + chunk_geometry(&[0x2000_0001, 8], 4, &space(&[10]), 8).unwrap(), + (1, vec![0x2000_0001]) + ); + assert!(chunk_geometry(&[0xFFFF_FFFF, 0xFFFF_FFFF, 1], 4, &space(&[10, 10]), 1).is_ok()); + } + fn make_f64_type() -> Datatype { Datatype::FloatingPoint { size: 8, @@ -1863,8 +2056,8 @@ mod tests { let file_data = vec![0u8; 64]; let result = read_chunked_data(&file_data, &layout, &dataspace, &datatype, None, 8, 8); assert!( - matches!(result, Err(FormatError::ChunkedReadError(_))), - "expected a clean ChunkedReadError, got {result:?}" + matches!(result, Err(FormatError::InvalidChunkDimensions(_))), + "expected a clean InvalidChunkDimensions, got {result:?}" ); } diff --git a/crates/clawhdf5-format/src/chunked_write.rs b/crates/clawhdf5-format/src/chunked_write.rs index ff5081e..05383a6 100644 --- a/crates/clawhdf5-format/src/chunked_write.rs +++ b/crates/clawhdf5-format/src/chunked_write.rs @@ -12,8 +12,9 @@ use crate::chunk_grid::ChunkGrid; use crate::ea_writer; use crate::error::FormatError; use crate::filter_pipeline::{ - FILTER_DEFLATE, FILTER_FLETCHER32, FILTER_LZ4, FILTER_PCODEC, FILTER_PCODEC_NAME, - FILTER_SHUFFLE, FILTER_ZSTD, FilterDescription, FilterPipeline, + FILTER_BITSHUFFLE, FILTER_BLOSC, FILTER_BZIP2, FILTER_DEFLATE, FILTER_FLETCHER32, FILTER_LZ4, + FILTER_LZF, FILTER_PCODEC, FILTER_PCODEC_NAME, FILTER_SHUFFLE, FILTER_ZSTD, FilterDescription, + FilterPipeline, }; use crate::filters::compress_chunk; /// Round a file offset up to the next cache-line boundary. @@ -48,6 +49,167 @@ pub struct ChunkOptions { /// Pcodec lossless numerical compression. Private, unregistered filter /// ID [`FILTER_PCODEC`] (480): only clawhdf5 can read it. pub pcodec: bool, + /// A plugin compression filter (LZF, ...). Takes priority over the + /// codecs above. Each needs its cargo feature to be written. + pub plugin: Option, +} + +/// A compression filter from the common HDF5 plugin set, written in the +/// format the libhdf5 plugin (h5py / hdf5plugin) reads. +#[derive(Debug, Clone, PartialEq, Eq)] +#[non_exhaustive] +pub enum PluginFilter { + /// LZF (filter 32000), h5py's built-in `compression="lzf"`. Needs the + /// `lzf` feature. + Lzf, + /// Bitshuffle (filter 32008): a bit transpose of each block of + /// `block_size` elements (0 = bitshuffle's default, else a multiple of + /// 8), optionally compressed. Needs the `bitshuffle` feature. + Bitshuffle { + /// Block size in elements; 0 for the default. + block_size: u32, + /// Compression after the transpose. + compression: BitshuffleCompression, + }, + /// bzip2 (filter 307) at block size `level` (1-9). Needs the `bzip2` + /// feature. + Bzip2 { + /// Block size 1-9 (9 = hdf5plugin's default). + level: u32, + }, + /// Blosc 1 (filter 32001): `codec` at `level` (0-9; 0 stores), after + /// `shuffle`. Needs the `blosc` feature. + Blosc { + /// The codec inside the Blosc frame. + codec: BloscCodec, + /// Compression level 0-9 (0 stores the data uncompressed). + level: u32, + /// The shuffle Blosc applies first. + shuffle: BloscShuffle, + }, +} + +/// The codec inside a Blosc frame that clawhdf5 can write. (It reads +/// BloscLZ too, but cannot write it.) +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum BloscCodec { + /// LZ4. + Lz4, + /// Snappy. + Snappy, + /// Zlib, at the Blosc level. + Zlib, + /// Zstandard (clawhdf5's pure-Rust encoder has one level, about zstd 1). + Zstd, +} + +/// The shuffle Blosc applies before compressing. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum BloscShuffle { + /// None. + None, + /// Byte shuffle (Blosc's default). + Byte, + /// Bit shuffle. + Bit, +} + +/// What bitshuffle compresses its blocks with. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum BitshuffleCompression { + /// Transpose only. + None, + /// LZ4 (bitshuffle's `cname="lz4"`, the common choice). + Lz4, + /// Zstandard. clawhdf5's pure-Rust encoder has a single level (about + /// zstd's level 1); `level` is recorded in the file for other writers. + Zstd { + /// Level recorded in `cd_values[5]`. + level: u32, + }, +} + +impl PluginFilter { + /// Whether the filter reorders bytes itself, so the automatic shuffle + /// pre-filter would only get in its way. + fn shuffles_itself(&self) -> bool { + match self { + PluginFilter::Lzf => false, + PluginFilter::Bitshuffle { .. } => true, + PluginFilter::Bzip2 { .. } => false, + PluginFilter::Blosc { .. } => true, + } + } + + /// The pipeline entry for this filter. `chunk_bytes` is one chunk's + /// uncompressed size (0 if unknown). + fn description(&self, element_size: u32, chunk_bytes: u32) -> FilterDescription { + match self { + // h5py's lzf_set_local: filter version, liblzf version, chunk + // size in bytes. Optional, as h5py flags it: a chunk the filter + // cannot shrink may then be stored unfiltered. + PluginFilter::Lzf => FilterDescription { + filter_id: FILTER_LZF, + name: Some("lzf".into()), + flags: 1, + client_data: vec![4, 0x0105, chunk_bytes], + }, + // bshuf_h5_set_local: version 0.4, element size, block size, + // compression (0 none, 2 LZ4, 3 Zstandard), Zstandard level. + // hdf5-blosc's blosc_set_local: filter revision 2, Blosc format + // 2, type size, chunk size, then level, shuffle, compressor. + PluginFilter::Blosc { + codec, + level, + shuffle, + } => FilterDescription { + filter_id: FILTER_BLOSC, + name: Some("blosc".into()), + flags: 1, + client_data: vec![ + 2, + 2, + element_size, + chunk_bytes, + (*level).min(9), + match shuffle { + BloscShuffle::None => 0, + BloscShuffle::Byte => 1, + BloscShuffle::Bit => 2, + }, + match codec { + BloscCodec::Lz4 => 1, + BloscCodec::Snappy => 3, + BloscCodec::Zlib => 4, + BloscCodec::Zstd => 5, + }, + ], + }, + PluginFilter::Bzip2 { level } => FilterDescription { + filter_id: FILTER_BZIP2, + name: Some("bzip2".into()), + flags: 1, + client_data: vec![(*level).clamp(1, 9)], + }, + PluginFilter::Bitshuffle { + block_size, + compression, + } => { + let mut cd = vec![0, 4, element_size, *block_size]; + match compression { + BitshuffleCompression::None => cd.push(0), + BitshuffleCompression::Lz4 => cd.push(2), + BitshuffleCompression::Zstd { level } => cd.extend([3, *level]), + } + FilterDescription { + filter_id: FILTER_BITSHUFFLE, + name: Some("bitshuffle; see https://github.com/kiyo-masui/bitshuffle".into()), + flags: 1, + client_data: cd, + } + } + } + } } /// Largest chunk the automatic choice produces, in bytes. @@ -92,14 +254,33 @@ impl ChunkOptions { || self.lz4 || self.zstd_level.is_some() || self.pcodec + || self.plugin.is_some() } /// Build a FilterPipeline from the options. pub fn build_pipeline(&self, element_size: u32) -> Option { + self.build_pipeline_for_chunk(element_size, 0) + } + + /// Build a FilterPipeline for chunks of `chunk_bytes` uncompressed bytes + /// (0 if unknown). Some plugin filters record the chunk size in their + /// client data. + pub fn build_pipeline_for_chunk( + &self, + element_size: u32, + chunk_bytes: u32, + ) -> Option { let mut filters = Vec::new(); - let has_compression = - self.deflate_level.is_some() || self.zstd_level.is_some() || self.lz4 || self.pcodec; + let plugin_shuffles = self + .plugin + .as_ref() + .is_some_and(PluginFilter::shuffles_itself); + let has_compression = self.deflate_level.is_some() + || self.zstd_level.is_some() + || self.lz4 + || self.pcodec + || (self.plugin.is_some() && !plugin_shuffles); // Shuffle before compression. Applied if explicitly requested OR if compression // is active and the caller hasn't disabled it — matches h5py default behavior @@ -113,8 +294,11 @@ impl ChunkOptions { }); } - // Compression filters (mutually exclusive, priority: pcodec > zstd > lz4 > deflate) - if self.pcodec { + // Compression filters (mutually exclusive, priority: plugin > pcodec > + // zstd > lz4 > deflate) + if let Some(plugin) = &self.plugin { + filters.push(plugin.description(element_size, chunk_bytes)); + } else if self.pcodec { filters.push(FilterDescription { filter_id: FILTER_PCODEC, name: Some(FILTER_PCODEC_NAME.into()), @@ -383,39 +567,7 @@ fn serialize_v4_single_chunk( let ndims = chunk_dims.len() as u8 + 1; buf.push(ndims); - // dim_size_encoded_length: how many bytes per dimension - // We need to figure out the minimum encoding width - let max_dim = chunk_dims - .iter() - .map(|&d| d as u64) - .chain(core::iter::once(element_size as u64)) - .max() - .unwrap_or(1); - let dim_encoded_len: u8 = if max_dim <= 0xFF { - 1 - } else if max_dim <= 0xFFFF { - 2 - } else { - 4 - }; - buf.push(dim_encoded_len); - - // dimension sizes (chunk dims + element size) - for &d in chunk_dims { - match dim_encoded_len { - 1 => buf.push(d as u8), - 2 => buf.extend_from_slice(&(d as u16).to_le_bytes()), - 4 => buf.extend_from_slice(&d.to_le_bytes()), - _ => {} - } - } - // Element size dimension - match dim_encoded_len { - 1 => buf.push(element_size as u8), - 2 => buf.extend_from_slice(&(element_size as u16).to_le_bytes()), - 4 => buf.extend_from_slice(&element_size.to_le_bytes()), - _ => {} - } + push_v4_chunk_dims(&mut buf, chunk_dims, element_size); // chunk index type = 1 (single chunk) buf.push(1); @@ -465,6 +617,25 @@ fn serialize_v4_fixed_array( /// The part of a v4 chunked layout message before the chunk index type: /// version, class, flags and the chunk dimensions (plus the element size). +/// Append a v4 layout's dimension width and its dimensions (the chunk +/// dimensions, then the element size). Each takes the fewest bytes that hold +/// the largest, as libhdf5 computes it (`H5D__chunk_set_sizes`: +/// `(log2(dim) + 8) / 8`); HDF5 2.0.0 refuses any other width. +pub(crate) fn push_v4_chunk_dims(buf: &mut Vec, chunk_dims: &[u32], element_size: u32) { + let max_dim = chunk_dims + .iter() + .copied() + .chain(core::iter::once(element_size)) + .max() + .unwrap_or(1) + .max(1); + let width = (32 - max_dim.leading_zeros()).div_ceil(8) as usize; + buf.push(width as u8); + for &d in chunk_dims.iter().chain(core::iter::once(&element_size)) { + buf.extend_from_slice(&d.to_le_bytes()[..width]); + } +} + fn layout_v4_chunked_prefix(chunk_dims: &[u32], element_size: u32) -> Vec { let mut buf = Vec::new(); buf.push(4); // version @@ -476,35 +647,7 @@ fn layout_v4_chunked_prefix(chunk_dims: &[u32], element_size: u32) -> Vec { let ndims = chunk_dims.len() as u8 + 1; buf.push(ndims); - let max_dim = chunk_dims - .iter() - .map(|&d| d as u64) - .chain(core::iter::once(element_size as u64)) - .max() - .unwrap_or(1); - let dim_encoded_len: u8 = if max_dim <= 0xFF { - 1 - } else if max_dim <= 0xFFFF { - 2 - } else { - 4 - }; - buf.push(dim_encoded_len); - - for &d in chunk_dims { - match dim_encoded_len { - 1 => buf.push(d as u8), - 2 => buf.extend_from_slice(&(d as u16).to_le_bytes()), - 4 => buf.extend_from_slice(&d.to_le_bytes()), - _ => {} - } - } - match dim_encoded_len { - 1 => buf.push(element_size as u8), - 2 => buf.extend_from_slice(&(element_size as u16).to_le_bytes()), - 4 => buf.extend_from_slice(&element_size.to_le_bytes()), - _ => {} - } + push_v4_chunk_dims(&mut buf, chunk_dims, element_size); buf } @@ -675,7 +818,12 @@ pub fn precompress_chunks( element_size: usize, options: &ChunkOptions, ) -> Result { - let pipeline = options.build_pipeline(element_size as u32); + let chunk_bytes = chunk_dims + .iter() + .try_fold(element_size as u64, |acc, &d| acc.checked_mul(d)) + .and_then(|b| u32::try_from(b).ok()) + .unwrap_or(0); + let pipeline = options.build_pipeline_for_chunk(element_size as u32, chunk_bytes); let has_filters = pipeline.is_some(); let pipeline_message = pipeline.as_ref().map(|pl| pl.serialize()); @@ -1569,6 +1717,35 @@ mod tests { assert_eq!(pl.filters[1].client_data, vec![3]); } + #[test] + fn chunk_options_pipeline_lzf() { + let options = ChunkOptions { + plugin: Some(PluginFilter::Lzf), + ..Default::default() + }; + assert!(options.is_chunked()); + let pl = options.build_pipeline_for_chunk(8, 800).unwrap(); + assert_eq!(pl.filters.len(), 2); + assert_eq!(pl.filters[0].filter_id, FILTER_SHUFFLE); + assert_eq!(pl.filters[1].filter_id, FILTER_LZF); + assert_eq!(pl.filters[1].client_data, vec![4, 0x0105, 800]); + } + + #[test] + fn chunk_options_pipeline_bitshuffle_has_no_auto_shuffle() { + let options = ChunkOptions { + plugin: Some(PluginFilter::Bitshuffle { + block_size: 0, + compression: BitshuffleCompression::Zstd { level: 5 }, + }), + ..Default::default() + }; + let pl = options.build_pipeline(4).unwrap(); + assert_eq!(pl.filters.len(), 1); + assert_eq!(pl.filters[0].filter_id, FILTER_BITSHUFFLE); + assert_eq!(pl.filters[0].client_data, vec![0, 4, 4, 0, 3, 5]); + } + #[test] fn chunk_options_zstd_priority_over_deflate() { let options = ChunkOptions { diff --git a/crates/clawhdf5-format/src/data_layout.rs b/crates/clawhdf5-format/src/data_layout.rs index 60a8243..59a9065 100644 --- a/crates/clawhdf5-format/src/data_layout.rs +++ b/crates/clawhdf5-format/src/data_layout.rs @@ -24,6 +24,34 @@ pub struct VdsMapping { pub virtual_selection: Vec, } +/// Most dimensions a layout message can list (libhdf5 `H5O_LAYOUT_NDIMS`): +/// 32 dataspace dimensions plus the element size. +const MAX_LAYOUT_NDIMS: usize = 33; + +/// libhdf5's checks on a chunked layout message's dimensions +/// (`H5O__layout_decode`): at most [`MAX_LAYOUT_NDIMS`], no dimension 0, and +/// before version 4 at least one dataspace dimension plus the element size. +/// A zero chunk dimension used to read the dataset as all fill values. +fn check_chunk_dims(dims: Vec, layout_version: u8) -> Result, FormatError> { + if dims.len() > MAX_LAYOUT_NDIMS { + return Err(FormatError::InvalidChunkDimensions( + "dimensionality is too large".into(), + )); + } + if layout_version < 4 && dims.len() < 2 { + return Err(FormatError::InvalidChunkDimensions( + "bad dimensions for chunked storage".into(), + )); + } + if let Some(u) = dims.iter().position(|&d| d == 0) { + return Err(FormatError::InvalidChunkDimensions(format!( + "bad chunk dimension value when parsing layout message - chunk dimension must be \ + positive: mesg->u.chunk.dim[{u}] = 0" + ))); + } + Ok(dims) +} + /// Parsed HDF5 data layout message. #[derive(Debug, Clone, PartialEq)] pub enum DataLayout { @@ -394,7 +422,7 @@ impl DataLayout { Ok(DataLayout::Contiguous { address, size }) } _ => Ok(DataLayout::Chunked { - chunk_dimensions: dims, + chunk_dimensions: check_chunk_dims(dims, 2)?, btree_address: address, version: 3, chunk_index_type: None, @@ -457,7 +485,7 @@ impl DataLayout { p += 4; } Ok(DataLayout::Chunked { - chunk_dimensions, + chunk_dimensions: check_chunk_dims(chunk_dimensions, 3)?, btree_address, version: 3, chunk_index_type: None, @@ -506,47 +534,40 @@ impl DataLayout { let dimensionality = data[pos + 1] as usize; let dim_size_encoded_length = data[pos + 2] as usize; let mut p = pos + 3; + if dimensionality > MAX_LAYOUT_NDIMS { + return Err(FormatError::InvalidChunkDimensions( + "dimensionality is too large".into(), + )); + } - // dimension sizes + // Each dimension takes 1 to 8 bytes (libhdf5 writes the + // fewest that hold the largest one, so 3, 5, 6 and 7 occur: + // a chunk dimension of 70 000 takes 3). libhdf5 refuses 0 + // and more than 8. + if dim_size_encoded_length == 0 || dim_size_encoded_length > 8 { + return Err(FormatError::InvalidChunkDimensions( + "encoded chunk dimension size is too large".into(), + )); + } ensure_len(data, p, dimensionality * dim_size_encoded_length)?; let mut chunk_dimensions = Vec::with_capacity(dimensionality); for _ in 0..dimensionality { - let val = match dim_size_encoded_length { - 1 => data[p] as u32, - 2 => u16::from_le_bytes([data[p], data[p + 1]]) as u32, - 4 => u32::from_le_bytes([data[p], data[p + 1], data[p + 2], data[p + 3]]), - 8 => { - // V4 chunked encodes dimension sizes as 8 bytes, but - // our ChunkedStorageV4 stores them as u32. We read only - // the low 4 bytes (little-endian). This silently - // truncates dimensions > 4 GiB, which are not expected - // in practice (HDF5 chunk dimensions are always small). - // If the high bytes are non-zero, the file is malformed - // or uses dimensions we cannot represent. - let high = u32::from_le_bytes([ - data[p + 4], - data[p + 5], - data[p + 6], - data[p + 7], - ]); - if high != 0 { - return Err(FormatError::UnexpectedEof { - expected: p + 8, - available: data.len(), - }); - } - u32::from_le_bytes([data[p], data[p + 1], data[p + 2], data[p + 3]]) - } - _ => { - return Err(FormatError::UnexpectedEof { - expected: p + dim_size_encoded_length, - available: data.len(), - }); - } - }; + let val = data[p..p + dim_size_encoded_length] + .iter() + .rev() + .fold(0u64, |acc, &b| (acc << 8) | u64::from(b)); + // Chunk dimensions are held as u32; HDF5 2.0 can write + // larger ones (layout version 5), which are refused + // rather than truncated. + let val = u32::try_from(val).map_err(|_| { + FormatError::InvalidChunkDimensions(format!( + "chunk dimension {val} is larger than 2^32 - 1, which is not supported" + )) + })?; chunk_dimensions.push(val); p += dim_size_encoded_length; } + let chunk_dimensions = check_chunk_dims(chunk_dimensions, 4)?; // chunk index type ensure_len(data, p, 1)?; @@ -755,6 +776,100 @@ mod tests { ); } + /// A v3 chunked layout message with these dims (element size last). + fn v3_chunked_msg(dims: &[u32]) -> Vec { + let mut buf = vec![3u8, 2, dims.len() as u8]; + buf.extend_from_slice(&0x1000u64.to_le_bytes()); + for d in dims { + buf.extend_from_slice(&d.to_le_bytes()); + } + buf + } + + #[test] + fn chunk_dimensions_are_checked_when_the_layout_is_parsed() { + assert!(DataLayout::parse(&v3_chunked_msg(&[4, 4, 8]), 8, 8).is_ok()); + // A zero chunk dimension used to read as all fill values. + let err = DataLayout::parse(&v3_chunked_msg(&[4, 0, 8]), 8, 8).unwrap_err(); + assert!( + matches!(&err, FormatError::InvalidChunkDimensions(m) if m.contains("dim[1] = 0")), + "{err:?}" + ); + // Only the element-size dimension: libhdf5 "bad dimensions". + assert_eq!( + DataLayout::parse(&v3_chunked_msg(&[8]), 8, 8).unwrap_err(), + FormatError::InvalidChunkDimensions("bad dimensions for chunked storage".into()) + ); + assert_eq!( + DataLayout::parse(&v3_chunked_msg(&[1; 34]), 8, 8).unwrap_err(), + FormatError::InvalidChunkDimensions("dimensionality is too large".into()) + ); + // v1/v2 and v4 messages get the zero check too. + let mut v1 = v1v2_header(1, 2, 2); + v1.extend_from_slice(&0x1000u64.to_le_bytes()); + v1.extend_from_slice(&0u32.to_le_bytes()); + v1.extend_from_slice(&8u32.to_le_bytes()); + assert!(matches!( + DataLayout::parse(&v1, 8, 8), + Err(FormatError::InvalidChunkDimensions(_)) + )); + let mut v4 = vec![4u8, 2, 0, 2, 4]; + v4.extend_from_slice(&0u32.to_le_bytes()); + v4.extend_from_slice(&8u32.to_le_bytes()); + v4.push(3); // fixed array index + v4.push(0); // page bits + v4.extend_from_slice(&0x1000u64.to_le_bytes()); + assert!(matches!( + DataLayout::parse(&v4, 8, 8), + Err(FormatError::InvalidChunkDimensions(_)) + )); + } + + /// A v4 chunked layout (fixed array index) whose `dims` are each + /// encoded in `width` bytes. + fn v4_chunked_msg(width: u8, dims: &[u64]) -> Vec { + let mut m = vec![4u8, 2, 0, dims.len() as u8, width]; + for &d in dims { + m.extend_from_slice(&d.to_le_bytes()[..width.min(8) as usize]); + } + m.push(3); // fixed array index + m.push(0); // page bits + m.extend_from_slice(&0x1000u64.to_le_bytes()); + m + } + + #[test] + fn v4_chunk_dimensions_take_1_to_8_bytes() { + // libhdf5 encodes each dimension in the fewest bytes that hold the + // largest: a chunk dimension of 70 000 takes 3, and 3, 5, 6 and 7 + // were refused ("UnexpectedEof"). + for width in 1..=8u8 { + let dims = [if width >= 3 { 70_000 } else { 200 }, 8]; + let layout = DataLayout::parse(&v4_chunked_msg(width, &dims), 8, 8) + .unwrap_or_else(|e| panic!("width {width}: {e:?}")); + assert!( + matches!(&layout, DataLayout::Chunked { chunk_dimensions, .. } + if chunk_dimensions.iter().map(|&d| u64::from(d)).eq(dims)), + "width {width}: {layout:?}" + ); + } + // libhdf5 refuses 0 and more than 8 bytes. + for width in [0u8, 9] { + assert_eq!( + DataLayout::parse(&v4_chunked_msg(width, &[4, 8]), 8, 8).unwrap_err(), + FormatError::InvalidChunkDimensions( + "encoded chunk dimension size is too large".into() + ) + ); + } + // A dimension past u32 cannot be represented and is refused, not + // truncated. + assert!(matches!( + DataLayout::parse(&v4_chunked_msg(5, &[1 << 32, 8]), 8, 8), + Err(FormatError::InvalidChunkDimensions(m)) if m.contains("2^32") + )); + } + #[test] fn v1v2_rejects_bad_class_dimensionality_and_truncation() { assert_eq!( diff --git a/crates/clawhdf5-format/src/data_read.rs b/crates/clawhdf5-format/src/data_read.rs index 76c2b0e..b8f2ffe 100644 --- a/crates/clawhdf5-format/src/data_read.rs +++ b/crates/clawhdf5-format/src/data_read.rs @@ -303,6 +303,7 @@ pub fn read_raw_data_selection( use crate::selection::Selection; crate::partial_read::validate(selection, &dataspace.dimensions)?; + crate::chunked_read::check_chunk_element_size(layout, datatype, offset_size)?; // Read only what the selection's bounding box touches when that is // possible; everything below is the decode-everything-then-pick path, @@ -360,6 +361,7 @@ pub fn read_raw_data_selection( chunk_index_type, .. } => { + crate::chunked_read::chunk_geometry(chunk_dimensions, *version, dataspace, elem_size)?; // For chunked data, only read chunks that intersect the selection let chunk_dims: Vec = chunk_dimensions.iter().map(|&d| d as u64).collect(); let rank = dims.len(); @@ -400,10 +402,10 @@ pub fn read_raw_data_selection( } else { // v3: B-tree v1 if let Some(addr) = btree_address { - crate::chunked_read::collect_chunk_info( + crate::chunked_read::collect_chunk_info_checked( file_data, *addr, - rank + 1, + chunk_dimensions, offset_size, length_size, )? diff --git a/crates/clawhdf5-format/src/datatype.rs b/crates/clawhdf5-format/src/datatype.rs index ba85afa..12e427f 100644 --- a/crates/clawhdf5-format/src/datatype.rs +++ b/crates/clawhdf5-format/src/datatype.rs @@ -208,6 +208,31 @@ fn offset_bytes_for_size(compound_size: u32) -> usize { } /// Read an unsigned integer of 1, 2, 4, or 8 bytes (LE). +/// The size field of the datatype message at `pos`, as stored (a +/// variable-length type's stored size is not modelled in [`Datatype`]). +fn stored_type_size(data: &[u8], pos: usize) -> Result { + ensure_len(data, pos, 8)?; + Ok(LittleEndian::read_u32(&data[pos + 4..pos + 8])) +} + +/// libhdf5 refuses an array type of more than `H5S_MAX_RANK` (32) +/// dimensions. +fn check_array_rank(ndims: usize) -> Result<(), FormatError> { + if ndims > 32 { + return Err(invalid("too many dimensions for array datatype")); + } + Ok(()) +} + +/// A zero-sized array dimension makes a zero-sized type, which libhdf5 +/// cannot open ("unable to retrieve size of datatype"). +fn check_array_dims(dims: &[u32]) -> Result<(), FormatError> { + if dims.contains(&0) { + return Err(invalid("zero-sized dimension specified")); + } + Ok(()) +} + fn read_uint(data: &[u8], offset: usize, nbytes: usize) -> Result { ensure_len(data, offset, nbytes)?; let slice = &data[offset..offset + nbytes]; @@ -232,10 +257,104 @@ fn read_uint(data: &[u8], offset: usize, nbytes: usize) -> Result) -> FormatError { + FormatError::InvalidDatatype(why.into()) +} + +/// libhdf5's bounds checks on an integer type's bit offset and precision +/// (`H5O__dtype_decode_helper`): both must lie inside the type. (Newer +/// libhdf5 checks bit fields the same way; HDF5 2.0, which h5py 3.16 ships, +/// does not, and opens such a type.) +fn check_integer_bits(size: u32, bit_offset: u16, bit_precision: u16) -> Result<(), FormatError> { + let bits = u64::from(size) * 8; + if u64::from(bit_offset) >= bits { + return Err(invalid("integer offset out of bounds")); + } + if bit_precision == 0 { + return Err(invalid("precision is zero")); + } + if u64::from(bit_offset) + u64::from(bit_precision) > bits { + return Err(invalid("integer offset+precision out of bounds")); + } + Ok(()) +} + +/// Whether the closed bit ranges `[a0, a1]` and `[b0, b1]` share a bit. +fn ranges_overlap(a0: u64, a1: u64, b0: u64, b1: u64) -> bool { + a0 <= b1 && b0 <= a1 +} + +/// libhdf5's checks on a floating-point type's fields: exponent and mantissa +/// must lie inside the type, be non-empty, and not overlap each other or the +/// sign bit. (libhdf5 does not check a float's bit offset and precision.) +/// +/// One libhdf5 check is left out on purpose: a sign bit position outside the +/// type ("sign bit position out of bounds"). clawhdf5 up to v2.7.0 wrote 63 +/// there for every float, so every `f32` it wrote (every agent store's +/// embeddings) would stop opening. The position is not used to decode an +/// IEEE float, so reading such a type returns the right values. +fn check_float_fields( + size: u32, + sign: u8, + epos: u8, + esize: u8, + mpos: u8, + msize: u8, +) -> Result<(), FormatError> { + let bits = u64::from(size) * 8; + let (sign, epos, esize, mpos, msize) = ( + u64::from(sign), + u64::from(epos), + u64::from(esize), + u64::from(mpos), + u64::from(msize), + ); + if esize == 0 { + return Err(invalid("exponent size can't be zero")); + } + if epos >= bits { + return Err(invalid("exponent starting position out of bounds")); + } + if epos + esize > bits { + return Err(invalid("exponent range out of bounds")); + } + if msize == 0 { + return Err(invalid("mantissa size can't be zero")); + } + if mpos >= bits { + return Err(invalid("mantissa starting position out of bounds")); + } + if mpos + msize > bits { + return Err(invalid("mantissa range out of bounds")); + } + let (e_end, m_end) = (epos + esize - 1, mpos + msize - 1); + if ranges_overlap(sign, sign, epos, e_end) { + return Err(invalid("exponent and sign positions overlap")); + } + if ranges_overlap(sign, sign, mpos, m_end) { + return Err(invalid("mantissa and sign positions overlap")); + } + if ranges_overlap(epos, e_end, mpos, m_end) { + return Err(invalid("mantissa and exponent positions overlap")); + } + Ok(()) +} + impl Datatype { /// Parse a datatype message from raw bytes. /// /// Returns `(Datatype, bytes_consumed)` for recursive parsing. + /// + /// A type libhdf5 refuses to decode is refused here too, with + /// [`FormatError::InvalidDatatype`] carrying libhdf5's reason: size 0, + /// integer/bit-field/float bit fields outside the type or overlapping, + /// a compound with no members, a member outside its compound, a + /// duplicate or overlapping member, an enum whose size differs from its + /// base type's or with an empty name, an array of more than 32 + /// dimensions or a zero-sized one, an unaligned opaque tag length. + /// Reading such a type used to return data from a corrupt file. Checks + /// newer libhdf5 releases add but HDF5 2.0 (h5py 3.16) lacks are left + /// out, so a file h5py opens still opens here. pub fn parse(data: &[u8]) -> Result<(Datatype, usize), FormatError> { Self::parse_with_depth(data, 0) } @@ -259,6 +378,14 @@ impl Datatype { let size = LittleEndian::read_u32(&data[4..8]); let mut pos = 8; + // libhdf5 refuses size 0 for every class. A fixed-length string is + // exempt: clawhdf5 up to v2.7.0 wrote an empty-string attribute + // with a size-0 string type, and refusing it would fail every + // attribute of such objects, while reading it (an empty string) is + // harmless. + if size == 0 && class_id != 3 { + return Err(invalid("invalid datatype size")); + } match class_id { 0 => { @@ -272,6 +399,7 @@ impl Datatype { let signed = (bf0 >> 3) & 0x01 == 1; let bit_offset = LittleEndian::read_u16(&data[pos..pos + 2]); let bit_precision = LittleEndian::read_u16(&data[pos + 2..pos + 4]); + check_integer_bits(size, bit_offset, bit_precision)?; pos += 4; Ok(( Datatype::FixedPoint { @@ -289,13 +417,23 @@ impl Datatype { ensure_len(data, pos, 12)?; let bo_low = bf0 & 0x01; let bo_high = (bf0 >> 6) & 0x01; + // Bit 6 (with bit 0) is VAX order, defined by version 3; libhdf5 + // ignores bit 6 in older versions, which this read as VAX, + // byte-swapping a little-endian float. + let bo_high = if version >= 3 { bo_high } else { 0 }; let byte_order = match (bo_high, bo_low) { (0, 0) => DatatypeByteOrder::LittleEndian, (0, 1) => DatatypeByteOrder::BigEndian, - (1, 0) => DatatypeByteOrder::Vax, + (1, 0) => { + return Err(invalid("bad byte order for datatype message")); + } (1, 1) => DatatypeByteOrder::Vax, _ => unreachable!(), }; + // Bits 4-5: mantissa normalization; 3 is undefined. + if (bf0 >> 4) & 0x03 == 3 { + return Err(invalid("unknown floating-point normalization")); + } let bit_offset = LittleEndian::read_u16(&data[pos..pos + 2]); let bit_precision = LittleEndian::read_u16(&data[pos + 2..pos + 4]); let exponent_location = data[pos + 4]; @@ -303,6 +441,14 @@ impl Datatype { let mantissa_location = data[pos + 6]; let mantissa_size = data[pos + 7]; let exponent_bias = LittleEndian::read_u32(&data[pos + 8..pos + 12]); + check_float_fields( + size, + bf1, + exponent_location, + exponent_size, + mantissa_location, + mantissa_size, + )?; pos += 12; Ok(( Datatype::FloatingPoint { @@ -371,6 +517,10 @@ impl Datatype { 5 => { // Opaque let tag_len = bf0 as usize; + // libhdf5 writes the NUL-padded length, a multiple of 8. + if !tag_len.is_multiple_of(8) { + return Err(invalid("opaque flag field must be aligned")); + } ensure_len(data, pos, tag_len)?; // The stored tag is NUL-padded to a multiple of 8 bytes; the // tag itself ends at the first NUL (libhdf5 reads it with @@ -384,7 +534,45 @@ impl Datatype { 6 => { // Compound let num_members = (bf0 as u16) | ((bf1 as u16) << 8); - let mut members = Vec::with_capacity(num_members as usize); + if num_members == 0 { + return Err(invalid("invalid number of members: 0")); + } + let mut members: Vec = Vec::with_capacity(num_members as usize); + // Each member's size in the compound as libhdf5 decodes it: + // its stored size, times a v1 member's array dimensions. A + // variable-length member takes 4 + offset size + 4 bytes on + // disk, not the 16 of `Datatype::type_size`. + let mut member_sizes: Vec = Vec::with_capacity(num_members as usize); + // libhdf5 checks each member as it is decoded: it must fit in + // the compound (by its own stored size, before a v1 member's + // array dimensions are applied), and must not repeat a name + // or overlap an earlier member (by its final size). + let check_member = |members: &[CompoundMember], + member_sizes: &[u64], + name: &str, + byte_offset: u64, + stored_size: u32, + final_size: u64| + -> Result<(), FormatError> { + if byte_offset + u64::from(stored_size) > u64::from(size) { + return Err(invalid( + "member type extends outside its parent compound type", + )); + } + if let Some(j) = members.iter().position(|m| m.name == name) { + return Err(invalid(format!( + "duplicated compound field name '{name}', for fields {j} and {}", + members.len() + ))); + } + let end = byte_offset + final_size; + if members.iter().zip(member_sizes).any(|(m, &m_size)| { + byte_offset < m.byte_offset + m_size && m.byte_offset < end + }) { + return Err(invalid("member overlaps with previous member")); + } + Ok(()) + }; if (3..=5).contains(&version) { // v3, v4 and v5 share the compact member encoding (name, @@ -396,9 +584,20 @@ impl Datatype { pos += name_len; let byte_offset = read_uint(data, pos, ob)?; pos += ob; + let stored_size = stored_type_size(data, pos)?; let (member_dt, consumed) = Self::parse_with_depth(&data[pos..], depth + 1)?; pos += consumed; + let final_size = u64::from(stored_size); + check_member( + &members, + &member_sizes, + &name, + byte_offset, + stored_size, + final_size, + )?; + member_sizes.push(final_size); members.push(CompoundMember { name, byte_offset, @@ -438,11 +637,11 @@ impl Datatype { let at = pos + 12 + 4 * j; LittleEndian::read_u32(&data[at..at + 4]) == 0 }); - if ndims > 4 || zero_dim { - return Err(FormatError::InvalidDatatypeVersion { - class: class_id, - version, - }); + if ndims > 4 { + return Err(invalid("invalid number of dimensions for array")); + } + if zero_dim { + return Err(invalid("zero-sized dimension specified")); } array_dims = (0..ndims) .map(|j| { @@ -452,15 +651,28 @@ impl Datatype { .collect(); pos += 28; } + let stored_size = stored_type_size(data, pos)?; let (mut member_dt, consumed) = Self::parse_with_depth(&data[pos..], depth + 1)?; pos += consumed; + let final_size = array_dims.iter().fold(u64::from(stored_size), |a, &d| { + a.saturating_mul(u64::from(d)) + }); if !array_dims.is_empty() { member_dt = Datatype::Array { base_type: Box::new(member_dt), dimensions: array_dims, }; } + check_member( + &members, + &member_sizes, + &name, + byte_offset, + stored_size, + final_size, + )?; + member_sizes.push(final_size); members.push(CompoundMember { name, byte_offset, @@ -499,6 +711,9 @@ impl Datatype { let (base_type, base_consumed) = Self::parse_with_depth(&data[pos..], depth + 1)?; pos += base_consumed; let base_size = base_type.type_size(); + if base_size != size { + return Err(invalid("ENUM datatype size does not match parent")); + } let mut members = Vec::with_capacity(num_members as usize); // Enum layout: base_type, then all names (null-terminated), then all values // v1/v2: names are padded to 8-byte boundaries @@ -506,6 +721,9 @@ impl Datatype { let mut member_names = Vec::with_capacity(num_members as usize); for _ in 0..num_members { let (name, name_len) = read_null_terminated_string(data, pos)?; + if name.is_empty() { + return Err(invalid("0 length enum name")); + } if version < 3 { let padded = (name_len + 7) & !7; pos += padded; @@ -566,6 +784,7 @@ impl Datatype { if version == 2 { ensure_len(data, pos, 4)?; let ndims = data[pos] as usize; + check_array_rank(ndims)?; pos += 4; // ndims(1) + reserved(3) ensure_len(data, pos, ndims * 4 + ndims * 4)?; let mut dimensions = Vec::with_capacity(ndims); @@ -573,6 +792,7 @@ impl Datatype { dimensions.push(LittleEndian::read_u32(&data[pos..pos + 4])); pos += 4; } + check_array_dims(&dimensions)?; // skip permutation indices pos += ndims * 4; let (base_type, consumed) = Self::parse_with_depth(&data[pos..], depth + 1)?; @@ -589,6 +809,7 @@ impl Datatype { // type); HDF5 1.14+/2.0 with `libver=latest` emits v5. ensure_len(data, pos, 1)?; let ndims = data[pos] as usize; + check_array_rank(ndims)?; pos += 1; ensure_len(data, pos, ndims * 4)?; let mut dimensions = Vec::with_capacity(ndims); @@ -596,6 +817,7 @@ impl Datatype { dimensions.push(LittleEndian::read_u32(&data[pos..pos + 4])); pos += 4; } + check_array_dims(&dimensions)?; let (base_type, consumed) = Self::parse_with_depth(&data[pos..], depth + 1)?; pos += consumed; Ok(( @@ -652,6 +874,70 @@ impl Datatype { } } + /// [`Self::parse`] for the datatype message of an object whose header + /// has version `header_version`: a version-1 header, which has no + /// checksum, additionally gets [`Self::check_unused_bits`], as libhdf5 + /// does. Use this wherever the header is at hand. + pub fn parse_in_header( + data: &[u8], + header_version: u8, + ) -> Result<(Datatype, usize), FormatError> { + let parsed = Self::parse(data)?; + if header_version == 1 { + parsed.0.check_unused_bits()?; + } + Ok(parsed) + } + + /// libhdf5's guard against a corrupt numeric type in a header without + /// a checksum (`H5T_is_numeric_with_unusual_unused_bits`, HDF5 1.14.4+): + /// an integer, float or bit field wider than a byte whose precision and + /// offset leave more than half its bits unused is taken for corruption + /// (e.g. a 3-bit integer in 4 bytes, `cve-2024-29162`, or a 32-bit float + /// in 65525 bytes, `cve-2024-32614`), anywhere in the type. libhdf5 + /// skips the check for checksummed (version-2) headers and when the + /// file is opened with `H5Pset_relax_file_integrity_checks`; so does + /// [`Self::parse_in_header`], which has no such option. + pub fn check_unused_bits(&self) -> Result<(), FormatError> { + match self { + Datatype::FixedPoint { + size, + bit_offset, + bit_precision, + .. + } + | Datatype::FloatingPoint { + size, + bit_offset, + bit_precision, + .. + } + | Datatype::BitField { + size, + bit_offset, + bit_precision, + .. + } => { + let bits = u64::from(*size) * 8; + let prec = u64::from(*bit_precision); + if *size > 1 && prec < bits && bits > 2 * (prec + u64::from(*bit_offset)) { + return Err(invalid(format!( + "datatype has unusually large # of unused bits (prec = {prec} bits, \ + size = {size} bytes), possibly corrupted file" + ))); + } + Ok(()) + } + Datatype::Compound { members, .. } => members + .iter() + .try_for_each(|m| m.datatype.check_unused_bits()), + Datatype::Enumeration { base_type, .. } + | Datatype::VariableLength { base_type, .. } + | Datatype::Array { base_type, .. } => base_type.check_unused_bits(), + _ => Ok(()), + } + } + /// Serialize datatype to HDF5 message bytes. pub fn serialize(&self) -> Vec { match self { @@ -863,9 +1149,23 @@ impl Datatype { } /// Check that this datatype can be written: every part of it has an - /// on-disk encoding. [`Self::serialize`] cannot report errors, so the - /// writer calls this first. + /// on-disk encoding, and the encoding is one the reader (and libhdf5) + /// accepts. [`Self::serialize`] cannot report errors, so the writer calls + /// this first. A compound with no fields or a repeated field name, or an + /// enum member with an empty name, is refused here: libhdf5 and h5py + /// refuse such types, and so does [`Self::parse`], so writing one made a + /// file that could not be read back. pub fn check_encodable(&self) -> Result<(), FormatError> { + self.check_encodable_parts()?; + Self::parse(&self.serialize()).map_err(|e| { + FormatError::SerializationError(format!( + "datatype cannot be written: HDF5 readers refuse it ({e})" + )) + })?; + Ok(()) + } + + fn check_encodable_parts(&self) -> Result<(), FormatError> { match self { Datatype::Opaque { tag, .. } if opaque_tag_text(tag).len() > MAX_OPAQUE_TAG_LEN => { Err(FormatError::SerializationError(format!( @@ -878,10 +1178,10 @@ impl Datatype { )), Datatype::Compound { members, .. } => members .iter() - .try_for_each(|m| m.datatype.check_encodable()), + .try_for_each(|m| m.datatype.check_encodable_parts()), Datatype::Enumeration { base_type, .. } | Datatype::VariableLength { base_type, .. } - | Datatype::Array { base_type, .. } => base_type.check_encodable(), + | Datatype::Array { base_type, .. } => base_type.check_encodable_parts(), _ => Ok(()), } } @@ -985,7 +1285,8 @@ mod tests { ) -> Vec { // LE byte order: bo_low=0, bo_high=0 let bf0 = 0x00u8; - let bf1 = 0x00u8; + // Sign bit: the top bit. + let bf1 = (size * 8 - 1) as u8; // mantissa norm = 2 (MSB not stored) in bits 24-31... wait, that's bf2 let bf2 = 0x02u8; // norm = 2 let mut buf = build_dt_header(1, 1, [bf0, bf1, bf2], size); @@ -1012,7 +1313,7 @@ mod tests { let levels = MAX_DATATYPE_DEPTH as usize + 10; let mut data = Vec::new(); for _ in 0..levels { - data.extend_from_slice(&build_dt_header(9, 3, [0, 0, 0], 0)); + data.extend_from_slice(&build_dt_header(9, 3, [0, 0, 0], 16)); } data.extend_from_slice(&build_fixed_point(4, false, false, 0, 32)); @@ -1026,7 +1327,7 @@ mod tests { let levels = MAX_DATATYPE_DEPTH as usize - 1; let mut data = Vec::new(); for _ in 0..levels { - data.extend_from_slice(&build_dt_header(9, 3, [0, 0, 0], 0)); + data.extend_from_slice(&build_dt_header(9, 3, [0, 0, 0], 16)); } data.extend_from_slice(&build_fixed_point(4, false, false, 0, 32)); @@ -1176,8 +1477,8 @@ mod tests { #[test] fn test_opaque() { - // tag_len = 4, tag = "BLOB" - let mut buf = build_dt_header(5, 1, [4, 0, 0], 64); + // tag = "BLOB"; the stored length is the NUL-padded length, 8 + let mut buf = build_dt_header(5, 1, [8, 0, 0], 64); buf.extend_from_slice(b"BLOB"); // Pad to 8 bytes buf.extend_from_slice(&[0, 0, 0, 0]); @@ -2049,4 +2350,273 @@ mod tests { }; assert_eq!(dt.type_size(), 48); } + /// Every check here mirrors one in libhdf5's `H5O__dtype_decode_helper`; + /// the error text is libhdf5's. + fn invalid_reason(data: &[u8]) -> String { + match Datatype::parse(data) { + Err(FormatError::InvalidDatatype(why)) => why, + other => panic!("expected InvalidDatatype, got {other:?}"), + } + } + + #[test] + fn size_zero_is_refused() { + // cve-2017-17508: a variable-length string member of stored size 0. + let mut data = build_dt_header(9, 1, [1, 0, 0], 0); + data.extend_from_slice(&build_fixed_point(1, false, false, 0, 8)); + assert_eq!(invalid_reason(&data), "invalid datatype size"); + // Except a fixed-length string, which clawhdf5 <= v2.7.0 wrote for an + // empty-string attribute. + assert!(Datatype::parse(&build_dt_header(3, 1, [0, 0, 0], 0)).is_ok()); + assert_eq!( + invalid_reason(&build_fixed_point(0, false, false, 0, 0)), + "invalid datatype size" + ); + } + + #[test] + fn integer_bits_must_lie_inside_the_type() { + assert_eq!( + invalid_reason(&build_fixed_point(4, false, false, 32, 1)), + "integer offset out of bounds" + ); + assert_eq!( + invalid_reason(&build_fixed_point(4, false, false, 0, 0)), + "precision is zero" + ); + assert_eq!( + invalid_reason(&build_fixed_point(4, false, false, 8, 25)), + "integer offset+precision out of bounds" + ); + // A partial-precision integer inside its bytes is fine. + assert!(Datatype::parse(&build_fixed_point(4, false, false, 12, 8)).is_ok()); + } + + #[test] + fn float_fields_must_lie_inside_the_type_and_not_overlap() { + // (sign, epos, esize, mpos, msize) on an f32 + let f32_with = |sign: u8, epos: u8, esize: u8, mpos: u8, msize: u8| { + let mut data = build_dt_header(1, 1, [0x20, sign, 0], 4); + data.extend_from_slice(&0u16.to_le_bytes()); + data.extend_from_slice(&32u16.to_le_bytes()); + data.extend_from_slice(&[epos, esize, mpos, msize]); + data.extend_from_slice(&127u32.to_le_bytes()); + data + }; + assert!(Datatype::parse(&f32_with(31, 23, 8, 0, 23)).is_ok()); + for (fields, why) in [ + ((31, 23, 0, 0, 23), "exponent size can't be zero"), + ( + (31, 32, 8, 0, 23), + "exponent starting position out of bounds", + ), + ((31, 30, 8, 0, 23), "exponent range out of bounds"), + ((31, 23, 8, 0, 0), "mantissa size can't be zero"), + ( + (31, 23, 8, 40, 1), + "mantissa starting position out of bounds", + ), + // cve-2024-29163: a 128-bit mantissa in a 4-byte float. + ((31, 23, 8, 0, 128), "mantissa range out of bounds"), + ((23, 23, 8, 0, 23), "exponent and sign positions overlap"), + ((0, 23, 8, 0, 23), "mantissa and sign positions overlap"), + // cve-2026-34734. + ( + (31, 20, 8, 0, 23), + "mantissa and exponent positions overlap", + ), + ] { + let (sign, epos, esize, mpos, msize) = fields; + assert_eq!( + invalid_reason(&f32_with(sign, epos, esize, mpos, msize)), + why, + "{fields:?}" + ); + } + // Normalization 3 is undefined; bit 6 (VAX) needs bit 0 from v3. + let mut data = f32_with(31, 23, 8, 0, 23); + data[1] = 0x30; + assert_eq!( + invalid_reason(&data), + "unknown floating-point normalization" + ); + let mut data = f32_with(31, 23, 8, 0, 23); + data[0] = 0x31; // version 3 + data[1] = 0x60; + assert_eq!(invalid_reason(&data), "bad byte order for datatype message"); + } + + #[test] + fn unusual_unused_bits_are_refused_in_version_1_headers_only() { + // cve-2024-29162: a 3-bit integer in 4 bytes. + let data = build_fixed_point(4, false, true, 0, 3); + assert!(Datatype::parse_in_header(&data, 2).is_ok()); + assert_eq!( + match Datatype::parse_in_header(&data, 1) { + Err(FormatError::InvalidDatatype(why)) => why, + other => panic!("{other:?}"), + }, + "datatype has unusually large # of unused bits (prec = 3 bits, size = 4 bytes), \ + possibly corrupted file" + ); + // Half the bits used (with the offset) is not unusual; nor is a + // 1-byte type; nor a full-precision one. + for (size, offset, prec) in [(4u32, 0u16, 16u16), (4, 8, 8), (1, 0, 1), (8, 0, 64)] { + let data = build_fixed_point(size, false, true, offset, prec); + assert!( + Datatype::parse_in_header(&data, 1).is_ok(), + "{size} {offset} {prec}" + ); + } + // Nested: a compound member's type is checked too. + let member = build_fixed_point(4, false, true, 0, 15); + let data = compound_v3(4, &[("a", 0, member)]); + assert!(Datatype::parse_in_header(&data, 2).is_ok()); + assert!(Datatype::parse_in_header(&data, 1).is_err()); + } + + #[test] + fn f32_written_by_clawhdf5_up_to_2_7_0_still_parses() { + // Those versions put the sign bit at 63 whatever the float's size; + // libhdf5 refuses it ("sign bit position out of bounds"). + let mut data = build_dt_header(1, 1, [0x20, 63, 0], 4); + data.extend_from_slice(&0u16.to_le_bytes()); + data.extend_from_slice(&32u16.to_le_bytes()); + data.extend_from_slice(&[23, 8, 0, 23]); + data.extend_from_slice(&127u32.to_le_bytes()); + assert!(Datatype::parse(&data).is_ok()); + } + + #[test] + fn float_bit_6_is_vax_order_only_from_version_3() { + // h5py opens a v1 float with bit 6 set as an ordinary little-endian + // float; it used to be read as VAX order. + let mut data = build_float(4, 23, 8, 0, 23, 127); + data[1] |= 0x40; + match Datatype::parse(&data).unwrap().0 { + Datatype::FloatingPoint { byte_order, .. } => { + assert_eq!(byte_order, DatatypeByteOrder::LittleEndian) + } + other => panic!("{other:?}"), + } + data[0] = 0x31; + data[1] |= 0x01; + match Datatype::parse(&data).unwrap().0 { + Datatype::FloatingPoint { byte_order, .. } => { + assert_eq!(byte_order, DatatypeByteOrder::Vax) + } + other => panic!("{other:?}"), + } + } + + #[test] + fn opaque_tag_length_must_be_padded() { + let mut data = build_dt_header(5, 1, [4, 0, 0], 4); + data.extend_from_slice(b"BLOB"); + assert_eq!(invalid_reason(&data), "opaque flag field must be aligned"); + } + + /// A v3 compound of `size` bytes with `(name, offset, member)` members. + fn compound_v3(size: u32, members: &[(&str, u8, Vec)]) -> Vec { + let n = members.len() as u8; + let mut data = build_dt_header(6, 3, [n, 0, 0], size); + for (name, off, dt) in members { + data.extend_from_slice(name.as_bytes()); + data.push(0); + data.push(*off); + data.extend_from_slice(dt); + } + data + } + + #[test] + fn compound_members_are_checked() { + let i4 = build_fixed_point(4, false, true, 0, 32); + // cve-2016-4332: no members. + assert_eq!( + invalid_reason(&compound_v3(8, &[])), + "invalid number of members: 0" + ); + assert_eq!( + invalid_reason(&compound_v3( + 8, + &[("a", 0, i4.clone()), ("b", 6, i4.clone())] + )), + "member type extends outside its parent compound type" + ); + assert_eq!( + invalid_reason(&compound_v3( + 8, + &[("a", 0, i4.clone()), ("a", 4, i4.clone())] + )), + "duplicated compound field name 'a', for fields 0 and 1" + ); + assert_eq!( + invalid_reason(&compound_v3( + 8, + &[("a", 0, i4.clone()), ("b", 2, i4.clone())] + )), + "member overlaps with previous member" + ); + assert_eq!( + invalid_reason(&compound_v3( + 8, + &[("b", 4, i4.clone()), ("a", 2, i4.clone())] + )), + "member overlaps with previous member" + ); + // Members out of offset order, and gaps, are fine. + assert!(Datatype::parse(&compound_v3(12, &[("b", 8, i4.clone()), ("a", 0, i4)])).is_ok()); + } + + #[test] + fn enum_is_checked() { + let base = build_fixed_point(4, false, true, 0, 32); + let enum_of = |size: u32, names: &[&str]| { + let mut data = build_dt_header(8, 3, [names.len() as u8, 0, 0], size); + data.extend_from_slice(&base); + for n in names { + data.extend_from_slice(n.as_bytes()); + data.push(0); + } + for i in 0..names.len() as u32 { + data.extend_from_slice(&i.to_le_bytes()); + } + data + }; + assert!(Datatype::parse(&enum_of(4, &["RED", "GREEN"])).is_ok()); + // cve-2024-32618. + assert_eq!( + invalid_reason(&enum_of(4, &["", "GREEN"])), + "0 length enum name" + ); + assert_eq!( + invalid_reason(&enum_of(2, &["RED"])), + "ENUM datatype size does not match parent" + ); + } + + #[test] + fn array_dimensions_are_checked() { + let base = build_fixed_point(4, false, true, 0, 32); + let array_v3 = |dims: &[u32]| { + let n = dims.iter().product::().max(1); + let mut data = build_dt_header(10, 3, [0, 0, 0], 4 * n); + data.push(dims.len() as u8); + for d in dims { + data.extend_from_slice(&d.to_le_bytes()); + } + data.extend_from_slice(&base); + data + }; + assert!(Datatype::parse(&array_v3(&[2, 3])).is_ok()); + assert_eq!( + invalid_reason(&array_v3(&[2, 0])), + "zero-sized dimension specified" + ); + assert_eq!( + invalid_reason(&array_v3(&[1; 33])), + "too many dimensions for array datatype" + ); + } } diff --git a/crates/clawhdf5-format/src/ea_writer.rs b/crates/clawhdf5-format/src/ea_writer.rs index e0f0375..f669534 100644 --- a/crates/clawhdf5-format/src/ea_writer.rs +++ b/crates/clawhdf5-format/src/ea_writer.rs @@ -7,7 +7,9 @@ extern crate alloc; use alloc::{vec, vec::Vec}; use crate::checksum::jenkins_lookup3; -use crate::chunked_write::{WrittenChunk, filtered_chunk_size_len, push_addr, push_index_element}; +use crate::chunked_write::{ + WrittenChunk, filtered_chunk_size_len, push_addr, push_index_element, push_v4_chunk_dims, +}; /// Serialize a v4 Extensible Array layout message. pub(crate) fn serialize_v4_extensible_array( @@ -24,35 +26,7 @@ pub(crate) fn serialize_v4_extensible_array( let ndims = chunk_dims.len() as u8 + 1; buf.push(ndims); - let max_dim = chunk_dims - .iter() - .map(|&d| d as u64) - .chain(core::iter::once(element_size as u64)) - .max() - .unwrap_or(1); - let dim_encoded_len: u8 = if max_dim <= 0xFF { - 1 - } else if max_dim <= 0xFFFF { - 2 - } else { - 4 - }; - buf.push(dim_encoded_len); - - for &d in chunk_dims { - match dim_encoded_len { - 1 => buf.push(d as u8), - 2 => buf.extend_from_slice(&(d as u16).to_le_bytes()), - 4 => buf.extend_from_slice(&d.to_le_bytes()), - _ => unreachable!("unexpected dim_encoded_len: {dim_encoded_len}"), - } - } - match dim_encoded_len { - 1 => buf.push(element_size as u8), - 2 => buf.extend_from_slice(&(element_size as u16).to_le_bytes()), - 4 => buf.extend_from_slice(&element_size.to_le_bytes()), - _ => unreachable!("unexpected dim_encoded_len: {dim_encoded_len}"), - } + push_v4_chunk_dims(&mut buf, chunk_dims, element_size); // chunk index type = 4 (Extensible Array) buf.push(4); diff --git a/crates/clawhdf5-format/src/error.rs b/crates/clawhdf5-format/src/error.rs index bb81939..0056b14 100644 --- a/crates/clawhdf5-format/src/error.rs +++ b/crates/clawhdf5-format/src/error.rs @@ -201,6 +201,28 @@ pub enum FormatError { DuplicateDatasetName(String), /// Integer overflow in size computation (malformed data protection). Overflow(String), + /// An object header that libhdf5 refuses to load (the reason is + /// libhdf5's own error text): a misaligned or overrunning message, a + /// wrong message count, contradictory message flags, a message of a + /// class that cannot be shared flagged shareable, … + InvalidObjectHeader(&'static str), + /// A datatype message libhdf5 refuses to decode (the reason is + /// libhdf5's own error text): size 0, bit fields outside the type, + /// an empty enum name, a compound member outside its compound, … + InvalidDatatype(String), + /// A chunked layout whose chunk dimensions libhdf5 refuses: a zero + /// dimension, a rank that does not match the dataspace, an element size + /// that is not the datatype's, or a chunk of 4 GiB or more indexed by a + /// version-1 B-tree. + InvalidChunkDimensions(String), + /// The superblock's end-of-file address lies past the end of the file: + /// the file was truncated (libhdf5 refuses to open it). + TruncatedFile { + /// End of file recorded in the superblock (relative to byte 0). + stored_eof: u64, + /// The file's actual length in bytes. + actual_len: u64, + }, } impl fmt::Display for FormatError { @@ -406,9 +428,17 @@ impl fmt::Display for FormatError { FormatError::InvalidFilterPipelineVersion(v) => { write!(f, "invalid filter pipeline version: {v}") } - FormatError::UnsupportedFilter(id) => { - write!(f, "unsupported filter: {id}") - } + FormatError::UnsupportedFilter(id) => match crate::filter_registry::known_filter(*id) { + Some((name, Some(feature))) => write!( + f, + "unsupported filter: {id} ({name}; this build lacks the `{feature}` feature)" + ), + Some((name, None)) => write!( + f, + "unsupported filter: {id} ({name}, not implemented by clawhdf5)" + ), + None => write!(f, "unsupported filter: {id}"), + }, FormatError::FilterError(msg) => { write!(f, "filter error: {msg}") } @@ -445,6 +475,25 @@ impl fmt::Display for FormatError { FormatError::Overflow(msg) => { write!(f, "integer overflow: {msg}") } + FormatError::InvalidObjectHeader(why) => { + write!(f, "corrupt object header: {why}") + } + FormatError::InvalidDatatype(why) => { + write!(f, "invalid datatype: {why}") + } + FormatError::InvalidChunkDimensions(why) => { + write!(f, "invalid chunk dimensions: {why}") + } + FormatError::TruncatedFile { + stored_eof, + actual_len, + } => { + write!( + f, + "truncated file: the superblock records end of file {stored_eof}, \ + but the file is {actual_len} bytes" + ) + } } } } diff --git a/crates/clawhdf5-format/src/filter_pipeline.rs b/crates/clawhdf5-format/src/filter_pipeline.rs index 74bc727..13ed69f 100644 --- a/crates/clawhdf5-format/src/filter_pipeline.rs +++ b/crates/clawhdf5-format/src/filter_pipeline.rs @@ -19,6 +19,18 @@ pub const FILTER_SCALEOFFSET: u16 = 6; pub const FILTER_LZ4: u16 = 32004; /// Zstandard compression. pub const FILTER_ZSTD: u16 = 32015; +/// bzip2 (registered by PyTables; hdf5plugin's `BZip2`). +pub const FILTER_BZIP2: u16 = 307; +/// LZF — h5py's built-in `compression="lzf"`. +pub const FILTER_LZF: u16 = 32000; +/// Blosc 1 (hdf5-blosc; hdf5plugin's `Blosc`). +pub const FILTER_BLOSC: u16 = 32001; +/// Bitshuffle, optionally with LZ4 or Zstandard (hdf5plugin's `Bitshuffle`). +pub const FILTER_BITSHUFFLE: u16 = 32008; +/// ZFP lossy floating-point compression (hdf5plugin's `Zfp`). Not supported. +pub const FILTER_ZFP: u16 = 32013; +/// Blosc 2 (hdf5plugin's `Blosc2`). +pub const FILTER_BLOSC2: u16 = 32026; /// Pcodec lossless numerical codec — a **private, unregistered** clawhdf5 /// filter. Pcodec has no ID in the HDF Group's filter registry (checked /// 2026-09-25, `hdf5_plugins/docs/RegisteredFilterPlugins.md`), so it uses an diff --git a/crates/clawhdf5-format/src/filter_registry.rs b/crates/clawhdf5-format/src/filter_registry.rs new file mode 100644 index 0000000..6d39613 --- /dev/null +++ b/crates/clawhdf5-format/src/filter_registry.rs @@ -0,0 +1,477 @@ +//! Filter registry: every filter is looked up here by its HDF5 filter ID. +//! +//! Two tiers: +//! +//! * **Built-in filters** — a static table of the filters compiled into this +//! build: the HDF5 standard filters (deflate, shuffle, Fletcher32, szip, +//! N-Bit, scale-offset) and the plugin filters whose cargo features are +//! enabled (LZ4, Zstandard, pcodec, LZF, bitshuffle, bzip2, blosc). +//! [`builtin_filters`] lists them. +//! * **Registered filters** (`std` only) — codecs the application supplies +//! for any other ID with [`register_filter`] (a [`FilterCodec`], or just a +//! decoding closure). A registered codec cannot shadow a built-in one, +//! except under 32023: that ID belongs to Granular BitRound, and the +//! built-in entry there only reads the pcodec chunks clawhdf5 <= 2.7.0 +//! wrote (filter name `"pcodec"`), so a codec registered for 32023 handles +//! every other chunk with that ID, and writes. +//! +//! An ID in neither tier fails with [`FormatError::UnsupportedFilter`], as it +//! always has. +//! +//! ``` +//! # #[cfg(feature = "std")] { +//! use clawhdf5_format::filter_registry::{self, FilterContext}; +//! use clawhdf5_format::error::FormatError; +//! +//! // A toy filter in the private-use range: every byte XORed with 0x5A. +//! filter_registry::register_filter(300, |input: &[u8], _ctx: &FilterContext<'_>| { +//! Ok::<_, FormatError>(input.iter().map(|b| b ^ 0x5A).collect()) +//! }) +//! .unwrap(); +//! assert!(filter_registry::is_filter_available(300)); +//! filter_registry::unregister_filter(300); +//! # } +//! ``` + +#[cfg(not(feature = "std"))] +extern crate alloc; + +#[cfg(not(feature = "std"))] +use alloc::vec::Vec; + +use crate::error::FormatError; +use crate::filter_pipeline::FilterDescription; + +/// What a codec is told about the filter it is applying. +#[derive(Debug, Clone, Copy)] +pub struct FilterContext<'a> { + /// The filter as recorded in the dataset's filter pipeline: its ID, name, + /// flags and client data (`cd_values`). + pub filter: &'a FilterDescription, + /// Size in bytes of one dataset element (the datatype's size). + pub element_size: usize, + /// Decoding only: the most bytes this stage may produce — what entered + /// the filter when the chunk was written. 0 means unknown; a decoder then + /// falls back to a fixed ceiling. Always 0 when encoding. + pub max_output: usize, +} + +impl FilterContext<'_> { + /// The filter's client data (`cd_values`). + pub fn client_data(&self) -> &[u32] { + &self.filter.client_data + } + + /// The largest output a decoder should allow: [`Self::max_output`], or + /// 256 MiB when that is unknown. + pub fn output_limit(&self) -> usize { + if self.max_output != 0 { + self.max_output + } else { + crate::filters::MAX_DECOMPRESS_SIZE + } + } +} + +/// A filter implementation. +/// +/// `decode` undoes the filter (the read direction). `encode` applies it (the +/// write direction); the default refuses with +/// [`FormatError::UnsupportedFilter`], which is right for a read-only codec. +pub trait FilterCodec: Send + Sync { + /// Undo the filter on one chunk. The output must not exceed + /// [`FilterContext::output_limit`]; the pipeline rejects a larger one. + fn decode(&self, input: &[u8], ctx: &FilterContext<'_>) -> Result, FormatError>; + + /// Apply the filter to one chunk. + fn encode(&self, input: &[u8], ctx: &FilterContext<'_>) -> Result, FormatError> { + let _ = input; + Err(FormatError::UnsupportedFilter(ctx.filter.filter_id)) + } +} + +/// Any `Fn(&[u8], &FilterContext) -> Result, FormatError>` is a +/// decode-only codec. +impl FilterCodec for F +where + F: Fn(&[u8], &FilterContext<'_>) -> Result, FormatError> + Send + Sync, +{ + fn decode(&self, input: &[u8], ctx: &FilterContext<'_>) -> Result, FormatError> { + self(input, ctx) + } +} + +/// Signature of a built-in filter's decoder or encoder. +pub type BuiltinFn = fn(&[u8], &FilterContext<'_>) -> Result, FormatError>; + +/// A filter compiled into this build. +#[derive(Debug, Clone, Copy)] +pub struct BuiltinFilter { + /// HDF5 filter ID. + pub id: u16, + /// Human-readable name. + pub name: &'static str, + /// Decoder. + pub(crate) decode: BuiltinFn, + /// Encoder, if this build can write the filter. + pub(crate) encode: Option, +} + +impl BuiltinFilter { + /// Whether this build can write the filter as well as read it. + pub fn can_encode(&self) -> bool { + self.encode.is_some() + } + + /// Whether the built-in entry only borrows its ID for some chunks, so a + /// registered codec may take the rest: the legacy pcodec entry under + /// Granular BitRound's 32023, which claims only chunks named `"pcodec"`. + fn is_shared(&self) -> bool { + self.id == crate::filter_pipeline::FILTER_PCODEC_LEGACY + } + + /// Whether this entry decodes chunks written with `filter`. + fn claims(&self, filter: &crate::filter_pipeline::FilterDescription) -> bool { + !self.is_shared() + || filter.name.as_deref() == Some(crate::filter_pipeline::FILTER_PCODEC_LEGACY_NAME) + } +} + +/// The filters compiled into this build, in ID order. +pub fn builtin_filters() -> &'static [BuiltinFilter] { + crate::filters::BUILTIN_FILTERS +} + +/// The built-in filter with this ID, if it is compiled in. +pub fn builtin_filter(id: u16) -> Option<&'static BuiltinFilter> { + builtin_filters().iter().find(|f| f.id == id) +} + +/// Why a filter ID may be missing from this build: the filter's name, and +/// the cargo feature that provides it (`None`: clawhdf5 does not implement +/// it — register a codec for it with [`register_filter`]). `None` for an ID +/// clawhdf5 knows nothing about. +pub fn known_filter(id: u16) -> Option<(&'static str, Option<&'static str>)> { + Some(match id { + 1 => ("deflate", Some("deflate")), + 4 => ("SZIP", Some("szip")), + 307 => ("bzip2", Some("bzip2")), + 480 => ("pcodec", Some("pcodec")), + 32000 => ("LZF", Some("lzf")), + 32001 => ("Blosc", Some("blosc")), + 32004 => ("LZ4", Some("lz4")), + 32008 => ("bitshuffle", Some("bitshuffle")), + 32013 => ("ZFP", None), + 32015 => ("Zstandard", Some("zstd")), + 32019 => ("JPEG", None), + 32022 => ("BitGroom", None), + 32023 => ("Granular BitRound", None), + 32026 => ("Blosc2", None), + _ => return None, + }) +} + +/// Whether a chunk filtered with `id` can be decoded: a built-in filter or a +/// registered one. +pub fn is_filter_available(id: u16) -> bool { + if builtin_filter(id).is_some() { + return true; + } + #[cfg(feature = "std")] + { + registered(id).is_some() + } + #[cfg(not(feature = "std"))] + { + false + } +} + +#[cfg(feature = "std")] +mod custom { + use super::FilterCodec; + use std::collections::BTreeMap; + use std::sync::{Arc, PoisonError, RwLock}; + + pub(super) type Registry = BTreeMap>; + + static REGISTRY: RwLock = RwLock::new(BTreeMap::new()); + + pub(super) fn with_read(f: impl FnOnce(&Registry) -> R) -> R { + // A panic while holding the lock cannot leave the map half-updated + // (every update is a single insert/remove), so poisoning is ignored. + f(®ISTRY.read().unwrap_or_else(PoisonError::into_inner)) + } + + pub(super) fn with_write(f: impl FnOnce(&mut Registry) -> R) -> R { + f(&mut REGISTRY.write().unwrap_or_else(PoisonError::into_inner)) + } +} + +/// Register a codec for filter `id`, process-wide. It is used for every +/// chunk read (and, if it implements [`FilterCodec::encode`], written) with +/// that filter ID, by every file. +/// +/// A plain closure `Fn(&[u8], &FilterContext) -> Result, FormatError>` +/// registers a decoder. Replaces (and returns) an earlier registration for +/// the same ID. Fails with [`FormatError::FilterError`] if `id` is a built-in +/// filter of this build: those cannot be overridden. The exception is 32023 +/// (Granular BitRound): with the `pcodec` feature the built-in entry there +/// reads only chunks whose filter is named `"pcodec"` (clawhdf5 <= 2.7.0's +/// files); a codec registered for 32023 decodes every other chunk with that +/// ID and does all the writing. +#[cfg(feature = "std")] +pub fn register_filter( + id: u16, + codec: C, +) -> Result>, FormatError> +where + C: FilterCodec + 'static, +{ + if let Some(builtin) = builtin_filter(id).filter(|b| !b.is_shared()) { + return Err(FormatError::FilterError(format!( + "filter {id} ({}) is built in and cannot be re-registered", + builtin.name + ))); + } + let codec: std::sync::Arc = std::sync::Arc::new(codec); + Ok(custom::with_write(|r| r.insert(id, codec))) +} + +/// Remove the codec registered for `id`. Returns whether one was registered. +#[cfg(feature = "std")] +pub fn unregister_filter(id: u16) -> bool { + custom::with_write(|r| r.remove(&id).is_some()) +} + +/// The codec registered for `id`, if any. +#[cfg(feature = "std")] +pub fn registered(id: u16) -> Option> { + custom::with_read(|r| r.get(&id).cloned()) +} + +/// Undo filter `ctx.filter` on `input`: the built-in decoder if there is one +/// that claims the chunk, else a registered one, else the built-in decoder's +/// own refusal or [`FormatError::UnsupportedFilter`]. +pub(crate) fn decode(input: &[u8], ctx: &FilterContext<'_>) -> Result, FormatError> { + let id = ctx.filter.filter_id; + let builtin = builtin_filter(id); + if let Some(builtin) = builtin.filter(|b| b.claims(ctx.filter)) { + return (builtin.decode)(input, ctx); + } + #[cfg(feature = "std")] + if let Some(codec) = registered(id) { + let out = codec.decode(input, ctx)?; + // A registered codec is outside our control: hold it to the same + // bound the built-in decoders enforce. + if out.len() > ctx.output_limit() { + return Err(FormatError::DecompressionError(format!( + "filter {id}: decoded {} bytes, more than the {} the chunk can hold", + out.len(), + ctx.output_limit() + ))); + } + return Ok(out); + } + match builtin { + Some(builtin) => (builtin.decode)(input, ctx), + None => Err(FormatError::UnsupportedFilter(id)), + } +} + +/// Apply filter `ctx.filter` to `input`. +pub(crate) fn encode(input: &[u8], ctx: &FilterContext<'_>) -> Result, FormatError> { + let id = ctx.filter.filter_id; + #[cfg(feature = "std")] + if builtin_filter(id).is_some_and(|b| b.is_shared()) + && let Some(codec) = registered(id) + { + return codec.encode(input, ctx); + } + if let Some(builtin) = builtin_filter(id) { + return match builtin.encode { + Some(encode) => encode(input, ctx), + None => Err(FormatError::UnsupportedFilter(id)), + }; + } + #[cfg(feature = "std")] + if let Some(codec) = registered(id) { + return codec.encode(input, ctx); + } + Err(FormatError::UnsupportedFilter(id)) +} + +#[cfg(all(test, feature = "std"))] +pub(crate) mod tests { + use super::*; + use crate::filter_pipeline::{FILTER_FLETCHER32, FILTER_SHUFFLE, FilterPipeline}; + use crate::filters::{compress_chunk, decompress_chunk}; + + fn pipeline(id: u16) -> FilterPipeline { + FilterPipeline { + version: 2, + filters: vec![FilterDescription { + filter_id: id, + name: Some("test".into()), + flags: 0, + client_data: vec![7], + }], + } + } + + struct Xor; + impl FilterCodec for Xor { + fn decode(&self, input: &[u8], ctx: &FilterContext<'_>) -> Result, FormatError> { + let k = ctx.client_data()[0] as u8; + Ok(input.iter().map(|b| b ^ k).collect()) + } + fn encode(&self, input: &[u8], ctx: &FilterContext<'_>) -> Result, FormatError> { + self.decode(input, ctx) + } + } + + // Each test uses its own ID: the registry is process-wide and tests run + // in parallel. + + #[test] + fn unknown_filter_keeps_its_error() { + let err = decompress_chunk(b"abc", &pipeline(311), 3, 1).unwrap_err(); + assert_eq!(err, FormatError::UnsupportedFilter(311)); + let err = compress_chunk(b"abc", &pipeline(311), 1).unwrap_err(); + assert_eq!(err, FormatError::UnsupportedFilter(311)); + } + + #[test] + fn registered_codec_round_trips_through_the_pipeline() { + assert!(!is_filter_available(312)); + assert!(register_filter(312, Xor).unwrap().is_none()); + assert!(is_filter_available(312)); + let data = b"hello, registry".to_vec(); + let enc = compress_chunk(&data, &pipeline(312), 1).unwrap(); + assert_ne!(enc, data); + assert_eq!( + decompress_chunk(&enc, &pipeline(312), data.len(), 1).unwrap(), + data + ); + assert!(unregister_filter(312)); + assert!(!unregister_filter(312)); + assert_eq!( + decompress_chunk(&enc, &pipeline(312), data.len(), 1).unwrap_err(), + FormatError::UnsupportedFilter(312) + ); + } + + #[test] + fn closure_registers_a_decoder_only() { + register_filter(313, |input: &[u8], _ctx: &FilterContext<'_>| { + Ok(input.iter().rev().copied().collect()) + }) + .unwrap(); + assert_eq!( + decompress_chunk(b"abc", &pipeline(313), 3, 1).unwrap(), + b"cba" + ); + assert_eq!( + compress_chunk(b"abc", &pipeline(313), 1).unwrap_err(), + FormatError::UnsupportedFilter(313) + ); + unregister_filter(313); + } + + #[test] + fn registered_decoder_output_is_bounded() { + register_filter(314, |_input: &[u8], _ctx: &FilterContext<'_>| { + Ok(vec![0u8; 1000]) + }) + .unwrap(); + let err = decompress_chunk(b"abc", &pipeline(314), 10, 1).unwrap_err(); + assert!(matches!(err, FormatError::DecompressionError(_)), "{err:?}"); + unregister_filter(314); + } + + #[test] + fn builtins_cannot_be_overridden() { + for id in [FILTER_SHUFFLE, FILTER_FLETCHER32] { + let Err(err) = register_filter(id, Xor) else { + panic!("built-in filter {id} was re-registered"); + }; + assert!(matches!(err, FormatError::FilterError(_)), "{err:?}"); + } + assert!(builtin_filter(FILTER_SHUFFLE).is_some()); + } + + /// Serialises the tests that register or read filter 32023 (the + /// registry is process-wide). + pub(crate) static ID_32023: std::sync::Mutex<()> = std::sync::Mutex::new(()); + + /// 32023 is Granular BitRound's ID; the `pcodec` build's built-in entry + /// there reads only clawhdf5 <= 2.7.0's pcodec chunks (named "pcodec"), + /// so a codec can be registered for the rest, and writes with it. + #[test] + fn a_codec_can_be_registered_for_granular_bitround() { + let _guard = ID_32023 + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner); + let named = |name: Option<&str>| FilterPipeline { + version: 2, + filters: vec![FilterDescription { + filter_id: 32023, + name: name.map(Into::into), + flags: 0, + client_data: vec![7], + }], + }; + let prev = register_filter(32023, Xor).expect("32023 must be registrable"); + assert!(prev.is_none()); + let data = b"granular bitround".to_vec(); + for name in [None, Some("granular_bitround"), Some("test")] { + let pl = named(name); + let enc = compress_chunk(&data, &pl, 1).unwrap(); + assert_ne!(enc, data); + assert_eq!(decompress_chunk(&enc, &pl, data.len(), 1).unwrap(), data); + } + // clawhdf5 <= 2.7.0's pcodec chunks still go to the built-in reader. + #[cfg(feature = "pcodec")] + { + let raw: Vec = (0..64) + .flat_map(|i| (f64::from(i) * 0.5).to_le_bytes()) + .collect(); + let comp = crate::filters::pcodec_compress(&raw, 8).unwrap(); + let mut pl = named(Some("pcodec")); + pl.filters[0].client_data = vec![8]; + assert_eq!(decompress_chunk(&comp, &pl, raw.len(), 8).unwrap(), raw); + } + assert!(unregister_filter(32023)); + let pl = named(None); + assert!(matches!( + decompress_chunk(&data, &pl, data.len(), 1), + Err(FormatError::UnsupportedFilter(32023)) + )); + } + + #[test] + fn unsupported_filter_error_names_the_filter() { + let msg = FormatError::UnsupportedFilter(32026).to_string(); + assert!( + msg.contains("Blosc2") && msg.contains("not implemented"), + "{msg}" + ); + let msg = FormatError::UnsupportedFilter(32013).to_string(); + assert!(msg.contains("ZFP"), "{msg}"); + let msg = FormatError::UnsupportedFilter(32000).to_string(); + assert!(msg.contains("LZF") && msg.contains("`lzf`"), "{msg}"); + assert_eq!( + FormatError::UnsupportedFilter(399).to_string(), + "unsupported filter: 399" + ); + } + + #[test] + fn builtin_table_is_sorted_and_unique() { + let ids: Vec = builtin_filters().iter().map(|f| f.id).collect(); + let mut sorted = ids.clone(); + sorted.sort_unstable(); + sorted.dedup(); + assert_eq!(ids, sorted); + } +} diff --git a/crates/clawhdf5-format/src/filters.rs b/crates/clawhdf5-format/src/filters.rs index 042aca9..36f2e96 100644 --- a/crates/clawhdf5-format/src/filters.rs +++ b/crates/clawhdf5-format/src/filters.rs @@ -4,14 +4,23 @@ extern crate alloc; #[cfg(not(feature = "std"))] -use alloc::{boxed::Box, vec, vec::Vec}; +use alloc::{boxed::Box, format, vec, vec::Vec}; use crate::error::FormatError; +#[cfg(feature = "deflate")] +use crate::filter_pipeline::FILTER_DEFLATE; +#[cfg(feature = "lz4")] +use crate::filter_pipeline::FILTER_LZ4; +#[cfg(feature = "szip")] +use crate::filter_pipeline::FILTER_SZIP; +#[cfg(feature = "zstd")] +use crate::filter_pipeline::FILTER_ZSTD; use crate::filter_pipeline::{ - FILTER_DEFLATE, FILTER_FLETCHER32, FILTER_LZ4, FILTER_NBIT, FILTER_PCODEC, - FILTER_PCODEC_LEGACY, FILTER_PCODEC_LEGACY_NAME, FILTER_SCALEOFFSET, FILTER_SHUFFLE, - FILTER_SZIP, FILTER_ZSTD, FilterPipeline, + FILTER_FLETCHER32, FILTER_NBIT, FILTER_SCALEOFFSET, FILTER_SHUFFLE, FilterPipeline, }; +#[cfg(feature = "pcodec")] +use crate::filter_pipeline::{FILTER_PCODEC, FILTER_PCODEC_LEGACY, FILTER_PCODEC_LEGACY_NAME}; +use crate::filter_registry::{self, BuiltinFilter, FilterContext}; /// Absolute ceiling on a single decompressed chunk's output size, used only /// when the pipeline's declared `chunk_size` is unavailable (0). Prevents @@ -98,33 +107,47 @@ pub fn decompress_chunk_masked( if filter_skipped(filter_mask, i) { continue; } - let bound = bounds[i]; - data = match filter.filter_id { - FILTER_SHUFFLE => shuffle_decompress(&data, element_size as usize)?, - // `bound` caps the decoded size so these decoders can't be forced - // into unbounded allocation by a hostile or corrupted payload. - FILTER_DEFLATE => deflate_decompress(&data, bound)?, - FILTER_LZ4 => lz4_decompress(&data, bound)?, - FILTER_ZSTD => zstd_decompress(&data, bound)?, - FILTER_FLETCHER32 => fletcher32_verify(&data)?, - FILTER_PCODEC => pcodec_decompress(&data, element_size as usize, bound)?, - // Pcodec chunks written by clawhdf5 <= 2.7.0 under the ID registered - // to Granular BitRound; recognised by the name those versions wrote. - FILTER_PCODEC_LEGACY if filter.name.as_deref() == Some(FILTER_PCODEC_LEGACY_NAME) => { - pcodec_decompress(&data, element_size as usize, bound)? - } - // These decoders also reject an element count that would - // over-allocate past `bound`. - FILTER_SCALEOFFSET => scaleoffset_decompress(&data, &filter.client_data, bound)?, - FILTER_NBIT => nbit_decompress(&data, &filter.client_data, bound)?, - FILTER_SZIP => crate::filters_szip::szip_decompress(&data, &filter.client_data, bound)?, - other => return Err(FormatError::UnsupportedFilter(other)), + // `max_output` caps the decoded size so a decoder can't be forced + // into unbounded allocation by a hostile or corrupted payload. + let ctx = FilterContext { + filter, + element_size: element_size as usize, + max_output: bounds[i], }; + data = filter_registry::decode(&data, &ctx)?; } Ok(data) } +/// Decode one stored chunk of a chunked dataset: [`decompress_chunk_masked`], +/// then require exactly `chunk_size` bytes (when `chunk_size` is known). +/// +/// HDF5 stores every chunk at the full chunk size — edge chunks are padded +/// before they are filtered, and an edge chunk left unfiltered is written +/// full-size too — so a pipeline that decodes to fewer bytes means a +/// corrupt chunk. libhdf5 fails such a read (or returns uninitialised +/// memory); it must never read back as zeros. `coords` (the chunk's offset +/// in the dataset) is named in the error. +pub fn decompress_chunk_exact( + compressed: &[u8], + pipeline: &FilterPipeline, + chunk_size: usize, + element_size: u32, + filter_mask: u32, + coords: &[u64], +) -> Result, FormatError> { + let data = + decompress_chunk_masked(compressed, pipeline, chunk_size, element_size, filter_mask)?; + if chunk_size != 0 && data.len() != chunk_size { + return Err(FormatError::ChunkedReadError(format!( + "chunk at {coords:?} decoded to {} bytes, expected {chunk_size}", + data.len() + ))); + } + Ok(data) +} + /// Apply a filter pipeline to compress a chunk. /// Filters are applied in FORWARD order for compression. pub fn compress_chunk( @@ -135,26 +158,128 @@ pub fn compress_chunk( let mut result = data.to_vec(); for filter in &pipeline.filters { - result = match filter.filter_id { - FILTER_SHUFFLE => shuffle_compress(&result, element_size as usize)?, - FILTER_DEFLATE => { - let level = filter.client_data.first().copied().unwrap_or(6); - deflate_compress(&result, level)? - } - FILTER_LZ4 => lz4_compress(&result, &filter.client_data)?, - FILTER_ZSTD => { - let level = filter.client_data.first().copied().unwrap_or(3); - zstd_compress(&result, level)? - } - FILTER_FLETCHER32 => fletcher32_append(&result)?, - FILTER_PCODEC => pcodec_compress(&result, element_size as usize)?, - other => return Err(FormatError::UnsupportedFilter(other)), + let ctx = FilterContext { + filter, + element_size: element_size as usize, + max_output: 0, }; + result = filter_registry::encode(&result, &ctx)?; } Ok(result) } +/// The filters compiled into this build, sorted by ID (see +/// [`crate::filter_registry`]). A filter whose cargo feature is off is left +/// out, so it fails as [`FormatError::UnsupportedFilter`] like any unknown ID. +pub(crate) static BUILTIN_FILTERS: &[BuiltinFilter] = &[ + #[cfg(feature = "deflate")] + BuiltinFilter { + id: FILTER_DEFLATE, + name: "deflate", + decode: |d, c| deflate_decompress(d, c.max_output), + encode: Some(|d, c| deflate_compress(d, c.client_data().first().copied().unwrap_or(6))), + }, + BuiltinFilter { + id: FILTER_SHUFFLE, + name: "shuffle", + decode: |d, c| shuffle_decompress(d, c.element_size), + encode: Some(|d, c| shuffle_compress(d, c.element_size)), + }, + BuiltinFilter { + id: FILTER_FLETCHER32, + name: "fletcher32", + decode: |d, _| fletcher32_verify(d), + encode: Some(|d, _| fletcher32_append(d)), + }, + #[cfg(feature = "szip")] + BuiltinFilter { + id: FILTER_SZIP, + name: "szip", + decode: |d, c| crate::filters_szip::szip_decompress(d, c.client_data(), c.max_output), + encode: None, + }, + // These decoders also reject an element count that would over-allocate + // past `max_output`. + BuiltinFilter { + id: FILTER_NBIT, + name: "nbit", + decode: |d, c| nbit_decompress(d, c.client_data(), c.max_output), + encode: None, + }, + BuiltinFilter { + id: FILTER_SCALEOFFSET, + name: "scaleoffset", + decode: |d, c| scaleoffset_decompress(d, c.client_data(), c.max_output), + encode: None, + }, + #[cfg(feature = "bzip2")] + BuiltinFilter { + id: crate::filter_pipeline::FILTER_BZIP2, + name: "bzip2", + decode: crate::filters_bzip2::bzip2_decode, + encode: Some(crate::filters_bzip2::bzip2_encode), + }, + #[cfg(feature = "pcodec")] + BuiltinFilter { + id: FILTER_PCODEC, + name: "pcodec (clawhdf5 private)", + decode: |d, c| pcodec_decompress(d, c.element_size, c.max_output), + encode: Some(|d, c| pcodec_compress(d, c.element_size)), + }, + #[cfg(feature = "lzf")] + BuiltinFilter { + id: crate::filter_pipeline::FILTER_LZF, + name: "lzf", + decode: crate::filters_lzf::lzf_decode, + encode: Some(crate::filters_lzf::lzf_encode), + }, + #[cfg(feature = "blosc")] + BuiltinFilter { + id: crate::filter_pipeline::FILTER_BLOSC, + name: "blosc", + decode: crate::filters_blosc::blosc_decode, + encode: Some(crate::filters_blosc::blosc_encode), + }, + #[cfg(feature = "lz4")] + BuiltinFilter { + id: FILTER_LZ4, + name: "lz4", + decode: |d, c| lz4_decompress(d, c.max_output), + encode: Some(|d, c| lz4_compress(d, c.client_data())), + }, + #[cfg(feature = "bitshuffle")] + BuiltinFilter { + id: crate::filter_pipeline::FILTER_BITSHUFFLE, + name: "bitshuffle", + decode: crate::filters_bitshuffle::bitshuffle_decode, + encode: Some(crate::filters_bitshuffle::bitshuffle_encode), + }, + #[cfg(feature = "zstd")] + BuiltinFilter { + id: FILTER_ZSTD, + name: "zstd", + decode: |d, c| zstd_decompress(d, c.max_output), + encode: Some(|d, c| zstd_compress(d, c.client_data().first().copied().unwrap_or(3))), + }, + // Pcodec chunks written by clawhdf5 <= 2.7.0 under the ID registered to + // Granular BitRound; recognised by the name those versions wrote, and + // never written. + #[cfg(feature = "pcodec")] + BuiltinFilter { + id: FILTER_PCODEC_LEGACY, + name: "pcodec (clawhdf5 <= 2.7.0)", + decode: |d, c| { + if c.filter.name.as_deref() == Some(FILTER_PCODEC_LEGACY_NAME) { + pcodec_decompress(d, c.element_size, c.max_output) + } else { + Err(FormatError::UnsupportedFilter(FILTER_PCODEC_LEGACY)) + } + }, + encode: None, + }, +]; + /// Decode the HDF5 scale-offset filter (id 6). /// /// Supports all three scale-offset variants: @@ -877,11 +1002,6 @@ mod sysz { } } -#[cfg(not(feature = "deflate"))] -fn deflate_decompress(_data: &[u8], _expected_bytes: usize) -> Result, FormatError> { - Err(FormatError::UnsupportedFilter(FILTER_DEFLATE)) -} - /// Compress data with zlib. #[cfg(feature = "deflate")] fn deflate_compress(data: &[u8], level: u32) -> Result, FormatError> { @@ -922,11 +1042,6 @@ pub(crate) fn deflate_bounded(data: &[u8], level: u32) -> Result, String } } -#[cfg(not(feature = "deflate"))] -fn deflate_compress(_data: &[u8], _level: u32) -> Result, FormatError> { - Err(FormatError::UnsupportedFilter(FILTER_DEFLATE)) -} - /// Default LZ4 block size of the registered HDF5 LZ4 filter (`H5Zlz4.c`, /// `DEFAULT_BLOCK_SIZE`): 1 GiB, so an HDF5 chunk is normally one block. #[cfg(feature = "lz4")] @@ -1028,11 +1143,6 @@ fn lz4_decompress_hdf5( Ok(out) } -#[cfg(not(feature = "lz4"))] -fn lz4_decompress(_data: &[u8], _expected_bytes: usize) -> Result, FormatError> { - Err(FormatError::UnsupportedFilter(FILTER_LZ4)) -} - /// Compress data in the registered HDF5 LZ4 filter format (see /// [`lz4_decompress`]), so libhdf5 with the LZ4 plugin (e.g. hdf5plugin) can /// read it. `cd[0]`, when present and non-zero, is the block size in bytes, @@ -1064,11 +1174,6 @@ fn lz4_compress(data: &[u8], cd: &[u32]) -> Result, FormatError> { Ok(result) } -#[cfg(not(feature = "lz4"))] -fn lz4_compress(_data: &[u8], _cd: &[u32]) -> Result, FormatError> { - Err(FormatError::UnsupportedFilter(FILTER_LZ4)) -} - /// Decompress zstd data. /// /// `expected_bytes` bounds the output (or [`MAX_DECOMPRESS_SIZE`] when @@ -1097,11 +1202,6 @@ fn zstd_decompress(data: &[u8], expected_bytes: usize) -> Result, Format Ok(out) } -#[cfg(not(feature = "zstd"))] -fn zstd_decompress(_data: &[u8], _expected_bytes: usize) -> Result, FormatError> { - Err(FormatError::UnsupportedFilter(FILTER_ZSTD)) -} - /// Compress data with zstd as one frame whose header records the content /// size. The registered HDF5 Zstandard filter (`H5Zzstd.c`, used by /// libhdf5 + hdf5plugin) sizes its output buffer from @@ -1113,11 +1213,6 @@ fn zstd_compress(data: &[u8], level: u32) -> Result, FormatError> { .map_err(|e| FormatError::CompressionError(format!("zstd: {e}"))) } -#[cfg(not(feature = "zstd"))] -fn zstd_compress(_data: &[u8], _level: u32) -> Result, FormatError> { - Err(FormatError::UnsupportedFilter(FILTER_ZSTD)) -} - /// Unshuffle (decompress direction): reconstruct interleaved element bytes. /// On disk: all byte-0s of each element together, then all byte-1s, etc. /// Output: elements in natural order. @@ -1351,7 +1446,7 @@ fn fletcher32_append(data: &[u8]) -> Result, FormatError> { // --------------------------------------------------------------------------- #[cfg(feature = "pcodec")] -fn pcodec_compress(data: &[u8], element_size: usize) -> Result, FormatError> { +pub(crate) fn pcodec_compress(data: &[u8], element_size: usize) -> Result, FormatError> { use pco::ChunkConfig; use pco::standalone::simple_compress; let config = ChunkConfig::default(); @@ -1389,11 +1484,6 @@ fn pcodec_compress(data: &[u8], element_size: usize) -> Result, FormatEr } } -#[cfg(not(feature = "pcodec"))] -fn pcodec_compress(_data: &[u8], _element_size: usize) -> Result, FormatError> { - Err(FormatError::UnsupportedFilter(FILTER_PCODEC)) -} - /// `expected_bytes` bounds the number of elements decoded: the output buffer /// is pre-sized to exactly `expected_bytes / element_size` elements and /// `simple_decompress_into` never writes past it, so a corrupted/hostile pco @@ -1452,17 +1542,41 @@ fn pcodec_decompress( } } -#[cfg(not(feature = "pcodec"))] -fn pcodec_decompress( - _data: &[u8], - _element_size: usize, - _expected_bytes: usize, -) -> Result, FormatError> { - Err(FormatError::UnsupportedFilter(FILTER_PCODEC)) -} - #[cfg(test)] mod tests { + + /// A chunk whose pipeline decodes to fewer bytes than the chunk holds is + /// an error naming the chunk, never a short buffer the reader pads. + #[test] + fn short_decoded_chunk_is_an_error() { + let pipeline = FilterPipeline { + version: 2, + filters: vec![FilterDescription { + filter_id: FILTER_SHUFFLE, + name: None, + flags: 0, + client_data: vec![4], + }], + }; + let full = [7u8; 32]; + assert_eq!( + decompress_chunk_exact(&full, &pipeline, 32, 4, 0, &[8]).unwrap(), + full + ); + let err = decompress_chunk_exact(&full[..16], &pipeline, 32, 4, 0, &[8, 0]).unwrap_err(); + let msg = err.to_string(); + assert!( + msg.contains("[8, 0]") && msg.contains("16") && msg.contains("32"), + "{msg}" + ); + // Every filter skipped: the stored bytes are the chunk, still checked. + assert!(decompress_chunk_exact(&full[..16], &pipeline, 32, 4, 1, &[0]).is_err()); + // Unknown chunk size: not checked. + assert_eq!( + decompress_chunk_exact(&full[..16], &pipeline, 0, 4, 0, &[0]).unwrap(), + &full[..16] + ); + } use super::*; use crate::filter_pipeline::FilterDescription; @@ -1677,7 +1791,7 @@ mod tests { // An unsupported filter is fine when the chunk skipped it. let unknown = FilterPipeline { version: 2, - filters: vec![filter(32000), filter(FILTER_DEFLATE)], + filters: vec![filter(32013), filter(FILTER_DEFLATE)], }; assert_eq!( decompress_chunk_masked(&deflated, &unknown, n, 8, 0b01).unwrap(), @@ -2644,6 +2758,10 @@ mod tests { #[cfg(feature = "pcodec")] fn pcodec_uses_private_id_and_reads_legacy_32023() { use crate::chunked_write::ChunkOptions; + #[cfg(feature = "std")] + let _guard = crate::filter_registry::tests::ID_32023 + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner); let opts = ChunkOptions { pcodec: true, ..Default::default() diff --git a/crates/clawhdf5-format/src/filters_bitshuffle.rs b/crates/clawhdf5-format/src/filters_bitshuffle.rs new file mode 100644 index 0000000..d20835e --- /dev/null +++ b/crates/clawhdf5-format/src/filters_bitshuffle.rs @@ -0,0 +1,438 @@ +//! Bitshuffle (HDF5 filter 32008) and the bit transpose it shares with blosc. +//! +//! **The transform.** A block of `n` elements (`n` a multiple of 8) of +//! `es` bytes each is viewed as an `n × 8·es` bit matrix — row *i* is +//! element *i*, column `8·j + k` is bit *k* (LSB first) of its byte *j* — and +//! transposed: the output is `8·es` rows of `n` bits, row `8·j + k` holding +//! bit *k* of byte *j* of every element in order, packed LSB first. That is +//! what `bshuf_trans_bit_elem` produces (checked against hdf5plugin's +//! library bit for bit). +//! +//! **The filter** (`bshuf_h5filter.c`). `cd_values`: `[0..2]` bitshuffle +//! version, `[2]` element size, `[3]` block size in elements (0 = default: +//! 8192 bytes' worth, rounded down to a multiple of 8, at least 128), +//! `[4]` compression (0 none, 2 LZ4, 3 Zstandard), `[5]` Zstandard level. +//! The chunk is cut into blocks of `block size` elements; the tail shorter +//! than a block is transposed as one block rounded down to a multiple of 8 +//! elements, and the last `n mod 8` elements are stored as they are. +//! Uncompressed, that is the whole chunk. Compressed, the chunk starts with a +//! 12-byte header — the decoded size (u64 big-endian) and the block size in +//! bytes (u32 big-endian) — and each transposed block is stored as a u32 +//! big-endian length and an LZ4 block / Zstandard frame; the untransposed +//! tail follows the last block. + +#[cfg(not(feature = "std"))] +extern crate alloc; +#[cfg(not(feature = "std"))] +use alloc::{format, vec, vec::Vec}; + +use crate::error::FormatError; +#[cfg(feature = "bitshuffle")] +use crate::filter_registry::FilterContext; + +/// Transpose an 8×8 bit matrix packed in a u64 (byte *r* = row *r*, bit *c* +/// of that byte = column *c*). An involution. +#[inline] +fn transpose8(mut x: u64) -> u64 { + let t = (x ^ (x >> 7)) & 0x00AA_00AA_00AA_00AA; + x = x ^ t ^ (t << 7); + let t = (x ^ (x >> 14)) & 0x0000_CCCC_0000_CCCC; + x = x ^ t ^ (t << 14); + let t = (x ^ (x >> 28)) & 0x0000_0000_F0F0_F0F0; + x ^ t ^ (t << 28) +} + +/// Bit-transpose one block: `input` and `out` are `n * es` bytes, `n` a +/// multiple of 8. +pub(crate) fn bitshuffle_block(input: &[u8], out: &mut [u8], n: usize, es: usize) { + debug_assert!(n.is_multiple_of(8) && input.len() == n * es && out.len() == n * es); + let row = n / 8; + for j in 0..es { + for g in 0..row { + let mut x = 0u64; + for t in 0..8 { + x |= u64::from(input[(8 * g + t) * es + j]) << (8 * t); + } + let y = transpose8(x); + for k in 0..8 { + out[(8 * j + k) * row + g] = (y >> (8 * k)) as u8; + } + } + } +} + +/// Undo [`bitshuffle_block`]. +pub(crate) fn bitunshuffle_block(input: &[u8], out: &mut [u8], n: usize, es: usize) { + debug_assert!(n.is_multiple_of(8) && input.len() == n * es && out.len() == n * es); + let row = n / 8; + for j in 0..es { + for g in 0..row { + let mut y = 0u64; + for k in 0..8 { + y |= u64::from(input[(8 * j + k) * row + g]) << (8 * k); + } + let x = transpose8(y); + for t in 0..8 { + out[(8 * g + t) * es + j] = (x >> (8 * t)) as u8; + } + } + } +} + +/// `bshuf_default_block_size`: 8 KiB of elements, a multiple of 8, >= 128. +#[cfg(feature = "bitshuffle")] +fn default_block_size(es: usize) -> usize { + ((8192 / es) / 8 * 8).max(128) +} + +#[cfg(feature = "bitshuffle")] +fn err(msg: &str) -> FormatError { + FormatError::DecompressionError(format!("bitshuffle: {msg}")) +} + +/// `cd_values[4]`: the compression bitshuffle applies after the transpose. +#[cfg(feature = "bitshuffle")] +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum Codec { + None, + Lz4, + Zstd, +} + +#[cfg(feature = "bitshuffle")] +fn codec(cd: &[u32]) -> Result { + match cd.get(4).copied().unwrap_or(0) { + 0 => Ok(Codec::None), + 2 => Ok(Codec::Lz4), + 3 => Ok(Codec::Zstd), + other => Err(FormatError::FilterError(format!( + "bitshuffle: unknown compression {other}" + ))), + } +} + +/// The element counts of the transposed blocks for `size` elements. +#[cfg(feature = "bitshuffle")] +fn blocks(size: usize, block: usize) -> impl Iterator { + let full = size / block; + let last = (size % block) / 8 * 8; + core::iter::repeat_n(block, full).chain((last > 0).then_some(last)) +} + +/// Decode a bitshuffle-filtered chunk. +#[cfg(feature = "bitshuffle")] +pub(crate) fn bitshuffle_decode( + input: &[u8], + ctx: &FilterContext<'_>, +) -> Result, FormatError> { + let cd = ctx.client_data(); + let es = match cd.get(2) { + Some(&e) if e != 0 => e as usize, + _ => return Err(err("missing element size")), + }; + let codec = codec(cd)?; + let limit = ctx.output_limit(); + if codec == Codec::None { + if input.len() > limit { + return Err(err("output exceeds the chunk size")); + } + let block = match cd.get(3) { + Some(&b) if b != 0 => b as usize, + _ => default_block_size(es), + }; + if !block.is_multiple_of(8) { + return Err(err("block size is not a multiple of 8")); + } + if !input.len().is_multiple_of(es) { + return Err(err("chunk is not a whole number of elements")); + } + let size = input.len() / es; + let mut out = vec![0u8; input.len()]; + let mut pos = 0; + for n in blocks(size, block) { + let bytes = n * es; + bitunshuffle_block(&input[pos..pos + bytes], &mut out[pos..pos + bytes], n, es); + pos += bytes; + } + out[pos..].copy_from_slice(&input[pos..]); + return Ok(out); + } + + let header = input.get(..12).ok_or_else(|| err("truncated header"))?; + let total = u64::from_be_bytes(header[..8].try_into().unwrap()); + let block_bytes = u32::from_be_bytes(header[8..12].try_into().unwrap()) as usize; + let total = usize::try_from(total) + .ok() + .filter(|&t| t <= limit) + .ok_or_else(|| err("decoded size exceeds the chunk size"))?; + if !total.is_multiple_of(es) { + return Err(err("chunk is not a whole number of elements")); + } + if block_bytes == 0 || !block_bytes.is_multiple_of(es) { + return Err(err("bad block size")); + } + let block = block_bytes / es; + if !block.is_multiple_of(8) { + return Err(err("block size is not a multiple of 8")); + } + let size = total / es; + let mut out = vec![0u8; total]; + let mut tmp = vec![0u8; block_bytes.min(total)]; + let mut ip = 12usize; + let mut op = 0usize; + let mut zstd = None; + for n in blocks(size, block) { + let bytes = n * es; + let len = input + .get(ip..ip + 4) + .map(|b| u32::from_be_bytes(b.try_into().unwrap()) as usize) + .ok_or_else(|| err("truncated block header"))?; + ip += 4; + let comp = input + .get(ip..ip.saturating_add(len)) + .ok_or_else(|| err("truncated block"))?; + ip += len; + let dst = &mut tmp[..bytes]; + let got = match codec { + Codec::Lz4 => lz4_flex::block::decompress_into(comp, dst) + .map_err(|e| err(&format!("lz4: {e}")))?, + Codec::Zstd => zstd_decode_into( + zstd.get_or_insert_with(ruzstd::decoding::FrameDecoder::new), + comp, + dst, + )?, + Codec::None => unreachable!(), + }; + if got != bytes { + return Err(err("block decoded to the wrong size")); + } + bitunshuffle_block(dst, &mut out[op..op + bytes], n, es); + op += bytes; + } + let tail = total - op; + let rest = input + .get(ip..ip + tail) + .ok_or_else(|| err("truncated trailing elements"))?; + out[op..].copy_from_slice(rest); + Ok(out) +} + +/// Decode Zstandard frames into exactly `dst`, failing if they hold more. +#[cfg(any(feature = "bitshuffle", feature = "blosc"))] +pub(crate) fn zstd_decode_into( + decoder: &mut ruzstd::decoding::FrameDecoder, + frames: &[u8], + dst: &mut [u8], +) -> Result { + decoder + .decode_all(frames, dst) + .map_err(|e| FormatError::DecompressionError(format!("zstd: {e}"))) +} + +/// Compress with ruzstd. It implements one level (roughly zstd's level 1), +/// so the requested level only matters to other encoders. +#[cfg(any(feature = "bitshuffle", feature = "blosc"))] +pub(crate) fn zstd_encode(data: &[u8]) -> Vec { + ruzstd::encoding::compress_to_vec(data, ruzstd::encoding::CompressionLevel::Fastest) +} + +/// Encode a chunk with the bitshuffle filter. +#[cfg(feature = "bitshuffle")] +pub(crate) fn bitshuffle_encode( + input: &[u8], + ctx: &FilterContext<'_>, +) -> Result, FormatError> { + let cd = ctx.client_data(); + let es = match cd.get(2) { + Some(&e) if e != 0 => e as usize, + _ => ctx.element_size.max(1), + }; + let codec = codec(cd)?; + let block = match cd.get(3) { + Some(&b) if b != 0 => b as usize, + _ => default_block_size(es), + }; + let cerr = |m: &str| FormatError::CompressionError(format!("bitshuffle: {m}")); + if !block.is_multiple_of(8) { + return Err(cerr("block size is not a multiple of 8")); + } + if !input.len().is_multiple_of(es) { + return Err(cerr("chunk is not a whole number of elements")); + } + let size = input.len() / es; + let mut out = Vec::with_capacity(input.len() + 12 + input.len() / 64); + if codec != Codec::None { + out.extend_from_slice(&(input.len() as u64).to_be_bytes()); + let block_bytes = + u32::try_from(block * es).map_err(|_| cerr("block size does not fit in 32 bits"))?; + out.extend_from_slice(&block_bytes.to_be_bytes()); + } + let mut tmp = vec![0u8; (block * es).min(input.len())]; + let mut pos = 0; + for n in blocks(size, block) { + let bytes = n * es; + let dst = &mut tmp[..bytes]; + bitshuffle_block(&input[pos..pos + bytes], dst, n, es); + match codec { + Codec::None => out.extend_from_slice(dst), + Codec::Lz4 | Codec::Zstd => { + let comp = if codec == Codec::Lz4 { + lz4_flex::block::compress(dst) + } else { + zstd_encode(dst) + }; + out.extend_from_slice(&(comp.len() as u32).to_be_bytes()); + out.extend_from_slice(&comp); + } + } + pos += bytes; + } + out.extend_from_slice(&input[pos..]); + Ok(out) +} + +#[cfg(test)] +mod tests { + use super::*; + + /// The definition, one bit at a time. + fn naive(input: &[u8], n: usize, es: usize) -> Vec { + let mut out = vec![0u8; n * es]; + for i in 0..n { + for j in 0..es { + for k in 0..8 { + if input[i * es + j] >> k & 1 == 1 { + let p = (8 * j + k) * n + i; + out[p / 8] |= 1 << (p % 8); + } + } + } + } + out + } + + #[test] + fn transpose_matches_the_definition_and_inverts() { + for (n, es) in [(8, 1), (16, 2), (24, 4), (128, 8), (64, 3), (8, 16)] { + let input: Vec = (0..n * es) + .map(|i| (i as u32).wrapping_mul(2_654_435_761).rotate_left(7) as u8) + .collect(); + let mut out = vec![0u8; n * es]; + bitshuffle_block(&input, &mut out, n, es); + assert_eq!(out, naive(&input, n, es), "n={n} es={es}"); + let mut back = vec![0u8; n * es]; + bitunshuffle_block(&out, &mut back, n, es); + assert_eq!(back, input); + } + } + + #[cfg(feature = "bitshuffle")] + fn ctx_for(cd: Vec) -> crate::filter_pipeline::FilterDescription { + crate::filter_pipeline::FilterDescription { + filter_id: crate::filter_pipeline::FILTER_BITSHUFFLE, + name: None, + flags: 0, + client_data: cd, + } + } + + #[cfg(feature = "bitshuffle")] + #[test] + fn filter_round_trips_every_mode() { + for es in [1usize, 2, 4, 8] { + for n in [0usize, 1, 7, 8, 100, 1000, 5003] { + let data: Vec = (0..n * es) + .map(|i| (i % 97) as u8 ^ (i / 300) as u8) + .collect(); + for (comp, block) in [(0, 0), (0, 16), (2, 0), (2, 64), (3, 0), (3, 1024)] { + let f = ctx_for(vec![0, 4, es as u32, block, comp]); + let ctx = FilterContext { + filter: &f, + element_size: es, + max_output: data.len(), + }; + let enc = bitshuffle_encode(&data, &ctx).unwrap(); + let dec = bitshuffle_decode(&enc, &ctx).unwrap(); + assert_eq!(dec, data, "es={es} n={n} comp={comp} block={block}"); + } + } + } + } + + #[cfg(feature = "bitshuffle")] + #[test] + fn rejects_oversized_and_truncated_chunks() { + let data = vec![5u8; 4096]; + let f = ctx_for(vec![0, 4, 4, 0, 2]); + let mut ctx = FilterContext { + filter: &f, + element_size: 4, + max_output: data.len(), + }; + let enc = bitshuffle_encode(&data, &ctx).unwrap(); + assert!(bitshuffle_decode(&enc[..enc.len() - 1], &ctx).is_err()); + ctx.max_output = 100; + assert!(bitshuffle_decode(&enc, &ctx).is_err()); + } + + /// Random and mutated chunks, in every mode, and hostile `cd_values`: + /// errors are fine, panics are not. + #[cfg(feature = "bitshuffle")] + #[test] + fn fuzzed_chunks_never_panic() { + use crate::test_fuzz::{Rng, fuzz_decoder}; + let data: Vec = (0..3001u32) + .flat_map(|i| ((i / 7) as u16).to_le_bytes()) + .collect(); + for (comp, block) in [(0, 0), (0, 16), (2, 0), (2, 64), (3, 0), (3, 1024)] { + let f = ctx_for(vec![0, 4, 2, block, comp]); + let ctx = FilterContext { + filter: &f, + element_size: 2, + max_output: data.len(), + }; + let seeds = vec![ + bitshuffle_encode(&data, &ctx).unwrap(), + bitshuffle_encode(&data[..34], &ctx).unwrap(), + bitshuffle_encode(&data[..512], &ctx).unwrap(), + ]; + fuzz_decoder( + 0xb5 + comp as u64 * 7 + block as u64, + &seeds, + 4_000, + data.len(), + |s| bitshuffle_decode(s, &ctx), + ); + } + // Hostile filter parameters on a valid chunk. + let mut rng = Rng::new(0xcd); + let good = ctx_for(vec![0, 4, 2, 0, 2]); + let enc = bitshuffle_encode( + &data, + &FilterContext { + filter: &good, + element_size: 2, + max_output: data.len(), + }, + ) + .unwrap(); + for _ in 0..3_000 { + let cd: Vec = (0..rng.below(7)) + .map(|_| match rng.below(4) { + 0 => rng.below(5) as u32, + 1 => u32::MAX - rng.below(4) as u32, + 2 => 1 << rng.below(32), + _ => rng.next_u64() as u32, + }) + .collect(); + let f = ctx_for(cd); + let ctx = FilterContext { + filter: &f, + element_size: 2, + max_output: data.len(), + }; + let _ = bitshuffle_decode(&enc, &ctx); + let _ = bitshuffle_decode(&data, &ctx); + } + } +} diff --git a/crates/clawhdf5-format/src/filters_blosc.rs b/crates/clawhdf5-format/src/filters_blosc.rs new file mode 100644 index 0000000..3edc964 --- /dev/null +++ b/crates/clawhdf5-format/src/filters_blosc.rs @@ -0,0 +1,711 @@ +//! Blosc 1 (HDF5 filter 32001, `hdf5-blosc`, hdf5plugin's `Blosc`), in pure +//! Rust: the Blosc 1 frame, its byte shuffle and bit shuffle, and the +//! BloscLZ, LZ4/LZ4HC, Snappy, Zlib and Zstandard codecs inside it. +//! +//! **Frame** (c-blosc 1.x, format version 2). A 16-byte header — version +//! (2), codec format version (1), flags, type size, then little-endian `u32` +//! decoded size, block size and frame size. Flags: bit 0 byte shuffle, bit +//! 1 stored raw ("memcpyed": the data follows the header), bit 2 bit +//! shuffle, bit 4 "do not split", bits 5-7 the codec (0 BloscLZ, 1 LZ4 and +//! LZ4HC, 2 Snappy, 3 Zlib, 4 Zstandard). Unless stored raw, a table of +//! `u32` block offsets follows, one per block of `block size` bytes (the +//! last one may be shorter). A block is one stream, or — when the "do not +//! split" flag is clear, the type size is at most 16, the block holds at +//! least 128 elements, and it is not the short last block — `type size` +//! streams, one per byte plane. Each stream is a `u32` length and the +//! codec's output; a length equal to the stream's decoded size means the +//! bytes are stored raw. The decoded block is then unshuffled (byte shuffle +//! for type size > 1; bit shuffle when the block holds a multiple of 8 +//! elements, the trailing partial element copied as is). +//! +//! **Filter** (`blosc_filter.c`) `cd_values`: `[0]` filter revision, `[1]` +//! Blosc format version, `[2]` type size, `[3]` chunk size in bytes, `[4]` +//! compression level, `[5]` shuffle (0 none, 1 byte, 2 bit), `[6]` +//! compressor (0 blosclz, 1 lz4, 2 lz4hc, 3 snappy, 4 zlib, 5 zstd). The +//! decoder needs only the frame. + +use crate::error::FormatError; +use crate::filter_registry::FilterContext; +use crate::filters_bitshuffle::{bitshuffle_block, bitunshuffle_block}; + +const HEADER: usize = 16; +const FLAG_SHUFFLE: u8 = 0x01; +const FLAG_MEMCPYED: u8 = 0x02; +const FLAG_BITSHUFFLE: u8 = 0x04; +const FLAG_FUTURE: u8 = 0x08; +const FLAG_DONT_SPLIT: u8 = 0x10; +const MAX_SPLITS: usize = 16; +const MIN_BUFFERSIZE: usize = 128; + +fn err(msg: &str) -> FormatError { + FormatError::DecompressionError(format!("blosc: {msg}")) +} + +fn le32(b: &[u8], at: usize) -> Result { + b.get(at..at + 4) + .map(|s| u32::from_le_bytes(s.try_into().unwrap()) as usize) + .ok_or_else(|| err("truncated frame")) +} + +/// The codec inside a Blosc frame (flags bits 5-7). +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum Codec { + BloscLz, + Lz4, + Snappy, + Zlib, + Zstd, +} + +impl Codec { + fn from_flags(flags: u8) -> Result { + match flags >> 5 { + 0 => Ok(Codec::BloscLz), + 1 => Ok(Codec::Lz4), + 2 => Ok(Codec::Snappy), + 3 => Ok(Codec::Zlib), + 4 => Ok(Codec::Zstd), + other => Err(err(&format!("unknown codec {other}"))), + } + } +} + +/// Decode one codec stream into exactly `dst`. +fn decode_stream( + codec: Codec, + src: &[u8], + dst: &mut [u8], + zstd: &mut Option, +) -> Result<(), FormatError> { + let n = match codec { + Codec::BloscLz => blosclz_decompress(src, dst), + Codec::Lz4 => { + lz4_flex::block::decompress_into(src, dst).map_err(|e| err(&format!("lz4: {e}")))? + } + Codec::Snappy => { + let len = snap::raw::decompress_len(src).map_err(|e| err(&format!("snappy: {e}")))?; + if len != dst.len() { + return Err(err("snappy stream has the wrong size")); + } + snap::raw::Decoder::new() + .decompress(src, dst) + .map_err(|e| err(&format!("snappy: {e}")))? + } + Codec::Zlib => { + let out = crate::filters::inflate_bounded(src, dst.len(), dst.len()) + .map_err(|e| err(&format!("zlib: {e}")))?; + let n = out.len(); + if n == dst.len() { + dst.copy_from_slice(&out); + } + n + } + Codec::Zstd => crate::filters_bitshuffle::zstd_decode_into( + zstd.get_or_insert_with(ruzstd::decoding::FrameDecoder::new), + src, + dst, + )?, + }; + if n != dst.len() { + return Err(err("stream decoded to the wrong size")); + } + Ok(()) +} + +/// Decode a Blosc-filtered chunk: one Blosc 1 frame. +/// +/// An HDF5 chunk is never empty, so a frame that decodes to nothing where +/// the chunk size is known is corrupt (libhdf5's filter fails it too). +pub(crate) fn blosc_decode(input: &[u8], ctx: &FilterContext<'_>) -> Result, FormatError> { + let out = blosc_decompress(input, ctx.output_limit())?; + if out.is_empty() && ctx.max_output != 0 { + return Err(err("empty frame for a non-empty chunk")); + } + Ok(out) +} + +/// Decompress a Blosc 1 frame, refusing more than `limit` bytes of output. +pub fn blosc_decompress(input: &[u8], limit: usize) -> Result, FormatError> { + if input.len() < HEADER { + return Err(err("truncated header")); + } + let version = input[0]; + let codec_version = input[1]; + let flags = input[2]; + let typesize = input[3] as usize; + let nbytes = le32(input, 4)?; + let blocksize = le32(input, 8)?; + let cbytes = le32(input, 12)?; + if version != 1 && version != 2 { + return Err(err(&format!( + "frame format version {version} is not Blosc 1 (a Blosc 2 chunk?)" + ))); + } + if flags & FLAG_FUTURE != 0 { + return Err(err("unknown header flags")); + } + if nbytes > limit { + return Err(err("decoded size exceeds the chunk size")); + } + if cbytes > input.len() { + return Err(err("frame is longer than the chunk")); + } + if cbytes < HEADER { + return Err(err("truncated frame")); + } + let src = &input[..cbytes]; + if nbytes == 0 { + return Ok(Vec::new()); + } + if blocksize == 0 || typesize == 0 { + return Err(err("bad block or type size")); + } + let mut out = vec![0u8; nbytes]; + if flags & FLAG_MEMCPYED != 0 { + if cbytes != nbytes + HEADER { + return Err(err("stored frame has the wrong size")); + } + out.copy_from_slice(&src[HEADER..]); + return Ok(out); + } + let codec = Codec::from_flags(flags)?; + if codec_version != 1 { + return Err(err(&format!( + "unsupported {codec:?} format version {codec_version}" + ))); + } + let nblocks = nbytes.div_ceil(blocksize); + let leftover = nbytes % blocksize; + if nblocks > (cbytes - HEADER) / 4 { + return Err(err("block table is truncated")); + } + let block_len = blocksize.min(nbytes); + let mut tmp = vec![0u8; block_len]; + let mut zstd = None; + let dont_split = flags & FLAG_DONT_SPLIT != 0; + for j in 0..nblocks { + let is_leftover = j == nblocks - 1 && leftover > 0; + let bsize = if is_leftover { leftover } else { blocksize }; + let nsplits = if !dont_split + && typesize <= MAX_SPLITS + && bsize / typesize >= MIN_BUFFERSIZE + && !is_leftover + { + typesize + } else { + 1 + }; + let neblock = bsize / nsplits; + let mut pos = le32(src, HEADER + 4 * j)?; + let tmp = &mut tmp[..bsize]; + for s in 0..nsplits { + let clen = src + .get(pos..) + .and_then(|rest| rest.get(..4)) + .map(|b| u32::from_le_bytes(b.try_into().unwrap()) as usize) + .ok_or_else(|| err("block offset out of range"))?; + pos += 4; + let stream = src + .get(pos..pos.saturating_add(clen)) + .ok_or_else(|| err("stream runs past the frame"))?; + let dst = &mut tmp[s * neblock..(s + 1) * neblock]; + if clen == neblock { + dst.copy_from_slice(stream); + } else { + decode_stream(codec, stream, dst, &mut zstd)?; + } + pos += clen; + } + // `bsize` is a whole number of splits by construction (`nsplits` > 1 + // only for full blocks, and c-blosc sizes those in whole elements). + if nsplits * neblock != bsize { + return Err(err("block is not a whole number of streams")); + } + let dest = &mut out[j * blocksize..j * blocksize + bsize]; + unshuffle_block(flags, typesize, tmp, dest); + } + Ok(out) +} + +/// Undo the frame's shuffle on one decoded block. +fn unshuffle_block(flags: u8, typesize: usize, src: &[u8], dest: &mut [u8]) { + let bsize = src.len(); + if flags & FLAG_SHUFFLE != 0 && typesize > 1 { + let n = bsize / typesize; + for i in 0..n { + for b in 0..typesize { + dest[i * typesize + b] = src[b * n + i]; + } + } + dest[n * typesize..].copy_from_slice(&src[n * typesize..]); + } else if flags & FLAG_BITSHUFFLE != 0 && bsize >= typesize { + let n = bsize / typesize; + if n.is_multiple_of(8) { + let body = n * typesize; + bitunshuffle_block(&src[..body], &mut dest[..body], n, typesize); + dest[body..].copy_from_slice(&src[body..]); + } else { + dest.copy_from_slice(src); + } + } else { + dest.copy_from_slice(src); + } +} + +/// BloscLZ decompression (c-blosc 1.21 `blosclz_decompress`): returns the +/// number of bytes written, or 0 on malformed input — exactly as the C +/// decoder, including stopping before a match that ends the stream, so a +/// stream libblosc rejects is rejected here too. +/// +/// Instructions: a control byte `ctrl`. Below 32, a literal run of +/// `ctrl + 1` bytes. Otherwise a match: length `(ctrl >> 5) + 2`, extended +/// by following bytes while they are 255 when the top three bits are all +/// set; distance `((ctrl & 31) << 8) + next byte + 1`, or — when that byte +/// is 255 and the high bits are 31 — a 16-bit big-endian distance plus 8192. +/// The first instruction is always a literal. +pub(crate) fn blosclz_decompress(input: &[u8], out: &mut [u8]) -> usize { + const MAX_DISTANCE: usize = 8191; + let limit = input.len(); + if limit == 0 { + return 0; + } + let mut ip = 1usize; + let mut op = 0usize; + let mut ctrl = (input[0] & 31) as usize; + loop { + if ctrl >= 32 { + let mut len = (ctrl >> 5) - 1; + let ofs = (ctrl & 31) << 8; + if len == 6 { + loop { + if ip + 1 >= limit { + return 0; + } + let code = input[ip] as usize; + ip += 1; + len += code; + if code != 255 { + break; + } + } + } else if ip + 1 >= limit { + return 0; + } + let code = input[ip] as usize; + ip += 1; + len += 3; + // The copy source is `distance` bytes back. + let mut distance = ofs + code + 1; + if code == 255 && ofs == 31 << 8 { + if ip + 1 >= limit { + return 0; + } + let far = ((input[ip] as usize) << 8) + input[ip + 1] as usize; + ip += 2; + distance = far + MAX_DISTANCE + 1; + } + if op + len > out.len() { + return 0; + } + if distance > op { + return 0; + } + if ip >= limit { + break; + } + ctrl = input[ip] as usize; + ip += 1; + let start = op - distance; + if distance >= len { + out.copy_within(start..start + len, op); + } else { + for k in 0..len { + out[op + k] = out[start + k]; + } + } + op += len; + } else { + let run = ctrl + 1; + if op + run > out.len() || ip + run > limit { + return 0; + } + out[op..op + run].copy_from_slice(&input[ip..ip + run]); + op += run; + ip += run; + if ip >= limit { + break; + } + ctrl = input[ip] as usize; + ip += 1; + } + } + op +} + +/// The codec our encoder puts inside the frame. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) enum EncodeCodec { + Lz4, + Snappy, + Zlib, + Zstd, +} + +impl EncodeCodec { + /// From the filter's `cd_values[6]` compressor code. + fn from_cd(code: u32) -> Result { + match code { + 1 | 2 => Ok(EncodeCodec::Lz4), + 3 => Ok(EncodeCodec::Snappy), + 4 => Ok(EncodeCodec::Zlib), + 5 => Ok(EncodeCodec::Zstd), + 0 => Err(FormatError::CompressionError( + "blosc: clawhdf5 cannot write BloscLZ; choose lz4, snappy, zlib or zstd".into(), + )), + other => Err(FormatError::CompressionError(format!( + "blosc: unknown compressor {other}" + ))), + } + } + + fn flags(self) -> u8 { + (match self { + EncodeCodec::Lz4 => 1, + EncodeCodec::Snappy => 2, + EncodeCodec::Zlib => 3, + EncodeCodec::Zstd => 4, + }) << 5 + } + + fn encode(self, data: &[u8], level: u32) -> Result, FormatError> { + match self { + EncodeCodec::Lz4 => Ok(lz4_flex::block::compress(data)), + EncodeCodec::Snappy => snap::raw::Encoder::new() + .compress_vec(data) + .map_err(|e| FormatError::CompressionError(format!("blosc: snappy: {e}"))), + EncodeCodec::Zlib => crate::filters::deflate_bounded(data, level.min(9)) + .map_err(|e| FormatError::CompressionError(format!("blosc: zlib: {e}"))), + EncodeCodec::Zstd => Ok(crate::filters_bitshuffle::zstd_encode(data)), + } + } +} + +/// Block size our encoder uses: at most 256 KiB, a whole number of +/// elements (and, for bit shuffle, of 8-element groups). +fn encode_block_size(nbytes: usize, typesize: usize, bitshuffle: bool) -> usize { + let unit = if bitshuffle { 8 * typesize } else { typesize }; + let target = (256 * 1024).min(nbytes); + if target < unit { + return nbytes.max(1); + } + target / unit * unit +} + +/// Encode a chunk as one Blosc 1 frame. `cd_values` as hdf5-blosc: +/// `[2]` type size, `[4]` level (0 = store), `[5]` shuffle, `[6]` codec. +pub(crate) fn blosc_encode(input: &[u8], ctx: &FilterContext<'_>) -> Result, FormatError> { + let cd = ctx.client_data(); + let cerr = |m: &str| FormatError::CompressionError(format!("blosc: {m}")); + let typesize = match cd.get(2) { + Some(&t) if t != 0 => t as usize, + _ => ctx.element_size.max(1), + }; + // Blosc records the type size in one byte; c-blosc treats larger types + // as bytes. + let typesize = if typesize > 255 { 1 } else { typesize }; + let level = cd.get(4).copied().unwrap_or(5); + let shuffle = cd.get(5).copied().unwrap_or(1); + let codec = EncodeCodec::from_cd(cd.get(6).copied().unwrap_or(1))?; + let nbytes = input.len(); + if nbytes > i32::MAX as usize - HEADER { + return Err(cerr("chunk too large for a Blosc frame")); + } + let mut flags = codec.flags(); + match shuffle { + 0 => {} + 1 => flags |= FLAG_SHUFFLE, + 2 => flags |= FLAG_BITSHUFFLE, + other => return Err(cerr(&format!("unknown shuffle mode {other}"))), + } + let blocksize = encode_block_size(nbytes, typesize, shuffle == 2); + let header = |flags: u8, blocksize: usize, cbytes: usize| { + let mut h = Vec::with_capacity(HEADER); + h.extend_from_slice(&[2, 1, flags, typesize as u8]); + h.extend_from_slice(&(nbytes as u32).to_le_bytes()); + h.extend_from_slice(&(blocksize as u32).to_le_bytes()); + h.extend_from_slice(&(cbytes as u32).to_le_bytes()); + h + }; + let stored = || { + let mut out = header( + FLAG_MEMCPYED | (flags & !(FLAG_SHUFFLE | FLAG_BITSHUFFLE)), + blocksize, + nbytes + HEADER, + ); + out.extend_from_slice(input); + out + }; + if level == 0 || nbytes == 0 { + return Ok(stored()); + } + + let nblocks = nbytes.div_ceil(blocksize); + let leftover = nbytes % blocksize; + let mut body = Vec::with_capacity(nbytes / 2); + let mut starts = Vec::with_capacity(nblocks); + let table_end = HEADER + 4 * nblocks; + let mut shuffled = vec![0u8; blocksize]; + for j in 0..nblocks { + let is_leftover = j == nblocks - 1 && leftover > 0; + let bsize = if is_leftover { leftover } else { blocksize }; + let block = &input[j * blocksize..j * blocksize + bsize]; + let sh = &mut shuffled[..bsize]; + shuffle_block(flags, typesize, block, sh); + starts.push(table_end + body.len()); + let nsplits = if typesize <= MAX_SPLITS + && bsize / typesize >= MIN_BUFFERSIZE + && !is_leftover + && bsize.is_multiple_of(typesize) + { + typesize + } else { + 1 + }; + let neblock = bsize / nsplits; + for s in 0..nsplits { + let part = &sh[s * neblock..(s + 1) * neblock]; + let comp = codec.encode(part, level)?; + if comp.len() < neblock { + body.extend_from_slice(&(comp.len() as u32).to_le_bytes()); + body.extend_from_slice(&comp); + } else { + body.extend_from_slice(&(neblock as u32).to_le_bytes()); + body.extend_from_slice(part); + } + } + if table_end + body.len() >= nbytes + HEADER { + // Incompressible: store instead, as c-blosc does. + return Ok(stored()); + } + } + // A split block must decode as split: the decoder infers splitting from + // the same rule, which requires a whole number of elements per block. + let cbytes = table_end + body.len(); + let mut out = header(flags, blocksize, cbytes); + for s in starts { + out.extend_from_slice(&(s as u32).to_le_bytes()); + } + out.extend_from_slice(&body); + Ok(out) +} + +/// Apply the frame's shuffle to one block (the inverse of +/// [`unshuffle_block`]). +fn shuffle_block(flags: u8, typesize: usize, src: &[u8], dest: &mut [u8]) { + let bsize = src.len(); + if flags & FLAG_SHUFFLE != 0 && typesize > 1 { + let n = bsize / typesize; + for i in 0..n { + for b in 0..typesize { + dest[b * n + i] = src[i * typesize + b]; + } + } + dest[n * typesize..].copy_from_slice(&src[n * typesize..]); + } else if flags & FLAG_BITSHUFFLE != 0 && bsize >= typesize { + let n = bsize / typesize; + if n.is_multiple_of(8) { + let body = n * typesize; + bitshuffle_block(&src[..body], &mut dest[..body], n, typesize); + dest[body..].copy_from_slice(&src[body..]); + } else { + dest.copy_from_slice(src); + } + } else { + dest.copy_from_slice(src); + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::filter_pipeline::{FILTER_BLOSC, FilterDescription}; + + /// A blosclz stream: literal "abc", then a 9-byte match 3 back (a run + /// of "abc"), then literal "Z". + #[test] + fn blosclz_decodes_literals_and_overlapping_matches() { + // Match: length (ctrl >> 5) + 2 = 8, distance ofs + code + 1 = 3. + let stream = [2, b'a', b'b', b'c', (6 << 5), 2, 0, b'Z']; + let mut out = [0u8; 12]; + assert_eq!(blosclz_decompress(&stream, &mut out), 12); + assert_eq!(&out, b"abcabcabcabZ"); + // A stream cut inside a match is malformed. + let mut out = [0u8; 11]; + assert_eq!(blosclz_decompress(&stream[..6], &mut out), 0); + // A match before the start of the output is malformed. + assert_eq!(blosclz_decompress(&[0, b'a', 32, 5, 0, b'x'], &mut out), 0); + } + + fn desc(cd: Vec) -> FilterDescription { + FilterDescription { + filter_id: FILTER_BLOSC, + name: None, + flags: 0, + client_data: cd, + } + } + + #[test] + fn frame_round_trips_every_codec_and_shuffle() { + for ts in [1usize, 2, 4, 8, 3, 32] { + for n in [0usize, 5, 100, 1000, 70_000, 300_001] { + if n * ts > 1 << 20 && ts > 1 { + continue; + } + let data: Vec = (0..n * ts) + .map(|i| ((i / ts) % 200) as u8 ^ (i % ts) as u8) + .collect(); + for codec in [1u32, 3, 4, 5] { + for shuffle in [0u32, 1, 2] { + for level in [0u32, 5] { + let f = desc(vec![2, 2, ts as u32, 0, level, shuffle, codec]); + let ctx = FilterContext { + filter: &f, + element_size: ts, + max_output: data.len(), + }; + let enc = blosc_encode(&data, &ctx).unwrap(); + let dec = blosc_decode(&enc, &ctx).unwrap_or_else(|e| { + panic!("ts={ts} n={n} codec={codec} shuffle={shuffle}: {e}") + }); + assert!( + dec == data, + "ts={ts} n={n} codec={codec} shuffle={shuffle} level={level}" + ); + } + } + } + } + } + } + + #[test] + fn rejects_bad_frames() { + let data = vec![9u8; 50_000]; + let f = desc(vec![2, 2, 4, 0, 5, 1, 1]); + let ctx = FilterContext { + filter: &f, + element_size: 4, + max_output: data.len(), + }; + let enc = blosc_encode(&data, &ctx).unwrap(); + assert!(blosc_decode(&enc[..enc.len() - 3], &ctx).is_err()); + let small = FilterContext { + max_output: 49_999, + ..ctx + }; + assert!(blosc_decode(&enc, &small).is_err()); + let mut v3 = enc.clone(); + v3[0] = 3; + assert!(blosc_decode(&v3, &ctx).is_err()); + let f0 = desc(vec![2, 2, 4, 0, 5, 1, 0]); + let ctx0 = FilterContext { filter: &f0, ..ctx }; + assert!(blosc_encode(&data, &ctx0).is_err()); + } + + /// A frame that declares no data, for a chunk that has some. + #[test] + fn empty_frame_for_a_non_empty_chunk_is_an_error() { + let mut frame = vec![2u8, 1, 0x20, 4]; + for v in [0u32, 64, 16] { + frame.extend_from_slice(&v.to_le_bytes()); + } + assert_eq!(blosc_decompress(&frame, 64).unwrap(), b""); + let f = desc(vec![2, 2, 4, 64, 5, 1, 1]); + let ctx = FilterContext { + filter: &f, + element_size: 4, + max_output: 64, + }; + assert!(blosc_decode(&frame, &ctx).is_err()); + } + + /// A frame whose header claims a compressed size smaller than the + /// header itself, not stored raw: an error, not an arithmetic overflow + /// (it panicked in debug builds). + #[test] + fn frame_size_below_the_header_is_an_error() { + let mut frame = vec![2u8, 1, 1 << 5, 4]; + for v in [64u32, 64, 8] { + frame.extend_from_slice(&v.to_le_bytes()); + } + frame.extend_from_slice(&[0; 40]); + assert!(blosc_decompress(&frame, 1000).is_err()); + for cbytes in 0..16u32 { + frame[12..16].copy_from_slice(&cbytes.to_le_bytes()); + assert!(blosc_decompress(&frame, 1000).is_err(), "cbytes={cbytes}"); + } + } + + /// A BloscLZ frame (our encoder cannot write one): a single block, + /// one stream, no shuffle. + fn blosclz_frame() -> Vec { + let stream = [2, b'a', b'b', b'c', (6 << 5), 2, 0, b'Z']; + let mut f = vec![2u8, 1, 0, 1]; + for v in [12u32, 12, (HEADER + 4 + 4 + stream.len()) as u32] { + f.extend_from_slice(&v.to_le_bytes()); + } + f.extend_from_slice(&((HEADER + 4) as u32).to_le_bytes()); + f.extend_from_slice(&(stream.len() as u32).to_le_bytes()); + f.extend_from_slice(&stream); + f + } + + /// Random and mutated frames, every codec and shuffle: errors are fine, + /// panics are not. + #[test] + fn fuzzed_frames_never_panic() { + let limit = 6000; + let data: Vec = (0..1500u32).flat_map(|i| (i / 5).to_le_bytes()).collect(); + let mut seeds = vec![blosclz_frame()]; + for codec in [1u32, 3, 4, 5] { + for shuffle in [0u32, 1, 2] { + for (ts, n) in [(4usize, data.len()), (4, 520), (1, 300), (2, 4)] { + let f = desc(vec![2, 2, ts as u32, 0, 5, shuffle, codec]); + let ctx = FilterContext { + filter: &f, + element_size: ts, + max_output: n, + }; + seeds.push(blosc_encode(&data[..n], &ctx).unwrap()); + } + } + } + // Stored raw. + let f = desc(vec![2, 2, 4, 0, 0, 1, 1]); + let ctx = FilterContext { + filter: &f, + element_size: 4, + max_output: 64, + }; + seeds.push(blosc_encode(&data[..64], &ctx).unwrap()); + crate::test_fuzz::fuzz_decoder(0xb10, &seeds, 30_000, limit, |s| { + blosc_decompress(s, limit) + }); + } + + /// BloscLZ streams on their own, random and mutated. + #[test] + fn fuzzed_blosclz_streams_never_panic() { + let seed = blosclz_frame()[HEADER + 8..].to_vec(); + let mut out = [0u8; 64]; + crate::test_fuzz::fuzz_decoder(0xb11, &[seed], 30_000, 64, |s| { + let n = blosclz_decompress(s, &mut out); + if n == 0 { + Err(err("malformed")) + } else { + Ok(out[..n].to_vec()) + } + }); + } +} diff --git a/crates/clawhdf5-format/src/filters_bzip2.rs b/crates/clawhdf5-format/src/filters_bzip2.rs new file mode 100644 index 0000000..93e33a3 --- /dev/null +++ b/crates/clawhdf5-format/src/filters_bzip2.rs @@ -0,0 +1,132 @@ +//! bzip2 (HDF5 filter 307, PyTables' `H5Zbzip2.c`, hdf5plugin's `BZip2`). +//! +//! The chunk is one bzip2 stream; `cd_values[0]` is the block size (1-9, +//! the compression level). Decoded with the `bzip2` crate's default backend, +//! `libbz2-rs-sys`, a pure-Rust port of libbzip2. + +use crate::error::FormatError; +use crate::filter_registry::FilterContext; + +fn err(msg: &str) -> FormatError { + FormatError::DecompressionError(format!("bzip2: {msg}")) +} + +/// Decode a bzip2-filtered chunk, refusing output beyond the chunk size. +pub(crate) fn bzip2_decode(input: &[u8], ctx: &FilterContext<'_>) -> Result, FormatError> { + use bzip2::{Decompress, Status}; + let limit = ctx.output_limit(); + let max_capacity = limit.saturating_add(1); + let hint = if ctx.max_output != 0 { + ctx.max_output + } else { + input.len().saturating_mul(4) + }; + let mut out = Vec::new(); + out.try_reserve_exact(hint.clamp(1, max_capacity)) + .map_err(|_| err("cannot allocate the output buffer"))?; + let mut dec = Decompress::new(false); + loop { + let (in_before, out_before) = (dec.total_in(), dec.total_out()); + let status = dec + .decompress_vec(&input[in_before as usize..], &mut out) + .map_err(|e| err(&e.to_string()))?; + if out.len() > limit { + return Err(err("output exceeds the chunk size")); + } + if status == Status::StreamEnd { + return Ok(out); + } + if out.len() == out.capacity() { + let grow = out + .capacity() + .min(max_capacity.saturating_sub(out.capacity())) + .max(1); + out.try_reserve_exact(grow) + .map_err(|_| err("cannot allocate the output buffer"))?; + } else if dec.total_in() as usize >= input.len() + || (dec.total_in(), dec.total_out()) == (in_before, out_before) + { + return Err(err("truncated stream")); + } + } +} + +/// Encode a chunk as one bzip2 stream at block size `cd_values[0]` +/// (default 9, as hdf5plugin). +pub(crate) fn bzip2_encode(input: &[u8], ctx: &FilterContext<'_>) -> Result, FormatError> { + use bzip2::{Action, Compress, Compression, Status}; + let level = ctx.client_data().first().copied().unwrap_or(9).clamp(1, 9); + let cerr = |m: String| FormatError::CompressionError(format!("bzip2: {m}")); + let mut enc = Compress::new(Compression::new(level), 0); + // bzip2's worst case is about 1% + 600 bytes over the input. + let mut out = Vec::with_capacity(input.len() + input.len() / 100 + 600); + loop { + let consumed = enc.total_in() as usize; + let status = enc + .compress_vec(&input[consumed..], &mut out, Action::Finish) + .map_err(|e| cerr(e.to_string()))?; + if status == Status::StreamEnd { + return Ok(out); + } + if out.len() == out.capacity() { + out.reserve(out.capacity().max(4096)); + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::filter_pipeline::{FILTER_BZIP2, FilterDescription}; + + fn desc(level: u32) -> FilterDescription { + FilterDescription { + filter_id: FILTER_BZIP2, + name: None, + flags: 0, + client_data: vec![level], + } + } + + #[test] + fn round_trips_and_bounds() { + let data: Vec = (0..100_000u32) + .flat_map(|i| (i % 777).to_le_bytes()) + .collect(); + for level in [1, 5, 9] { + let f = desc(level); + let ctx = FilterContext { + filter: &f, + element_size: 4, + max_output: data.len(), + }; + let enc = bzip2_encode(&data, &ctx).unwrap(); + assert!(enc.len() < data.len() / 4); + assert_eq!(bzip2_decode(&enc, &ctx).unwrap(), data); + // Truncated, and larger than the chunk: errors, not data. + assert!(bzip2_decode(&enc[..enc.len() / 2], &ctx).is_err()); + let small = FilterContext { + max_output: data.len() - 1, + ..ctx + }; + assert!(bzip2_decode(&enc, &small).is_err()); + } + } + + /// Random and mutated streams: errors are fine, panics are not. + #[test] + fn fuzzed_streams_never_panic() { + let f = desc(9); + let data: Vec = (0..4000u32).flat_map(|i| (i % 91).to_le_bytes()).collect(); + let ctx = FilterContext { + filter: &f, + element_size: 4, + max_output: data.len(), + }; + let seeds = vec![ + bzip2_encode(&data, &ctx).unwrap(), + bzip2_encode(&data[..40], &ctx).unwrap(), + ]; + crate::test_fuzz::fuzz_decoder(0xb2, &seeds, 3_000, data.len(), |s| bzip2_decode(s, &ctx)); + } +} diff --git a/crates/clawhdf5-format/src/filters_lzf.rs b/crates/clawhdf5-format/src/filters_lzf.rs new file mode 100644 index 0000000..acb9763 --- /dev/null +++ b/crates/clawhdf5-format/src/filters_lzf.rs @@ -0,0 +1,260 @@ +//! LZF (HDF5 filter 32000) — h5py's built-in compression filter +//! (`compression="lzf"`), in pure Rust. +//! +//! The chunk is one raw LZF stream (liblzf 3.x format, no header). The +//! stream is a sequence of instructions, each starting with a control byte: +//! +//! * `000LLLLL` — a literal run: the next `L + 1` bytes (1..=32) are copied. +//! * `LLLOOOOO [E] OOOOOOOO` — a back reference: copy `len + 2` bytes from +//! `distance` bytes back, where `len` is the top three bits (1..=6), or +//! `7 + E` when they are all ones, and `distance` is the 13-bit offset +//! (high five bits in the control byte, low eight in the last byte) plus 1. +//! +//! h5py's filter (`lzf_filter.c`) records the chunk's size in bytes in +//! `cd_values[2]` (slots 0 and 1 hold the filter and liblzf versions) and +//! sizes its output buffer from it. + +#[cfg(not(feature = "std"))] +extern crate alloc; +#[cfg(not(feature = "std"))] +use alloc::{format, vec, vec::Vec}; + +use crate::error::FormatError; +use crate::filter_registry::FilterContext; + +/// `H5PY_FILTER_LZF_VERSION`, written to `cd_values[0]`. +pub const LZF_FILTER_VERSION: u32 = 4; +/// `LZF_VERSION` (liblzf 1.5), written to `cd_values[1]`. +pub const LZF_API_VERSION: u32 = 0x0105; + +const MAX_LITERAL: usize = 32; +const MAX_OFFSET: usize = 1 << 13; +const MAX_REF: usize = (1 << 8) + (1 << 3); +const HASH_LOG: u32 = 14; + +fn err(msg: &str) -> FormatError { + FormatError::DecompressionError(format!("lzf: {msg}")) +} + +/// Decode an LZF-filtered chunk. +pub(crate) fn lzf_decode(input: &[u8], ctx: &FilterContext<'_>) -> Result, FormatError> { + let limit = ctx.output_limit(); + let hint = match ctx.client_data().get(2) { + Some(&n) if n != 0 => n as usize, + _ => input.len().saturating_mul(2), + }; + lzf_decompress(input, hint.min(limit), limit) +} + +/// Decompress a raw LZF stream, refusing to produce more than `limit` bytes. +pub fn lzf_decompress( + input: &[u8], + size_hint: usize, + limit: usize, +) -> Result, FormatError> { + let mut out: Vec = Vec::new(); + out.try_reserve(size_hint) + .map_err(|_| err("cannot allocate the output buffer"))?; + let mut ip = 0usize; + while ip < input.len() { + let ctrl = input[ip] as usize; + ip += 1; + if ctrl < 32 { + let run = ctrl + 1; + let lit = input + .get(ip..ip + run) + .ok_or_else(|| err("literal run past the end of the input"))?; + if out.len() + run > limit { + return Err(err("output exceeds the chunk size")); + } + out.extend_from_slice(lit); + ip += run; + } else { + let mut len = ctrl >> 5; + if len == 7 { + len += *input + .get(ip) + .ok_or_else(|| err("truncated back reference"))? + as usize; + ip += 1; + } + let low = *input + .get(ip) + .ok_or_else(|| err("truncated back reference"))? as usize; + ip += 1; + let distance = ((ctrl & 0x1f) << 8) + low + 1; + let len = len + 2; + if distance > out.len() { + return Err(err("back reference before the start of the output")); + } + if out.len() + len > limit { + return Err(err("output exceeds the chunk size")); + } + let start = out.len() - distance; + if distance >= len { + out.extend_from_within(start..start + len); + } else { + // Overlapping copy: repeats the last `distance` bytes. + for k in 0..len { + let b = out[start + k]; + out.push(b); + } + } + } + } + Ok(out) +} + +/// Encode a chunk with the LZF filter. +pub(crate) fn lzf_encode(input: &[u8], _ctx: &FilterContext<'_>) -> Result, FormatError> { + Ok(lzf_compress(input)) +} + +fn hash3(b: &[u8]) -> usize { + let v = (u32::from(b[0]) << 16) | (u32::from(b[1]) << 8) | u32::from(b[2]); + (v.wrapping_mul(2_654_435_761) >> (32 - HASH_LOG)) as usize +} + +fn flush_literals(out: &mut Vec, lit: &[u8]) { + for run in lit.chunks(MAX_LITERAL) { + out.push((run.len() - 1) as u8); + out.extend_from_slice(run); + } +} + +/// Compress `input` into a raw LZF stream any liblzf decoder reads. +/// +/// Incompressible input grows by one byte per 32. (h5py's own filter gives +/// up on such a chunk and stores it unfiltered; storing the slightly larger +/// stream is equally readable.) +pub fn lzf_compress(input: &[u8]) -> Vec { + let n = input.len(); + let mut out = Vec::with_capacity(n + n / MAX_LITERAL + 1); + let mut table = vec![0u32; 1 << HASH_LOG]; + let mut lit_start = 0usize; + let mut i = 0usize; + while i + 2 < n { + let h = hash3(&input[i..]); + let cand = table[h] as usize; + table[h] = (i + 1) as u32; + if cand != 0 { + let r = cand - 1; + let distance = i - r; + if distance <= MAX_OFFSET && input[r..r + 3] == input[i..i + 3] { + let max_len = (n - i).min(MAX_REF); + let mut len = 3; + while len < max_len && input[r + len] == input[i + len] { + len += 1; + } + flush_literals(&mut out, &input[lit_start..i]); + let code = len - 2; + let off = distance - 1; + if code < 7 { + out.push(((code << 5) | (off >> 8)) as u8); + } else { + out.push(((7 << 5) | (off >> 8)) as u8); + out.push((code - 7) as u8); + } + out.push((off & 0xff) as u8); + // Index the positions the match covered so later data can + // refer back into it. + let end = i + len; + let mut j = i + 1; + while j < end && j + 2 < n { + table[hash3(&input[j..])] = (j + 1) as u32; + j += 1; + } + i = end; + lit_start = i; + continue; + } + } + i += 1; + } + flush_literals(&mut out, &input[lit_start..]); + out +} + +#[cfg(test)] +mod tests { + use super::*; + + fn round_trip(data: &[u8]) { + let c = lzf_compress(data); + assert_eq!(lzf_decompress(&c, data.len(), data.len()).unwrap(), data); + } + + #[test] + fn round_trips() { + round_trip(b""); + round_trip(b"a"); + round_trip(b"abcabcabcabcabcabcabcabcabcabcabcabc"); + round_trip(&[7u8; 10_000]); + let noise: Vec = (0..70_000u32) + .map(|i| (i.wrapping_mul(2_654_435_761) >> 13) as u8) + .collect(); + round_trip(&noise); + let ramp: Vec = (0..100_000u32) + .flat_map(|i| (i % 1000).to_le_bytes()) + .collect(); + round_trip(&ramp); + } + + #[test] + fn compresses_repetitive_data() { + let data = [42u8; 4096]; + assert!(lzf_compress(&data).len() < 100); + } + + /// The chunk h5py 3.16's bundled liblzf writes for + /// `b"hello hello hello hello"` (read back with `read_direct_chunk`): a + /// 7-byte literal, a 14-byte back reference 6 bytes back (extended + /// length), and a 2-byte literal. + #[test] + fn decodes_liblzf_output() { + let stream = b"\x06hello h\xe0\x05\x05\x01lo"; + assert_eq!( + lzf_decompress(stream, 23, 23).unwrap(), + b"hello hello hello hello" + ); + } + + #[test] + fn rejects_corrupt_streams() { + // Back reference before the start. + assert!(lzf_decompress(&[0x20, 0x00], 10, 10).is_err()); + // Literal run past the end. + assert!(lzf_decompress(&[0x05, 1, 2], 10, 10).is_err()); + // Output over the limit. + let c = lzf_compress(&[1u8; 100]); + assert!(lzf_decompress(&c, 10, 99).is_err()); + } + + /// Random and mutated streams: errors are fine, panics are not. + #[test] + fn fuzzed_streams_never_panic() { + let seeds: Vec> = [ + b"hello hello hello hello".to_vec(), + vec![7u8; 3000], + (0..2000u32).flat_map(|i| (i % 37).to_le_bytes()).collect(), + (0..500u32) + .map(|i| (i.wrapping_mul(2_654_435_761) >> 13) as u8) + .collect(), + ] + .iter() + .map(|d| lzf_compress(d)) + .collect(); + for limit in [0usize, 23, 4096, 8000] { + crate::test_fuzz::fuzz_decoder( + 0x1f2 + limit as u64, + &seeds[..1], + 5_000, + limit.max(23), + |s| lzf_decompress(s, limit, limit.max(23)), + ); + } + crate::test_fuzz::fuzz_decoder(0x1f3, &seeds, 20_000, 8000, |s| { + lzf_decompress(s, 8000, 8000) + }); + } +} diff --git a/crates/clawhdf5-format/src/filters_szip.rs b/crates/clawhdf5-format/src/filters_szip.rs index 4e7fd71..7e381cf 100644 --- a/crates/clawhdf5-format/src/filters_szip.rs +++ b/crates/clawhdf5-format/src/filters_szip.rs @@ -34,6 +34,7 @@ const SZ_NN_OPTION_MASK: u32 = 32; /// /// The chunk is a 4-byte little-endian uncompressed size followed by the /// szlib stream. +#[cfg_attr(not(feature = "szip"), allow(dead_code))] pub(crate) fn szip_decompress( _data: &[u8], _cd: &[u32], diff --git a/crates/clawhdf5-format/src/lib.rs b/crates/clawhdf5-format/src/lib.rs index ef09c1b..905ce84 100644 --- a/crates/clawhdf5-format/src/lib.rs +++ b/crates/clawhdf5-format/src/lib.rs @@ -43,6 +43,14 @@ //! | `checksum` | yes | Jenkins lookup3 checksum validation | //! | `deflate` | yes | Deflate (gzip) compression via `flate2` | //! | `provenance` | yes | SHINES provenance — SHA-256 hashing & verification | +//! | `lzf` | yes | LZF filter (32000), h5py's `compression="lzf"` | +//! | `bitshuffle` | no | Bitshuffle filter (32008), none/LZ4/Zstandard | +//! | `bzip2` | no | bzip2 filter (307) | +//! | `blosc` | no | Blosc 1 filter (32001) | +//! | `plugin-filters` | no | The four above | +//! +//! Filters are looked up by ID in [`filter_registry`], which also takes +//! codecs registered at run time for other IDs. #![cfg_attr(not(feature = "std"), no_std)] @@ -71,7 +79,16 @@ pub mod extensible_array; pub mod file_writer; pub mod fill_value; pub mod filter_pipeline; +pub mod filter_registry; pub mod filters; +#[cfg(any(feature = "bitshuffle", feature = "blosc"))] +mod filters_bitshuffle; +#[cfg(feature = "blosc")] +pub mod filters_blosc; +#[cfg(feature = "bzip2")] +mod filters_bzip2; +#[cfg(feature = "lzf")] +pub mod filters_lzf; mod filters_szip; pub mod fixed_array; pub mod float16; @@ -100,6 +117,16 @@ pub mod shared_message; pub mod signature; pub mod superblock; pub mod symbol_table; +#[cfg(all( + test, + any( + feature = "lzf", + feature = "bitshuffle", + feature = "bzip2", + feature = "blosc" + ) +))] +mod test_fuzz; pub mod type_builders; pub mod vds; pub mod vl_data; diff --git a/crates/clawhdf5-format/src/object_header.rs b/crates/clawhdf5-format/src/object_header.rs index 4b8067f..2c3c819 100644 --- a/crates/clawhdf5-format/src/object_header.rs +++ b/crates/clawhdf5-format/src/object_header.rs @@ -108,10 +108,20 @@ impl ObjectHeader { return Err(FormatError::InvalidObjectHeaderVersion(version)); } - let num_messages = LittleEndian::read_u16(&data[offset + 2..offset + 4]); + let num_messages = LittleEndian::read_u16(&data[offset + 2..offset + 4]) as usize; let reference_count = LittleEndian::read_u32(&data[offset + 4..offset + 8]); let header_data_size = LittleEndian::read_u32(&data[offset + 8..offset + 12]) as usize; + // libhdf5 (H5O__prefix_deserialize): a header with messages needs room + // for at least one message header, and one without has an empty chunk. + if (num_messages > 0 && header_data_size < V1_MSG_HEADER_SIZE) + || (num_messages == 0 && header_data_size > 0) + { + return Err(FormatError::InvalidObjectHeader( + "bad object header chunk size", + )); + } + // Pad to 8-byte alignment: header prefix is 12 bytes, pad to 16 let padding = 4; // pad 12-byte prefix to 16-byte alignment let msg_start = offset @@ -124,64 +134,23 @@ impl ObjectHeader { ensure_len(data, msg_start, header_data_size)?; let mut messages = Vec::new(); - let mut pos = msg_start; - let msg_end = - msg_start - .checked_add(header_data_size) - .ok_or(FormatError::UnexpectedEof { - expected: usize::MAX, - available: data.len(), - })?; - - for _ in 0..num_messages { - if pos + 8 > msg_end { - break; - } - let msg_type_raw = LittleEndian::read_u16(&data[pos..pos + 2]); - let msg_data_size = LittleEndian::read_u16(&data[pos + 2..pos + 4]) as usize; - let msg_flags = data[pos + 4]; - // reserved(3) at pos+5..pos+8 - pos += 8; - - ensure_len(data, pos, msg_data_size)?; - let msg_type = MessageType::from_u16(msg_type_raw); - - check_unknown_message(msg_type, msg_flags)?; - - if msg_type != MessageType::Nil { - messages.push(HeaderMessage { - msg_type, - size: msg_data_size, - flags: msg_flags, - creation_order: None, - data: data[pos..pos + msg_data_size].to_vec(), - }); - } - - pos += msg_data_size; - - // Follow continuations - if msg_type == MessageType::ObjectHeaderContinuation { - let cont_msg_data = &messages - .last() - .ok_or(FormatError::InvalidObjectHeaderSignature)? - .data; - if cont_msg_data.len() >= (offset_size as usize + length_size as usize) { - let cont_offset = read_offset(cont_msg_data, 0, offset_size)? as usize; - let cont_length = - read_offset(cont_msg_data, offset_size as usize, length_size)? as usize; - // Parse continuation block (v1: just raw messages, no signature) - let cont_msgs = Self::parse_v1_continuation( - data, - cont_offset, - cont_length, - offset_size, - length_size, - 32, // max continuation depth - )?; - messages.extend(cont_msgs); - } - } + let chunk0_count = Self::parse_v1_chunk( + data, + msg_start, + header_data_size, + offset_size, + length_size, + MAX_V1_CONTINUATION_DEPTH, + &mut messages, + )?; + // libhdf5 reads every message in the first chunk and refuses a header + // whose prefix claims fewer than that (continuation chunks are read + // later and not held to the count). Stopping after the claimed number + // silently dropped the rest. + if chunk0_count > num_messages { + return Err(FormatError::InvalidObjectHeader( + "bad object header message count", + )); } Ok(ObjectHeader { @@ -196,72 +165,87 @@ impl ObjectHeader { }) } - fn parse_v1_continuation( + /// Parse the messages of one version-1 chunk (`length` bytes at + /// `offset`, no signature), following continuation messages as they are + /// met. Returns how many messages (NIL ones included) this chunk itself + /// holds. + /// + /// A version-1 chunk is filled with messages whose sizes are multiples of + /// 8; libhdf5 refuses a message that is not aligned, that runs past the + /// end of the chunk, or leftover bytes too few for a message header (a + /// "gap", which only version 2 allows). + #[allow(clippy::too_many_arguments)] + fn parse_v1_chunk( data: &[u8], offset: usize, length: usize, offset_size: u8, length_size: u8, depth_remaining: u16, - ) -> Result, FormatError> { + messages: &mut Vec, + ) -> Result { if depth_remaining == 0 { return Err(FormatError::NestingDepthExceeded); } ensure_len(data, offset, length)?; - let mut messages = Vec::new(); + let end = offset + length; let mut pos = offset; - let end = offset.saturating_add(length); + let mut count = 0usize; - while pos + 8 <= end { + while pos < end { + if end - pos < V1_MSG_HEADER_SIZE { + return Err(FormatError::InvalidObjectHeader( + "gap found in early version of file format", + )); + } let msg_type_raw = LittleEndian::read_u16(&data[pos..pos + 2]); let msg_data_size = LittleEndian::read_u16(&data[pos + 2..pos + 4]) as usize; let msg_flags = data[pos + 4]; - pos += 8; + // reserved(3) at pos+5..pos+8 + pos += V1_MSG_HEADER_SIZE; - if pos + msg_data_size > end { - break; + if !msg_data_size.is_multiple_of(8) { + return Err(FormatError::InvalidObjectHeader("message not aligned")); } + if msg_data_size > end - pos { + return Err(FormatError::InvalidObjectHeader( + "message size exceeds buffer end", + )); + } + let body = &data[pos..pos + msg_data_size]; + check_message(1, msg_type_raw, msg_flags, body, offset_size, length_size)?; + count += 1; let msg_type = MessageType::from_u16(msg_type_raw); - - check_unknown_message(msg_type, msg_flags)?; - if msg_type != MessageType::Nil { messages.push(HeaderMessage { msg_type, size: msg_data_size, flags: msg_flags, creation_order: None, - data: data[pos..pos + msg_data_size].to_vec(), + data: body.to_vec(), }); } - pos += msg_data_size; - // Recursive continuations + // Follow continuations (v1 continuation chunks are just raw + // messages, no signature); check_message has checked the body. if msg_type == MessageType::ObjectHeaderContinuation { - let cont_msg_data = &messages - .last() - .ok_or(FormatError::InvalidObjectHeaderSignature)? - .data; - if cont_msg_data.len() >= (offset_size as usize + length_size as usize) { - let cont_offset = read_offset(cont_msg_data, 0, offset_size)? as usize; - let cont_length = - read_offset(cont_msg_data, offset_size as usize, length_size)? as usize; - let cont_msgs = Self::parse_v1_continuation( - data, - cont_offset, - cont_length, - offset_size, - length_size, - depth_remaining - 1, - )?; - messages.extend(cont_msgs); - } + let cont_offset = read_offset(body, 0, offset_size)? as usize; + let cont_length = read_offset(body, offset_size as usize, length_size)? as usize; + Self::parse_v1_chunk( + data, + cont_offset, + cont_length, + offset_size, + length_size, + depth_remaining - 1, + messages, + )?; } } - Ok(messages) + Ok(count) } fn parse_v2( @@ -278,6 +262,11 @@ impl ObjectHeader { return Err(FormatError::InvalidObjectHeaderVersion(version)); } let flags = data[offset + 5]; + if flags & !V2_HDR_ALL_FLAGS != 0 { + return Err(FormatError::InvalidObjectHeader( + "unknown object header status flag(s)", + )); + } let mut pos = offset + 6; @@ -297,7 +286,14 @@ impl ObjectHeader { // Optional attribute storage thresholds (flags bit 4) if flags & 0x10 != 0 { ensure_len(data, pos, 4)?; - // max_compact_attrs(2) + min_dense_attrs(2) — read but don't store for now + // max_compact_attrs(2) + min_dense_attrs(2) — checked, not stored + let max_compact = LittleEndian::read_u16(&data[pos..pos + 2]); + let min_dense = LittleEndian::read_u16(&data[pos + 2..pos + 4]); + if max_compact < min_dense { + return Err(FormatError::InvalidObjectHeader( + "bad object header attribute phase change values", + )); + } pos += 4; } @@ -312,6 +308,14 @@ impl ObjectHeader { ensure_len(data, pos, chunk_size_width as usize)?; let chunk0_size = read_offset(data, pos, chunk_size_width)? as usize; pos += chunk_size_width as usize; + // Bit 2: attribute creation order tracked → messages include creation order field + let has_creation_order = flags & 0x04 != 0; + let msg_header_size = if has_creation_order { 6 } else { 4 }; + if chunk0_size > 0 && chunk0_size < msg_header_size { + return Err(FormatError::InvalidObjectHeader( + "bad object header chunk size", + )); + } let chunk0_msg_start = pos; let chunk0_msg_end = pos @@ -335,9 +339,6 @@ impl ObjectHeader { } } - // Bit 2: attribute creation order tracked → messages include creation order field - let has_creation_order = flags & 0x04 != 0; - // Parse messages from chunk0 let mut messages = Vec::new(); let mut continuations = Vec::new(); @@ -396,8 +397,20 @@ impl ObjectHeader { ) -> Result<(), FormatError> { let msg_header_size = if has_creation_order { 6 } else { 4 }; let mut pos = start; + let mut null_count = 0usize; - while pos + msg_header_size <= end { + while pos < end { + // Leftover bytes too few for a message header are a gap, which + // libhdf5 allows only in a chunk without NIL messages (a writer + // that leaves a gap had no NIL message to put the space in). + if end - pos < msg_header_size { + if null_count != 0 { + return Err(FormatError::InvalidObjectHeader( + "gap in chunk with no null messages", + )); + } + break; + } let msg_type_raw = data[pos] as u16; let msg_data_size = LittleEndian::read_u16(&data[pos + 1..pos + 3]) as usize; let msg_flags = data[pos + 3]; @@ -408,32 +421,36 @@ impl ObjectHeader { }; pos += msg_header_size; - if pos + msg_data_size > end { - // Could be padding at end of chunk - break; + // `end` is where the messages stop and the checksum starts. + // libhdf5 bounds a message by the chunk including its checksum, + // but a message that runs into the checksum still fails there: + // its loop stops at the checksum, and reading the checksum from + // past its start overruns the chunk ("ran off end of input + // buffer while decoding"). Both refuse it; only the text + // differs. + if msg_data_size > end - pos { + return Err(FormatError::InvalidObjectHeader( + "message size exceeds buffer end", + )); } + let body = &data[pos..pos + msg_data_size]; + check_message(2, msg_type_raw, msg_flags, body, offset_size, length_size)?; let msg_type = MessageType::from_u16(msg_type_raw); - - check_unknown_message(msg_type, msg_flags)?; - - let msg_data = data[pos..pos + msg_data_size].to_vec(); - if msg_type == MessageType::ObjectHeaderContinuation { - // Parse continuation offset/length from message data - if msg_data.len() >= (offset_size as usize + length_size as usize) { - let cont_off = read_offset(&msg_data, 0, offset_size)? as usize; - let cont_len = - read_offset(&msg_data, offset_size as usize, length_size)? as usize; - continuations.push((cont_off, cont_len)); - } - } else if msg_type != MessageType::Nil { + // check_message has checked the body holds both fields. + let cont_off = read_offset(body, 0, offset_size)? as usize; + let cont_len = read_offset(body, offset_size as usize, length_size)? as usize; + continuations.push((cont_off, cont_len)); + } else if msg_type == MessageType::Nil { + null_count += 1; + } else { messages.push(HeaderMessage { msg_type, size: msg_data_size, flags: msg_flags, creation_order, - data: msg_data, + data: body.to_vec(), }); } @@ -496,22 +513,149 @@ impl ObjectHeader { } } -/// Header message flag bit 7: fail if the message is unknown, always. +/// Size of a version-1 message header: type(2) + size(2) + flags(1) + reserved(3). +const V1_MSG_HEADER_SIZE: usize = 8; + +/// How deep version-1 continuation chunks may chain (malformed-data guard). +const MAX_V1_CONTINUATION_DEPTH: u16 = 32; + +/// Every defined version-2 object header status flag (libhdf5 +/// `H5O_HDR_ALL_FLAGS`): chunk-0 size width (bits 0-1), attribute creation +/// order tracked/indexed, attribute phase-change values, times stored. +const V2_HDR_ALL_FLAGS: u8 = 0x3F; + +// Header message flag bits (libhdf5 `H5O_MSG_FLAG_*`). Bit 0 (constant) needs +// no check. Bit 3 (fail if unknown and the file is opened for writing) never +// fails a read: the parser only ever reads, as libhdf5 ignores it for a +// read-only open. +const MSG_FLAG_SHARED: u8 = 0x02; +const MSG_FLAG_DONTSHARE: u8 = 0x04; +const MSG_FLAG_FAIL_IF_UNKNOWN_AND_OPEN_FOR_WRITE: u8 = 0x08; +const MSG_FLAG_MARK_IF_UNKNOWN: u8 = 0x10; +const MSG_FLAG_WAS_UNKNOWN: u8 = 0x20; +const MSG_FLAG_SHAREABLE: u8 = 0x40; +/// Fail if the message is unknown, whatever the access mode. const MSG_FLAG_FAIL_IF_UNKNOWN_ALWAYS: u8 = 0x80; -/// Refuse an unknown message the file says no reader may skip. +/// Message type ids libhdf5 has a class for (`H5O_msg_class_g`): 0x00-0x18 +/// except 0x09 (a test-only "bogus" message). Anything else is an unknown +/// message. +fn is_known_message(id: u16) -> bool { + id <= 0x18 && id != 0x09 +} + +/// Message classes that may be shared (`H5O_SHARE_IS_SHARABLE`): dataspace, +/// datatype, the two fill-value messages, filter pipeline and attribute. +fn is_shareable_message(id: u16) -> bool { + matches!(id, 0x01 | 0x03 | 0x04 | 0x05 | 0x0B | 0x0C) +} + +/// Check one header message the way libhdf5 does while it loads an object +/// header (`H5O__chunk_deserialize`), so an object libhdf5 refuses to open is +/// refused here too instead of being read from a corrupt header: /// -/// The parser only ever reads, so bit 3 (fail only when opened for writing) -/// is ignored, as libhdf5 ignores it for a read-only open; bit 7 fails -/// regardless of access mode. This had the two the wrong way round, failing -/// objects libhdf5 reads and reading ones it refuses (`tbogus.h5`). -fn check_unknown_message(msg_type: MessageType, msg_flags: u8) -> Result<(), FormatError> { - match msg_type { - MessageType::Unknown(id) if msg_flags & MSG_FLAG_FAIL_IF_UNKNOWN_ALWAYS != 0 => { - Err(FormatError::UnsupportedMessage(id)) - } - _ => Ok(()), +/// - contradictory flag combinations; +/// - an unknown message the file says no reader may skip (bit 7). This had +/// bits 3 and 7 the wrong way round once, failing objects libhdf5 reads +/// and reading ones it refuses (`tbogus.h5`); +/// - a known message whose class cannot be shared, flagged shared or +/// shareable (`cve-2016-4332`); +/// - the messages libhdf5 decodes while loading the header, whose decode +/// errors fail the load: continuation, reference count (which a version-1 +/// header cannot hold), and both modification-time messages. +fn check_message( + header_version: u8, + id: u16, + flags: u8, + body: &[u8], + offset_size: u8, + length_size: u8, +) -> Result<(), FormatError> { + let bad_flags = FormatError::InvalidObjectHeader("bad flag combination for message"); + if flags & MSG_FLAG_SHARED != 0 && flags & MSG_FLAG_DONTSHARE != 0 { + return Err(bad_flags); } + if flags & MSG_FLAG_WAS_UNKNOWN != 0 + && (flags & MSG_FLAG_FAIL_IF_UNKNOWN_AND_OPEN_FOR_WRITE != 0 + || flags & MSG_FLAG_MARK_IF_UNKNOWN == 0) + { + return Err(bad_flags); + } + + if !is_known_message(id) { + if flags & MSG_FLAG_FAIL_IF_UNKNOWN_ALWAYS != 0 { + return Err(FormatError::UnsupportedMessage(id)); + } + return Ok(()); + } + if flags & (MSG_FLAG_SHARED | MSG_FLAG_SHAREABLE) != 0 && !is_shareable_message(id) { + return Err(FormatError::InvalidObjectHeader( + "message of unshareable class flagged as shareable", + )); + } + + let overrun = FormatError::InvalidObjectHeader("ran off end of input buffer while decoding"); + match id { + // Continuation: address + length, and the chunk cannot be empty. + 0x10 => { + if body.len() < offset_size as usize + length_size as usize { + return Err(overrun); + } + if read_offset(body, offset_size as usize, length_size)? == 0 { + return Err(FormatError::InvalidObjectHeader( + "invalid continuation chunk size (0)", + )); + } + } + // Reference count: version-2 headers only; version 0 then a u32. + 0x16 => { + if header_version == 1 { + return Err(FormatError::InvalidObjectHeader( + "object header version does not support reference count message", + )); + } + match body.first() { + None => return Err(overrun), + Some(0) => {} + Some(_) => { + return Err(FormatError::InvalidObjectHeader( + "bad version number for reference count message", + )); + } + } + if body.len() < 5 { + return Err(overrun); + } + } + // Old modification time: "YYYYMMDDhhmmss" and 2 reserved bytes. + 0x0E => { + if body.len() < 16 { + return Err(overrun); + } + if !body[..14].iter().all(u8::is_ascii_digit) { + return Err(FormatError::InvalidObjectHeader( + "badly formatted modification time message", + )); + } + } + // New modification time: version 1, 3 reserved bytes, u32 seconds. + 0x12 => { + match body.first() { + None => return Err(overrun), + Some(1) => {} + Some(_) => { + return Err(FormatError::InvalidObjectHeader( + "bad version number for mtime message", + )); + } + } + if body.len() < 8 { + return Err(overrun); + } + } + _ => {} + } + Ok(()) } #[cfg(test)] @@ -528,11 +672,14 @@ mod tests { // Calculate total header message data size let mut msg_bytes = Vec::new(); for (mtype, mdata, mflags) in messages { + // v1 message sizes are multiples of 8 (the data is zero-padded). + let padded = mdata.len().div_ceil(8) * 8; msg_bytes.extend_from_slice(&mtype.to_le_bytes()); // type(2) - msg_bytes.extend_from_slice(&(mdata.len() as u16).to_le_bytes()); // size(2) + msg_bytes.extend_from_slice(&(padded as u16).to_le_bytes()); // size(2) msg_bytes.push(*mflags); // flags(1) msg_bytes.extend_from_slice(&[0u8; 3]); // reserved(3) msg_bytes.extend_from_slice(mdata); // data + msg_bytes.resize(msg_bytes.len() + padded - mdata.len(), 0); } let mut buf = Vec::new(); @@ -622,9 +769,10 @@ mod tests { let hdr = ObjectHeader::parse(&data, 0, 8, 8).unwrap(); assert_eq!(hdr.messages.len(), 2); assert_eq!(hdr.messages[0].msg_type, MessageType::Dataspace); - assert_eq!(hdr.messages[0].data, vec![1, 2, 3, 4]); + // v1 message data is padded to a multiple of 8 bytes. + assert_eq!(hdr.messages[0].data, vec![1, 2, 3, 4, 0, 0, 0, 0]); assert_eq!(hdr.messages[1].msg_type, MessageType::DataLayout); - assert_eq!(hdr.messages[1].data, vec![5, 6]); + assert_eq!(hdr.messages[1].data[..2], [5, 6]); } #[test] @@ -650,7 +798,7 @@ mod tests { // Bit 3 = fail if unknown *and the file is opened for writing*. This // parser only reads, so libhdf5 (read-only) opens such an object and // so must we. Bits 4/5 (mark if unknown / was unknown) never fail. - for flags in [0x08u8, 0x10, 0x20, 0x38] { + for flags in [0x08u8, 0x10, 0x30] { let messages = [(0x00FFu16, &[0xAA][..], flags)]; let data = build_v1_header(&messages, 8, 8); let hdr = ObjectHeader::parse(&data, 0, 8, 8).unwrap(); @@ -658,6 +806,231 @@ mod tests { } } + #[test] + fn contradictory_message_flags_are_refused() { + // libhdf5: "bad flag combination for message" for shared + don't + // share, was-unknown without mark-if-unknown, and was-unknown with + // fail-if-unknown-on-write. + for flags in [0x06u8, 0x20, 0x38] { + let data = build_v1_header(&[(0x00FFu16, &[0xAA][..], flags)], 8, 8); + assert_eq!( + ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), + FormatError::InvalidObjectHeader("bad flag combination for message"), + "flags {flags:#x}" + ); + let data = build_v2_header(0x00, &[(0xF0, &[1, 2], flags)], None); + assert_eq!( + ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), + FormatError::InvalidObjectHeader("bad flag combination for message"), + "flags {flags:#x}" + ); + } + } + + #[test] + fn unshareable_message_flagged_shareable_is_refused() { + // A layout (0x08) or modification time (0x12) message cannot be + // shared; bit 1 (shared) or bit 6 (shareable) on one is corruption + // (cve-2016-4332). A datatype (0x03) may be shareable. + let mtime = [1u8, 0, 0, 0, 0x10, 0x20, 0x30, 0x40]; + for (id, flags) in [(0x08u16, 0x40u8), (0x08, 0x02), (0x12, 0x40)] { + let data = build_v1_header(&[(id, &mtime[..], flags)], 8, 8); + assert_eq!( + ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), + FormatError::InvalidObjectHeader( + "message of unshareable class flagged as shareable" + ), + "id {id:#x} flags {flags:#x}" + ); + } + let data = build_v1_header(&[(0x03, &[0u8; 8][..], 0x40)], 8, 8); + assert!(ObjectHeader::parse(&data, 0, 8, 8).is_ok()); + // An unknown message is never checked for shareability. + let data = build_v1_header(&[(0x00FF, &[0u8; 8][..], 0x40)], 8, 8); + assert!(ObjectHeader::parse(&data, 0, 8, 8).is_ok()); + } + + #[test] + fn v1_message_must_be_aligned() { + // cve-2018-13873: a v1 message whose size is not a multiple of 8. + let mut data = build_v1_header(&[(0x01, &[0u8; 8][..], 0)], 8, 8); + data[16 + 2] = 7; // size field of the only message + assert_eq!( + ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), + FormatError::InvalidObjectHeader("message not aligned") + ); + } + + #[test] + fn message_overrunning_its_chunk_is_refused() { + // It used to end the chunk quietly, dropping this message and any + // after it. + let mut data = build_v1_header(&[(0x01, &[0u8; 8][..], 0)], 8, 8); + data[16 + 2] = 16; + data.resize(data.len() + 64, 0); + assert_eq!( + ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), + FormatError::InvalidObjectHeader("message size exceeds buffer end") + ); + let mut data = build_v2_header(0x00, &[(0x01, &[1, 2], 0)], None); + data[7 + 1] = 9; // size of the only message (after OHDR, ver, flags, chunk size) + let chk = crate::checksum::jenkins_lookup3(&data[..data.len() - 4]); + let n = data.len(); + data[n - 4..].copy_from_slice(&chk.to_le_bytes()); + assert_eq!( + ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), + FormatError::InvalidObjectHeader("message size exceeds buffer end") + ); + } + + #[test] + fn v1_gap_after_last_message_is_refused() { + // Fewer than 8 bytes left over: a gap, which only version 2 allows. + let mut data = build_v1_header(&[(0x01, &[0u8; 8][..], 0)], 8, 8); + data[8] += 4; // header_data_size + data.extend_from_slice(&[0u8; 4]); + assert_eq!( + ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), + FormatError::InvalidObjectHeader("gap found in early version of file format") + ); + } + + #[test] + fn v1_chunk_holding_more_messages_than_the_prefix_says_is_refused() { + // cve-2024-32619: the prefix says 1 message, the chunk holds 2. The + // second used to be dropped silently. + let mut data = build_v1_header(&[(0x01, &[0u8; 8][..], 0), (0x03, &[0u8; 8][..], 0)], 8, 8); + data[2] = 1; + assert_eq!( + ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), + FormatError::InvalidObjectHeader("bad object header message count") + ); + // Fewer in the chunk than the prefix says is fine (the rest may be in + // continuation chunks; libhdf5 only enforces that with strict checks). + data[2] = 3; + assert_eq!( + ObjectHeader::parse(&data, 0, 8, 8).unwrap().messages.len(), + 2 + ); + } + + #[test] + fn v1_prefix_chunk_size_must_fit_the_message_count() { + let mut data = build_v1_header(&[], 8, 8); + data[8] = 8; // no messages but a non-empty chunk + data.extend_from_slice(&[0u8; 8]); + assert_eq!( + ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), + FormatError::InvalidObjectHeader("bad object header chunk size") + ); + } + + #[test] + fn v1_header_cannot_hold_a_reference_count_message() { + // cve-2018-11204. + let data = build_v1_header(&[(0x16, &[0, 2, 0, 0, 0][..], 0)], 8, 8); + assert_eq!( + ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), + FormatError::InvalidObjectHeader( + "object header version does not support reference count message" + ) + ); + let data = build_v2_header(0x00, &[(0x16, &[0, 2, 0, 0, 0], 0)], None); + assert!(ObjectHeader::parse(&data, 0, 8, 8).is_ok()); + } + + #[test] + fn modification_time_messages_are_decoded_with_the_header() { + // cve-2024-33873 (version 0) and cve-2024-33874 (empty message). + for (body, why) in [ + ( + &[0u8, 0, 0, 0, 1, 2, 3, 4][..], + "bad version number for mtime message", + ), + (&[][..], "ran off end of input buffer while decoding"), + ] { + let data = build_v2_header(0x00, &[(0x12, body, 0)], None); + assert_eq!( + ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), + FormatError::InvalidObjectHeader(why) + ); + } + let data = build_v2_header(0x00, &[(0x12, &[1, 0, 0, 0, 1, 2, 3, 4], 0)], None); + assert!(ObjectHeader::parse(&data, 0, 8, 8).is_ok()); + // The old (0x0E) message is 14 ASCII digits and 2 reserved bytes. + let data = build_v1_header(&[(0x0E, &b"20110414214255\0\0"[..], 0)], 8, 8); + assert!(ObjectHeader::parse(&data, 0, 8, 8).is_ok()); + let data = build_v1_header(&[(0x0E, &b"2011041421425x\0\0"[..], 0)], 8, 8); + assert_eq!( + ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), + FormatError::InvalidObjectHeader("badly formatted modification time message") + ); + } + + #[test] + fn continuation_message_must_hold_a_nonempty_chunk() { + let mut cont = [0u8; 16]; + cont[..8].copy_from_slice(&64u64.to_le_bytes()); + let data = build_v2_header(0x00, &[(0x10, &cont, 0)], None); + assert_eq!( + ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), + FormatError::InvalidObjectHeader("invalid continuation chunk size (0)") + ); + let data = build_v2_header(0x00, &[(0x10, &cont[..8], 0)], None); + assert_eq!( + ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), + FormatError::InvalidObjectHeader("ran off end of input buffer while decoding") + ); + } + + #[test] + fn v2_prefix_is_checked() { + let data = build_v2_header(0x40, &[(0x01, &[1], 0)], None); + assert_eq!( + ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), + FormatError::InvalidObjectHeader("unknown object header status flag(s)") + ); + // build_v2_header writes max_compact 8, min_dense 6; swap them. + let mut data = build_v2_header(0x10, &[(0x01, &[1], 0)], None); + data[6] = 6; + data[8] = 8; + let chk = crate::checksum::jenkins_lookup3(&data[..data.len() - 4]); + let n = data.len(); + data[n - 4..].copy_from_slice(&chk.to_le_bytes()); + assert_eq!( + ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), + FormatError::InvalidObjectHeader("bad object header attribute phase change values") + ); + } + + #[test] + fn v2_gap_is_allowed_only_without_nil_messages() { + // Three bytes after the last message: a gap (a message header is 4). + let mut data = build_v2_header(0x00, &[(0x01, &[1, 2, 3], 0), (0x03, &[], 0)], None); + // Turn the empty datatype message (4 header bytes) into a 3-byte gap + // by shrinking the chunk. + let n = data.len(); + data.truncate(n - 5); + data[6] -= 1; + let chk = crate::checksum::jenkins_lookup3(&data); + data.extend_from_slice(&chk.to_le_bytes()); + assert_eq!( + ObjectHeader::parse(&data, 0, 8, 8).unwrap().messages.len(), + 1 + ); + + let mut data = build_v2_header(0x00, &[(0x00, &[1, 2, 3], 0), (0x03, &[], 0)], None); + let n = data.len(); + data.truncate(n - 5); + data[6] -= 1; + let chk = crate::checksum::jenkins_lookup3(&data); + data.extend_from_slice(&chk.to_le_bytes()); + assert_eq!( + ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), + FormatError::InvalidObjectHeader("gap in chunk with no null messages") + ); + } + #[test] fn parse_v2_unknown_message_flags() { let data = build_v2_header(0x00, &[(0xF0, &[1, 2], 0x08)], None); diff --git a/crates/clawhdf5-format/src/parallel_read.rs b/crates/clawhdf5-format/src/parallel_read.rs index 0bb2785..6593132 100644 --- a/crates/clawhdf5-format/src/parallel_read.rs +++ b/crates/clawhdf5-format/src/parallel_read.rs @@ -10,7 +10,7 @@ use crate::chunked_read::ChunkInfo; use crate::error::FormatError; use crate::filter_pipeline::FilterPipeline; -use crate::filters::decompress_chunk_masked; +use crate::filters::decompress_chunk_exact; use crate::lane_partition::{self, LaneStats, PartitionStats}; /// Threshold: only use parallel decompression when chunk count exceeds this. @@ -84,12 +84,13 @@ pub fn decompress_chunks_lane_partitioned( } let raw_chunk = &file_data[c_addr..c_addr + size]; - let decompressed = decompress_chunk_masked( + let decompressed = decompress_chunk_exact( raw_chunk, pipeline, chunk_total_bytes, element_size, chunk_info.filter_mask, + &chunk_info.offsets, )?; stats.chunks_processed += 1; @@ -160,12 +161,13 @@ pub fn decompress_chunks_parallel( } let raw_chunk = &file_data[c_addr..c_addr + size]; - let decompressed = decompress_chunk_masked( + let decompressed = decompress_chunk_exact( raw_chunk, pipeline, chunk_total_bytes, element_size, chunk_info.filter_mask, + &chunk_info.offsets, )?; Ok(DecompressedChunk { @@ -204,12 +206,13 @@ pub fn decompress_chunks_sequential( let raw_chunk = &file_data[c_addr..c_addr + size]; let decompressed = if let Some(pl) = pipeline { - decompress_chunk_masked( + decompress_chunk_exact( raw_chunk, pl, chunk_total_bytes, element_size, chunk_info.filter_mask, + &chunk_info.offsets, )? } else { raw_chunk.to_vec() @@ -218,3 +221,60 @@ pub fn decompress_chunks_sequential( } Ok(result) } + +#[cfg(test)] +mod tests { + use super::*; + use crate::filter_pipeline::{FILTER_SHUFFLE, FilterDescription}; + + /// Eight shuffled 32-byte chunks; chunk 5 is stored short when `short`. + fn chunks(short: bool) -> (Vec, Vec) { + let mut file = Vec::new(); + let mut infos = Vec::new(); + for i in 0..8u64 { + let len = if short && i == 5 { 16 } else { 32 }; + infos.push(ChunkInfo { + chunk_size: len as u32, + filter_mask: 0, + offsets: vec![i * 8], + address: file.len() as u64, + }); + file.extend(core::iter::repeat_n(i as u8, len)); + } + (file, infos) + } + + /// Every parallel decoder refuses a chunk that decodes short, naming it. + #[test] + fn short_decoded_chunk_is_an_error() { + let pipeline = FilterPipeline { + version: 2, + filters: vec![FilterDescription { + filter_id: FILTER_SHUFFLE, + name: None, + flags: 0, + client_data: vec![4], + }], + }; + let (file, good) = chunks(false); + assert_eq!( + decompress_chunks_parallel(&file, &good, &pipeline, 32, 4).unwrap()[5], + [5u8; 32] + ); + let (file, bad) = chunks(true); + let errs = [ + decompress_chunks_lane_partitioned(&file, &bad, &pipeline, 32, 4, 1, Some(3)) + .map(|_| ()) + .unwrap_err(), + decompress_chunks_parallel(&file, &bad, &pipeline, 32, 4) + .map(|_| ()) + .unwrap_err(), + decompress_chunks_sequential(&file, &bad, Some(&pipeline), 32, 4) + .map(|_| ()) + .unwrap_err(), + ]; + for e in errs { + assert!(e.to_string().contains("[40]"), "{e}"); + } + } +} diff --git a/crates/clawhdf5-format/src/partial_read.rs b/crates/clawhdf5-format/src/partial_read.rs index 5cc3aaa..7d73599 100644 --- a/crates/clawhdf5-format/src/partial_read.rs +++ b/crates/clawhdf5-format/src/partial_read.rs @@ -22,7 +22,7 @@ use crate::data_read::extract_selection_from_buffer; use crate::dataspace::Dataspace; use crate::error::FormatError; use crate::filter_pipeline::FilterPipeline; -use crate::filters::{all_filters_skipped, decompress_chunk_masked}; +use crate::filters::{all_filters_skipped, decompress_chunk_exact}; use crate::selection::Selection; /// The smallest axis-aligned box containing every selected element, as @@ -330,12 +330,13 @@ pub fn read_selection( let decoded; let data: &[u8] = match pipeline { Some(pl) if !all_filters_skipped(pl, chunk.filter_mask) => { - decoded = decompress_chunk_masked( + decoded = decompress_chunk_exact( raw, pl, chunk_bytes, elem_size as u32, chunk.filter_mask, + &chunk.offsets[..rank], )?; &decoded } diff --git a/crates/clawhdf5-format/src/superblock.rs b/crates/clawhdf5-format/src/superblock.rs index d2ec4d6..a565321 100644 --- a/crates/clawhdf5-format/src/superblock.rs +++ b/crates/clawhdf5-format/src/superblock.rs @@ -100,6 +100,42 @@ pub mod swmr_flags { } impl Superblock { + /// Where the HDF5 data ends, relative to the superblock, for a file of + /// `file_len` bytes whose superblock is at `user_block` (both counted + /// from the start of the file), with libhdf5's truncation check + /// (`H5F__super_read`). + /// + /// The superblock records the end of the file's data as an absolute + /// address. A file shorter than that was truncated, and libhdf5 refuses + /// to open it ("truncated file"); so does this, with + /// [`FormatError::TruncatedFile`]. Bytes past that address are not part + /// of the file: libhdf5 fails any read of them ("addr overflow" / + /// "address plus size exceeds file eoa"), so a reader should parse only + /// the data up to the returned end. As libhdf5 does for a SWMR reader, + /// the check is skipped for a version-3 superblock whose writer is still + /// writing it in SWMR mode (it extends the file as it goes); the data + /// then ends at the end of the file. + /// + /// When the superblock's recorded base address differs from where the + /// superblock actually is (a user block added or removed after the file + /// was written), libhdf5 moves the recorded end of file by the same + /// amount, and so does this. + pub fn data_end(&self, user_block: u64, file_len: u64) -> Result { + let eof = + i128::from(self.eof_address) - i128::from(self.base_address) + i128::from(user_block); + if eof < 0 || eof > i128::from(file_len) { + if self.version >= 3 && self.is_swmr_write() { + return Ok(file_len.saturating_sub(user_block)); + } + return Err(FormatError::TruncatedFile { + stored_eof: u64::try_from(eof).unwrap_or(self.eof_address), + actual_len: file_len, + }); + } + // 0 <= eof <= file_len, so it fits a u64. + Ok((eof as u64).saturating_sub(user_block)) + } + /// Whether the file was opened with write access when the superblock was written. pub fn is_write_access(&self) -> bool { self.consistency_flags & swmr_flags::WRITE_ACCESS != 0 @@ -537,6 +573,30 @@ mod tests { buf } + #[test] + fn data_end_refuses_truncated_files_like_libhdf5() { + // build_v2_bytes records base 0, end of file 2048. + let sb = Superblock::parse(&build_v2_bytes(8, 2), 0).unwrap(); + assert_eq!(sb.data_end(0, 2048), Ok(2048)); + // Bytes past the recorded end are not part of the file. + assert_eq!(sb.data_end(0, 4096), Ok(2048)); + assert_eq!( + sb.data_end(0, 2047), + Err(FormatError::TruncatedFile { + stored_eof: 2048, + actual_len: 2047 + }) + ); + // A user block added in front after the file was written (the + // recorded base address is still 0): the end moves with it. + assert_eq!(sb.data_end(512, 2560), Ok(2048)); + assert!(sb.data_end(512, 2559).is_err()); + // A v3 superblock still being written in SWMR mode is not checked. + let mut swmr = Superblock::parse(&build_v2_bytes(8, 3), 0).unwrap(); + swmr.consistency_flags = swmr_flags::WRITE_ACCESS | swmr_flags::SWMR_WRITE; + assert_eq!(swmr.data_end(0, 1000), Ok(1000)); + } + #[test] fn parse_v0_8byte_offsets() { let data = build_v0_bytes(8); diff --git a/crates/clawhdf5-format/src/test_fuzz.rs b/crates/clawhdf5-format/src/test_fuzz.rs new file mode 100644 index 0000000..41515db --- /dev/null +++ b/crates/clawhdf5-format/src/test_fuzz.rs @@ -0,0 +1,149 @@ +//! Mutation fuzzing for the filter decoders (tests only). +//! +//! A decoder fed a random or mutated frame may fail, but must not panic — +//! tests build with overflow checks and debug assertions, so an unchecked +//! subtraction, multiplication or shift on a header field, or an +//! out-of-range slice, fails the test — and must not return more than its +//! output limit. + +#[cfg(not(feature = "std"))] +extern crate alloc; +#[cfg(not(feature = "std"))] +use alloc::vec::Vec; + +use crate::error::FormatError; + +/// xorshift64*: deterministic, so a failure reproduces. +pub(crate) struct Rng(u64); + +impl Rng { + pub(crate) fn new(seed: u64) -> Rng { + Rng(seed.max(1)) + } + + pub(crate) fn next_u64(&mut self) -> u64 { + let mut x = self.0; + x ^= x >> 12; + x ^= x << 25; + x ^= x >> 27; + self.0 = x; + x.wrapping_mul(0x2545_F491_4F6C_DD1D) + } + + /// Uniform in `0..n` (`n` > 0). + pub(crate) fn below(&mut self, n: usize) -> usize { + (self.next_u64() % n as u64) as usize + } + + pub(crate) fn bytes(&mut self, n: usize) -> Vec { + (0..n).map(|_| self.next_u64() as u8).collect() + } + + /// A u32 that tends to hit edge cases in size and offset fields. + fn interesting_u32(&mut self, len: usize) -> u32 { + match self.below(10) { + 0 => 0, + 1 => 1, + 2 => self.below(20) as u32, + 3 => 15 + self.below(3) as u32, + 4 => u32::MAX - self.below(16) as u32, + 5 => 1 << self.below(32), + 6 => (len as u32) + .wrapping_add(self.below(9) as u32) + .wrapping_sub(4), + 7 => i32::MAX as u32, + _ => self.next_u64() as u32, + } + } +} + +/// One to four random edits of `seed`. +pub(crate) fn mutate(rng: &mut Rng, seed: &[u8]) -> Vec { + let mut v = seed.to_vec(); + for _ in 0..1 + rng.below(4) { + let len = v.len(); + match rng.below(9) { + 0 if len > 0 => { + let i = rng.below(len); + v[i] ^= 1 << rng.below(8); + } + 1 if len > 0 => { + let i = rng.below(len); + v[i] = rng.next_u64() as u8; + } + 2 if len > 0 => { + let i = rng.below(len); + v[i] = [0, 0xff, 0x7f, 0x80, 0x20, 0x1f][rng.below(6)]; + } + // A size or offset field: little- or big-endian, anywhere, but + // most often in the first 32 bytes where headers live. + 3 | 4 if len >= 4 => { + let span = if rng.below(2) == 0 { len.min(32) } else { len }; + let i = rng.below(span - 3); + let x = rng.interesting_u32(len); + let b = if rng.below(2) == 0 { + x.to_le_bytes() + } else { + x.to_be_bytes() + }; + v[i..i + 4].copy_from_slice(&b); + } + 5 if len > 0 => v.truncate(rng.below(len)), + 6 => { + let n = 1 + rng.below(64); + let extra = rng.bytes(n); + v.extend_from_slice(&extra); + } + 7 if len > 1 => { + let a = rng.below(len); + let b = a + rng.below(len - a); + let copy = v[a..b].to_vec(); + let at = rng.below(len); + v.splice(at..at, copy); + } + _ if len > 0 => { + let i = rng.below(len); + v[i] = v[i].wrapping_add(1 + rng.below(3) as u8); + } + _ => v.push(rng.next_u64() as u8), + } + } + v +} + +/// Feed `iters` inputs to `decode`: mostly mutations of `seeds`, some pure +/// noise and some truncated seeds. Asserts only "no panic, output within +/// `limit`". +pub(crate) fn fuzz_decoder( + seed: u64, + seeds: &[Vec], + iters: usize, + limit: usize, + mut decode: impl FnMut(&[u8]) -> Result, FormatError>, +) { + assert!(!seeds.is_empty()); + let mut rng = Rng::new(seed); + for s in seeds { + // The seeds themselves must be valid, or the fuzz explores nothing. + decode(s).expect("seed frame must decode"); + } + for _ in 0..iters { + let input = match rng.below(16) { + 0 => { + let n = rng.below(96); + rng.bytes(n) + } + 1 => { + let s = &seeds[rng.below(seeds.len())]; + s[..rng.below(s.len() + 1)].to_vec() + } + _ => { + let s = &seeds[rng.below(seeds.len())]; + mutate(&mut rng, s) + } + }; + if let Ok(out) = decode(&input) { + assert!(out.len() <= limit, "decoded {} > limit {limit}", out.len()); + } + } +} diff --git a/crates/clawhdf5-format/src/type_builders.rs b/crates/clawhdf5-format/src/type_builders.rs index 9e46e66..f055a88 100644 --- a/crates/clawhdf5-format/src/type_builders.rs +++ b/crates/clawhdf5-format/src/type_builders.rs @@ -731,6 +731,61 @@ impl DatasetBuilder { self } + /// Compress with a plugin filter ([`PluginFilter`]), in the format the + /// libhdf5 plugin reads (h5py, hdf5plugin). Implies chunked storage. + /// Each filter needs its cargo feature (`lzf`, ...); writing fails with + /// `UnsupportedFilter` without it. + /// + /// [`PluginFilter`]: crate::chunked_write::PluginFilter + pub fn with_plugin_filter(&mut self, filter: crate::chunked_write::PluginFilter) -> &mut Self { + self.chunk_options.plugin = Some(filter); + self + } + + /// Enable LZF compression (filter 32000) — h5py's built-in + /// `compression="lzf"`. Implies chunked storage; shuffle is applied + /// first unless `.without_shuffle()`. Requires the `lzf` cargo feature. + pub fn with_lzf(&mut self) -> &mut Self { + self.with_plugin_filter(crate::chunked_write::PluginFilter::Lzf) + } + + /// Enable bitshuffle (filter 32008) with `compression` after the bit + /// transpose, in bitshuffle's default block size. Implies chunked + /// storage; no byte shuffle is added. Requires the `bitshuffle` cargo + /// feature. + pub fn with_bitshuffle( + &mut self, + compression: crate::chunked_write::BitshuffleCompression, + ) -> &mut Self { + self.with_plugin_filter(crate::chunked_write::PluginFilter::Bitshuffle { + block_size: 0, + compression, + }) + } + + /// Enable bzip2 (filter 307) at block size `level` (1-9). Implies + /// chunked storage; shuffle is applied first unless + /// `.without_shuffle()`. Requires the `bzip2` cargo feature. + pub fn with_bzip2(&mut self, level: u32) -> &mut Self { + self.with_plugin_filter(crate::chunked_write::PluginFilter::Bzip2 { level }) + } + + /// Enable Blosc (filter 32001) with `codec` at `level` (0-9) after + /// `shuffle`. Implies chunked storage; no extra HDF5 shuffle is added. + /// Requires the `blosc` cargo feature. + pub fn with_blosc( + &mut self, + codec: crate::chunked_write::BloscCodec, + level: u32, + shuffle: crate::chunked_write::BloscShuffle, + ) -> &mut Self { + self.with_plugin_filter(crate::chunked_write::PluginFilter::Blosc { + codec, + level, + shuffle, + }) + } + /// Enable Pcodec lossless numerical compression (private clawhdf5 filter /// ID 480). /// diff --git a/crates/clawhdf5-io/src/async_read.rs b/crates/clawhdf5-io/src/async_read.rs index d9e2ffe..30969fa 100644 --- a/crates/clawhdf5-io/src/async_read.rs +++ b/crates/clawhdf5-io/src/async_read.rs @@ -281,8 +281,13 @@ impl AsyncHDF5File { // HDF5 addresses are relative to the superblock: drop any user block // so they index `data` directly. let user_block = find_signature(&data)?; + let whole_len = data.len() as u64; data.drain(..user_block); let superblock = Superblock::parse(&data, 0)?; + // Refuse a truncated file, and keep nothing past the end of file the + // superblock records, as libhdf5 does. + let end = superblock.data_end(user_block as u64, whole_len)?; + data.truncate(end as usize); Ok(Self { data, superblock }) } @@ -312,7 +317,7 @@ impl AsyncHDF5File { let dt_msg = find_msg(&header, MessageType::Datatype).ok_or(FormatError::DatasetMissingData)?; - let (datatype, _) = Datatype::parse(&dt_msg.data)?; + let (datatype, _) = Datatype::parse_in_header(&dt_msg.data, header.version)?; let ds_msg = find_msg(&header, MessageType::Dataspace).ok_or(FormatError::DatasetMissingShape)?; @@ -605,6 +610,28 @@ mod tests { tokio::fs::remove_file(&path).await.ok(); } + /// As libhdf5 does: a truncated file is refused, and bytes past the end + /// of file the superblock records are dropped. + #[tokio::test] + async fn async_refuses_truncated_files_and_drops_trailing_bytes() { + let bytes = make_test_hdf5_f64("v", &[1.0, 2.0]); + let truncated = bytes[..bytes.len() - 8].to_vec(); + let err = AsyncHDF5File::from_bytes(truncated).err().unwrap(); + assert!( + matches!( + err, + AsyncHDF5Error::Format(FormatError::TruncatedFile { .. }) + ), + "{err}" + ); + + let mut appended = bytes.clone(); + appended.extend_from_slice(&[0xAB; 64]); + let file = AsyncHDF5File::from_bytes(appended).unwrap(); + assert_eq!(file.as_bytes().len(), bytes.len()); + assert_eq!(file.read_f64("v").await.unwrap(), vec![1.0, 2.0]); + } + #[tokio::test] async fn async_error_display() { let io_err = AsyncHDF5Error::Io(io::Error::new(io::ErrorKind::NotFound, "gone")); diff --git a/crates/clawhdf5-io/src/mpi_vol.rs b/crates/clawhdf5-io/src/mpi_vol.rs index 7dac39e..87a82ea 100644 --- a/crates/clawhdf5-io/src/mpi_vol.rs +++ b/crates/clawhdf5-io/src/mpi_vol.rs @@ -180,8 +180,7 @@ fn mpi_collective_read(vol: &MpiVol, location: &str, path: &str) -> Result Result Result, } +/// The HDF5 bytes of a whole file and its superblock: from the superblock +/// (addresses are relative to it, so any user block is skipped) to the end +/// of file the superblock records. A file shorter than that is truncated +/// and refused, and nothing past it is read, as in libhdf5. +pub(crate) fn hdf5_view( + whole: &[u8], +) -> Result<(&[u8], clawhdf5_format::superblock::Superblock), VolError> { + use clawhdf5_format::{signature::split_user_block, superblock::Superblock}; + let err = |e: clawhdf5_format::error::FormatError| VolError::DataError(e.to_string()); + let (user_block, data) = split_user_block(whole).map_err(err)?; + let sb = Superblock::parse(data, 0).map_err(err)?; + let end = sb + .data_end(user_block.len() as u64, whole.len() as u64) + .map_err(err)?; + // data_end is at most the file length less the user block. + Ok((&data[..end as usize], sb)) +} + impl NativeVol { /// Create a new native VOL connector. pub fn new() -> Self { @@ -264,6 +282,8 @@ impl VirtualObjectLayer for NativeVol { fn open(&mut self, location: &str) -> Result<(), VolError> { let data = std::fs::read(location)?; + // Refuse a truncated file at open, as libhdf5 does. + hdf5_view(&data)?; self.data = Some(data); self.location = Some(location.to_string()); Ok(()) @@ -283,13 +303,10 @@ impl VirtualObjectLayer for NativeVol { use clawhdf5_format::{ data_layout::DataLayout, data_read::read_raw_data_full, dataspace::Dataspace, datatype::Datatype, filter_pipeline::FilterPipeline, group_v2::resolve_path_any, - message_type::MessageType, object_header::ObjectHeader, signature::split_user_block, - superblock::Superblock, + message_type::MessageType, object_header::ObjectHeader, }; - // Addresses are relative to the superblock: skip any user block. - let (_, data) = split_user_block(data).map_err(|e| VolError::DataError(e.to_string()))?; - let sb = Superblock::parse(data, 0).map_err(|e| VolError::DataError(e.to_string()))?; + let (data, sb) = hdf5_view(data)?; let addr = resolve_path_any(data, &sb, path) .map_err(|e| VolError::NotFound(format!("{path}: {e}")))?; @@ -301,8 +318,8 @@ impl VirtualObjectLayer for NativeVol { .iter() .find(|m| m.msg_type == MessageType::Datatype) .ok_or_else(|| VolError::DataError("missing datatype".into()))?; - let (datatype, _) = - Datatype::parse(&dt_msg.data).map_err(|e| VolError::DataError(e.to_string()))?; + let (datatype, _) = Datatype::parse_in_header(&dt_msg.data, header.version) + .map_err(|e| VolError::DataError(e.to_string()))?; let ds_msg = header .messages @@ -387,6 +404,36 @@ mod tests { assert!(vol.as_bytes().is_none()); } + /// As libhdf5 does: a file shorter than the end of file its superblock + /// records is truncated and refused (at open, and when read from + /// memory), and bytes appended past that end are not part of the file. + #[test] + fn native_vol_refuses_truncated_files_and_ignores_trailing_bytes() { + use clawhdf5_format::file_writer::FileWriter as FmtWriter; + + let mut fw = FmtWriter::new(); + fw.create_dataset("x").with_f64_data(&[1.0, 2.0, 3.0]); + let bytes = fw.finish().unwrap(); + + let truncated = bytes[..bytes.len() - 8].to_vec(); + let err = NativeVol::from_bytes(truncated.clone()) + .read_dataset("x") + .unwrap_err(); + assert!(err.to_string().contains("truncated"), "{err}"); + let dir = std::env::temp_dir().join(format!("clawhdf5_vol_trunc_{}", std::process::id())); + std::fs::create_dir_all(&dir).unwrap(); + let path = dir.join("truncated.h5"); + std::fs::write(&path, &truncated).unwrap(); + let err = NativeVol::open_path(path.to_str().unwrap()).err().unwrap(); + assert!(err.to_string().contains("truncated"), "{err}"); + std::fs::remove_dir_all(&dir).ok(); + + let mut appended = bytes.clone(); + appended.extend_from_slice(&[0xAB; 64]); + let raw = NativeVol::from_bytes(appended).read_dataset("x").unwrap(); + assert_eq!(raw.len(), 24); + } + #[test] fn vol_error_display() { let err = VolError::Unsupported("read_dataset".into()); diff --git a/crates/clawhdf5-tools/Cargo.toml b/crates/clawhdf5-tools/Cargo.toml new file mode 100644 index 0000000..d54f39d --- /dev/null +++ b/crates/clawhdf5-tools/Cargo.toml @@ -0,0 +1,23 @@ +[package] +name = "clawhdf5-tools" +version = "2.7.0" +edition = "2024" +rust-version.workspace = true +license = "MIT" +description = "h5rs: pure-Rust HDF5 command-line tools (ls, dump, stat, diff, check) without libhdf5" +repository = "https://git.redclaw.dev/quantumclaw/clawhdf5" +keywords = ["hdf5", "h5dump", "h5ls", "cli", "science"] +categories = ["command-line-utilities", "science"] +readme = "README.md" + +[[bin]] +name = "h5rs" +path = "src/main.rs" + +[dependencies] +clawhdf5 = { path = "../clawhdf5", version = "2.7.0" } +clawhdf5-format = { path = "../clawhdf5-format", version = "2.7.0" } +serde_json = "1" + +[dev-dependencies] +tempfile = { workspace = true } diff --git a/crates/clawhdf5-tools/README.md b/crates/clawhdf5-tools/README.md new file mode 100644 index 0000000..79f775d --- /dev/null +++ b/crates/clawhdf5-tools/README.md @@ -0,0 +1,278 @@ +# clawhdf5-tools: `h5rs` + +HDF5 command-line tools in pure Rust, built only on the `clawhdf5` facade and +`clawhdf5-format`. No libhdf5 and no C code, so the binary also builds as a +fully static executable: `cargo build --release -p clawhdf5-tools --target +x86_64-unknown-linux-musl` gives a static-pie `h5rs` of 1.8 MB with no +shared-library dependencies (built and run on tank, 2026-09-26, Rust 1.98). + +| Command | Modelled on | What it does | +|---------|-------------|--------------| +| `h5rs ls` | `h5ls` | list objects: name, kind, shape, datatype; `-v` adds layout, chunking, storage, filters, attributes | +| `h5rs dump` | `h5dump` | the file's structure and values as DDL text, or as JSON (hdf5-json layout) | +| `h5rs stat` | `h5stat` | object, link, rank, layout, filter and attribute counts; raw-data and file size | +| `h5rs diff` | `h5diff` | structural and value differences between two files or objects | +| `h5rs check` | `h5check` | structural validator: every object and header message, the checksums of version 2+ structures, chunk index consistency | + +```bash +cargo install --path crates/clawhdf5-tools # or: cargo build --release -p clawhdf5-tools +h5rs --help +h5rs --help +``` + +Every command takes `--max-bytes N` where it reads values (default 1 GiB): a +dataset whose dataspace claims more than that is reported instead of read, so +a corrupt size cannot exhaust memory. + +## `h5rs ls` + +```console +$ h5rs ls -r data.h5 # an extract of the output +/ Group +/external External Link {other.h5//x} +/grp Group +/grp/ext2 Dataset {6/Inf, 10/Inf} float32 +/grp/gz Dataset {1000} int32 +/hard2 Group, same as /grp/sub +/named_t Type +/soft Soft Link {/contig} +``` + +The first two columns are h5ls's (`h5ls -r` prints the same text); h5rs adds +the datatype. `FILE/path` lists a group's members or one dataset, as h5ls +does. `-v` prints, per object: + +```console +$ h5rs ls -v data.h5/grp/gz +gz Dataset {1000/1000} + Address: 1120 + Links: 1 + Layout: chunked (fixed array index) + Chunks: {100} 400 bytes + Storage: 4000 logical bytes, 1237 allocated bytes, 323.36% utilization + Filter-0: shuffle-2 OPT {4} + Filter-1: deflate-1 OPT {4} + Filter-2: fletcher32-3 + Type: 32-bit little-endian integer +``` + +## `h5rs dump` + +```console +$ h5rs dump data.h5 # DDL, like h5dump +$ h5rs dump -A data.h5 # no dataset values (attributes still shown), like h5dump -A +$ h5rs dump -d /grp/gz data.h5 # one dataset +$ h5rs dump -p data.h5 # also STORAGE_LAYOUT and FILTERS blocks +$ h5rs dump --json data.h5 # hdf5-json +``` + +The DDL output is h5dump's: on the test files of `tests/gen_files.py` +(compact, contiguous and chunked datasets with every chunk index; v1 and v2 +groups; integers of both byte orders, floats, compound with an array member, +enum, fixed and variable-length strings; soft, external and hard links; a +named datatype; compact and dense attributes) `h5rs dump` and `h5rs dump -A` +print the same bytes as h5dump 1.14.6 and as Debian's h5dump 1.14.5 (the +`hdf5-tools` package CI installs in `rust:latest`; the whole interop suite +was run in that image on 2026-09-26) — `dump_matches_h5dump` in +`tests/h5rs_interop.rs` checks this, and `dump_shows_nul_padding_in_nested_strings` +that null-padded strings show their NULs (`"a\000b"`) at any depth, as +h5dump's do. Not covered by those tests: references, opaque, bitfield, +variable-length sequences and virtual datasets. Known differences from +h5dump: + +- Floats print at their own precision (a `float32` 0.1 prints as `0.1`), + which for some values is more digits than h5dump's `%g`. +- A compound nested in a compound prints inline (`{ 1, 2.5 }`) where + h5dump prints it as an indented block, one member per line; only the + outer compound is a block. +- `long double` (x87 80-bit) and other floats wider than 64 bits: the + datatype is printed as an `H5T_FLOAT { ... }` block instead of h5dump's + one-line description, and each value as ``; `dump` then + exits 1. The library cannot convert them (see `docs/known-issues.md`). +- The `-p` block is h5rs's own (it names the chunk index), not h5dump's. + +### JSON schema + +`--json` follows the HDF Group's [hdf5-json](https://github.com/HDFGroup/hdf5-json) +layout: + +```json +{ + "apiVersion": "1.1.1", + "root": "g-0000000000000060", + "groups": { "": { "alias": ["/grp"], "attributes": [...], "links": [...] } }, + "datasets": { "": { "alias": [...], "attributes": [...], "shape": {...}, + "type": {...}, "creationProperties": {...}, "value": ... } }, + "datatypes": { "": { "alias": [...], "attributes": [...], "type": {...} } } +} +``` + +- **ids** are `g-`/`d-`/`t-` plus the object header address in 16 hex digits + (hdf5-json uses UUIDs; these are stable for a given file). `alias` lists + every path that reaches the object. +- **links**: `{"class": "H5L_TYPE_HARD", "title", "collection", "id"}`, + `{"class": "H5L_TYPE_SOFT", "title", "h5path"}`, + `{"class": "H5L_TYPE_EXTERNAL", "title", "file", "h5path"}`, + `{"class": "H5L_TYPE_USER_DEFINED", "title", "linkClass"}`. +- **shape**: `{"class": "H5S_NULL"}`, `{"class": "H5S_SCALAR"}` or + `{"class": "H5S_SIMPLE", "dims": [...], "maxdims": [...]}` with + `"H5S_UNLIMITED"` for an unlimited dimension. +- **type**: `{"class": "H5T_INTEGER" | "H5T_FLOAT" | "H5T_BITFIELD", "base": + "H5T_STD_I32LE" ...}`, `{"class": "H5T_STRING", "charSet", "strPad", + "length": n | "H5T_VARIABLE"}`, `{"class": "H5T_COMPOUND", "fields": [{"name", + "type"}]}`, `{"class": "H5T_ARRAY", "base", "dims"}`, `{"class": "H5T_ENUM", + "base", "mapping": {"NAME": value}}`, `{"class": "H5T_VLEN", "base"}`, + `{"class": "H5T_OPAQUE", "size", "tag"}`, `{"class": "H5T_REFERENCE", "base": + "H5T_STD_REF_OBJ" | "H5T_STD_REF_DSETREG" | "H5T_STD_REF"}`. +- **value**: nested lists in the dataset's shape (a scalar is the bare value, + a null dataspace `null`). A compound element is a list of its members, an + enum element its integer value, a string a JSON string, opaque/bitfield data + a `0x...` hex string, an object reference the referenced object's path, + NaN/infinities the strings `"NaN"`, `"Infinity"`, `"-Infinity"`, and an + integer beyond 64 bits a decimal string. +- **creationProperties**: `layout` (`{"class": "H5D_CHUNKED", "dims": [...]}` + etc.) and `filters` (`[{"id", "name", "class", "parameters"}]`). + +A value that cannot be read is replaced by `"value_error": ""` and +the command exits 1. + +## `h5rs stat` + +Prints h5stat's report sections with the same labels for the facts it +computes — object and link counts, max links to an object, max objects in a +group, dataset ranks, layout counts, filter counts, attribute counts, total +raw data size and total file size (all equal to h5stat's on the test files; +`stat_matches_h5stat` checks them). It does not break metadata space down by +structure as h5stat does; it reports metadata and free space as one figure. + +## `h5rs diff` + +```console +$ h5rs diff a.h5 b.h5 # whole files +$ h5rs diff a.h5 b.h5 /grp # one object and everything below it +$ h5rs diff a.h5 b.h5 /x /y # different paths in each +$ h5rs diff -r a.h5 b.h5 /d # list every differing element +$ h5rs diff -d 0.001 a.h5 b.h5 # |a - b| > 0.001 is a difference +$ h5rs diff -p 0.01 a.h5 b.h5 # |a - b| / |a| > 1% is a difference +$ h5rs diff --follow-symlinks a.h5 b.h5 /lnk # what the soft link /lnk leads to +$ h5rs diff -r -n 10 a.h5 b.h5 # list at most 10 differing elements per object +``` + +The options are named as h5diff's: `-n N`/`--count=N` limits the listed +elements, `-c`/`--compare` (list objects that are not comparable) is +accepted and always in effect, and `--delta=D`, `--relative=R` work too. + +Integers are compared in integer arithmetic, with or without a tolerance, so +64-bit values beyond 2^53 lose no precision (`-d 0` tells 2^60 from +2^60 + 1). A relative tolerance below the f64 epsilon (2.2e-16) compares +exactly, as h5diff's does. + +Exit status: 0 no differences, 1 differences, 2 error — the same as h5diff's +on the cases `diff_exit_codes_match_h5diff` runs. Compared: which objects +exist, their kinds, datatypes and shapes, attribute sets and values, dataset +values, and soft/external link targets. A soft link — including an OBJ +that is itself a soft link — is compared as a link, by its target path, as +h5diff does; `--follow-symlinks` compares the objects soft links lead to +instead (and walks into soft-linked groups), and two dangling links are then +the same, as in h5diff. External links are always compared by target (file +and path): `--follow-symlinks` does not open other files, where h5diff's +does. Two dangling soft links with different targets are a difference +(h5diff reports them with `-r`/`-v` but exits 0 without). Every path is compared: an object +hard-linked under two names is compared under both, with everything below +it, so a file that shares one object between two names equals a file that +stores two identical copies. Differences from h5diff, on purpose: objects +that cannot be compared (different shapes or datatype classes) count as a +difference (h5diff warns and exits 0); two NaNs are equal; and the members +of a group reached by a second hard link are compared (h5diff lists them in +one file only and exits 1). + +## `h5rs check` + +```console +$ h5rs check data.h5 +checked data.h5: superblock v3, 33 objects (4 groups, 28 datasets, 1 named datatypes), 139 header messages, 34 chunks +checksums verified: superblock 1, v2 object headers 33, v2 B-trees 3, fractal heaps 3 (+3 blocks), chunk indexes 5 +no problems found + +$ h5rs check damaged.h5 # one bit flipped in /grp/ext1's chunk index header +problem: 0x79c /grp/ext1: chunk index (extensible array): checksum mismatch: expected 0xcfe2391f, computed 0x2dd59bed +checked damaged.h5: superblock v3, 33 objects (4 groups, 28 datasets, 1 named datatypes), 139 header messages, 27 chunks +checksums verified: superblock 1, v2 object headers 33, v2 B-trees 3, fractal heaps 3 (+3 blocks), chunk indexes 4 +1 problem found +``` + +Each problem line is `problem:
: `; addresses +are HDF5 addresses (relative to the superblock, as h5dump and h5ls print +them). + +It walks every object reachable from the root group (and the superblock +extension) and checks: + +- the superblock (and its checksum, version 2+) and that the file is not + shorter than the superblock's end-of-file address; +- every object header, with the checksum of version 2 headers and of their + continuation chunks, and every header message parsed by type (dataspace, + datatype, fill value, layout, filter pipeline, attributes, link info, + links, group info, symbol table), including shared messages; +- groups: the symbol table (v1 B-tree, local heap, symbol nodes), or the + links, and for dense storage the fractal heap — header, and every direct + and indirect block with its checksum, back-pointer and heap offset — and + the v2 B-tree name and creation-order indexes (every node's checksum, and + the record count against the header's); +- dense attribute storage the same way; +- datasets: the layout against the dataspace and datatype (compact and + contiguous sizes, chunk rank), and for chunked datasets the whole chunk + index (v1 B-tree, single chunk, implicit, fixed array, extensible array, + v2 B-tree, with the checksums of the last three): every chunk's offset must + be a multiple of the chunk size and inside the extent, appear once, and + have a plausible size; +- that all raw data (contiguous blocks and chunks) lies inside the file and + no two pieces overlap. + +`--data` also reads every dataset, decoding every chunk through its filters +(which catches corrupt compressed data and Fletcher-32 mismatches), and +follows every variable-length element (strings and sequences, also inside +compounds and arrays) of every dataset and attribute into its global heap +collection: a collection that does not parse, a missing heap object, or a +sequence longer than its heap object is a problem at the collection's +address. Data the +tool cannot decode (a filter it does not implement, such as szip, or a +dataset over `--max-bytes`) is a `note:`, not a problem. Every problem is +printed with the address of the structure involved; the exit status is 0 +when there are none, 1 when there are, 2 for a usage error or a missing +file. + +libhdf5's h5check understands only the HDF5 1.8 file format; `h5rs check` +also covers the structures HDF5 1.10+ writes (fixed/extensible array and v2 +B-tree chunk indexes, and version 3 superblocks). + +What it does not check: free-space manager and shared-message (SOHM) table +checksums, global heap collections no variable-length value points into (and +none at all without `--data`), and objects reachable only by external links. It validates with +clawhdf5's parsers, so it accepts what they accept: some header damage that +libhdf5 refuses goes unreported. Of the 150 CVE and fuzzer files of the +HDF Group's `cve_hdf5` corpus (`cvefiles/` and `fuzzerfiles/`), +`check --data` passes 16, and h5dump 1.14.6 rejects 9 of those (tank, +2026-09-26, `h5rs check --data F` and `h5dump F` per file; before the +library's header checks it passed 28, of which h5dump rejects 21). + +## Robustness + +A panic is a bug: `h5rs` catches it, prints `internal error`, and exits 3 +(`check` records it against the object and carries on). `scripts/h5rs-fuzz.sh` +runs every subcommand over every file of a corpus (by default the HDF Group's +CVE reproducers, fetched by `conformance/fetch-corpus.sh`), optionally with +byte-flipped copies (`MUTATE=N`), under a timeout and a memory limit, with +overflow checks on, and fails on any panic, crash or hang. +`scripts/h5rs-check-ok-files.sh` runs `check --data` over the conformance +files that both clawhdf5 and h5py read in full, which must all pass. + +## Tests + +```bash +CLAWHDF5_PYTHON=.venv/bin/python CLAWHDF5_REQUIRE_INTEROP=1 cargo test -p clawhdf5-tools +``` + +The interop tests write their files with h5py and compare with h5ls, h5stat, +h5dump and h5diff; each skips when what it needs is missing unless +`CLAWHDF5_REQUIRE_INTEROP=1`. diff --git a/crates/clawhdf5-tools/src/check.rs b/crates/clawhdf5-tools/src/check.rs new file mode 100644 index 0000000..8c56771 --- /dev/null +++ b/crates/clawhdf5-tools/src/check.rs @@ -0,0 +1,1011 @@ +//! `h5rs check`: a structural validator. +//! +//! It walks every object reachable from the root group (and the superblock +//! extension), parses every header message, verifies the checksum of every +//! checksummed (version 2+) structure it meets, checks each chunked +//! dataset's chunk index against the dataset's shape, and checks that raw +//! data lies inside the file without overlapping other raw data. Every +//! problem is reported with the address of the structure involved. + +use std::collections::{BTreeMap, HashSet}; +use std::panic::{self, AssertUnwindSafe}; + +use clawhdf5_format::attribute_info::AttributeInfoMessage; +use clawhdf5_format::btree_v2::{BTreeV2Header, collect_btree_v2_records}; +use clawhdf5_format::data_layout::DataLayout; +use clawhdf5_format::dataspace::{Dataspace, DataspaceType}; +use clawhdf5_format::datatype::Datatype; +use clawhdf5_format::group_info::GroupInfoMessage; +use clawhdf5_format::link_info::LinkInfoMessage; +use clawhdf5_format::message_type::MessageType; +use clawhdf5_format::object_header::ObjectHeader; +use clawhdf5_format::symbol_table::SymbolTableMessage; + +use crate::cli::{Args, Out}; +use crate::h5::{Error, ErrorKind, H5, Kind}; +use crate::info::{self, DsInfo}; + +pub const USAGE: &str = "\ +usage: h5rs check [--data] [-q] [--max-bytes N] FILE + +Validate FILE's structure: walk every object from the root group, parse +every header message, verify the checksums of version 2+ structures +(superblock, object headers and continuation chunks, v2 B-tree nodes, +fractal heap headers and blocks, extensible/fixed array chunk indexes), +check each chunked dataset's chunk index against its shape (offsets aligned +to the chunk size and inside the extent, no duplicates, sizes plausible), +and check that all raw data lies inside the file without overlaps. Every +problem is printed with the address of the structure involved. + + --data also read every dataset, decoding every chunk through its + filters (catches corrupt compressed data and Fletcher-32 + mismatches), and follow every variable-length element of + datasets and attributes into its global heap collection + -q, --quiet print only the problems, not the summary + --max-bytes N largest dataset read by --data (default 1 GiB) + +Exit status: 0 no problems, 1 problems found, 2 usage error or file not +found, 3 internal error."; + +const MAX_CHUNKS_CHECKED: usize = 10_000_000; + +/// Whether values of `dt` hold variable-length data (in the global heap). +fn has_vl(dt: &Datatype, depth: u32) -> bool { + if depth > 32 { + return false; + } + match dt { + Datatype::VariableLength { .. } => true, + Datatype::Compound { members, .. } => { + members.iter().any(|m| has_vl(&m.datatype, depth + 1)) + } + Datatype::Array { base_type, .. } => has_vl(base_type, depth + 1), + _ => false, + } +} + +#[derive(Default)] +struct Counts { + objects: u64, + groups: u64, + datasets: u64, + datatypes: u64, + messages: u64, + chunks: u64, + datasets_read: u64, + /// Global heap collections that variable-length data points into, + /// parsed without error (with --data). + global_heaps: u64, + sb_checksum: u64, + ohdr_v2: u64, + btree_v2: u64, + heaps: u64, + heap_block_checksums: u64, + chunk_index_checksummed: u64, +} + +struct Problem { + addr: u64, + path: String, + msg: String, +} + +struct Checker<'a> { + h5: &'a H5, + read_data: bool, + eof: u64, + problems: Vec, + /// Things not checked, which are not problems with the file. + notes: Vec, + counts: Counts, + /// Raw data extents: (start, end, owner path). + extents: Vec<(u64, u64, String)>, + heaps_seen: HashSet, + btrees_seen: HashSet, + /// Global heap collections already read (with --data). + gcols_seen: HashSet, + panicked: bool, +} + +pub fn run(args: &mut Args, out: &mut Out) -> std::io::Result { + let mut read_data = false; + let mut quiet = false; + let mut max_bytes = None; + let mut file = None; + while let Some(a) = args.next() { + match a.as_str() { + "--data" => read_data = true, + "-q" | "--quiet" => quiet = true, + "--max-bytes" => match args.number() { + Some(n) => max_bytes = Some(n), + None => return args.usage_error(out, "--max-bytes needs a number", USAGE), + }, + "-h" | "--help" => { + writeln!(out.o, "{USAGE}")?; + return Ok(0); + } + s if s.starts_with('-') && s.len() > 1 => { + return args.usage_error(out, &format!("unknown option {a}"), USAGE); + } + _ if file.is_none() => file = Some(a), + _ => return args.usage_error(out, &format!("unexpected argument {a}"), USAGE), + } + } + let Some(file) = file else { + return args.usage_error(out, "missing FILE", USAGE); + }; + let path = std::path::Path::new(&file); + if !path.is_file() { + writeln!(out.e, "h5rs check: {file}: no such file")?; + return Ok(2); + } + let mut h5 = match H5::open(path) { + Ok(h) => h, + Err(_) => return unopenable(path, out), + }; + if let Some(m) = max_bytes { + h5.max_bytes = m; + } + let mut c = Checker { + h5: &h5, + read_data, + eof: h5.data().len() as u64, + problems: Vec::new(), + notes: Vec::new(), + counts: Counts::default(), + extents: Vec::new(), + heaps_seen: HashSet::new(), + btrees_seen: HashSet::new(), + gcols_seen: HashSet::new(), + panicked: false, + }; + c.superblock(); + c.objects(); + c.overlaps(); + for p in &c.problems { + writeln!(out.o, "problem: {:#x} {}: {}", p.addr, p.path, p.msg)?; + } + if !quiet { + for n in &c.notes { + writeln!(out.o, "note: {:#x} {}: {}", n.addr, n.path, n.msg)?; + } + } + if !quiet { + c.summary(&file, out)?; + } + Ok(if c.panicked { + 3 + } else if c.problems.is_empty() { + 0 + } else { + 1 + }) +} + +/// The file could not be opened at all: say why, as precisely as possible. +fn unopenable(path: &std::path::Path, out: &mut Out) -> std::io::Result { + let data = match std::fs::read(path) { + Ok(d) => d, + Err(e) => { + writeln!(out.e, "h5rs check: {}: {e}", path.display())?; + return Ok(2); + } + }; + let msg = match clawhdf5_format::signature::find_signature(&data) { + Err(_) => ( + 0, + "no HDF5 signature at offset 0 or any power of two from 512".to_string(), + ), + Ok(off) => match clawhdf5_format::superblock::Superblock::parse(&data[off..], 0) { + Ok(sb) => match sb.data_end(off as u64, data.len() as u64) { + Err(clawhdf5_format::error::FormatError::TruncatedFile { + stored_eof, + actual_len, + }) => ( + 0, + format!( + "file is truncated: the superblock's end-of-file address is \ + {stored_eof:#x} but the file is {actual_len:#x} bytes long" + ), + ), + // Say what the library refused, not just that it did. + _ => match clawhdf5::File::from_bytes(data.clone()) { + Err(e) => (off as u64, format!("file cannot be opened: {e}")), + Ok(_) => (off as u64, "file cannot be opened".to_string()), + }, + }, + Err(e) => (off as u64, format!("superblock: {e}")), + }, + }; + writeln!(out.o, "problem: {:#x} /: {}", msg.0, msg.1)?; + writeln!( + out.o, + "checked {}: 1 problem found (file cannot be opened)", + path.display() + )?; + Ok(1) +} + +fn msg_name(t: MessageType) -> String { + match t { + MessageType::Unknown(n) => format!("message type {n:#x}"), + t => format!("{t:?} message"), + } +} + +impl Checker<'_> { + fn problem(&mut self, addr: u64, path: &str, msg: impl Into) { + self.problems.push(Problem { + addr, + path: path.to_string(), + msg: msg.into(), + }); + } + + fn err(&mut self, default_addr: u64, path: &str, e: &Error) { + // Reading attributes or links fails on a damaged heap or B-tree that + // the structure checks have already reported at the same address. + if let Some(a) = e.addr + && self.problems.iter().any(|p| p.addr == a && p.path == path) + { + return; + } + self.problem(e.addr.unwrap_or(default_addr), path, e.msg.clone()); + } + + /// Run `f`; a panic inside it becomes a problem instead of aborting the + /// whole check. + fn guarded(&mut self, addr: u64, path: &str, f: impl FnOnce(&mut Self)) { + let r = panic::catch_unwind(AssertUnwindSafe(|| f(self))); + if r.is_err() { + self.panicked = true; + self.problem( + addr, + path, + "internal error while checking this object (see stderr)", + ); + } + } + + fn superblock(&mut self) { + let sb = self.h5.sb().clone(); + if sb.version >= 2 { + // Superblock::parse verified it, or the file would not be open. + self.counts.sb_checksum += 1; + } + // libhdf5 stores the end-of-file address counting the user block + // (h5py's `userblock_size` files show it), unlike every other + // address, which is relative to the superblock. + let file_len = self.eof.saturating_add(self.h5.file.user_block_size()); + if sb.eof_address > file_len { + self.problem( + 0, + "/", + format!( + "file is truncated: the superblock's end-of-file address is {:#x} but the file \ + is {file_len:#x} bytes long", + sb.eof_address + ), + ); + } + if sb.root_group_address >= self.eof { + self.problem( + 0, + "/", + format!( + "root group address {:#x} is past the end of the file", + sb.root_group_address + ), + ); + } + let undef = if sb.offset_size >= 8 { + u64::MAX + } else { + (1u64 << (8 * u32::from(sb.offset_size))) - 1 + }; + if let Some(ext) = sb.superblock_extension_address.filter(|&a| a != undef) { + self.guarded(ext, "", |c| match c.h5.header(ext) { + Ok(h) => { + if h.version >= 2 { + c.counts.ohdr_v2 += 1; + } + c.messages(ext, "", &h); + } + Err(e) => c.err(ext, "", &e), + }); + } + } + + fn objects(&mut self) { + let h5 = self.h5; + // Collect first, then check: the walk borrows the file, the checks + // borrow `self` mutably. + let (items, walk) = h5.walk_collect(); + if let Err(e) = walk { + self.err(h5.root(), "/", &e); + } + for it in items { + let path = it.path; + // Soft, external and user-defined links have no header here. + let (Some(addr), None, Some(header)) = (it.addr, it.first_path, it.header) else { + continue; + }; + if addr >= self.eof { + self.problem( + addr, + &path, + "object header address is past the end of the file", + ); + continue; + } + let h = match header { + Ok(h) => h, + Err(e) => { + self.err(addr, &path, &e); + continue; + } + }; + self.counts.objects += 1; + self.guarded(addr, &path, |c| c.object(addr, &path, &h)); + } + } + + fn object(&mut self, addr: u64, path: &str, h: &ObjectHeader) { + if h.version >= 2 { + self.counts.ohdr_v2 += 1; + } + self.messages(addr, path, h); + match self.h5.attributes(h) { + Ok((attrs, errs)) => { + for e in errs { + self.problem(addr, path, format!("attribute: {e}")); + } + if self.read_data { + for a in &attrs { + let what = format!("attribute \"{}\": ", a.name); + self.vl_data(addr, path, &what, &a.datatype, &a.dataspace, &a.raw_data); + } + } + } + Err(e) => self.err(addr, path, &e), + } + let is_root = addr == self.h5.root(); + match Kind::of(h) { + Kind::Group => { + self.counts.groups += 1; + self.group(addr, path, h); + } + Kind::Dataset => { + self.counts.datasets += 1; + self.dataset(addr, path, h); + } + Kind::Datatype => self.counts.datatypes += 1, + Kind::Unknown if is_root => self.counts.groups += 1, + Kind::Unknown => { + self.problem( + addr, + path, + "object header describes no group, dataset or named datatype", + ); + } + } + } + + /// Parse every message of a header by type. + fn messages(&mut self, addr: u64, path: &str, h: &ObjectHeader) { + let (os, ls) = (self.h5.os(), self.h5.ls()); + for m in &h.messages { + self.counts.messages += 1; + let data = + match clawhdf5_format::shared_message::message_data(self.h5.data(), m, os, ls) { + Ok(d) => d.into_owned(), + Err(e) => { + self.problem(addr, path, format!("shared {}: {e}", msg_name(m.msg_type))); + continue; + } + }; + let r: Result<(), String> = match m.msg_type { + MessageType::Dataspace => Dataspace::parse(&data, ls) + .map(drop) + .map_err(|e| e.to_string()), + MessageType::Datatype => { + Datatype::parse(&data).map(drop).map_err(|e| e.to_string()) + } + MessageType::FillValue | MessageType::FillValueOld => { + let mut mm = m.clone(); + mm.data = data.clone(); + clawhdf5_format::fill_value::parse_fill_value(&mm) + .map(drop) + .map_err(|e| e.to_string()) + } + MessageType::DataLayout => DataLayout::parse(&data, os, ls) + .map(drop) + .map_err(|e| e.to_string()), + MessageType::FilterPipeline => { + clawhdf5_format::filter_pipeline::FilterPipeline::parse(&data) + .map(drop) + .map_err(|e| e.to_string()) + } + MessageType::Attribute => { + clawhdf5_format::attribute::AttributeMessage::parse_in_file( + &data, + self.h5.data(), + os, + ls, + ) + .map(drop) + .map_err(|e| e.to_string()) + } + MessageType::AttributeInfo => match AttributeInfoMessage::parse(&data, os) { + Ok(ai) => { + self.dense_storage( + addr, + path, + "attribute", + ai.fractal_heap_address, + [ai.btree_name_index_address, ai.btree_creation_order_address], + ); + Ok(()) + } + Err(e) => Err(e.to_string()), + }, + MessageType::LinkInfo => LinkInfoMessage::parse(&data, os) + .map(drop) + .map_err(|e| e.to_string()), + MessageType::Link => clawhdf5_format::link_message::LinkMessage::parse(&data, os) + .map(drop) + .or_else(|e| match e { + clawhdf5_format::error::FormatError::InvalidLinkType(t) if t >= 65 => { + Ok(()) + } + e => Err(e.to_string()), + }), + MessageType::GroupInfo => GroupInfoMessage::parse(&data) + .map(drop) + .map_err(|e| e.to_string()), + MessageType::SymbolTable => SymbolTableMessage::parse(&data, os) + .map(drop) + .map_err(|e| e.to_string()), + _ => Ok(()), + }; + if let Err(e) = r { + self.problem(addr, path, format!("{}: {e}", msg_name(m.msg_type))); + } + } + } + + /// Dense (fractal heap + v2 B-tree) link or attribute storage. + fn dense_storage( + &mut self, + addr: u64, + path: &str, + what: &str, + heap: Option, + btrees: [Option; 2], + ) { + if let Some(fh) = heap + && self.heaps_seen.insert(fh) + { + self.counts.heaps += 1; + let r = crate::heap_blocks::verify(self.h5, fh); + self.counts.heap_block_checksums += r.checksums as u64; + for e in r.problems { + self.err(fh, path, &e.context(&format!("dense {what} storage"))); + } + } + for bt in btrees.into_iter().flatten() { + if !self.btrees_seen.insert(bt) { + continue; + } + let (os, ls) = (self.h5.os(), self.h5.ls()); + let Ok(off) = usize::try_from(bt) else { + self.problem(bt, path, format!("{what} index address out of range")); + continue; + }; + match BTreeV2Header::parse(self.h5.data(), off, os, ls) { + Ok(hdr) => { + self.counts.btree_v2 += 1; + match collect_btree_v2_records(self.h5.data(), &hdr, os, ls) { + Ok(recs) => { + if recs.len() as u64 != hdr.total_records { + self.problem( + bt, + path, + format!( + "{what} index: v2 B-tree header counts {} records, the tree holds {}", + hdr.total_records, + recs.len() + ), + ); + } + } + Err(e) => self.problem(bt, path, format!("{what} index (v2 B-tree): {e}")), + } + } + Err(e) => self.problem(bt, path, format!("{what} index (v2 B-tree header): {e}")), + } + } + let _ = addr; + } + + fn group(&mut self, addr: u64, path: &str, h: &ObjectHeader) { + if let Some(m) = h + .messages + .iter() + .find(|m| m.msg_type == MessageType::LinkInfo) + && let Ok(li) = LinkInfoMessage::parse(&m.data, self.h5.os()) + { + self.dense_storage( + addr, + path, + "link", + li.fractal_heap_address, + [li.btree_name_index_address, li.btree_creation_order_address], + ); + } + if let Err(e) = self.h5.links(h) { + self.err(addr, path, &e.context("cannot list the group")); + } + if let Some(m) = h + .messages + .iter() + .find(|m| m.msg_type == MessageType::SymbolTable) + && let Ok(st) = SymbolTableMessage::parse(&m.data, self.h5.os()) + { + for (a, what) in [ + (st.btree_address, "B-tree"), + (st.local_heap_address, "local heap"), + ] { + if a >= self.eof { + self.problem( + addr, + path, + format!("symbol table {what} address {a:#x} is past the end of the file"), + ); + } + } + } + } + + fn dataset(&mut self, addr: u64, path: &str, h: &ObjectHeader) { + let info = DsInfo::read(self.h5, path, h); + for e in [ + info.dt.as_ref().err(), + info.ds.as_ref().err(), + info.layout.as_ref().err(), + info.filters.as_ref().err(), + ] + .into_iter() + .flatten() + { + let e = e.clone(); + self.err(addr, path, &e); + } + let (Ok(dt), Ok(ds), Ok(layout)) = (&info.dt, &info.ds, &info.layout) else { + return; + }; + let esize = u64::from(dt.type_size()); + if esize == 0 { + self.problem(addr, path, "datatype has size 0"); + return; + } + let need = match crate::h5::byte_len(ds, dt) { + Ok(n) => n, + Err(e) => { + self.err(addr, path, &e); + return; + } + }; + if let Some(max) = &ds.max_dimensions { + for (i, (&d, &m)) in ds.dimensions.iter().zip(max).enumerate() { + if m != u64::MAX && d > m { + self.problem( + addr, + path, + format!("dimension {i} is {d}, above its maximum {m}"), + ); + } + } + } + match layout { + DataLayout::Compact { data } => { + if (data.len() as u64) < need { + self.problem( + addr, + path, + format!( + "compact data holds {} bytes; the dataset needs {need}", + data.len() + ), + ); + } + } + DataLayout::Contiguous { address, size } => { + if let Some(a) = address { + if !info.external && *size < need { + self.problem( + *a, + path, + format!("contiguous storage is {size} bytes; the dataset needs {need}"), + ); + } + self.extent(*a, *size, path); + } + } + DataLayout::Chunked { .. } => self.chunked(addr, path, &info, dt, ds, layout), + DataLayout::Virtual { .. } => { + let mut l = layout.clone(); + if let Err(e) = l.resolve_vds_mappings(self.h5.data(), self.h5.ls()) { + self.problem(addr, path, format!("virtual dataset mappings: {e}")); + } + } + } + if self.read_data { + match self.h5.read_dataset(path, dt, ds) { + Ok(raw) => { + self.counts.datasets_read += 1; + self.vl_data(addr, path, "", dt, ds, &raw); + } + // Valid data this tool cannot decode is not a problem with + // the file. + Err(e) if e.kind != ErrorKind::Corrupt => self.notes.push(Problem { + addr, + path: path.to_string(), + msg: format!("data not read: {}", e.msg), + }), + Err(e) => self.problem(addr, path, format!("reading the data: {}", e.msg)), + } + } + } + + /// With --data: follow every variable-length element of `raw` (the + /// values of a dataset or attribute) into the global heap, so a damaged + /// collection, a missing heap object or a sequence longer than its heap + /// object is reported at the collection's address. Each bad collection + /// is reported once per object. + fn vl_data( + &mut self, + addr: u64, + path: &str, + what: &str, + dt: &Datatype, + ds: &Dataspace, + raw: &[u8], + ) { + if !has_vl(dt, 0) { + return; + } + let n = crate::h5::num_elements(ds).unwrap_or(0); + let size = dt.type_size() as usize; + let mut bad: BTreeMap = BTreeMap::new(); + for i in 0..n { + let Some(b) = usize::try_from(i) + .ok() + .and_then(|i| i.checked_mul(size)) + .and_then(|s| raw.get(s..s.checked_add(size)?)) + else { + self.problem( + addr, + path, + format!("{what}element {i} is past the data read"), + ); + break; + }; + self.vl_element(dt, b, 0, &mut bad); + if bad.len() >= 100 { + break; + } + } + for (a, msg) in bad { + self.problem(a, path, format!("{what}variable-length data: {msg}")); + } + } + + fn vl_element(&mut self, dt: &Datatype, b: &[u8], depth: u32, bad: &mut BTreeMap) { + if depth > 32 { + return; + } + match dt { + Datatype::VariableLength { + is_string, + base_type, + .. + } => { + let os = usize::from(self.h5.os()); + let (Some(lenb), Some(addrb), Some(idxb)) = + (b.get(..4), b.get(4..4 + os), b.get(4 + os..8 + os)) + else { + return; + }; + let le = |x: &[u8]| { + x.iter() + .enumerate() + .fold(0u64, |a, (i, &v)| a | (u64::from(v) << (8 * i))) + }; + let (len, gcol, idx) = (le(lenb), le(addrb), le(idxb)); + let undef = if os >= 8 { + u64::MAX + } else { + (1u64 << (8 * os)) - 1 + }; + if len == 0 || gcol == 0 || gcol == undef || bad.contains_key(&gcol) { + return; + } + let obj = match self.h5.heap_object(gcol, idx as u32) { + Ok(o) => o, + Err(e) => { + bad.insert(e.addr.unwrap_or(gcol), e.msg); + return; + } + }; + if self.gcols_seen.insert(gcol) { + self.counts.global_heaps += 1; + } + let bs = if *is_string { + 1 + } else { + u64::from(base_type.type_size()) + }; + if len + .checked_mul(bs) + .is_none_or(|need| need > obj.len() as u64) + { + bad.insert( + gcol, + format!( + "global heap object {idx} holds {} bytes; the element needs {len} x {bs}", + obj.len() + ), + ); + return; + } + if !*is_string && bs > 0 && has_vl(base_type, depth + 1) { + let bs = bs as usize; + for k in 0..len as usize { + self.vl_element(base_type, &obj[k * bs..(k + 1) * bs], depth + 1, bad); + } + } + } + Datatype::Compound { members, .. } => { + for m in members { + if let Some(mb) = usize::try_from(m.byte_offset).ok().and_then(|o| b.get(o..)) { + self.vl_element(&m.datatype, mb, depth + 1, bad); + } + } + } + Datatype::Array { + base_type, + dimensions, + } => { + let bs = base_type.type_size() as usize; + let n = dimensions + .iter() + .try_fold(1usize, |a, &d| a.checked_mul(d as usize)) + .unwrap_or(usize::MAX); + for k in 0..n { + match b.get(k * bs..(k + 1) * bs) { + Some(eb) => self.vl_element(base_type, eb, depth + 1, bad), + None => break, + } + } + } + _ => {} + } + } + + fn chunked( + &mut self, + addr: u64, + path: &str, + info: &DsInfo, + dt: &Datatype, + ds: &Dataspace, + layout: &DataLayout, + ) { + let DataLayout::Chunked { + chunk_dimensions, + btree_address, + chunk_index_type, + .. + } = layout + else { + return; + }; + let rank = match ds.space_type { + DataspaceType::Simple => ds.dimensions.len(), + _ => 0, + }; + if chunk_dimensions.len() != rank + 1 { + self.problem( + addr, + path, + format!( + "chunked layout has {} chunk dimensions for a rank-{rank} dataset (expected {})", + chunk_dimensions.len(), + rank + 1 + ), + ); + return; + } + let cdims = &chunk_dimensions[..rank]; + if cdims.contains(&0) { + self.problem(addr, path, "a chunk dimension is 0"); + return; + } + let Some(index_addr) = *btree_address else { + return; // no chunk allocated yet + }; + if index_addr >= self.eof { + self.problem( + addr, + path, + format!("chunk index address {index_addr:#x} is past the end of the file"), + ); + return; + } + let chunks = match info::chunks(self.h5, layout, ds, dt) { + Ok(c) => c, + Err(e) => { + self.err(index_addr, path, &e); + return; + } + }; + if matches!(chunk_index_type, Some(3..=5)) { + self.counts.chunk_index_checksummed += 1; + } + let filtered = matches!(&info.filters, Ok(Some(p)) if !p.filters.is_empty()); + let chunk_bytes = cdims.iter().try_fold(u64::from(dt.type_size()), |a, &d| { + a.checked_mul(u64::from(d)) + }); + let max = ds.max_dimensions.clone(); + let mut seen: HashSet> = HashSet::with_capacity(chunks.len().min(1 << 20)); + let mut reported = 0usize; + for (n, c) in chunks.iter().enumerate() { + if n >= MAX_CHUNKS_CHECKED { + self.problem(index_addr, path, "too many chunks; stopped checking them"); + break; + } + self.counts.chunks += 1; + let mut bad = Vec::new(); + if c.offsets.len() < rank { + bad.push(format!( + "has {} coordinates for a rank-{rank} dataset", + c.offsets.len() + )); + } else { + for (i, (&o, &cd)) in c.offsets.iter().zip(cdims).enumerate() { + if o % u64::from(cd) != 0 { + bad.push(format!( + "offset {o} in dimension {i} is not a multiple of the chunk size {cd}" + )); + } + let limit = match max.as_ref().and_then(|m| m.get(i)) { + Some(&u64::MAX) | None => ds.dimensions[i], + Some(&m) => m.max(ds.dimensions[i]), + }; + if o >= limit && !(limit == 0 && o == 0) { + bad.push(format!( + "offset {o} in dimension {i} is outside the extent {limit}" + )); + } + } + if !seen.insert(c.offsets[..rank].to_vec()) { + bad.push("appears twice in the chunk index".into()); + } + } + if c.chunk_size == 0 { + bad.push("has size 0".into()); + } else if !filtered + && let Some(cb) = chunk_bytes + && u64::from(c.chunk_size) != cb + { + bad.push(format!( + "is {} bytes; an unfiltered chunk is {cb}", + c.chunk_size + )); + } + if !bad.is_empty() { + reported += 1; + if reported <= 50 { + self.problem( + c.address, + path, + format!("chunk at {:?} {}", c.offsets, bad.join("; ")), + ); + } + } + self.extent(c.address, u64::from(c.chunk_size), path); + } + if reported > 50 { + self.problem( + index_addr, + path, + format!("{} more bad chunks not listed", reported - 50), + ); + } + } + + /// Record raw data at `[start, start + len)`, checking it is in the file. + fn extent(&mut self, start: u64, len: u64, path: &str) { + if len == 0 { + return; + } + match start.checked_add(len) { + Some(end) if end <= self.eof => self.extents.push((start, end, path.to_string())), + _ => self.problem( + start, + path, + format!("raw data ({len} bytes) extends past the end of the file"), + ), + } + } + + fn overlaps(&mut self) { + let mut ext = std::mem::take(&mut self.extents); + ext.sort_unstable_by_key(|e| (e.0, e.1)); + let mut reported = 0usize; + let mut far: Option<(u64, String)> = None; + let mut msgs = Vec::new(); + for (s, e, p) in &ext { + if let Some((end, owner)) = &far + && s < end + { + reported += 1; + if reported <= 50 { + msgs.push(( + *s, + p.clone(), + format!("raw data at {s:#x} overlaps raw data of {owner}"), + )); + } + } + if far.as_ref().is_none_or(|(end, _)| e > end) { + far = Some((*e, p.clone())); + } + } + for (a, p, m) in msgs { + self.problem(a, &p, m); + } + if reported > 50 { + self.problem( + 0, + "/", + format!("{} more raw data overlaps not listed", reported - 50), + ); + } + } + + fn summary(&self, file: &str, out: &mut Out) -> std::io::Result<()> { + let c = &self.counts; + let sb = self.h5.sb(); + writeln!( + out.o, + "checked {file}: superblock v{}, {} objects ({} groups, {} datasets, {} named datatypes), \ + {} header messages, {} chunks", + sb.version, c.objects, c.groups, c.datasets, c.datatypes, c.messages, c.chunks + )?; + writeln!( + out.o, + "checksums verified: superblock {}, v2 object headers {}, v2 B-trees {}, \ + fractal heaps {} (+{} blocks), chunk indexes {}", + c.sb_checksum, + c.ohdr_v2, + c.btree_v2, + c.heaps, + c.heap_block_checksums, + c.chunk_index_checksummed + )?; + if self.read_data { + writeln!( + out.o, + "datasets read: {}, global heap collections read: {}", + c.datasets_read, c.global_heaps + )?; + } + match self.problems.len() { + 0 => writeln!(out.o, "no problems found"), + 1 => writeln!(out.o, "1 problem found"), + n => writeln!(out.o, "{n} problems found"), + } + } +} diff --git a/crates/clawhdf5-tools/src/cli.rs b/crates/clawhdf5-tools/src/cli.rs new file mode 100644 index 0000000..a4f6cde --- /dev/null +++ b/crates/clawhdf5-tools/src/cli.rs @@ -0,0 +1,52 @@ +//! Argument handling and output plumbing shared by the subcommands. + +use std::io::Write; + +/// Where a subcommand writes: `o` for results, `e` for diagnostics. +pub struct Out<'a> { + pub o: &'a mut dyn Write, + pub e: &'a mut dyn Write, +} + +/// The arguments after the subcommand name. +pub struct Args { + cmd: &'static str, + rest: std::collections::VecDeque, +} + +impl Args { + pub fn new(cmd: &'static str, rest: Vec) -> Self { + Self { + cmd, + rest: rest.into(), + } + } + + #[allow(clippy::should_implement_trait)] + pub fn next(&mut self) -> Option { + self.rest.pop_front() + } + + /// Put an option and its value back, to be read next (for the + /// `--name=value` form). + pub fn push_front(&mut self, name: String, value: String) { + self.rest.push_front(value); + self.rest.push_front(name); + } + + /// The value after an option such as `--max-bytes`. + pub fn value(&mut self) -> Option { + self.rest.pop_front() + } + + /// A numeric option value. + pub fn number(&mut self) -> Option { + self.rest.pop_front().and_then(|s| s.parse().ok()) + } + + /// Report a usage problem; exit status 2. + pub fn usage_error(&self, out: &mut Out, msg: &str, usage: &str) -> std::io::Result { + writeln!(out.e, "h5rs {}: {msg}\n\n{usage}", self.cmd)?; + Ok(2) + } +} diff --git a/crates/clawhdf5-tools/src/diff.rs b/crates/clawhdf5-tools/src/diff.rs new file mode 100644 index 0000000..c7ad723 --- /dev/null +++ b/crates/clawhdf5-tools/src/diff.rs @@ -0,0 +1,904 @@ +//! `h5rs diff`: compare two files (or two objects) like h5diff. + +use std::cell::OnceCell; +use std::collections::{BTreeMap, HashMap}; +use std::rc::Rc; + +use clawhdf5_format::attribute::AttributeMessage; +use clawhdf5_format::dataspace::{Dataspace, DataspaceType}; +use clawhdf5_format::datatype::Datatype; +use clawhdf5_format::object_header::ObjectHeader; + +use crate::cli::{Args, Out}; +use crate::h5::{H5, Kind, Link, LinkKind}; +use crate::value::{self, Decoder, Value}; + +pub const USAGE: &str = "\ +usage: h5rs diff [options] FILE1 FILE2 [OBJ1 [OBJ2]] + +Compare FILE1 and FILE2 (or OBJ1 in FILE1 with OBJ2 in FILE2, and everything +below them): the objects present, their kinds, datatypes, shapes, attribute +sets and values, and soft/external link targets (an OBJ that is itself a +soft link is compared as a link, as h5diff does, unless --follow-symlinks). + + -r, --report list every differing element (position, values, difference) + -q, --quiet print nothing; only the exit status + --follow-symlinks compare the objects soft links lead to (and walk into + soft-linked groups) instead of the links' target paths; + two dangling links are the same + -d, --delta D numbers differ only when |a - b| > D + -p, --relative R numbers differ only when |a - b| / |a| > R (an R below + the f64 epsilon, 2.2e-16, compares exactly, as h5diff) + -n, --count N list at most N differing elements per object with -r + -c, --compare list objects that are not comparable (always done; + accepted for h5diff compatibility) + --max-bytes N largest dataset read (default 1 GiB); a larger one is an error + +Two NaNs compare equal. Unlike h5diff, objects that cannot be compared +(different kinds, datatype classes or shapes) count as a difference. + +Exit status: 0 no differences, 1 differences found, 2 error (a file or +object could not be opened or read)."; + +#[derive(Clone, Copy)] +enum Tol { + Exact, + Delta(f64), + Relative(f64), +} + +struct Opts { + report: bool, + quiet: bool, + tol: Tol, + count: usize, + /// Compare the objects soft links lead to, not the links' targets. + follow: bool, +} + +/// What a relative path names in one file. +#[derive(Clone)] +enum Entry { + Obj(u64, Kind), + Soft(String), + External(String, String), + UserDefined(u8), + Broken(String), +} + +struct Side<'a> { + h5: &'a H5, + label: String, + base: String, + /// Relative path -> what it is. + entries: BTreeMap, + /// Address -> relative path, for comparing references. + rel_of: OnceCell>, +} + +impl Side<'_> { + fn full(&self, rel: &str) -> String { + if rel.is_empty() { + if self.base.is_empty() { + "/".into() + } else { + self.base.clone() + } + } else { + format!("{}{rel}", self.base) + } + } + + fn rel_paths(&self) -> &HashMap { + self.rel_of.get_or_init(|| { + let mut m = HashMap::new(); + let _ = self.h5.walk(|it| { + if let (Some(a), None) = (it.addr, it.first_path) { + m.entry(a).or_insert_with(|| it.path.to_string()); + } + }); + m + }) + } +} + +struct Diff { + opts: Opts, + diffs: u64, + /// Differences already reported under an object heading. + per_object: u64, + errors: u64, +} + +pub fn run(args: &mut Args, out: &mut Out) -> std::io::Result { + let mut opts = Opts { + report: false, + quiet: false, + tol: Tol::Exact, + count: usize::MAX, + follow: false, + }; + let mut max_bytes = None; + let mut pos = Vec::new(); + while let Some(a) = args.next() { + match a.as_str() { + "-r" | "--report" => opts.report = true, + "-q" | "--quiet" => opts.quiet = true, + "--follow-symlinks" => opts.follow = true, + "-d" | "--delta" => match args.number::() { + Some(d) if d >= 0.0 => opts.tol = Tol::Delta(d), + _ => return args.usage_error(out, "--delta needs a number >= 0", USAGE), + }, + "-p" | "--relative" => match args.number::() { + // As h5diff: a relative tolerance below the f64 epsilon + // cannot be told from rounding, so it compares exactly. + Some(r) if r >= 0.0 => { + opts.tol = if r < f64::EPSILON { + Tol::Exact + } else { + Tol::Relative(r) + } + } + _ => return args.usage_error(out, "--relative needs a number >= 0", USAGE), + }, + // h5diff's -c: h5rs always lists them. + "-c" | "--compare" => {} + "-n" | "--count" => match args.number() { + Some(n) => opts.count = n, + None => return args.usage_error(out, "--count needs a number", USAGE), + }, + "--max-bytes" => match args.number() { + Some(n) => max_bytes = Some(n), + None => return args.usage_error(out, "--max-bytes needs a number", USAGE), + }, + "-h" | "--help" => { + writeln!(out.o, "{USAGE}")?; + return Ok(0); + } + // h5diff's --count=N, --delta=D, --relative=R forms. + s if s.starts_with("--") && s.contains('=') => { + let (k, v) = s.split_once('=').unwrap_or((s, "")); + if !matches!(k, "--count" | "--delta" | "--relative" | "--max-bytes") { + return args.usage_error(out, &format!("unknown option {a}"), USAGE); + } + args.push_front(k.to_string(), v.to_string()); + } + s if s.starts_with('-') && s.len() > 1 => { + return args.usage_error(out, &format!("unknown option {a}"), USAGE); + } + _ => pos.push(a), + } + } + if pos.len() < 2 || pos.len() > 4 { + return args.usage_error(out, "expected FILE1 FILE2 [OBJ1 [OBJ2]]", USAGE); + } + let mut files = Vec::new(); + for f in &pos[..2] { + match H5::open(std::path::Path::new(f)) { + Ok(mut h) => { + if let Some(m) = max_bytes { + h.max_bytes = m; + } + files.push(h); + } + Err(e) => { + writeln!(out.e, "h5rs diff: {e}")?; + return Ok(2); + } + } + } + let obj1 = pos.get(2).cloned().unwrap_or_else(|| "/".into()); + let obj2 = pos.get(3).cloned().unwrap_or_else(|| obj1.clone()); + let mut sides = Vec::new(); + for (h5, (obj, f)) in files.iter().zip([(&obj1, &pos[0]), (&obj2, &pos[1])]) { + let start = match start(h5, obj, opts.follow) { + Ok(x) => x, + Err(_) => { + writeln!( + out.e, + "h5rs diff: object <{obj}> could not be found in <{f}>" + )?; + return Ok(2); + } + }; + let base = obj.trim_end_matches('/').to_string(); + let base = if base.is_empty() || base.starts_with('/') { + base + } else { + format!("/{base}") + }; + let collected = match start { + Start::Obj(addr) => collect(h5, addr, &base, opts.follow), + Start::Link(e) => Ok(BTreeMap::from([(String::new(), e)])), + }; + let entries = match collected { + Ok(e) => e, + Err(e) => { + writeln!(out.e, "h5rs diff: {f}: {e}")?; + return Ok(2); + } + }; + sides.push(Side { + h5, + label: f.clone(), + base, + entries, + rel_of: OnceCell::new(), + }); + } + let (a, b) = (&sides[0], &sides[1]); + let mut d = Diff { + opts, + diffs: 0, + per_object: 0, + errors: 0, + }; + let mut names: Vec<&String> = a.entries.keys().chain(b.entries.keys()).collect(); + names.sort(); + names.dedup(); + for rel in names { + match (a.entries.get(rel), b.entries.get(rel)) { + (Some(_), None) => { + d.diffs += 1; + d.say( + out, + &format!("<{}> exists only in <{}>", a.full(rel), a.label), + )?; + } + (None, Some(_)) => { + d.diffs += 1; + d.say( + out, + &format!("<{}> exists only in <{}>", b.full(rel), b.label), + )?; + } + (Some(ea), Some(eb)) => d.entry(out, a, b, rel, ea, eb)?, + (None, None) => {} + } + } + // Per-object counts were printed with each object; add a total when + // they do not already tell the whole story. + if !d.opts.quiet && d.diffs > 0 && d.diffs != d.per_object { + writeln!(out.o, "{} difference(s) found in total", d.diffs)?; + } + Ok(if d.errors > 0 { + 2 + } else if d.diffs > 0 { + 1 + } else { + 0 + }) +} + +/// What OBJ1/OBJ2 names. +enum Start { + Obj(u64), + /// A soft (not followed, or dangling), external or user-defined link. + Link(Entry), +} + +/// Resolve OBJ1/OBJ2. Soft links on the way to it are followed, as in any +/// HDF5 path; a soft link that *is* the named object is compared as a link +/// (its target path), as h5diff does, unless `follow`. +fn start(h5: &H5, obj: &str, follow: bool) -> crate::h5::Result { + let p = obj.trim_matches('/'); + if !p.is_empty() { + let (parent, name) = p.rsplit_once('/').unwrap_or(("", p)); + let link = h5 + .resolve(parent) + .and_then(|a| h5.header(a)) + .and_then(|h| h5.links(&h)) + .ok() + .and_then(|ls| ls.into_iter().find(|l| l.name == name)); + if let Some(l) = link { + return Ok(match l.kind { + LinkKind::Hard(a) => Start::Obj(a), + LinkKind::Soft(t) if follow => match soft_target(h5, &format!("/{parent}"), &t) { + Ok(a) => Start::Obj(a), + Err(_) => Start::Link(Entry::Soft(t)), + }, + LinkKind::Soft(t) => Start::Link(Entry::Soft(t)), + LinkKind::External { file, path } => Start::Link(Entry::External(file, path)), + LinkKind::UserDefined(t) => Start::Link(Entry::UserDefined(t)), + }); + } + } + h5.resolve(obj).map(Start::Obj) +} + +/// The object a soft link in group `parent` (an absolute path) leads to. +fn soft_target(h5: &H5, parent: &str, target: &str) -> crate::h5::Result { + if target.starts_with('/') { + h5.resolve(target) + } else { + h5.resolve(&format!("{}/{target}", parent.trim_end_matches('/'))) + } +} + +/// One object as the path walk sees it: what its paths compare as, and +/// (for a group) its links. +struct Node { + entry: Entry, + links: Vec, +} + +/// Every path below the object at `start`, relative to it (`""` is `start` +/// itself). Each hard link is its own path, so an object linked under two +/// names is compared under both, with everything below it: two files that +/// hold the same values are equal whether one shares an object between +/// names and the other stores copies. A hard link back to an ancestor (a +/// cycle) is recorded, not descended into. With `follow`, a soft link is +/// walked like a hard link to its target (a dangling one stays a link); +/// `base` is the absolute path of `start`, for relative link targets. +fn collect( + h5: &H5, + start: u64, + base: &str, + follow: bool, +) -> crate::h5::Result> { + let mut cache: HashMap> = HashMap::new(); + let mut node = |addr: u64| -> Rc { + cache + .entry(addr) + .or_insert_with(|| { + Rc::new(match h5.header(addr) { + Err(e) => Node { + entry: Entry::Broken(e.to_string()), + links: Vec::new(), + }, + Ok(h) => match Kind::of(&h) { + Kind::Group => match h5.links(&h) { + Ok(links) => Node { + entry: Entry::Obj(addr, Kind::Group), + links, + }, + Err(e) => Node { + entry: Entry::Broken(format!("links: {e}")), + links: Vec::new(), + }, + }, + k => Node { + entry: Entry::Obj(addr, k), + links: Vec::new(), + }, + }, + }) + }) + .clone() + }; + let mut entries = BTreeMap::new(); + // (address, relative path, addresses of the groups above it) + let mut stack: Vec<(u64, String, Rc>)> = + vec![(start, String::new(), Rc::new(Vec::new()))]; + while let Some((addr, path, above)) = stack.pop() { + if entries.len() >= crate::h5::MAX_OBJECTS { + return Err(crate::h5::Error::new(format!( + "more than {} paths; stopped walking", + crate::h5::MAX_OBJECTS + ))); + } + let n = node(addr); + entries.insert(path.clone(), n.entry.clone()); + if n.links.is_empty() || above.contains(&addr) { + continue; + } + let mut chain = (*above).clone(); + chain.push(addr); + let chain = Rc::new(chain); + for l in &n.links { + let child = format!("{path}/{}", l.name); + match &l.kind { + LinkKind::Hard(a) => stack.push((*a, child, chain.clone())), + LinkKind::Soft(t) if follow => match soft_target(h5, &format!("{base}{path}"), t) { + Ok(a) => stack.push((a, child, chain.clone())), + Err(_) => { + entries.insert(child, Entry::Soft(t.clone())); + } + }, + LinkKind::Soft(t) => { + entries.insert(child, Entry::Soft(t.clone())); + } + LinkKind::External { file, path } => { + entries.insert(child, Entry::External(file.clone(), path.clone())); + } + LinkKind::UserDefined(t) => { + entries.insert(child, Entry::UserDefined(*t)); + } + } + } + } + Ok(entries) +} + +fn kind_word(k: Kind) -> &'static str { + match k { + Kind::Group => "group", + Kind::Dataset => "dataset", + Kind::Datatype => "datatype", + Kind::Unknown => "object", + } +} + +impl Diff { + fn say(&self, out: &mut Out, msg: &str) -> std::io::Result<()> { + if self.opts.quiet { + Ok(()) + } else { + writeln!(out.o, "{msg}") + } + } + + fn error(&mut self, out: &mut Out, msg: &str) -> std::io::Result<()> { + self.errors += 1; + writeln!(out.e, "h5rs diff: {msg}") + } + + fn entry( + &mut self, + out: &mut Out, + a: &Side, + b: &Side, + rel: &str, + ea: &Entry, + eb: &Entry, + ) -> std::io::Result<()> { + let (pa, pb) = (a.full(rel), b.full(rel)); + match (ea, eb) { + (Entry::Broken(e), _) => self.error(out, &format!("<{pa}> in <{}>: {e}", a.label)), + (_, Entry::Broken(e)) => self.error(out, &format!("<{pb}> in <{}>: {e}", b.label)), + (Entry::Soft(x), Entry::Soft(y)) => { + // Followed links that are left are dangling on both sides, + // which h5diff counts as the same. + if x != y && !self.opts.follow { + self.diffs += 1; + self.say(out, &format!("soft link: <{pa}> -> {x} and <{pb}> -> {y}"))?; + } + Ok(()) + } + (Entry::External(f1, p1), Entry::External(f2, p2)) => { + if (f1, p1) != (f2, p2) { + self.diffs += 1; + self.say( + out, + &format!("external link: <{pa}> -> {f1}:{p1} and <{pb}> -> {f2}:{p2}"), + )?; + } + Ok(()) + } + (Entry::UserDefined(x), Entry::UserDefined(y)) => { + if x != y { + self.diffs += 1; + self.say(out, &format!("user-defined link: <{pa}> and <{pb}> differ"))?; + } + Ok(()) + } + (Entry::Obj(aa, ka), Entry::Obj(ab, kb)) if ka == kb => { + let (ha, hb) = match (a.h5.header(*aa), b.h5.header(*ab)) { + (Ok(x), Ok(y)) => (x, y), + (Err(e), _) | (_, Err(e)) => return self.error(out, &format!("<{pa}>: {e}")), + }; + let before = self.diffs; + let mut rows = Vec::new(); + match ka { + Kind::Dataset => self.dataset(out, a, b, &pa, &pb, &ha, &hb, &mut rows)?, + Kind::Datatype => match (a.h5.datatype(&ha), b.h5.datatype(&hb)) { + (Ok(x), Ok(y)) => { + if x != y { + self.diffs += 1; + rows.push("datatypes differ".to_string()); + } + } + (Err(e), _) | (_, Err(e)) => self.error(out, &format!("<{pa}>: {e}"))?, + }, + _ => {} + } + self.attributes(out, a, b, &pa, &ha, &hb, &mut rows)?; + let n = self.diffs - before; + if n > 0 && !self.opts.quiet { + writeln!(out.o, "{}: <{pa}> and <{pb}>", kind_word(*ka))?; + for r in &rows { + writeln!(out.o, "{r}")?; + } + writeln!(out.o, "{n} difference(s) found")?; + self.per_object += n; + } + Ok(()) + } + _ => { + self.diffs += 1; + self.say( + out, + &format!( + "Not comparable: <{pa}> is a {} and <{pb}> is a {}", + entry_word(ea), + entry_word(eb) + ), + ) + } + } + } + + #[allow(clippy::too_many_arguments)] + fn dataset( + &mut self, + out: &mut Out, + a: &Side, + b: &Side, + pa: &str, + pb: &str, + ha: &ObjectHeader, + hb: &ObjectHeader, + rows: &mut Vec, + ) -> std::io::Result<()> { + let got = (|| -> crate::h5::Result<_> { + Ok(( + a.h5.datatype(ha)?, + a.h5.resolved_dataspace(pa, ha)?, + b.h5.datatype(hb)?, + b.h5.resolved_dataspace(pb, hb)?, + )) + })(); + let (dta, dsa, dtb, dsb) = match got { + Ok(x) => x, + Err(e) => return self.error(out, &format!("<{pa}>: {e}")), + }; + if let Some(why) = not_comparable(&dta, &dsa, &dtb, &dsb) { + self.diffs += 1; + rows.push(format!("Not comparable: {why}")); + return Ok(()); + } + let raw = + a.h5.read_dataset(pa, &dta, &dsa) + .and_then(|x| Ok((x, b.h5.read_dataset(pb, &dtb, &dsb)?))); + let (ra, rb) = match raw { + Ok(x) => x, + Err(e) => return self.error(out, &format!("<{pa}>: {e}")), + }; + self.values(out, a, b, pa, (&dta, &ra), (&dtb, &rb), &dsa, rows) + } + + #[allow(clippy::too_many_arguments)] + fn values( + &mut self, + out: &mut Out, + a: &Side, + b: &Side, + what: &str, + (dta, ra): (&Datatype, &[u8]), + (dtb, rb): (&Datatype, &[u8]), + ds: &Dataspace, + rows: &mut Vec, + ) -> std::io::Result<()> { + let n = crate::h5::num_elements(ds).unwrap_or(0) as usize; + let (da, db) = (Decoder::new(a.h5), Decoder::new(b.h5)); + let dims: Vec = match ds.space_type { + DataspaceType::Simple => ds.dimensions.clone(), + _ => vec![1], + }; + let mut found = 0u64; + let mut header = false; + for i in 0..n { + let va = da.element(dta, ra, i); + let vb = db.element(dtb, rb, i); + if let (Value::Error(e), _) | (_, Value::Error(e)) = (&va, &vb) { + return self.error(out, &format!("<{what}> element {i}: {e}")); + } + if self.equal(a, b, &va, &vb) { + continue; + } + found += 1; + if self.opts.report && (found as usize) <= self.opts.count { + if !header { + header = true; + rows.push(format!( + "{:<24}{:<24}{:<24}{}", + "position", "value 1", "value 2", "difference" + )); + rows.push("-".repeat(80)); + } + let pos = format!("[ {} ]", index(i as u64, &dims)); + let ta = value::text(&va, &|_| None); + let tb = value::text(&vb, &|_| None); + let dif = match (int_of(&va), int_of(&vb), number(&va), number(&vb)) { + (Some(x), Some(y), ..) => x.abs_diff(y).to_string(), + (_, _, Some(x), Some(y)) => value::fmt_float((x - y).abs(), 64), + _ => String::new(), + }; + rows.push(format!("{pos:<24}{ta:<24}{tb:<24}{dif}")); + } + } + self.diffs += found; + Ok(()) + } + + #[allow(clippy::too_many_arguments)] + fn attributes( + &mut self, + out: &mut Out, + a: &Side, + b: &Side, + path: &str, + ha: &ObjectHeader, + hb: &ObjectHeader, + rows: &mut Vec, + ) -> std::io::Result<()> { + let (la, lb) = match (a.h5.attributes(ha), b.h5.attributes(hb)) { + (Ok(x), Ok(y)) => (x, y), + (Err(e), _) | (_, Err(e)) => return self.error(out, &format!("<{path}>: {e}")), + }; + for e in la.1.iter().chain(lb.1.iter()) { + self.error(out, &format!("<{path}>: attribute: {e}"))?; + } + let ma: BTreeMap<&str, &AttributeMessage> = + la.0.iter().map(|x| (x.name.as_str(), x)).collect(); + let mb: BTreeMap<&str, &AttributeMessage> = + lb.0.iter().map(|x| (x.name.as_str(), x)).collect(); + let mut names: Vec<&str> = ma.keys().chain(mb.keys()).copied().collect(); + names.sort_unstable(); + names.dedup(); + for n in names { + match (ma.get(n), mb.get(n)) { + (Some(x), Some(y)) => { + if let Some(why) = + not_comparable(&x.datatype, &x.dataspace, &y.datatype, &y.dataspace) + { + self.diffs += 1; + rows.push(format!("attribute \"{n}\": not comparable: {why}")); + continue; + } + for (m, h5) in [(x, a.h5), (y, b.h5)] { + let need = + crate::h5::byte_len(&m.dataspace, &m.datatype).unwrap_or(u64::MAX); + if need > h5.max_bytes || (m.raw_data.len() as u64) < need { + return self.error( + out, + &format!( + "<{path}> attribute \"{n}\": value is truncated or too large" + ), + ); + } + } + let before = rows.len(); + let what = format!("{path}\" attribute \"{n}"); + self.values( + out, + a, + b, + &what, + (&x.datatype, &x.raw_data), + (&y.datatype, &y.raw_data), + &x.dataspace, + rows, + )?; + if rows.len() > before { + rows.insert(before, format!("attribute \"{n}\":")); + } + } + (Some(_), None) => { + self.diffs += 1; + rows.push(format!("attribute \"{n}\" exists only in <{}>", a.label)); + } + (None, Some(_)) => { + self.diffs += 1; + rows.push(format!("attribute \"{n}\" exists only in <{}>", b.label)); + } + (None, None) => {} + } + } + Ok(()) + } + + fn equal(&self, a: &Side, b: &Side, x: &Value, y: &Value) -> bool { + // Integers in integer arithmetic: through f64 they lose precision + // above 2^53, and values that differ would compare equal. + if let (Some(p), Some(q)) = (int_of(x), int_of(y)) { + return int_close(p, q, self.opts.tol); + } + if let (Some(p), Some(q)) = (number(x), number(y)) { + return close(p, q, self.opts.tol); + } + match (x, y) { + (Value::Str(p), Value::Str(q)) => p == q, + (Value::Bytes(p), Value::Bytes(q)) | (Value::OtherRef(p), Value::OtherRef(q)) => p == q, + (Value::Compound(p), Value::Compound(q)) => { + p.len() == q.len() + && p.iter() + .zip(q) + .all(|((_, u), (_, v))| self.equal(a, b, u, v)) + } + (Value::Array(p), Value::Array(q)) | (Value::Seq(p), Value::Seq(q)) => { + p.len() == q.len() && p.iter().zip(q).all(|(u, v)| self.equal(a, b, u, v)) + } + (Value::Ref(None), Value::Ref(None)) => true, + (Value::Ref(Some(p)), Value::Ref(Some(q))) => { + // Addresses mean nothing across files: compare the paths the + // references lead to. + let (pp, qq) = (a.rel_paths().get(p), b.rel_paths().get(q)); + pp.is_some() && pp == qq + } + _ => false, + } + } +} + +fn int_of(v: &Value) -> Option { + match v { + Value::Int(i) | Value::Enum(_, i) => Some(*i), + _ => None, + } +} + +fn number(v: &Value) -> Option { + match v { + Value::Int(i) | Value::Enum(_, i) => Some(*i as f64), + Value::Float(f, _) => Some(*f), + _ => None, + } +} + +fn close(a: f64, b: f64, tol: Tol) -> bool { + if a.is_nan() || b.is_nan() { + return a.is_nan() && b.is_nan(); + } + if a == b { + return true; + } + let d = (a - b).abs(); + match tol { + Tol::Exact => false, + Tol::Delta(t) => d <= t, + Tol::Relative(r) => a != 0.0 && d / a.abs() <= r, + } +} + +/// `close` for integers, exactly: the difference is taken in i128, so no +/// precision is lost at any 64-bit magnitude. +fn int_close(a: i128, b: i128, tol: Tol) -> bool { + if a == b { + return true; + } + let d = a.abs_diff(b); + match tol { + Tol::Exact => false, + // d is whole, so d <= t exactly when d <= floor(t) (saturating). + Tol::Delta(t) => d <= t.floor() as u128, + // d >= 1 here, so the quotient is never rounded to 0. + Tol::Relative(r) => a != 0 && d as f64 / a.unsigned_abs() as f64 <= r, + } +} + +fn entry_word(e: &Entry) -> &'static str { + match e { + Entry::Obj(_, k) => kind_word(*k), + Entry::Soft(_) => "soft link", + Entry::External(..) => "external link", + Entry::UserDefined(_) => "user-defined link", + Entry::Broken(_) => "unreadable object", + } +} + +/// Why two datasets/attributes cannot be compared element by element. +fn not_comparable( + dta: &Datatype, + dsa: &Dataspace, + dtb: &Datatype, + dsb: &Dataspace, +) -> Option { + let rank = |d: &Dataspace| match d.space_type { + DataspaceType::Simple => Some(d.dimensions.clone()), + DataspaceType::Scalar => Some(Vec::new()), + DataspaceType::Null => None, + }; + let (sa, sb) = (rank(dsa), rank(dsb)); + if sa != sb { + let show = |s: &Option>| match s { + None => "null".to_string(), + Some(d) if d.is_empty() => "scalar".to_string(), + Some(d) => format!("{d:?}"), + }; + return Some(format!("shapes differ: {} and {}", show(&sa), show(&sb))); + } + if !types_comparable(dta, dtb) { + return Some(format!( + "datatypes differ: {} and {}", + crate::dtype::short(dta), + crate::dtype::short(dtb) + )); + } + None +} + +fn types_comparable(a: &Datatype, b: &Datatype) -> bool { + use crate::dtype::class; + let numeric = |t: &Datatype| matches!(class(t), "integer" | "float"); + if numeric(a) && numeric(b) { + return class(a) == class(b); + } + if class(a) != class(b) { + return false; + } + match (a, b) { + (Datatype::Compound { members: ma, .. }, Datatype::Compound { members: mb, .. }) => { + ma.len() == mb.len() + && ma + .iter() + .zip(mb) + .all(|(x, y)| x.name == y.name && types_comparable(&x.datatype, &y.datatype)) + } + ( + Datatype::Array { + base_type: x, + dimensions: dx, + }, + Datatype::Array { + base_type: y, + dimensions: dy, + }, + ) => dx == dy && types_comparable(x, y), + ( + Datatype::VariableLength { base_type: x, .. }, + Datatype::VariableLength { base_type: y, .. }, + ) => class(a) == "string" || types_comparable(x, y), + ( + Datatype::Enumeration { base_type: x, .. }, + Datatype::Enumeration { base_type: y, .. }, + ) => types_comparable(x, y), + _ => true, + } +} + +fn index(mut i: u64, dims: &[u64]) -> String { + let mut idx = vec![0u64; dims.len()]; + for (k, &d) in dims.iter().enumerate().rev() { + let d = d.max(1); + idx[k] = i % d; + i /= d; + } + idx.iter() + .map(|x| x.to_string()) + .collect::>() + .join(" ") +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn tolerances() { + assert!(close(1.0, 1.0, Tol::Exact)); + assert!(!close(1.0, 1.001, Tol::Exact)); + assert!(close(1.0, 1.001, Tol::Delta(0.01))); + assert!(!close(1.0, 1.1, Tol::Delta(0.01))); + assert!(close(100.0, 101.0, Tol::Relative(0.02))); + assert!(!close(0.0, 1e-9, Tol::Relative(0.5))); + assert!(close(f64::NAN, f64::NAN, Tol::Exact)); + assert!(!close(f64::NAN, 1.0, Tol::Delta(1e9))); + } + + #[test] + fn integer_tolerances_are_exact_above_2_pow_53() { + let big = 1i128 << 60; + assert!(!int_close(big, big + 1, Tol::Delta(0.0))); + assert!(!int_close(big, big + 1, Tol::Delta(0.5))); + assert!(int_close(big, big + 1, Tol::Delta(1.0))); + assert!(!int_close(big, big + 200, Tol::Delta(199.9))); + assert!(!int_close(big, big + 1, Tol::Relative(0.0))); + assert!(!int_close(big, big + 1, Tol::Relative(1e-19))); + assert!(int_close(big, big + 1, Tol::Relative(1e-18))); + let umax = i128::from(u64::MAX); + assert!(!int_close(umax, umax - 1, Tol::Delta(0.0))); + assert!(int_close( + i128::from(i64::MIN), + umax, + Tol::Delta(f64::INFINITY) + )); + assert!(!int_close(0, 1, Tol::Relative(1e9))); + } + + #[test] + fn positions() { + assert_eq!(index(5, &[3, 4]), "1 1"); + assert_eq!(index(0, &[1]), "0"); + } +} diff --git a/crates/clawhdf5-tools/src/dtype.rs b/crates/clawhdf5-tools/src/dtype.rs new file mode 100644 index 0000000..c42c4eb --- /dev/null +++ b/crates/clawhdf5-tools/src/dtype.rs @@ -0,0 +1,529 @@ +//! Names for datatypes: a short one for listings, h5ls's long form, h5dump's +//! DDL and the HDF Group's hdf5-json type objects. + +use clawhdf5_format::datatype::{ + CharacterSet, Datatype, DatatypeByteOrder, ReferenceType, StringPadding, +}; +use serde_json::{Value as J, json}; + +fn be(o: &DatatypeByteOrder) -> bool { + matches!(o, DatatypeByteOrder::BigEndian) +} + +fn order_suffix(o: &DatatypeByteOrder) -> &'static str { + match o { + DatatypeByteOrder::LittleEndian => "LE", + DatatypeByteOrder::BigEndian => "BE", + DatatypeByteOrder::Vax => "VAX", + } +} + +fn order_word(o: &DatatypeByteOrder) -> &'static str { + match o { + DatatypeByteOrder::LittleEndian => "little-endian", + DatatypeByteOrder::BigEndian => "big-endian", + DatatypeByteOrder::Vax => "VAX-order", + } +} + +/// True when a float is laid out exactly as IEEE 754 binary16/32/64. +pub fn is_ieee(dt: &Datatype) -> bool { + let Datatype::FloatingPoint { + size, + bit_offset, + bit_precision, + exponent_location, + exponent_size, + mantissa_location, + mantissa_size, + exponent_bias, + .. + } = dt + else { + return false; + }; + let std = match size { + 2 => (16, 10, 5, 10, 15), + 4 => (32, 23, 8, 23, 127), + 8 => (64, 52, 11, 52, 1023), + _ => return false, + }; + *bit_offset == 0 + && *mantissa_location == 0 + && ( + *bit_precision, + *exponent_location, + *exponent_size, + *mantissa_size, + *exponent_bias, + ) == std +} + +/// Short name used by `ls`: `int32`, `float64-be`, `string[3]`, ... +pub fn short(dt: &Datatype) -> String { + match dt { + Datatype::FixedPoint { + size, + signed, + byte_order, + .. + } => format!( + "{}int{}{}", + if *signed { "" } else { "u" }, + u64::from(*size) * 8, + if be(byte_order) { "-be" } else { "" } + ), + Datatype::FloatingPoint { + size, byte_order, .. + } => format!( + "float{}{}", + u64::from(*size) * 8, + if be(byte_order) { "-be" } else { "" } + ), + Datatype::Time { size, .. } => format!("time{}", u64::from(*size) * 8), + Datatype::String { size, .. } => format!("string[{size}]"), + Datatype::BitField { size, .. } => format!("bitfield{}", u64::from(*size) * 8), + Datatype::Opaque { size, .. } => format!("opaque[{size}]"), + Datatype::Compound { members, .. } => format!( + "compound{{{}}}", + members + .iter() + .map(|m| format!("{}: {}", m.name, short(&m.datatype))) + .collect::>() + .join(", ") + ), + Datatype::Reference { ref_type, .. } => match ref_type { + ReferenceType::Object | ReferenceType::Object2 => "object-reference".into(), + ReferenceType::DatasetRegion | ReferenceType::DatasetRegion2 => { + "region-reference".into() + } + ReferenceType::Attribute => "attribute-reference".into(), + }, + Datatype::Enumeration { base_type, .. } => format!("enum<{}>", short(base_type)), + Datatype::VariableLength { + is_string: true, .. + } => "vlen-string".into(), + Datatype::VariableLength { base_type, .. } => format!("vlen<{}>", short(base_type)), + Datatype::Array { + base_type, + dimensions, + } => format!( + "array[{}]<{}>", + dimensions + .iter() + .map(|d| d.to_string()) + .collect::>() + .join(","), + short(base_type) + ), + } +} + +fn pad_word(p: &StringPadding) -> &'static str { + match p { + StringPadding::NullTerminate => "null-terminated", + StringPadding::NullPad => "null-padded", + StringPadding::SpacePad => "space-padded", + } +} + +fn cset_word(c: &CharacterSet) -> &'static str { + match c { + CharacterSet::Ascii => "ASCII", + CharacterSet::Utf8 => "UTF-8", + } +} + +/// h5ls -v style description. +pub fn long(dt: &Datatype) -> String { + match dt { + Datatype::FixedPoint { + size, + signed, + byte_order, + .. + } => format!( + "{}-bit {} {}integer", + u64::from(*size) * 8, + order_word(byte_order), + if *signed { "" } else { "unsigned " } + ), + Datatype::FloatingPoint { + size, byte_order, .. + } => { + if is_ieee(dt) { + format!( + "IEEE {}-bit {} float", + u64::from(*size) * 8, + order_word(byte_order) + ) + } else { + format!( + "{}-bit {} non-IEEE float", + u64::from(*size) * 8, + order_word(byte_order) + ) + } + } + Datatype::Time { size, .. } => format!("{}-bit time", u64::from(*size) * 8), + Datatype::String { + size, + padding, + charset, + } => format!( + "{size}-byte {} {} string", + pad_word(padding), + cset_word(charset) + ), + Datatype::BitField { + size, byte_order, .. + } => format!( + "{}-bit {} bitfield", + u64::from(*size) * 8, + order_word(byte_order) + ), + Datatype::Opaque { size, tag } => { + let tag = String::from_utf8_lossy(tag); + let tag = tag.trim_end_matches('\0'); + if tag.is_empty() { + format!("{size}-byte opaque type") + } else { + format!("{size}-byte opaque type (tag \"{tag}\")") + } + } + Datatype::Compound { size, members } => { + let mut s = String::from("struct {"); + for m in members { + s.push_str(&format!( + "\n \"{}\" +{} {}", + m.name, + m.byte_offset, + long(&m.datatype) + )); + } + s.push_str(&format!("\n }} {size} bytes")); + s + } + Datatype::Reference { ref_type, .. } => match ref_type { + ReferenceType::Object | ReferenceType::Object2 => "object reference".into(), + ReferenceType::DatasetRegion | ReferenceType::DatasetRegion2 => { + "dataset region reference".into() + } + ReferenceType::Attribute => "attribute reference".into(), + }, + Datatype::Enumeration { + base_type, members, .. + } => { + let vals: Vec = members + .iter() + .map(|m| format!("{} = {}", m.name, enum_value_text(base_type, &m.value))) + .collect(); + format!("enum {} {{{}}}", long(base_type), vals.join(", ")) + } + Datatype::VariableLength { + is_string: true, + padding, + charset, + .. + } => format!( + "variable-length {} {} string", + padding.as_ref().map(pad_word).unwrap_or("null-terminated"), + charset.as_ref().map(cset_word).unwrap_or("ASCII") + ), + Datatype::VariableLength { base_type, .. } => { + format!("variable length of {}", long(base_type)) + } + Datatype::Array { + base_type, + dimensions, + } => format!( + "[{}] {}", + dimensions + .iter() + .map(|d| d.to_string()) + .collect::>() + .join(","), + long(base_type) + ), + } +} + +/// The integer an enum member's value bytes hold, as text. +pub fn enum_value_text(base: &Datatype, bytes: &[u8]) -> String { + match crate::value::decode_int(base, bytes) { + Some(v) => v.to_string(), + None => crate::value::hex(bytes), + } +} + +/// h5dump's predefined name for an atomic type, if it has one. +fn atomic_ddl(dt: &Datatype) -> Option { + match dt { + Datatype::FixedPoint { + size, + signed, + byte_order, + .. + } => Some(format!( + "H5T_STD_{}{}{}", + if *signed { "I" } else { "U" }, + u64::from(*size) * 8, + order_suffix(byte_order) + )), + Datatype::FloatingPoint { + size, byte_order, .. + } if is_ieee(dt) => Some(format!( + "H5T_IEEE_F{}{}", + u64::from(*size) * 8, + order_suffix(byte_order) + )), + Datatype::BitField { + size, byte_order, .. + } => Some(format!( + "H5T_STD_B{}{}", + u64::from(*size) * 8, + order_suffix(byte_order) + )), + Datatype::Time { .. } => Some("H5T_TIME".into()), + _ => None, + } +} + +/// h5dump DDL for a datatype; `ind` is the indentation of the line the type +/// starts on (continuation lines are indented relative to it). +pub fn ddl(dt: &Datatype, ind: usize) -> String { + let pad = " ".repeat(ind + 3); + let end = " ".repeat(ind); + if let Some(s) = atomic_ddl(dt) { + return s; + } + match dt { + Datatype::FloatingPoint { + size, + byte_order, + bit_offset, + bit_precision, + exponent_location, + exponent_size, + mantissa_location, + mantissa_size, + exponent_bias, + } => format!( + "H5T_FLOAT {{\n{pad}SIZE {size};\n{pad}ORDER H5T_ORDER_{};\n{pad}OFFSET {bit_offset};\n\ + {pad}PRECISION {bit_precision};\n{pad}EXPONENT {exponent_location} {exponent_size} \ + BIAS {exponent_bias};\n{pad}MANTISSA {mantissa_location} {mantissa_size};\n{end}}}", + order_suffix(byte_order) + ), + Datatype::String { + size, + padding, + charset, + } => string_ddl(&size.to_string(), padding, charset, &pad, &end), + Datatype::VariableLength { + is_string: true, + padding, + charset, + .. + } => string_ddl( + "H5T_VARIABLE", + padding.as_ref().unwrap_or(&StringPadding::NullTerminate), + charset.as_ref().unwrap_or(&CharacterSet::Ascii), + &pad, + &end, + ), + Datatype::VariableLength { base_type, .. } => { + format!("H5T_VLEN {{ {} }}", ddl(base_type, ind)) + } + Datatype::Opaque { size, tag } => { + let tag = String::from_utf8_lossy(tag); + format!( + "H5T_OPAQUE {{\n{pad}OPAQUE_SIZE {size};\n{pad}OPAQUE_TAG \"{}\";\n{end}}}", + tag.trim_end_matches('\0') + ) + } + Datatype::Compound { members, .. } => { + let mut s = String::from("H5T_COMPOUND {\n"); + for m in members { + s.push_str(&format!( + "{pad}{} \"{}\";\n", + ddl(&m.datatype, ind + 3), + m.name + )); + } + s.push_str(&end); + s.push('}'); + s + } + Datatype::Reference { ref_type, .. } => match ref_type { + ReferenceType::Object => "H5T_REFERENCE { H5T_STD_REF_OBJECT }".into(), + ReferenceType::DatasetRegion => "H5T_REFERENCE { H5T_STD_REF_DSETREG }".into(), + _ => "H5T_REFERENCE { H5T_STD_REF }".into(), + }, + Datatype::Enumeration { + base_type, members, .. + } => { + let mut s = format!("H5T_ENUM {{\n{pad}{};\n", ddl(base_type, ind + 3)); + for m in members { + let name = format!("\"{}\"", m.name); + s.push_str(&format!( + "{pad}{name:<18} {};\n", + enum_value_text(base_type, &m.value) + )); + } + s.push_str(&end); + s.push('}'); + s + } + Datatype::Array { + base_type, + dimensions, + } => format!( + "H5T_ARRAY {{ {} {} }}", + dimensions + .iter() + .map(|d| format!("[{d}]")) + .collect::(), + ddl(base_type, ind) + ), + // Atomic types were handled above. + _ => short(dt), + } +} + +fn string_ddl( + size: &str, + padding: &StringPadding, + charset: &CharacterSet, + pad: &str, + end: &str, +) -> String { + let p = match padding { + StringPadding::NullTerminate => "H5T_STR_NULLTERM", + StringPadding::NullPad => "H5T_STR_NULLPAD", + StringPadding::SpacePad => "H5T_STR_SPACEPAD", + }; + let c = match charset { + CharacterSet::Ascii => "H5T_CSET_ASCII", + CharacterSet::Utf8 => "H5T_CSET_UTF8", + }; + format!( + "H5T_STRING {{\n{pad}STRSIZE {size};\n{pad}STRPAD {p};\n{pad}CSET {c};\n{pad}CTYPE H5T_C_S1;\n{end}}}" + ) +} + +/// hdf5-json type object. +pub fn json(dt: &Datatype) -> J { + match dt { + Datatype::FixedPoint { .. } => { + json!({"class": "H5T_INTEGER", "base": atomic_ddl(dt)}) + } + Datatype::FloatingPoint { .. } if is_ieee(dt) => { + json!({"class": "H5T_FLOAT", "base": atomic_ddl(dt)}) + } + Datatype::FloatingPoint { + size, byte_order, .. + } => json!({ + "class": "H5T_FLOAT", + "size": size, + "order": format!("H5T_ORDER_{}", order_suffix(byte_order)), + }), + Datatype::BitField { .. } => json!({"class": "H5T_BITFIELD", "base": atomic_ddl(dt)}), + Datatype::Time { size, .. } => json!({"class": "H5T_TIME", "size": size}), + Datatype::String { + size, + padding, + charset, + } => json!({ + "class": "H5T_STRING", + "charSet": cset_json(charset), + "strPad": pad_json(padding), + "length": size, + }), + Datatype::VariableLength { + is_string: true, + padding, + charset, + .. + } => json!({ + "class": "H5T_STRING", + "charSet": cset_json(charset.as_ref().unwrap_or(&CharacterSet::Ascii)), + "strPad": pad_json(padding.as_ref().unwrap_or(&StringPadding::NullTerminate)), + "length": "H5T_VARIABLE", + }), + Datatype::VariableLength { base_type, .. } => { + json!({"class": "H5T_VLEN", "base": json(base_type)}) + } + Datatype::Opaque { size, tag } => json!({ + "class": "H5T_OPAQUE", + "size": size, + "tag": String::from_utf8_lossy(tag).trim_end_matches('\0'), + }), + Datatype::Compound { members, .. } => json!({ + "class": "H5T_COMPOUND", + "fields": members + .iter() + .map(|m| json!({"name": m.name, "type": json(&m.datatype)})) + .collect::>(), + }), + Datatype::Reference { ref_type, .. } => json!({ + "class": "H5T_REFERENCE", + "base": match ref_type { + ReferenceType::Object => "H5T_STD_REF_OBJ", + ReferenceType::DatasetRegion => "H5T_STD_REF_DSETREG", + _ => "H5T_STD_REF", + }, + }), + Datatype::Enumeration { + base_type, members, .. + } => { + let mut mapping = serde_json::Map::new(); + for m in members { + let v = crate::value::decode_int(base_type, &m.value) + .and_then(|v| i64::try_from(v).ok()) + .map(J::from) + .unwrap_or_else(|| J::from(crate::value::hex(&m.value))); + mapping.insert(m.name.clone(), v); + } + json!({"class": "H5T_ENUM", "base": json(base_type), "mapping": mapping}) + } + Datatype::Array { + base_type, + dimensions, + } => json!({"class": "H5T_ARRAY", "base": json(base_type), "dims": dimensions}), + } +} + +fn cset_json(c: &CharacterSet) -> &'static str { + match c { + CharacterSet::Ascii => "H5T_CSET_ASCII", + CharacterSet::Utf8 => "H5T_CSET_UTF8", + } +} + +fn pad_json(p: &StringPadding) -> &'static str { + match p { + StringPadding::NullTerminate => "H5T_STR_NULLTERM", + StringPadding::NullPad => "H5T_STR_NULLPAD", + StringPadding::SpacePad => "H5T_STR_SPACEPAD", + } +} + +/// Class name used to decide whether two datatypes can be compared. +pub fn class(dt: &Datatype) -> &'static str { + match dt { + Datatype::FixedPoint { .. } => "integer", + Datatype::FloatingPoint { .. } => "float", + Datatype::Time { .. } => "time", + Datatype::String { .. } => "string", + Datatype::BitField { .. } => "bitfield", + Datatype::Opaque { .. } => "opaque", + Datatype::Compound { .. } => "compound", + Datatype::Reference { .. } => "reference", + Datatype::Enumeration { .. } => "enum", + Datatype::VariableLength { + is_string: true, .. + } => "string", + Datatype::VariableLength { .. } => "vlen", + Datatype::Array { .. } => "array", + } +} diff --git a/crates/clawhdf5-tools/src/dump.rs b/crates/clawhdf5-tools/src/dump.rs new file mode 100644 index 0000000..7b1d8e7 --- /dev/null +++ b/crates/clawhdf5-tools/src/dump.rs @@ -0,0 +1,1002 @@ +//! `h5rs dump`: the whole file (or one dataset) as h5dump-style DDL text or +//! as hdf5-json. + +use std::cell::OnceCell; +use std::collections::HashMap; + +use clawhdf5_format::attribute::AttributeMessage; +use clawhdf5_format::data_layout::DataLayout; +use clawhdf5_format::dataspace::{Dataspace, DataspaceType}; +use clawhdf5_format::datatype::{Datatype, StringPadding}; +use clawhdf5_format::object_header::ObjectHeader; +use serde_json::{Map, Value as J, json}; + +use crate::cli::{Args, Out}; +use crate::h5::{Error, H5, Kind, Link, LinkKind}; +use crate::info::{self, DsInfo}; +use crate::value::{self, Decoder, Value}; + +pub const USAGE: &str = "\ +usage: h5rs dump [--json] [-A] [-p] [-d PATH] [--max-bytes N] FILE + +Print FILE's groups, datasets, named datatypes, links and attributes, with +their values, as h5dump-style DDL text (default) or as JSON. + + --json hdf5-json layout (see the crate README for the schema) + -A, --header no dataset values (attribute values are still printed, + as with h5dump -A) + -p, --properties also print each dataset's storage layout and filters + -d, --dataset P dump only the dataset at path P + --max-bytes N largest dataset or attribute decoded (default 1 GiB); + a larger one is reported instead of read + +Exit status: 0 dumped, 1 something could not be read (reported on stderr +and marked in the output), 2 error."; + +struct Opts { + json: bool, + header_only: bool, + props: bool, +} + +struct Dump<'a> { + h5: &'a H5, + opts: Opts, + problems: usize, + paths: OnceCell>, +} + +pub fn run(args: &mut Args, out: &mut Out) -> std::io::Result { + let mut opts = Opts { + json: false, + header_only: false, + props: false, + }; + let mut dataset = None; + let mut max_bytes = None; + let mut file = None; + while let Some(a) = args.next() { + match a.as_str() { + "--json" | "-j" => opts.json = true, + "-A" | "--header" => opts.header_only = true, + "-p" | "--properties" => opts.props = true, + "-d" | "--dataset" => match args.value() { + Some(p) => dataset = Some(p), + None => return args.usage_error(out, "-d needs a path", USAGE), + }, + "--max-bytes" => match args.number() { + Some(n) => max_bytes = Some(n), + None => return args.usage_error(out, "--max-bytes needs a number", USAGE), + }, + "-h" | "--help" => { + writeln!(out.o, "{USAGE}")?; + return Ok(0); + } + s if s.starts_with('-') && s.len() > 1 => { + return args.usage_error(out, &format!("unknown option {a}"), USAGE); + } + _ if file.is_none() => file = Some(a), + _ => return args.usage_error(out, &format!("unexpected argument {a}"), USAGE), + } + } + let Some(file) = file else { + return args.usage_error(out, "missing FILE", USAGE); + }; + let mut h5 = match H5::open(std::path::Path::new(&file)) { + Ok(h) => h, + Err(e) => { + writeln!(out.e, "h5rs dump: {e}")?; + return Ok(2); + } + }; + if let Some(m) = max_bytes { + h5.max_bytes = m; + } + let mut d = Dump { + h5: &h5, + opts, + problems: 0, + paths: OnceCell::new(), + }; + let fname = std::path::Path::new(&file) + .file_name() + .map(|s| s.to_string_lossy().into_owned()) + .unwrap_or(file.clone()); + let code = if d.opts.json { + d.json(out, dataset.as_deref())? + } else { + d.ddl(out, &fname, dataset.as_deref())? + }; + if code != 0 { + return Ok(code); + } + Ok(if d.problems > 0 { 1 } else { 0 }) +} + +/// A full path for `name` inside the group at `base`. +fn join(base: &str, name: &str) -> String { + if base == "/" { + format!("/{name}") + } else { + format!("{base}/{name}") + } +} + +fn quote(s: &str) -> String { + s.replace('\\', "\\\\").replace('"', "\\\"") +} + +impl Dump<'_> { + /// Object address -> first path, for printing references. + fn paths(&self) -> &HashMap { + self.paths.get_or_init(|| { + let mut m = HashMap::new(); + let _ = self.h5.walk(|it| { + if let (Some(a), None) = (it.addr, it.first_path) { + m.entry(a).or_insert_with(|| it.path.to_string()); + } + }); + m + }) + } + + fn problem(&mut self, out: &mut Out, what: &str, e: &Error) -> std::io::Result<()> { + self.problems += 1; + writeln!(out.e, "h5rs dump: {what}: {e}") + } + + // ------------------------------------------------------------------ + // DDL + // ------------------------------------------------------------------ + + fn ddl(&mut self, out: &mut Out, fname: &str, only: Option<&str>) -> std::io::Result { + writeln!(out.o, "HDF5 \"{}\" {{", quote(fname))?; + if let Some(p) = only { + let h = match self.h5.resolve(p).and_then(|a| self.h5.header(a)) { + Ok(h) => h, + Err(e) => { + writeln!(out.o, "}}")?; + writeln!(out.e, "h5rs dump: {p}: {e}")?; + return Ok(2); + } + }; + if Kind::of(&h) != Kind::Dataset { + writeln!(out.o, "}}")?; + writeln!(out.e, "h5rs dump: {p}: not a dataset")?; + return Ok(2); + } + let full = if p.starts_with('/') { + p.to_string() + } else { + format!("/{p}") + }; + self.ddl_dataset(out, &full, &full, &h, 0)?; + } else { + let root = self.h5.root(); + let mut seen = HashMap::new(); + match self.h5.header(root) { + Ok(h) => self.ddl_group(out, "/", "/", root, &h, 0, &mut seen)?, + Err(e) => { + writeln!(out.o, "GROUP \"/\" {{\n}}")?; + self.problem(out, "/", &e)?; + } + } + } + writeln!(out.o, "}}")?; + Ok(0) + } + + #[allow(clippy::too_many_arguments)] + fn ddl_group( + &mut self, + out: &mut Out, + name: &str, + path: &str, + addr: u64, + h: &ObjectHeader, + ind: usize, + seen: &mut HashMap, + ) -> std::io::Result<()> { + let pad = " ".repeat(ind); + seen.insert(addr, path.to_string()); + writeln!(out.o, "{pad}GROUP \"{}\" {{", quote(name))?; + self.ddl_attributes(out, path, h, ind + 3)?; + let links = match self.h5.links(h) { + Ok(l) => l, + Err(e) => { + self.problem(out, path, &e)?; + Vec::new() + } + }; + // The DDL recursion follows the group nesting; bound it (the walk used + // by the other commands is iterative). + if seen.len() > crate::h5::MAX_OBJECTS || ind > 3 * MAX_DDL_DEPTH { + let e = Error::new("too many objects, or groups nested too deeply; stopped"); + self.problem(out, path, &e)?; + return writeln!(out.o, "{pad}}}"); + } + for l in &links { + self.ddl_link(out, path, l, ind + 3, seen)?; + } + writeln!(out.o, "{pad}}}") + } + + fn ddl_link( + &mut self, + out: &mut Out, + base: &str, + l: &Link, + ind: usize, + seen: &mut HashMap, + ) -> std::io::Result<()> { + let pad = " ".repeat(ind); + let inner = " ".repeat(ind + 3); + let name = quote(&l.name); + let path = join(base, &l.name); + match &l.kind { + LinkKind::Soft(t) => writeln!( + out.o, + "{pad}SOFTLINK \"{name}\" {{\n{inner}LINKTARGET \"{}\"\n{pad}}}", + quote(t) + ), + LinkKind::External { file, path: p } => writeln!( + out.o, + "{pad}EXTERNAL_LINK \"{name}\" {{\n{inner}TARGETFILE \"{}\"\n{inner}TARGETPATH \"{}\"\n{pad}}}", + quote(file), + quote(p) + ), + LinkKind::UserDefined(t) => writeln!( + out.o, + "{pad}USERDEFINED_LINK \"{name}\" {{\n{inner}LINKCLASS {t}\n{pad}}}" + ), + LinkKind::Hard(a) => { + let h = match self.h5.header(*a) { + Ok(h) => h, + Err(e) => { + self.problem(out, &path, &e)?; + return writeln!(out.o, "{pad}UNKNOWN_OBJECT \"{name}\" {{\n{pad}}}"); + } + }; + let kind = Kind::of(&h); + if let Some(first) = seen.get(a) { + let word = match kind { + Kind::Group => "GROUP", + Kind::Dataset => "DATASET", + Kind::Datatype => "DATATYPE", + Kind::Unknown => "OBJECT", + }; + return writeln!( + out.o, + "{pad}{word} \"{name}\" {{\n{inner}HARDLINK \"{}\"\n{pad}}}", + quote(first) + ); + } + match kind { + Kind::Group => self.ddl_group(out, &l.name, &path, *a, &h, ind, seen), + Kind::Dataset => { + seen.insert(*a, path.clone()); + self.ddl_dataset(out, &l.name, &path, &h, ind) + } + Kind::Datatype => { + seen.insert(*a, path.clone()); + match self.h5.datatype(&h) { + Ok(dt) => writeln!( + out.o, + "{pad}DATATYPE \"{name}\" {};", + crate::dtype::ddl(&dt, ind) + ), + Err(e) => { + self.problem(out, &path, &e)?; + writeln!(out.o, "{pad}DATATYPE \"{name}\" ?;") + } + } + } + Kind::Unknown => { + seen.insert(*a, path.clone()); + writeln!(out.o, "{pad}UNKNOWN_OBJECT \"{name}\" {{\n{pad}}}") + } + } + } + } + } + + fn ddl_dataset( + &mut self, + out: &mut Out, + name: &str, + path: &str, + h: &ObjectHeader, + ind: usize, + ) -> std::io::Result<()> { + let pad = " ".repeat(ind); + let inner = " ".repeat(ind + 3); + writeln!(out.o, "{pad}DATASET \"{}\" {{", quote(name))?; + let info = DsInfo::read(self.h5, path, h); + match &info.dt { + Ok(dt) => writeln!(out.o, "{inner}DATATYPE {}", crate::dtype::ddl(dt, ind + 3))?, + Err(e) => self.problem(out, path, e)?, + } + match &info.ds { + Ok(ds) => writeln!(out.o, "{inner}DATASPACE {}", info::dataspace_ddl(ds))?, + Err(e) => self.problem(out, path, e)?, + } + if self.opts.props { + self.ddl_properties(out, path, &info, ind + 3)?; + } + if !self.opts.header_only + && let (Ok(dt), Ok(ds)) = (&info.dt, &info.ds) + { + match self.h5.read_dataset(path, dt, ds) { + Ok(raw) => self.ddl_data(out, dt, ds, &raw, ind + 3)?, + Err(e) => { + writeln!(out.o, "{inner}DATA {{\n{inner}\n{inner}}}")?; + self.problem(out, path, &e)?; + } + } + } + self.ddl_attributes(out, path, h, ind + 3)?; + writeln!(out.o, "{pad}}}") + } + + fn ddl_properties( + &mut self, + out: &mut Out, + path: &str, + info: &DsInfo, + ind: usize, + ) -> std::io::Result<()> { + let pad = " ".repeat(ind); + let inner = " ".repeat(ind + 3); + let Ok(layout) = &info.layout else { + if let Err(e) = &info.layout { + self.problem(out, path, e)?; + } + return Ok(()); + }; + writeln!(out.o, "{pad}STORAGE_LAYOUT {{")?; + match layout { + DataLayout::Chunked { + chunk_dimensions, .. + } => { + let rank = chunk_dimensions.len().saturating_sub(1); + writeln!( + out.o, + "{inner}CHUNKED ( {} )", + chunk_dimensions[..rank] + .iter() + .map(|d| d.to_string()) + .collect::>() + .join(", ") + )?; + writeln!( + out.o, + "{inner}INDEX {}", + info::chunk_index_name(layout).to_uppercase() + )?; + } + DataLayout::Contiguous { address, size } => { + writeln!(out.o, "{inner}CONTIGUOUS")?; + match address { + Some(a) => writeln!(out.o, "{inner}OFFSET {a}")?, + None => writeln!(out.o, "{inner}NOT ALLOCATED")?, + } + writeln!(out.o, "{inner}SIZE {size}")?; + } + DataLayout::Compact { data } => { + writeln!(out.o, "{inner}COMPACT")?; + writeln!(out.o, "{inner}SIZE {}", data.len())?; + } + DataLayout::Virtual { .. } => writeln!(out.o, "{inner}VIRTUAL")?, + } + if let Ok(a) = info::allocated_bytes(self.h5, info) + && matches!(layout, DataLayout::Chunked { .. }) + { + writeln!(out.o, "{inner}SIZE {a}")?; + } + writeln!(out.o, "{pad}}}")?; + writeln!(out.o, "{pad}FILTERS {{")?; + match &info.filters { + Ok(Some(p)) if !p.filters.is_empty() => { + for f in &p.filters { + writeln!( + out.o, + "{inner}{} {{ ID {}; PARAMS {{ {} }} }}", + info::filter_name(f).to_uppercase(), + f.filter_id, + f.client_data + .iter() + .map(|c| c.to_string()) + .collect::>() + .join(" ") + )?; + } + } + Ok(_) => writeln!(out.o, "{inner}NONE")?, + Err(e) => self.problem(out, path, e)?, + } + writeln!(out.o, "{pad}}}") + } + + fn ddl_attributes( + &mut self, + out: &mut Out, + path: &str, + h: &ObjectHeader, + ind: usize, + ) -> std::io::Result<()> { + let (attrs, errs) = match self.h5.attributes(h) { + Ok(x) => x, + Err(e) => return self.problem(out, path, &e), + }; + for e in errs { + self.problem(out, path, &Error::new(format!("attribute: {e}")))?; + } + for a in &attrs { + self.ddl_attribute(out, path, a, ind)?; + } + Ok(()) + } + + fn ddl_attribute( + &mut self, + out: &mut Out, + path: &str, + a: &AttributeMessage, + ind: usize, + ) -> std::io::Result<()> { + let pad = " ".repeat(ind); + let inner = " ".repeat(ind + 3); + writeln!(out.o, "{pad}ATTRIBUTE \"{}\" {{", quote(&a.name))?; + writeln!( + out.o, + "{inner}DATATYPE {}", + crate::dtype::ddl(&a.datatype, ind + 3) + )?; + writeln!( + out.o, + "{inner}DATASPACE {}", + info::dataspace_ddl(&a.dataspace) + )?; + { + match self.checked_attr(a) { + Ok(()) => self.ddl_data(out, &a.datatype, &a.dataspace, &a.raw_data, ind + 3)?, + Err(e) => { + writeln!(out.o, "{inner}DATA {{\n{inner}\n{inner}}}")?; + self.problem(out, &format!("{path} attribute {}", a.name), &e)?; + } + } + } + writeln!(out.o, "{pad}}}") + } + + /// An attribute's value holds as many bytes as its dataspace says. + fn checked_attr(&self, a: &AttributeMessage) -> crate::h5::Result<()> { + let need = crate::h5::byte_len(&a.dataspace, &a.datatype)?; + if need > self.h5.max_bytes { + return Err(Error::new(format!( + "attribute is {need} bytes, over the {} byte limit (--max-bytes)", + self.h5.max_bytes + ))); + } + if (a.raw_data.len() as u64) < need { + return Err(Error::new(format!( + "attribute holds {} bytes, its dataspace needs {need}", + a.raw_data.len() + ))); + } + Ok(()) + } + + /// An h5dump DATA block: elements row by row, each line starting with + /// the index of its first element, wrapped where h5dump wraps (78 columns). + fn ddl_data( + &mut self, + out: &mut Out, + dt: &Datatype, + ds: &Dataspace, + raw: &[u8], + ind: usize, + ) -> std::io::Result<()> { + let pad = " ".repeat(ind); + writeln!(out.o, "{pad}DATA {{")?; + let dims: Vec = match ds.space_type { + DataspaceType::Null => { + return writeln!(out.o, "{pad}}}"); + } + DataspaceType::Scalar => vec![1], + DataspaceType::Simple => ds.dimensions.clone(), + }; + let n = crate::h5::num_elements(ds).unwrap_or(0) as usize; + let dec = Decoder::new(self.h5); + let last = dims.last().copied().unwrap_or(1).max(1) as usize; + let paths = |a: u64| self.paths().get(&a).cloned(); + let mut line = String::new(); + let mut errors = 0usize; + for i in 0..n { + let v = dec.element(dt, raw, i); + if matches!(v, Value::Error(_)) { + errors += 1; + } + let comma = if i + 1 < n { "," } else { "" }; + let esize = dt.type_size() as usize; + let eb = raw.get(i * esize..(i + 1) * esize).unwrap_or_default(); + if let (Value::Compound(_), Datatype::Compound { members, .. }) = (&v, dt) { + // h5dump prints each compound element as a block, one + // member per line. + if !line.is_empty() { + writeln!(out.o, "{line}")?; + line.clear(); + } + writeln!(out.o, "{pad}({}): {{", index_text(i as u64, &dims))?; + for (k, m) in members.iter().enumerate() { + let sep = if k + 1 < members.len() { "," } else { "" }; + let mb = usize::try_from(m.byte_offset) + .ok() + .and_then(|o| eb.get(o..)) + .unwrap_or_default(); + let t = ddl_text(&dec, &m.datatype, mb, &paths, 1); + writeln!(out.o, "{pad} {t}{sep}")?; + } + writeln!(out.o, "{pad} }}{comma}")?; + continue; + } + let t = if matches!(v, Value::Error(_)) { + format!("{}{comma}", value::text(&v, &paths)) + } else { + format!("{}{comma}", ddl_text(&dec, dt, eb, &paths, 0)) + }; + let row_start = i % last == 0; + if row_start || line.len() + 1 + t.len() > 77 { + if !line.is_empty() { + writeln!(out.o, "{line}")?; + } + line = format!("{pad}({}): {t}", index_text(i as u64, &dims)); + } else { + line.push(' '); + line.push_str(&t); + } + } + if !line.is_empty() { + writeln!(out.o, "{line}")?; + } + if errors > 0 { + self.problems += 1; + writeln!(out.e, "h5rs dump: {errors} element(s) could not be decoded")?; + } + writeln!(out.o, "{pad}}}") + } + + // ------------------------------------------------------------------ + // JSON + // ------------------------------------------------------------------ + + fn json(&mut self, out: &mut Out, only: Option<&str>) -> std::io::Result { + let mut groups = Map::new(); + let mut datasets = Map::new(); + let mut datatypes = Map::new(); + let mut doc = Map::new(); + doc.insert("apiVersion".into(), json!("1.1.1")); + if let Some(p) = only { + let full = if p.starts_with('/') { + p.to_string() + } else { + format!("/{p}") + }; + let r = self + .h5 + .resolve(&full) + .and_then(|a| Ok((a, self.h5.header(a)?))); + match r { + Ok((a, h)) if Kind::of(&h) == Kind::Dataset => { + let obj = self.json_dataset(out, &full, &h)?; + datasets.insert(obj_id(Kind::Dataset, a), obj); + } + Ok(_) => { + writeln!(out.e, "h5rs dump: {p}: not a dataset")?; + return Ok(2); + } + Err(e) => { + writeln!(out.e, "h5rs dump: {p}: {e}")?; + return Ok(2); + } + } + } else { + doc.insert("root".into(), json!(obj_id(Kind::Group, self.h5.root()))); + let (items, walk) = self.h5.walk_collect(); + if let Err(e) = walk { + self.problem(out, "/", &e)?; + } + // Aliases: every path an object is reachable by. + let mut aliases: HashMap> = HashMap::new(); + for it in &items { + if let Some(a) = it.addr { + aliases.entry(a).or_default().push(it.path.clone()); + } + } + for it in items { + let path = it.path; + let (Some(addr), None, Some(header)) = (it.addr, it.first_path, it.header) else { + continue; + }; + let h = match header { + Ok(h) => h, + Err(e) => { + self.problem(out, &path, &e)?; + continue; + } + }; + let kind = Kind::of(&h); + let alias = aliases.remove(&addr).unwrap_or_default(); + let mut obj = match kind { + Kind::Dataset => self.json_dataset(out, &path, &h)?, + Kind::Datatype => { + let mut m = Map::new(); + match self.h5.datatype(&h) { + Ok(dt) => { + m.insert("type".into(), crate::dtype::json(&dt)); + } + Err(e) => self.problem(out, &path, &e)?, + } + m.insert("attributes".into(), self.json_attributes(out, &path, &h)?); + J::Object(m) + } + _ => { + let mut m = Map::new(); + m.insert("attributes".into(), self.json_attributes(out, &path, &h)?); + let links = match self.h5.links(&h) { + Ok(l) => l, + Err(e) => { + if kind == Kind::Group { + self.problem(out, &path, &e)?; + } + Vec::new() + } + }; + let mut jl = Vec::new(); + for l in &links { + jl.push(self.json_link(l)); + } + m.insert("links".into(), J::Array(jl)); + J::Object(m) + } + }; + if let J::Object(m) = &mut obj { + m.insert("alias".into(), json!(alias)); + } + let id = obj_id(kind, addr); + match kind { + Kind::Dataset => datasets.insert(id, obj), + Kind::Datatype => datatypes.insert(id, obj), + _ => groups.insert(id, obj), + }; + } + doc.insert("groups".into(), J::Object(groups)); + } + doc.insert("datasets".into(), J::Object(datasets)); + if only.is_none() { + doc.insert("datatypes".into(), J::Object(datatypes)); + } + let text = serde_json::to_string_pretty(&J::Object(doc)).map_err(std::io::Error::other)?; + writeln!(out.o, "{text}")?; + Ok(0) + } + + fn json_link(&self, l: &Link) -> J { + match &l.kind { + LinkKind::Hard(a) => { + let kind = self + .h5 + .header(*a) + .map(|h| Kind::of(&h)) + .unwrap_or(Kind::Unknown); + let coll = match kind { + Kind::Dataset => "datasets", + Kind::Datatype => "datatypes", + _ => "groups", + }; + json!({"class": "H5L_TYPE_HARD", "title": l.name, "collection": coll, + "id": obj_id(kind, *a)}) + } + LinkKind::Soft(t) => json!({"class": "H5L_TYPE_SOFT", "title": l.name, "h5path": t}), + LinkKind::External { file, path } => json!({"class": "H5L_TYPE_EXTERNAL", + "title": l.name, "file": file, "h5path": path}), + LinkKind::UserDefined(t) => json!({"class": "H5L_TYPE_USER_DEFINED", + "title": l.name, "linkClass": t}), + } + } + + fn json_dataset(&mut self, out: &mut Out, path: &str, h: &ObjectHeader) -> std::io::Result { + let info = DsInfo::read(self.h5, path, h); + let mut m = Map::new(); + match &info.dt { + Ok(dt) => { + m.insert("type".into(), crate::dtype::json(dt)); + } + Err(e) => self.problem(out, path, e)?, + } + match &info.ds { + Ok(ds) => { + m.insert("shape".into(), shape_json(ds)); + } + Err(e) => self.problem(out, path, e)?, + } + let mut cp = Map::new(); + if let Ok(l) = &info.layout { + let mut lj = Map::new(); + lj.insert( + "class".into(), + json!(match l { + DataLayout::Compact { .. } => "H5D_COMPACT", + DataLayout::Contiguous { .. } => "H5D_CONTIGUOUS", + DataLayout::Chunked { .. } => "H5D_CHUNKED", + DataLayout::Virtual { .. } => "H5D_VIRTUAL", + }), + ); + if let DataLayout::Chunked { + chunk_dimensions, .. + } = l + { + let rank = chunk_dimensions.len().saturating_sub(1); + lj.insert("dims".into(), json!(chunk_dimensions[..rank])); + } + cp.insert("layout".into(), J::Object(lj)); + } else if let Err(e) = &info.layout { + self.problem(out, path, e)?; + } + if let Ok(Some(p)) = &info.filters { + let fl: Vec = p + .filters + .iter() + .map(|f| { + json!({"id": f.filter_id, "name": info::filter_name(f), + "class": filter_class(f.filter_id), "parameters": f.client_data}) + }) + .collect(); + cp.insert("filters".into(), J::Array(fl)); + } + m.insert("creationProperties".into(), J::Object(cp)); + if !self.opts.header_only + && let (Ok(dt), Ok(ds)) = (&info.dt, &info.ds) + { + // JSON values are built in memory, some 32+ bytes per element: + // hold them to the same budget as the raw data. + let n = crate::h5::num_elements(ds).unwrap_or(u64::MAX); + let read = if n.saturating_mul(JSON_BYTES_PER_ELEMENT) > self.h5.max_bytes { + Err(Error::new(format!( + "{n} elements are too many to hold as JSON within --max-bytes {}", + self.h5.max_bytes + )) + .with_kind(crate::h5::ErrorKind::Limit)) + } else { + self.h5.read_dataset(path, dt, ds) + }; + match read { + Ok(raw) => { + let v = self.json_values(out, dt, ds, &raw)?; + m.insert("value".into(), v); + } + Err(e) => { + m.insert("value_error".into(), json!(e.to_string())); + self.problem(out, path, &e)?; + } + } + } + m.insert("attributes".into(), self.json_attributes(out, path, h)?); + Ok(J::Object(m)) + } + + fn json_attributes( + &mut self, + out: &mut Out, + path: &str, + h: &ObjectHeader, + ) -> std::io::Result { + let (attrs, errs) = match self.h5.attributes(h) { + Ok(x) => x, + Err(e) => { + self.problem(out, path, &e)?; + return Ok(J::Array(Vec::new())); + } + }; + for e in errs { + self.problem(out, path, &Error::new(format!("attribute: {e}")))?; + } + let mut v = Vec::new(); + for a in &attrs { + let mut m = Map::new(); + m.insert("name".into(), json!(a.name)); + m.insert("shape".into(), shape_json(&a.dataspace)); + m.insert("type".into(), crate::dtype::json(&a.datatype)); + { + match self.checked_attr(a) { + Ok(()) => { + let val = self.json_values(out, &a.datatype, &a.dataspace, &a.raw_data)?; + m.insert("value".into(), val); + } + Err(e) => { + m.insert("value_error".into(), json!(e.to_string())); + self.problem(out, &format!("{path} attribute {}", a.name), &e)?; + } + } + } + v.push(J::Object(m)); + } + Ok(J::Array(v)) + } + + fn json_values( + &mut self, + out: &mut Out, + dt: &Datatype, + ds: &Dataspace, + raw: &[u8], + ) -> std::io::Result { + let n = crate::h5::num_elements(ds).unwrap_or(0) as usize; + let dec = Decoder::new(self.h5); + let paths = |a: u64| self.paths().get(&a).cloned(); + let mut flat = Vec::with_capacity(n); + let mut errors = 0usize; + for i in 0..n { + let v = dec.element(dt, raw, i); + if matches!(v, Value::Error(_)) { + errors += 1; + } + flat.push(value::to_json(&v, &paths)); + } + if errors > 0 { + self.problems += 1; + writeln!(out.e, "h5rs dump: {errors} element(s) could not be decoded")?; + } + Ok(match ds.space_type { + DataspaceType::Null => J::Null, + DataspaceType::Scalar => flat.into_iter().next().unwrap_or(J::Null), + DataspaceType::Simple => nest(flat, &ds.dimensions), + }) + } +} + +/// Deepest group nesting the DDL output follows. +const MAX_DDL_DEPTH: usize = 256; + +/// Memory budgeted per element for a dataset's value held as JSON. +const JSON_BYTES_PER_ELEMENT: u64 = 64; + +/// `(i,j,k)` index of flat element `i` of an array with `dims`. +/// One element as h5dump prints it in a DATA block: [`value::text`], except +/// that a null-padded fixed string shows its padding (every byte, NULs as +/// `\000`) at any depth, in compound members and array elements too. +fn ddl_text( + dec: &Decoder, + dt: &Datatype, + b: &[u8], + paths: &dyn Fn(u64) -> Option, + depth: u32, +) -> String { + let err = || value::text(&Value::Error("short element".into()), paths); + if depth > 32 { + return value::text(&dec.decode(dt, b, depth), paths); + } + match dt { + Datatype::String { + padding: StringPadding::NullPad, + size, + .. + } => match b.get(..*size as usize) { + Some(s) => value::quote_bytes(s), + None => err(), + }, + Datatype::Compound { members, .. } => format!( + "{{ {} }}", + members + .iter() + .map( + |m| match usize::try_from(m.byte_offset).ok().and_then(|o| b.get(o..)) { + Some(mb) => ddl_text(dec, &m.datatype, mb, paths, depth + 1), + None => err(), + } + ) + .collect::>() + .join(", ") + ), + Datatype::Array { + base_type, + dimensions, + } => { + let bs = base_type.type_size() as usize; + let n = dimensions + .iter() + .try_fold(1usize, |a, &d| a.checked_mul(d as usize)); + match n.filter(|n| n.checked_mul(bs).is_some_and(|t| t <= b.len())) { + Some(n) => format!( + "[ {} ]", + (0..n) + .map(|k| ddl_text(dec, base_type, &b[k * bs..], paths, depth + 1)) + .collect::>() + .join(", ") + ), + None => value::text(&dec.decode(dt, b, depth), paths), + } + } + _ => value::text(&dec.decode(dt, b, depth), paths), + } +} + +fn index_text(mut i: u64, dims: &[u64]) -> String { + let mut idx = vec![0u64; dims.len()]; + for (k, &d) in dims.iter().enumerate().rev() { + let d = d.max(1); + idx[k] = i % d; + i /= d; + } + idx.iter() + .map(|x| x.to_string()) + .collect::>() + .join(",") +} + +/// Turn a flat row-major list into nested lists of shape `dims`. +fn nest(flat: Vec, dims: &[u64]) -> J { + if dims.len() <= 1 { + return J::Array(flat); + } + let inner: usize = dims[1..].iter().map(|&d| d as usize).product(); + if inner == 0 { + return J::Array((0..dims[0]).map(|_| nest(Vec::new(), &dims[1..])).collect()); + } + let mut it = flat.into_iter(); + let mut outer = Vec::with_capacity(dims[0] as usize); + for _ in 0..dims[0] { + let part: Vec = it.by_ref().take(inner).collect(); + outer.push(nest(part, &dims[1..])); + } + J::Array(outer) +} + +fn shape_json(ds: &Dataspace) -> J { + match ds.space_type { + DataspaceType::Null => json!({"class": "H5S_NULL"}), + DataspaceType::Scalar => json!({"class": "H5S_SCALAR"}), + DataspaceType::Simple => { + let mut m = Map::new(); + m.insert("class".into(), json!("H5S_SIMPLE")); + m.insert("dims".into(), json!(ds.dimensions)); + if let Some(max) = &ds.max_dimensions { + let mx: Vec = max + .iter() + .map(|&d| { + if d == u64::MAX { + json!("H5S_UNLIMITED") + } else { + json!(d) + } + }) + .collect(); + m.insert("maxdims".into(), J::Array(mx)); + } + J::Object(m) + } + } +} + +fn filter_class(id: u16) -> &'static str { + match id { + 1 => "H5Z_FILTER_DEFLATE", + 2 => "H5Z_FILTER_SHUFFLE", + 3 => "H5Z_FILTER_FLETCHER32", + 4 => "H5Z_FILTER_SZIP", + 5 => "H5Z_FILTER_NBIT", + 6 => "H5Z_FILTER_SCALEOFFSET", + _ => "H5Z_FILTER_USER", + } +} + +/// Deterministic object id: kind prefix + header address (hdf5-json uses +/// UUIDs; these are stable for a given file instead). +pub fn obj_id(kind: Kind, addr: u64) -> String { + let p = match kind { + Kind::Dataset => 'd', + Kind::Datatype => 't', + _ => 'g', + }; + format!("{p}-{addr:016x}") +} diff --git a/crates/clawhdf5-tools/src/h5.rs b/crates/clawhdf5-tools/src/h5.rs new file mode 100644 index 0000000..e2e7931 --- /dev/null +++ b/crates/clawhdf5-tools/src/h5.rs @@ -0,0 +1,718 @@ +//! The file model the tools share: an open file, its objects, their links +//! and the messages that describe a dataset, all read through +//! `clawhdf5-format` (and the `clawhdf5` facade for dataset values). +//! +//! Nothing here trusts the file: every size is checked before it is used, +//! and every failure is an [`Error`] carrying the address it happened at. + +use std::cell::RefCell; +use std::collections::HashMap; +use std::path::{Path, PathBuf}; +use std::rc::Rc; + +use clawhdf5::File; +use clawhdf5_format::attribute::{AttributeMessage, extract_attributes_tolerant}; +use clawhdf5_format::attribute_info::AttributeInfoMessage; +use clawhdf5_format::btree_v2::{BTreeV2Header, collect_btree_v2_records}; +use clawhdf5_format::data_layout::DataLayout; +use clawhdf5_format::dataspace::{Dataspace, DataspaceType}; +use clawhdf5_format::datatype::Datatype; +use clawhdf5_format::error::FormatError; +use clawhdf5_format::filter_pipeline::FilterPipeline; +use clawhdf5_format::fractal_heap::FractalHeapHeader; +use clawhdf5_format::global_heap::GlobalHeapCollection; +use clawhdf5_format::group_v1; +use clawhdf5_format::link_info::LinkInfoMessage; +use clawhdf5_format::link_message::{LinkMessage, LinkTarget}; +use clawhdf5_format::message_type::MessageType; +use clawhdf5_format::object_header::ObjectHeader; +use clawhdf5_format::superblock::Superblock; +use clawhdf5_format::symbol_table::SymbolTableMessage; + +/// Largest dataset or attribute (in bytes) the tools will decode by default. +/// A corrupt dataspace can claim far more elements than the file holds; the +/// limit keeps such a file from exhausting memory. `--max-bytes` changes it. +pub const DEFAULT_MAX_BYTES: u64 = 1 << 30; + +/// Upper bound on the objects one traversal visits, so a crafted file with +/// millions of links cannot make a walk run for ever. +pub const MAX_OBJECTS: usize = 1_000_000; + +/// A tool error: what went wrong and, when known, the file address (relative +/// to the superblock, as every HDF5 address is) of the structure involved. +#[derive(Debug, Clone)] +pub struct Error { + pub addr: Option, + pub msg: String, + pub kind: ErrorKind, +} + +/// Whether an error says something is wrong with the file. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum ErrorKind { + /// The file is damaged or not valid HDF5 (as far as clawhdf5 knows). + Corrupt, + /// Valid HDF5 that clawhdf5 cannot decode (a filter it does not + /// implement, external raw data files, ...). + Unsupported, + /// Refused by a limit of the tool (`--max-bytes`). + Limit, +} + +impl Error { + pub fn new(msg: impl Into) -> Self { + Self { + addr: None, + msg: msg.into(), + kind: ErrorKind::Corrupt, + } + } + + pub fn at(addr: u64, msg: impl Into) -> Self { + Self { + addr: Some(addr), + msg: msg.into(), + kind: ErrorKind::Corrupt, + } + } + + pub fn with_kind(mut self, kind: ErrorKind) -> Self { + self.kind = kind; + self + } + + /// The same error with `prefix: ` in front of its message. + pub fn context(mut self, prefix: &str) -> Self { + self.msg = format!("{prefix}: {}", self.msg); + self + } +} + +fn format_error_kind(e: &FormatError) -> ErrorKind { + match e { + FormatError::UnsupportedFilter(_) + | FormatError::ExternalDataFilesUnsupported + | FormatError::ExternalLinkUnsupported { .. } => ErrorKind::Unsupported, + _ => ErrorKind::Corrupt, + } +} + +impl std::fmt::Display for Error { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self.addr { + Some(a) => write!(f, "{} (at address {a:#x})", self.msg), + None => write!(f, "{}", self.msg), + } + } +} + +impl From for Error { + fn from(e: FormatError) -> Self { + Error::new(e.to_string()).with_kind(format_error_kind(&e)) + } +} + +impl From for Error { + fn from(e: clawhdf5::Error) -> Self { + let kind = match &e { + clawhdf5::Error::Format(f) => format_error_kind(f), + _ => ErrorKind::Corrupt, + }; + Error::new(e.to_string()).with_kind(kind) + } +} + +pub type Result = std::result::Result; + +/// Attach an address to a format error. +pub fn fe_at(addr: u64) -> impl Fn(FormatError) -> Error { + move |e| Error::at(addr, e.to_string()) +} + +/// What an object header describes. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Kind { + Group, + Dataset, + Datatype, + Unknown, +} + +impl Kind { + pub fn of(h: &ObjectHeader) -> Kind { + let has = |t: MessageType| h.messages.iter().any(|m| m.msg_type == t); + if has(MessageType::DataLayout) { + Kind::Dataset + } else if has(MessageType::SymbolTable) + || has(MessageType::LinkInfo) + || has(MessageType::Link) + || has(MessageType::GroupInfo) + { + Kind::Group + } else if has(MessageType::Datatype) { + Kind::Datatype + } else { + Kind::Unknown + } + } + + pub fn name(self) -> &'static str { + match self { + Kind::Group => "Group", + Kind::Dataset => "Dataset", + Kind::Datatype => "Type", + Kind::Unknown => "Unknown", + } + } +} + +/// Where a link points. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum LinkKind { + Hard(u64), + Soft(String), + External { + file: String, + path: String, + }, + /// A user-defined link class (type 65-255): its target means something + /// only to the application that registered the class. + UserDefined(u8), +} + +#[derive(Debug, Clone)] +pub struct Link { + pub name: String, + pub kind: LinkKind, +} + +/// An open file. +pub struct H5 { + pub path: PathBuf, + pub file: File, + pub max_bytes: u64, + heaps: RefCell, String>>>, + /// Fractal heaps whose blocks were verified: `None` = sound. + verified_heaps: RefCell>>, +} + +impl H5 { + pub fn open(path: &Path) -> Result
{ + if !path.exists() { + return Err(Error::new(format!("{}: no such file", path.display()))); + } + let file = File::open(path).map_err(|e| { + Error::new(format!( + "{}: not an HDF5 file this tool can open: {e}", + path.display() + )) + })?; + Ok(H5 { + path: path.to_path_buf(), + file, + max_bytes: DEFAULT_MAX_BYTES, + heaps: RefCell::new(HashMap::new()), + verified_heaps: RefCell::new(HashMap::new()), + }) + } + + /// The file's bytes from the superblock on: what every address indexes. + pub fn data(&self) -> &[u8] { + self.file.as_bytes() + } + + pub fn sb(&self) -> &Superblock { + self.file.superblock() + } + + pub fn os(&self) -> u8 { + self.sb().offset_size + } + + pub fn ls(&self) -> u8 { + self.sb().length_size + } + + pub fn root(&self) -> u64 { + self.sb().root_group_address + } + + pub fn header(&self, addr: u64) -> Result { + let off = usize::try_from(addr).map_err(|_| Error::at(addr, "address out of range"))?; + ObjectHeader::parse(self.data(), off, self.os(), self.ls()) + .map_err(|e| Error::at(addr, format!("object header: {e}"))) + } + + /// The payload of the first message of type `t`, resolving a shared + /// message to the message it points at. + pub fn payload(&self, h: &ObjectHeader, t: MessageType) -> Result>> { + match h.messages.iter().find(|m| m.msg_type == t) { + None => Ok(None), + Some(m) => { + clawhdf5_format::shared_message::message_data(self.data(), m, self.os(), self.ls()) + .map(|c| Some(c.into_owned())) + .map_err(|e| Error::new(format!("{t:?} message: {e}"))) + } + } + } + + pub fn datatype(&self, h: &ObjectHeader) -> Result { + let b = self + .payload(h, MessageType::Datatype)? + .ok_or_else(|| Error::new("no datatype message"))?; + Datatype::parse(&b) + .map(|(d, _)| d) + .map_err(|e| Error::new(format!("datatype message: {e}"))) + } + + pub fn dataspace(&self, h: &ObjectHeader) -> Result { + let b = self + .payload(h, MessageType::Dataspace)? + .ok_or_else(|| Error::new("no dataspace message"))?; + Dataspace::parse(&b, self.ls()).map_err(|e| Error::new(format!("dataspace message: {e}"))) + } + + pub fn layout(&self, h: &ObjectHeader) -> Result { + let m = h + .messages + .iter() + .find(|m| m.msg_type == MessageType::DataLayout) + .ok_or_else(|| Error::new("no layout message"))?; + DataLayout::parse(&m.data, self.os(), self.ls()) + .map_err(|e| Error::new(format!("layout message: {e}"))) + } + + pub fn filters(&self, h: &ObjectHeader) -> Result> { + match self.payload(h, MessageType::FilterPipeline)? { + None => Ok(None), + Some(b) => FilterPipeline::parse(&b) + .map(Some) + .map_err(|e| Error::new(format!("filter pipeline message: {e}"))), + } + } + + /// Every attribute that can be read, plus one error per attribute that + /// cannot. + pub fn attributes(&self, h: &ObjectHeader) -> Result<(Vec, Vec)> { + if let Some(m) = h + .messages + .iter() + .find(|m| m.msg_type == MessageType::AttributeInfo) + && let Ok(ai) = AttributeInfoMessage::parse(&m.data, self.os()) + && let Some(fh) = ai.fractal_heap_address + { + self.verified_heap(fh) + .map_err(|e| e.context("dense attribute storage"))?; + } + let (mut attrs, errs) = extract_attributes_tolerant(self.data(), h, self.os(), self.ls()) + .map_err(|e| Error::new(format!("attributes: {e}")))?; + attrs.sort_by(|a, b| a.name.cmp(&b.name)); + Ok((attrs, errs.iter().map(|e| e.to_string()).collect())) + } + + /// Every link of the group whose header is `h`, sorted by name. An + /// object that is not a group has none. + pub fn links(&self, h: &ObjectHeader) -> Result> { + let os = self.os(); + let ls = self.ls(); + let data = self.data(); + let mut out = Vec::new(); + if let Some(m) = h + .messages + .iter() + .find(|m| m.msg_type == MessageType::SymbolTable) + { + let stm = SymbolTableMessage::parse(&m.data, os) + .map_err(|e| Error::new(format!("symbol table message: {e}")))?; + let entries = group_v1::resolve_v1_group_entries(data, &stm, os, ls) + .map_err(|e| Error::at(stm.btree_address, format!("symbol table: {e}")))?; + let has_soft = entries.iter().any(group_v1::is_v1_soft_link); + for e in entries { + if !group_v1::is_v1_soft_link(&e) { + out.push(Link { + name: e.name, + kind: LinkKind::Hard(e.object_header_address), + }); + } + } + if has_soft { + let soft = group_v1::v1_soft_links(data, &stm, os, ls) + .map_err(|e| Error::at(stm.btree_address, format!("soft links: {e}")))?; + for (name, target) in soft { + out.push(Link { + name, + kind: LinkKind::Soft(target), + }); + } + } + } else { + let li = match h + .messages + .iter() + .find(|m| m.msg_type == MessageType::LinkInfo) + { + Some(m) => Some( + LinkInfoMessage::parse(&m.data, os) + .map_err(|e| Error::new(format!("link info message: {e}")))?, + ), + None => None, + }; + for m in h + .messages + .iter() + .filter(|m| m.msg_type == MessageType::Link) + { + out.push(link_from_message(&m.data, os)?); + } + if let Some(li) = li + && let Some(fh) = li.fractal_heap_address + { + for bytes in self.dense_heap_objects(fh, li.btree_name_index_address, 5)? { + out.push(link_from_message(&bytes, os)?); + } + } + } + out.sort_by(|a, b| a.name.cmp(&b.name)); + Ok(out) + } + + /// The objects a dense-storage B-tree (`btree`, of record type + /// `name_type`: 5 for links, 8 for attributes) indexes in the fractal + /// heap at `heap`. + pub fn dense_heap_objects( + &self, + heap: u64, + btree: Option, + name_type: u8, + ) -> Result>> { + self.verified_heap(heap)?; + let data = self.data(); + let os = self.os(); + let ls = self.ls(); + let fh = FractalHeapHeader::parse(data, to_usize(heap)?, os, ls) + .map_err(|e| Error::at(heap, format!("fractal heap header: {e}")))?; + let bt = btree.ok_or_else(|| Error::at(heap, "dense storage without a name index"))?; + let hdr = BTreeV2Header::parse(data, to_usize(bt)?, os, ls) + .map_err(|e| Error::at(bt, format!("v2 B-tree header: {e}")))?; + let recs = collect_btree_v2_records(data, &hdr, os, ls) + .map_err(|e| Error::at(bt, format!("v2 B-tree: {e}")))?; + // Name-index records: hash(4) + heap ID; creation-order ones: order(8) + heap ID. + let skip = if hdr.tree_type == name_type { 4 } else { 8 }; + let idlen = usize::from(fh.heap_id_length); + let mut out = Vec::with_capacity(recs.len()); + for r in &recs { + let id = r + .data + .get(skip..skip + idlen) + .ok_or_else(|| Error::at(bt, "v2 B-tree record shorter than a heap ID"))?; + let obj = fh + .read_managed_object(data, id, os) + .map_err(|e| Error::at(heap, format!("fractal heap object: {e}")))?; + out.push(obj); + } + Ok(out) + } + + /// Verify every block of the fractal heap at `addr` (once per heap): + /// the library reads heap objects without checking the blocks' + /// checksums, which libhdf5 does, so a damaged block would otherwise be + /// read as if it were sound. + pub fn verified_heap(&self, addr: u64) -> Result<()> { + if let Some(r) = self.verified_heaps.borrow().get(&addr) { + return r.clone().map_or(Ok(()), Err); + } + let report = crate::heap_blocks::verify(self, addr); + let n = report.problems.len(); + let r = report.problems.into_iter().next().map(|mut e| { + if n > 1 { + e.msg = format!("{} (and {} more problems in this heap)", e.msg, n - 1); + } + e + }); + self.verified_heaps.borrow_mut().insert(addr, r.clone()); + r.map_or(Ok(()), Err) + } + + /// The global heap object `idx` of the collection at `addr` (cached per + /// collection). + pub fn heap_object(&self, addr: u64, idx: u32) -> Result> { + let coll = { + let mut cache = self.heaps.borrow_mut(); + cache + .entry(addr) + .or_insert_with(|| match usize::try_from(addr) { + Ok(a) => GlobalHeapCollection::parse(self.data(), a, self.ls()) + .map(Rc::new) + .map_err(|e| e.to_string()), + Err(_) => Err("address out of range".into()), + }) + .clone() + .map_err(|e| Error::at(addr, format!("global heap: {e}")))? + }; + let idx16 = u16::try_from(idx) + .map_err(|_| Error::at(addr, format!("global heap object index {idx} out of range")))?; + coll.get_object(idx16) + .map(|o| o.data.clone()) + .ok_or_else(|| Error::at(addr, format!("global heap has no object {idx}"))) + } + + /// The dataspace of the dataset at `path` with a virtual dataset's + /// extent resolved from its sources (as libhdf5 reports it) instead of + /// the stored one. + pub fn resolved_dataspace(&self, path: &str, h: &ObjectHeader) -> Result { + let mut ds = self.dataspace(h)?; + if matches!(self.layout(h), Ok(DataLayout::Virtual { .. })) { + let d = self.file.dataset(path)?; + ds.dimensions = d.shape()?; + if ds.space_type == DataspaceType::Simple + && let Some(m) = &ds.max_dimensions + && m.len() != ds.dimensions.len() + { + ds.max_dimensions = None; + } + } + Ok(ds) + } + + /// Read a dataset's values (raw, in file byte order) through the + /// `clawhdf5` facade, refusing one larger than `max_bytes`. `ds` must be + /// the [resolved](Self::resolved_dataspace) dataspace. + pub fn read_dataset(&self, path: &str, dt: &Datatype, ds: &Dataspace) -> Result> { + let need = byte_len(ds, dt)?; + if need > self.max_bytes { + return Err(Error::new(format!( + "dataset is {need} bytes, over the {} byte limit (--max-bytes)", + self.max_bytes + )) + .with_kind(ErrorKind::Limit)); + } + let d = self.file.dataset(path)?; + let raw = d.read_selection(&clawhdf5::Selection::All)?; + if raw.len() as u64 != need { + return Err(Error::new(format!( + "read {} bytes, expected {need}", + raw.len() + ))); + } + Ok(raw) + } + + /// Walk every object reachable by hard links from the root, depth first + /// in name order, calling `visit` once per link (the root is visited + /// first with an empty link name). Each object is described once; later + /// hard links to it are reported with `first_path` set. + pub fn walk(&self, mut visit: impl FnMut(&WalkItem<'_>)) -> Result<()> { + self.walk_from(self.root(), "/", &mut visit) + } + + pub fn walk_from( + &self, + start: u64, + start_path: &str, + visit: &mut dyn FnMut(&WalkItem<'_>), + ) -> Result<()> { + let mut seen: HashMap = HashMap::new(); + let mut stack: Vec<(u64, String, Option, usize)> = + vec![(start, start_path.to_string(), None, 0)]; + let mut count = 0usize; + while let Some((addr, path, link, depth)) = stack.pop() { + count += 1; + if count > MAX_OBJECTS { + return Err(Error::new(format!( + "more than {MAX_OBJECTS} links; stopped walking" + ))); + } + // Soft, external and user-defined links have no address. + if let Some(Link { kind, .. }) = &link + && !matches!(kind, LinkKind::Hard(_)) + { + visit(&WalkItem { + path: &path, + link: link.as_ref(), + addr: None, + first_path: None, + header: None, + depth, + }); + continue; + } + if let Some(first) = seen.get(&addr) { + visit(&WalkItem { + path: &path, + link: link.as_ref(), + addr: Some(addr), + first_path: Some(first), + header: None, + depth, + }); + continue; + } + seen.insert(addr, path.clone()); + let header = self.header(addr); + visit(&WalkItem { + path: &path, + link: link.as_ref(), + addr: Some(addr), + first_path: None, + header: Some(&header), + depth, + }); + let Ok(h) = &header else { continue }; + if Kind::of(h) != Kind::Group { + continue; + } + // Link errors are the visitor's to report (it sees the header). + let Ok(links) = self.links(h) else { continue }; + let base = if path == "/" { "" } else { path.as_str() }; + for l in links.into_iter().rev() { + let child = format!("{base}/{}", l.name); + let a = match l.kind { + LinkKind::Hard(a) => a, + _ => 0, + }; + stack.push((a, child, Some(l), depth + 1)); + } + } + Ok(()) + } + + /// Resolve an absolute object path to its header address, following + /// soft links like the library does. + pub fn resolve(&self, path: &str) -> Result { + let p = path.trim_matches('/'); + if p.is_empty() { + return Ok(self.root()); + } + clawhdf5_format::group_v2::resolve_path_any(self.data(), self.sb(), p) + .map_err(|e| Error::new(format!("{path}: {e}"))) + } +} + +/// An owned copy of a [`WalkItem`], from [`H5::walk_collect`]. +pub struct Visited { + pub path: String, + pub addr: Option, + pub first_path: Option, + pub header: Option>, +} + +impl H5 { + /// Every step of [`H5::walk`], collected, and the walk's own error. + pub fn walk_collect(&self) -> (Vec, Result<()>) { + let mut items = Vec::new(); + let r = self.walk(|it| { + items.push(Visited { + path: it.path.to_string(), + addr: it.addr, + first_path: it.first_path.map(str::to_string), + header: it.header.cloned(), + }); + }); + (items, r) + } +} + +/// One step of [`H5::walk`]. +pub struct WalkItem<'a> { + pub path: &'a str, + /// The link that led here (`None` for the start object). + pub link: Option<&'a Link>, + /// Header address (`None` for a soft/external/user-defined link). + pub addr: Option, + /// Set when this object was already visited under another path. + pub first_path: Option<&'a str>, + /// The parsed header, for an object seen for the first time. + pub header: Option<&'a Result>, + pub depth: usize, +} + +fn link_from_message(data: &[u8], os: u8) -> Result { + match LinkMessage::parse(data, os) { + Ok(l) => Ok(Link { + name: l.name, + kind: match l.link_target { + LinkTarget::Hard { + object_header_address, + } => LinkKind::Hard(object_header_address), + LinkTarget::Soft { target_path } => LinkKind::Soft(target_path), + LinkTarget::External { + filename, + object_path, + } => LinkKind::External { + file: filename, + path: object_path, + }, + }, + }), + Err(FormatError::InvalidLinkType(t)) if t >= 65 => Ok(Link { + name: user_defined_link_name(data).unwrap_or_else(|| "?".into()), + kind: LinkKind::UserDefined(t), + }), + Err(e) => Err(Error::new(format!("link message: {e}"))), + } +} + +/// The name of a user-defined link, which `LinkMessage::parse` refuses. +fn user_defined_link_name(d: &[u8]) -> Option { + // version(1) flags(1) [type(1)] [corder(8)] [cset(1)] len(1|2|4|8) name + let flags = *d.get(1)?; + let mut p = 2usize; + if flags & 0x08 != 0 { + p += 1; + } + if flags & 0x04 != 0 { + p += 8; + } + if flags & 0x10 != 0 { + p += 1; + } + let w = 1usize << (flags & 0x03); + let mut n = 0usize; + for i in 0..w { + n |= usize::from(*d.get(p + i)?) << (8 * i); + } + p += w; + let name = d.get(p..p.checked_add(n)?)?; + Some(String::from_utf8_lossy(name).into_owned()) +} + +pub fn to_usize(a: u64) -> Result { + usize::try_from(a).map_err(|_| Error::at(a, "address out of range")) +} + +/// Number of elements a dataspace holds (0 for a null dataspace). +pub fn num_elements(ds: &Dataspace) -> Result { + match ds.space_type { + DataspaceType::Null => Ok(0), + DataspaceType::Scalar => Ok(1), + DataspaceType::Simple => ds + .dimensions + .iter() + .try_fold(1u64, |a, &d| a.checked_mul(d)) + .ok_or_else(|| Error::new("dataspace element count overflows")), + } +} + +/// Bytes needed for all of a dataspace's elements of type `dt`. +pub fn byte_len(ds: &Dataspace, dt: &Datatype) -> Result { + num_elements(ds)? + .checked_mul(u64::from(dt.type_size())) + .ok_or_else(|| Error::new("dataset byte size overflows")) +} + +/// Split a `FILE[/object/path]` argument the way h5ls does: the longest +/// prefix that is an existing file is the file. +pub fn split_file_arg(arg: &str) -> (String, Option) { + if Path::new(arg).is_file() { + return (arg.to_string(), None); + } + let mut idx: Vec = arg.match_indices('/').map(|(i, _)| i).collect(); + idx.reverse(); + for i in idx { + let (f, rest) = arg.split_at(i); + if !f.is_empty() && Path::new(f).is_file() { + return (f.to_string(), Some(rest.to_string())); + } + } + (arg.to_string(), None) +} diff --git a/crates/clawhdf5-tools/src/heap_blocks.rs b/crates/clawhdf5-tools/src/heap_blocks.rs new file mode 100644 index 0000000..a1c40cd --- /dev/null +++ b/crates/clawhdf5-tools/src/heap_blocks.rs @@ -0,0 +1,308 @@ +//! Walk every block of a fractal heap and verify it: signatures, the +//! back-pointer to the heap header, each block's heap offset, and the +//! checksums (always present on indirect blocks; on direct blocks when the +//! heap header's flag says so). The library reads only the blocks an object +//! lives in and does not verify block checksums, so `check` does it here. + +use std::collections::HashSet; + +use clawhdf5_format::checksum::jenkins_lookup3; +use clawhdf5_format::fractal_heap::FractalHeapHeader; + +use crate::h5::{Error, H5}; + +/// Heap header flag bit 1: direct blocks carry a checksum. +const FLAG_CHECKSUM_DBLOCKS: u8 = 0x02; +const MAX_DEPTH: u32 = 16; +const MAX_BLOCKS: usize = 1 << 20; + +#[derive(Default, Debug)] +pub struct HeapReport { + pub direct_blocks: usize, + pub indirect_blocks: usize, + /// Blocks whose checksum was verified. + pub checksums: usize, + pub problems: Vec, +} + +struct Walk<'a> { + data: &'a [u8], + heap: u64, + fh: FractalHeapHeader, + checksum_dblocks: bool, + boff_bytes: usize, + os: usize, + ls: usize, + seen: HashSet, + r: HeapReport, +} + +fn le(b: &[u8]) -> u64 { + b.iter() + .take(8) + .enumerate() + .fold(0u64, |a, (i, &x)| a | (u64::from(x) << (8 * i))) +} + +fn undefined(v: u64, os: usize) -> bool { + if os >= 8 { + v == u64::MAX + } else { + v == (1u64 << (8 * os)) - 1 + } +} + +fn log2(v: u64) -> u32 { + 63u32.saturating_sub(v.max(1).leading_zeros()) +} + +/// Verify the fractal heap whose header is at `heap`. The header itself is +/// parsed (and its checksum verified) by the library; an error there is +/// returned as the only problem. +pub fn verify(h5: &H5, heap: u64) -> HeapReport { + let data = h5.data(); + let Ok(off) = usize::try_from(heap) else { + return HeapReport { + problems: vec![Error::at(heap, "fractal heap address out of range")], + ..Default::default() + }; + }; + let fh = match FractalHeapHeader::parse(data, off, h5.os(), h5.ls()) { + Ok(f) => f, + Err(e) => { + return HeapReport { + problems: vec![Error::at(heap, format!("fractal heap header: {e}"))], + ..Default::default() + }; + } + }; + // Flags: signature(4) version(1) heap ID length(2) filter length(2) flags(1). + let flags = data.get(off + 9).copied().unwrap_or(0); + let mut w = Walk { + data, + heap, + checksum_dblocks: flags & FLAG_CHECKSUM_DBLOCKS != 0 && fh.filter_pipeline.is_none(), + boff_bytes: usize::from(fh.max_heap_size).div_ceil(8), + os: usize::from(h5.os()), + ls: usize::from(h5.ls()), + fh, + seen: HashSet::new(), + r: HeapReport::default(), + }; + if w.fh.table_width == 0 || w.fh.starting_block_size == 0 || !w.fh.table_width.is_power_of_two() + { + w.problem(heap, "fractal heap header: invalid doubling table geometry"); + return w.r; + } + let root = w.fh.root_block_address; + if !undefined(root, w.os) { + if w.fh.current_rows_in_root_indirect_block == 0 { + let size = w.fh.starting_block_size; + w.direct(root, size, 0); + } else { + let rows = w.fh.current_rows_in_root_indirect_block; + w.indirect(root, rows, 0, 0); + } + } + w.r +} + +impl Walk<'_> { + fn problem(&mut self, addr: u64, msg: impl Into) { + self.r.problems.push(Error::at(addr, msg)); + } + + fn row_size(&self, row: usize) -> Option { + let s = self.fh.starting_block_size; + if row <= 1 { + Some(s) + } else { + let sh = u32::try_from(row - 1).ok()?; + s.checked_mul(1u64.checked_shl(sh)?) + } + } + + fn max_direct_rows(&self) -> usize { + let ratio = (self.fh.max_direct_block_size / self.fh.starting_block_size).max(1); + log2(ratio) as usize + 2 + } + + fn rows_for_size(&self, size: u64) -> u16 { + let first = log2(self.fh.starting_block_size) + log2(u64::from(self.fh.table_width)); + (log2(size).saturating_sub(first) + 1) as u16 + } + + /// Common block prefix: signature, version, heap header address and + /// block offset. Returns the position after it, or `None` after + /// recording a problem. + fn prefix(&mut self, addr: u64, sig: &[u8; 4], what: &str, heap_offset: u64) -> Option { + if !self.seen.insert(addr) { + self.problem( + addr, + format!("fractal heap {what} block reached twice (cycle)"), + ); + return None; + } + if self.seen.len() > MAX_BLOCKS { + self.problem(self.heap, "fractal heap has too many blocks; stopped"); + return None; + } + let Ok(start) = usize::try_from(addr) else { + self.problem( + addr, + format!("fractal heap {what} block address out of range"), + ); + return None; + }; + let hdr_len = 5 + self.os + self.boff_bytes; + let Some(b) = start + .checked_add(hdr_len) + .and_then(|e| self.data.get(start..e)) + else { + self.problem( + addr, + format!("fractal heap {what} block lies past the end of the file"), + ); + return None; + }; + if &b[..4] != sig { + self.problem(addr, format!("fractal heap {what} block: bad signature")); + return None; + } + if b[4] != 0 { + self.problem(addr, format!("fractal heap {what} block: version {}", b[4])); + return None; + } + let back = le(&b[5..5 + self.os]); + if back != self.heap { + self.problem( + addr, + format!( + "fractal heap {what} block points at heap header {back:#x}, not {:#x}", + self.heap + ), + ); + } + let boff = le(&b[5 + self.os..hdr_len]); + if boff != heap_offset { + self.problem( + addr, + format!("fractal heap {what} block has heap offset {boff}, expected {heap_offset}"), + ); + } + Some(start + hdr_len) + } + + fn direct(&mut self, addr: u64, size: u64, heap_offset: u64) { + let Some(pos) = self.prefix(addr, b"FHDB", "direct", heap_offset) else { + return; + }; + self.r.direct_blocks += 1; + if self.fh.filter_pipeline.is_some() { + return; // stored filtered: its size on disk is not the block size + } + let start = pos - (5 + self.os + self.boff_bytes); + let Some(end) = usize::try_from(size) + .ok() + .and_then(|s| start.checked_add(s)) + else { + self.problem(addr, "fractal heap direct block size out of range"); + return; + }; + let Some(block) = self.data.get(start..end) else { + self.problem( + addr, + "fractal heap direct block extends past the end of the file", + ); + return; + }; + if self.checksum_dblocks { + let Some(stored) = block.get(pos - start..pos - start + 4) else { + self.problem(addr, "fractal heap direct block too small for its checksum"); + return; + }; + let stored = u32::from_le_bytes([stored[0], stored[1], stored[2], stored[3]]); + let mut copy = block.to_vec(); + copy[pos - start..pos - start + 4].fill(0); + let computed = jenkins_lookup3(©); + self.r.checksums += 1; + if computed != stored { + self.problem( + addr, + format!( + "fractal heap direct block: checksum mismatch: stored {stored:#010x}, computed {computed:#010x}" + ), + ); + } + } + } + + fn indirect(&mut self, addr: u64, nrows: u16, heap_offset: u64, depth: u32) { + if depth > MAX_DEPTH { + self.problem(addr, "fractal heap indirect blocks nested too deeply"); + return; + } + let Some(mut pos) = self.prefix(addr, b"FHIB", "indirect", heap_offset) else { + return; + }; + self.r.indirect_blocks += 1; + let start = pos - (5 + self.os + self.boff_bytes); + let width = usize::from(self.fh.table_width); + let filtered = self.fh.filter_pipeline.is_some(); + let direct_rows = self.max_direct_rows(); + let mut children: Vec<(u64, bool, u64, u64)> = Vec::new(); // addr, direct, size/rows, offset + let mut off = heap_offset; + for row in 0..usize::from(nrows) { + let Some(rs) = self.row_size(row) else { + self.problem(addr, "fractal heap row size overflows"); + return; + }; + let direct = row < direct_rows; + for _ in 0..width { + let Some(b) = self.data.get(pos..pos + self.os) else { + self.problem( + addr, + "fractal heap indirect block extends past the end of the file", + ); + return; + }; + let child = le(b); + pos += self.os; + if direct && filtered { + pos += self.ls + 4; + } + if !undefined(child, self.os) { + children.push((child, direct, rs, off)); + } + off = off.saturating_add(rs); + } + } + let Some(stored) = self.data.get(pos..pos + 4) else { + self.problem( + addr, + "fractal heap indirect block extends past the end of the file", + ); + return; + }; + let stored = u32::from_le_bytes([stored[0], stored[1], stored[2], stored[3]]); + let computed = jenkins_lookup3(&self.data[start..pos]); + self.r.checksums += 1; + if computed != stored { + self.problem( + addr, + format!( + "fractal heap indirect block: checksum mismatch: stored {stored:#010x}, computed {computed:#010x}" + ), + ); + return; // its child pointers cannot be trusted + } + for (child, direct, size, off) in children { + if direct { + self.direct(child, size, off); + } else { + let rows = self.rows_for_size(size); + self.indirect(child, rows, off, depth + 1); + } + } + } +} diff --git a/crates/clawhdf5-tools/src/info.rs b/crates/clawhdf5-tools/src/info.rs new file mode 100644 index 0000000..606457b --- /dev/null +++ b/crates/clawhdf5-tools/src/info.rs @@ -0,0 +1,245 @@ +//! Dataset facts shared by `ls`, `dump`, `stat` and `check`: shape text, +//! layout, filters and storage. + +use clawhdf5_format::chunked_read::{ChunkInfo, list_chunks}; +use clawhdf5_format::data_layout::DataLayout; +use clawhdf5_format::dataspace::{Dataspace, DataspaceType}; +use clawhdf5_format::datatype::Datatype; +use clawhdf5_format::filter_pipeline::{FilterDescription, FilterPipeline}; +use clawhdf5_format::message_type::MessageType; +use clawhdf5_format::object_header::ObjectHeader; + +use crate::h5::{Error, H5, Result}; + +/// h5ls's `{10/Inf, 20}` shape text. `always_max` prints `cur/max` for every +/// dimension (h5ls -v). +pub fn shape_text(ds: &Dataspace, always_max: bool) -> String { + match ds.space_type { + DataspaceType::Null => "{NULL}".into(), + DataspaceType::Scalar => "{SCALAR}".into(), + DataspaceType::Simple => { + let dims: Vec = ds + .dimensions + .iter() + .enumerate() + .map(|(i, &d)| { + let m = ds.max_dimensions.as_ref().and_then(|m| m.get(i).copied()); + match m { + Some(u64::MAX) => format!("{d}/Inf"), + Some(m) if m != d || always_max => format!("{d}/{m}"), + None if always_max => format!("{d}/{d}"), + _ => d.to_string(), + } + }) + .collect(); + format!("{{{}}}", dims.join(", ")) + } + } +} + +/// h5dump's `SIMPLE { ( 3, 4 ) / ( 3, H5S_UNLIMITED ) }`. +pub fn dataspace_ddl(ds: &Dataspace) -> String { + match ds.space_type { + DataspaceType::Null => "NULL".into(), + DataspaceType::Scalar => "SCALAR".into(), + DataspaceType::Simple => { + let cur: Vec = ds.dimensions.iter().map(|d| d.to_string()).collect(); + let max: Vec = ds + .dimensions + .iter() + .enumerate() + .map( + |(i, &d)| match ds.max_dimensions.as_ref().and_then(|m| m.get(i).copied()) { + Some(u64::MAX) => "H5S_UNLIMITED".into(), + Some(m) => m.to_string(), + None => d.to_string(), + }, + ) + .collect(); + format!( + "SIMPLE {{ ( {} ) / ( {} ) }}", + cur.join(", "), + max.join(", ") + ) + } + } +} + +pub fn filter_name(f: &FilterDescription) -> String { + let known = match f.filter_id { + 1 => "deflate", + 2 => "shuffle", + 3 => "fletcher32", + 4 => "szip", + 5 => "nbit", + 6 => "scaleoffset", + 307 => "bzip2", + 32000 => "lzf", + 32001 => "blosc", + 32004 => "lz4", + 32008 => "bitshuffle", + 32013 => "zfp", + 32015 => "zstd", + 32026 => "blosc2", + _ => "", + }; + if !known.is_empty() { + return known.into(); + } + match &f.name { + Some(n) if !n.is_empty() => n.clone(), + _ => "user-defined".into(), + } +} + +/// `deflate-1 OPT {4}` as h5ls prints a filter. +pub fn filter_text(f: &FilterDescription) -> String { + let opt = if f.flags & 1 != 0 { " OPT" } else { "" }; + let cd = if f.client_data.is_empty() { + String::new() + } else { + format!( + " {{{}}}", + f.client_data + .iter() + .map(|c| c.to_string()) + .collect::>() + .join(", ") + ) + }; + format!("{}-{}{opt}{cd}", filter_name(f), f.filter_id) +} + +pub fn layout_name(l: &DataLayout) -> &'static str { + match l { + DataLayout::Compact { .. } => "compact", + DataLayout::Contiguous { .. } => "contiguous", + DataLayout::Chunked { .. } => "chunked", + DataLayout::Virtual { .. } => "virtual", + } +} + +/// Chunk index kind of a chunked layout. +pub fn chunk_index_name(l: &DataLayout) -> &'static str { + match l { + DataLayout::Chunked { + version, + chunk_index_type, + .. + } => { + if *version < 4 { + return "v1 B-tree"; + } + match chunk_index_type { + Some(1) => "single chunk", + Some(2) => "implicit", + Some(3) => "fixed array", + Some(4) => "extensible array", + Some(5) => "v2 B-tree", + _ => "unknown", + } + } + _ => "", + } +} + +/// Everything about one dataset that can be learned without reading its +/// values. +pub struct DsInfo { + pub dt: Result, + pub ds: Result, + pub layout: Result, + pub filters: Result>, + pub external: bool, +} + +impl DsInfo { + pub fn read(h5: &H5, path: &str, h: &ObjectHeader) -> DsInfo { + DsInfo { + dt: h5.datatype(h), + ds: h5.resolved_dataspace(path, h), + layout: h5.layout(h), + filters: h5.filters(h), + external: h + .messages + .iter() + .any(|m| m.msg_type == MessageType::ExternalDataFiles), + } + } + + pub fn logical_bytes(&self) -> Option { + let (Ok(dt), Ok(ds)) = (&self.dt, &self.ds) else { + return None; + }; + crate::h5::byte_len(ds, dt).ok() + } +} + +/// Every allocated chunk (empty when none are). Errors only for a corrupt +/// chunk index. +pub fn chunks( + h5: &H5, + layout: &DataLayout, + ds: &Dataspace, + dt: &Datatype, +) -> Result> { + let DataLayout::Chunked { btree_address, .. } = layout else { + return Ok(Vec::new()); + }; + let Some(addr) = *btree_address else { + return Ok(Vec::new()); + }; + list_chunks( + h5.data(), + layout, + ds, + dt.type_size() as usize, + h5.os(), + h5.ls(), + ) + .map(|(c, _)| c) + .map_err(|e| { + Error::at( + addr, + format!("chunk index ({}): {e}", chunk_index_name(layout)), + ) + }) +} + +/// Bytes of raw data the dataset has allocated in the file (what libhdf5's +/// `H5Dget_storage_size` reports). +pub fn allocated_bytes(h5: &H5, info: &DsInfo) -> Result { + let layout = info.layout.as_ref().map_err(Clone::clone)?; + Ok(match layout { + DataLayout::Compact { data } => data.len() as u64, + DataLayout::Contiguous { address, size } => { + if address.is_some() { + *size + } else { + 0 + } + } + DataLayout::Chunked { .. } => { + let dt = info.dt.as_ref().map_err(Clone::clone)?; + let ds = info.ds.as_ref().map_err(Clone::clone)?; + chunks(h5, layout, ds, dt)? + .iter() + .map(|c| u64::from(c.chunk_size)) + .sum() + } + DataLayout::Virtual { .. } => 0, + }) +} + +/// Number of hard links to an object, as its header records it. +pub fn link_count(h: &ObjectHeader) -> u64 { + if let Some(rc) = h.reference_count { + return u64::from(rc); + } + h.messages + .iter() + .find(|m| m.msg_type == MessageType::ObjectReferenceCount) + .and_then(|m| m.data.get(1..5)) + .map(|b| u64::from(u32::from_le_bytes([b[0], b[1], b[2], b[3]]))) + .unwrap_or(1) +} diff --git a/crates/clawhdf5-tools/src/lib.rs b/crates/clawhdf5-tools/src/lib.rs new file mode 100644 index 0000000..693ea9a --- /dev/null +++ b/crates/clawhdf5-tools/src/lib.rs @@ -0,0 +1,63 @@ +//! `h5rs`: HDF5 command-line tools built on clawhdf5 alone, no libhdf5. +//! +//! The binary's subcommands (`ls`, `dump`, `stat`, `diff`, `check`) live in +//! the modules below; [`run`] dispatches to them. See the crate README for +//! the command reference. + +pub mod check; +pub mod cli; +pub mod diff; +pub mod dtype; +pub mod dump; +pub mod h5; +pub mod heap_blocks; +pub mod info; +pub mod ls; +pub mod stat; +pub mod value; + +use cli::{Args, Out}; + +pub const USAGE: &str = "\ +h5rs: HDF5 tools in pure Rust (clawhdf5, no libhdf5) + +usage: h5rs [options] ... + +commands: + ls list objects (like h5ls) + dump print a file's structure and values as DDL or JSON (like h5dump) + stat object, layout, filter and storage statistics (like h5stat) + diff compare two files (like h5diff) + check validate a file's structure and checksums + +Run `h5rs --help` for a command's options."; + +/// Run `h5rs` with `argv` (without the program name). Returns the exit +/// status. +pub fn run(argv: Vec, out: &mut Out) -> std::io::Result { + let mut it = argv.into_iter(); + let Some(cmd) = it.next() else { + writeln!(out.e, "{USAGE}")?; + return Ok(2); + }; + let rest: Vec = it.collect(); + match cmd.as_str() { + "ls" => ls::run(&mut Args::new("ls", rest), out), + "dump" => dump::run(&mut Args::new("dump", rest), out), + "stat" => stat::run(&mut Args::new("stat", rest), out), + "diff" => diff::run(&mut Args::new("diff", rest), out), + "check" => check::run(&mut Args::new("check", rest), out), + "-h" | "--help" | "help" => { + writeln!(out.o, "{USAGE}")?; + Ok(0) + } + "-V" | "--version" => { + writeln!(out.o, "h5rs {}", env!("CARGO_PKG_VERSION"))?; + Ok(0) + } + other => { + writeln!(out.e, "h5rs: unknown command {other:?}\n\n{USAGE}")?; + Ok(2) + } + } +} diff --git a/crates/clawhdf5-tools/src/ls.rs b/crates/clawhdf5-tools/src/ls.rs new file mode 100644 index 0000000..ef966f6 --- /dev/null +++ b/crates/clawhdf5-tools/src/ls.rs @@ -0,0 +1,385 @@ +//! `h5rs ls`: list a file's objects like h5ls. + +use std::collections::HashMap; + +use clawhdf5_format::attribute::AttributeMessage; +use clawhdf5_format::dataspace::DataspaceType; +use clawhdf5_format::object_header::ObjectHeader; + +use crate::cli::{Args, Out}; +use crate::h5::{H5, Kind, Link, LinkKind, split_file_arg}; +use crate::info::{self, DsInfo}; +use crate::value::{self, Decoder}; + +pub const USAGE: &str = "\ +usage: h5rs ls [-r] [-v] [--max-bytes N] FILE[/OBJECT] + +List the objects in FILE (or in the group OBJECT), one per line, like h5ls: +name, kind, shape and, for datasets, the datatype. + + -r, --recursive list every object below, with full paths + -v, --verbose also print address, link count, layout, chunking, + storage, filters, datatype and attributes + --max-bytes N largest attribute value decoded for -v (default 1 GiB) + +Exit status: 0 listed, 1 some object could not be described, 2 error."; + +struct Ls<'a> { + h5: &'a H5, + verbose: bool, + kinds: HashMap, + problems: usize, +} + +pub fn run(args: &mut Args, out: &mut Out) -> std::io::Result { + let mut recursive = false; + let mut verbose = false; + let mut max_bytes = None; + let mut target = None; + while let Some(a) = args.next() { + match a.as_str() { + "-r" | "--recursive" => recursive = true, + "-v" | "--verbose" => verbose = true, + "-rv" | "-vr" => { + recursive = true; + verbose = true; + } + "--max-bytes" => match args.number() { + Some(n) => max_bytes = Some(n), + None => return args.usage_error(out, "--max-bytes needs a number", USAGE), + }, + "-h" | "--help" => { + writeln!(out.o, "{USAGE}")?; + return Ok(0); + } + s if s.starts_with('-') && s.len() > 1 => { + return args.usage_error(out, &format!("unknown option {a}"), USAGE); + } + _ if target.is_none() => target = Some(a), + _ => return args.usage_error(out, &format!("unexpected argument {a}"), USAGE), + } + } + let Some(target) = target else { + return args.usage_error(out, "missing FILE", USAGE); + }; + let (file, obj) = split_file_arg(&target); + let mut h5 = match H5::open(std::path::Path::new(&file)) { + Ok(h) => h, + Err(e) => { + writeln!(out.e, "h5rs ls: {e}")?; + return Ok(2); + } + }; + if let Some(m) = max_bytes { + h5.max_bytes = m; + } + let mut ls = Ls { + h5: &h5, + verbose, + kinds: HashMap::new(), + problems: 0, + }; + let obj_path = obj.unwrap_or_else(|| "/".into()); + let addr = match h5.resolve(&obj_path) { + Ok(a) => a, + Err(e) => { + writeln!(out.e, "h5rs ls: {obj_path}: not found: {e}")?; + return Ok(2); + } + }; + let header = match h5.header(addr) { + Ok(h) => h, + Err(e) => { + writeln!(out.e, "h5rs ls: {obj_path}: {e}")?; + return Ok(2); + } + }; + let is_root = obj_path.trim_matches('/').is_empty(); + if Kind::of(&header) != Kind::Group && !is_root { + // h5ls names a single object by its base name, or with -r by its + // path as given (without the leading slash). + let name = if recursive { + obj_path.trim_start_matches('/').to_string() + } else { + obj_path.rsplit('/').next().unwrap_or(&obj_path).to_string() + }; + ls.object_line(out, &name, &obj_path, addr, &Ok(header))?; + } else if recursive { + let start = if is_root { "/" } else { "" }; + let mut io_err = None; + let res = h5.walk_from(addr, start, &mut |item| { + if io_err.is_some() || (item.link.is_none() && !is_root) { + return; // listing a group's contents, not the group + } + let path = if item.path.is_empty() { "/" } else { item.path }; + let full = if is_root { + path.to_string() + } else { + format!("{}{path}", obj_path.trim_end_matches('/')) + }; + if let Err(e) = ls.item( + out, + path, + &full, + item.link, + item.addr, + item.first_path, + item.header, + ) { + io_err = Some(e); + } + }); + if let Some(e) = io_err { + return Err(e); + } + if let Err(e) = res { + writeln!(out.e, "h5rs ls: {e}")?; + ls.problems += 1; + } + } else { + match h5.links(&header) { + Ok(links) => { + for l in &links { + let (a, hdr) = match l.kind { + LinkKind::Hard(a) => (Some(a), Some(h5.header(a))), + _ => (None, None), + }; + let full = format!("{}/{}", obj_path.trim_end_matches('/'), l.name); + ls.item(out, &l.name, &full, Some(l), a, None, hdr.as_ref())?; + } + } + Err(e) => { + writeln!(out.e, "h5rs ls: {obj_path}: cannot list group: {e}")?; + ls.problems += 1; + } + } + } + Ok(if ls.problems > 0 { 1 } else { 0 }) +} + +impl Ls<'_> { + #[allow(clippy::too_many_arguments)] + fn item( + &mut self, + out: &mut Out, + name: &str, + full: &str, + link: Option<&Link>, + addr: Option, + first: Option<&str>, + header: Option<&crate::h5::Result>, + ) -> std::io::Result<()> { + if let Some(l) = link { + match &l.kind { + LinkKind::Soft(t) => return writeln!(out.o, "{name:<24} Soft Link {{{t}}}"), + LinkKind::External { file, path } => { + return writeln!(out.o, "{name:<24} External Link {{{file}/{path}}}"); + } + LinkKind::UserDefined(t) => { + return writeln!(out.o, "{name:<24} User-defined link (type {t})"); + } + LinkKind::Hard(_) => {} + } + } + let Some(addr) = addr else { return Ok(()) }; + if let Some(first) = first { + let k = self.kinds.get(&addr).copied().unwrap_or(Kind::Unknown); + return writeln!(out.o, "{name:<24} {}, same as {first}", k.name()); + } + match header { + Some(h) => self.object_line(out, name, full, addr, h), + None => Ok(()), + } + } + + fn object_line( + &mut self, + out: &mut Out, + name: &str, + full: &str, + addr: u64, + header: &crate::h5::Result, + ) -> std::io::Result<()> { + let h = match header { + Ok(h) => h, + Err(e) => { + self.problems += 1; + writeln!(out.o, "{name:<24} ** error **")?; + return writeln!(out.e, "h5rs ls: {name}: {e}"); + } + }; + let kind = Kind::of(h); + self.kinds.insert(addr, kind); + match kind { + Kind::Dataset => { + let info = DsInfo::read(self.h5, full, h); + let shape = match &info.ds { + Ok(ds) => info::shape_text(ds, self.verbose), + Err(_) => "{?}".into(), + }; + if self.verbose { + writeln!(out.o, "{name:<24} Dataset {shape}")?; + } else { + let t = match &info.dt { + Ok(dt) => crate::dtype::short(dt), + Err(_) => "?".into(), + }; + writeln!(out.o, "{name:<24} Dataset {shape} {t}")?; + } + for e in [ + info.dt.as_ref().err(), + info.ds.as_ref().err(), + info.layout.as_ref().err(), + info.filters.as_ref().err(), + ] + .into_iter() + .flatten() + { + self.problems += 1; + writeln!(out.e, "h5rs ls: {name}: {e}")?; + } + if self.verbose { + self.dataset_details(out, addr, h, &info)?; + } + } + k => { + writeln!(out.o, "{name:<24} {}", k.name())?; + if self.verbose { + writeln!(out.o, " Address: {addr}")?; + writeln!(out.o, " Links: {}", info::link_count(h))?; + if k == Kind::Datatype + && let Ok(dt) = self.h5.datatype(h) + { + writeln!(out.o, " Type: {}", crate::dtype::long(&dt))?; + } + self.attributes(out, name, h)?; + } + } + } + Ok(()) + } + + fn dataset_details( + &mut self, + out: &mut Out, + addr: u64, + h: &ObjectHeader, + info: &DsInfo, + ) -> std::io::Result<()> { + writeln!(out.o, " Address: {addr}")?; + writeln!(out.o, " Links: {}", info::link_count(h))?; + if let Ok(l) = &info.layout { + let idx = info::chunk_index_name(l); + if idx.is_empty() { + writeln!(out.o, " Layout: {}", info::layout_name(l))?; + } else { + writeln!( + out.o, + " Layout: {} ({idx} index)", + info::layout_name(l) + )?; + } + if let clawhdf5_format::data_layout::DataLayout::Chunked { + chunk_dimensions, .. + } = l + { + let rank = chunk_dimensions.len().saturating_sub(1); + let dims = &chunk_dimensions[..rank]; + let esize = info + .dt + .as_ref() + .map(|d| u64::from(d.type_size())) + .unwrap_or(0); + let bytes = dims + .iter() + .try_fold(esize, |a, &d| a.checked_mul(u64::from(d))) + .map(|b| b.to_string()) + .unwrap_or_else(|| "?".into()); + writeln!( + out.o, + " Chunks: {{{}}} {bytes} bytes", + dims.iter() + .map(|d| d.to_string()) + .collect::>() + .join(", ") + )?; + } + } + if info.external { + writeln!(out.o, " External: raw data in external files")?; + } + let logical = info.logical_bytes(); + match (logical, info::allocated_bytes(self.h5, info)) { + (Some(l), Ok(a)) => { + if a > 0 { + writeln!( + out.o, + " Storage: {l} logical bytes, {a} allocated bytes, {:.2}% utilization", + l as f64 * 100.0 / a as f64 + )?; + } else { + writeln!(out.o, " Storage: {l} logical bytes, 0 allocated bytes")?; + } + } + (_, Err(e)) => { + self.problems += 1; + writeln!(out.e, "h5rs ls: {e}")?; + } + _ => {} + } + if let Ok(Some(p)) = &info.filters { + for (i, f) in p.filters.iter().enumerate() { + writeln!(out.o, " Filter-{i}: {}", info::filter_text(f))?; + } + } + if let Ok(dt) = &info.dt { + writeln!(out.o, " Type: {}", crate::dtype::long(dt))?; + } + self.attributes(out, "", h) + } + + fn attributes(&mut self, out: &mut Out, name: &str, h: &ObjectHeader) -> std::io::Result<()> { + let (attrs, errs) = match self.h5.attributes(h) { + Ok(x) => x, + Err(e) => { + self.problems += 1; + return writeln!(out.e, "h5rs ls: {name}: {e}"); + } + }; + for e in errs { + self.problems += 1; + writeln!(out.e, "h5rs ls: {name}: attribute: {e}")?; + } + for a in &attrs { + self.attribute(out, a)?; + } + Ok(()) + } + + fn attribute(&mut self, out: &mut Out, a: &AttributeMessage) -> std::io::Result<()> { + let shape = match a.dataspace.space_type { + DataspaceType::Scalar => "scalar".to_string(), + DataspaceType::Null => "null".to_string(), + DataspaceType::Simple => info::shape_text(&a.dataspace, false), + }; + writeln!(out.o, " Attribute: {} {shape}", a.name)?; + writeln!( + out.o, + " Type: {}", + crate::dtype::long(&a.datatype) + )?; + let n = crate::h5::num_elements(&a.dataspace).unwrap_or(0); + if n == 0 { + return Ok(()); + } + let dec = Decoder::new(self.h5); + let shown = n.min(8) as usize; + let mut vals = Vec::with_capacity(shown); + for i in 0..shown { + let v = dec.element(&a.datatype, &a.raw_data, i); + vals.push(value::text(&v, &|_| None)); + } + let more = if n > 8 { ", ..." } else { "" }; + writeln!(out.o, " Data: {}{more}", vals.join(", ")) + } +} diff --git a/crates/clawhdf5-tools/src/main.rs b/crates/clawhdf5-tools/src/main.rs new file mode 100644 index 0000000..4e6d431 --- /dev/null +++ b/crates/clawhdf5-tools/src/main.rs @@ -0,0 +1,58 @@ +//! The `h5rs` binary. A panic anywhere is a bug: it is caught, reported as +//! an internal error and turned into exit status 3, never a crash. + +use std::io::Write; +use std::panic::{self, AssertUnwindSafe}; + +use clawhdf5_tools::cli::Out; + +/// Exit status for an internal error (a caught panic). +const INTERNAL_ERROR: i32 = 3; + +fn main() { + panic::set_hook(Box::new(|info| { + let msg = if let Some(s) = info.payload().downcast_ref::<&str>() { + (*s).to_string() + } else if let Some(s) = info.payload().downcast_ref::() { + s.clone() + } else { + "unknown panic".into() + }; + let at = info + .location() + .map(|l| format!(" at {}:{}", l.file(), l.line())) + .unwrap_or_default(); + eprintln!("h5rs: internal error (this is a bug; please report it): {msg}{at}"); + })); + let argv: Vec = std::env::args().skip(1).collect(); + let stdout = std::io::stdout(); + let mut o = std::io::BufWriter::new(stdout.lock()); + let mut e = std::io::stderr(); + let result = panic::catch_unwind(AssertUnwindSafe(|| { + let mut out = Out { + o: &mut o, + e: &mut e, + }; + clawhdf5_tools::run(argv, &mut out) + })); + let code = match result { + Ok(Ok(code)) => match o.flush() { + Ok(()) => code, + Err(err) if err.kind() == std::io::ErrorKind::BrokenPipe => code, + Err(err) => { + eprintln!("h5rs: {err}"); + 2 + } + }, + Ok(Err(err)) if err.kind() == std::io::ErrorKind::BrokenPipe => 0, + Ok(Err(err)) => { + eprintln!("h5rs: {err}"); + 2 + } + Err(_) => { + let _ = o.flush(); + INTERNAL_ERROR + } + }; + std::process::exit(code); +} diff --git a/crates/clawhdf5-tools/src/stat.rs b/crates/clawhdf5-tools/src/stat.rs new file mode 100644 index 0000000..c7e44da --- /dev/null +++ b/crates/clawhdf5-tools/src/stat.rs @@ -0,0 +1,305 @@ +//! `h5rs stat`: object, layout, filter and storage statistics, laid out like +//! h5stat's report. + +use std::collections::BTreeMap; + +use clawhdf5_format::data_layout::DataLayout; +use clawhdf5_format::dataspace::DataspaceType; + +use crate::cli::{Args, Out}; +use crate::h5::{H5, Kind, LinkKind}; +use crate::info::{self, DsInfo}; + +pub const USAGE: &str = "\ +usage: h5rs stat FILE + +Print statistics for FILE, in the layout of h5stat's report: object counts, +links, dataset ranks, layouts, filters, attribute counts, raw-data size and +a file-space summary. Metadata space is not broken down by structure (h5stat +does); it is reported as one figure together with free space. + +Exit status: 0 printed, 1 some object could not be read (counted as +\"unreadable\"), 2 error."; + +#[derive(Default)] +struct Stats { + groups: u64, + datasets: u64, + datatypes: u64, + other: u64, + unreadable: u64, + links: u64, + max_links_to_object: u64, + max_objects_in_group: u64, + group_sizes: BTreeMap, + ranks: BTreeMap, + max_1d: u64, + layout: [u64; 4], + external: u64, + no_filter: u64, + filter_counts: BTreeMap<&'static str, u64>, + raw_data: u64, + raw_errors: u64, + attr_objects: u64, + max_attrs: u64, + attr_counts: BTreeMap, +} + +const FILTER_KINDS: [&str; 8] = [ + "GZIP", + "SHUFFLE", + "FLETCHER32", + "SZIP", + "NBIT", + "SCALEOFFSET", + "USER-DEFINED", + "", +]; + +fn filter_kind(id: u16) -> &'static str { + match id { + 1 => "GZIP", + 2 => "SHUFFLE", + 3 => "FLETCHER32", + 4 => "SZIP", + 5 => "NBIT", + 6 => "SCALEOFFSET", + _ => "USER-DEFINED", + } +} + +pub fn run(args: &mut Args, out: &mut Out) -> std::io::Result { + let mut file = None; + while let Some(a) = args.next() { + match a.as_str() { + "-h" | "--help" => { + writeln!(out.o, "{USAGE}")?; + return Ok(0); + } + s if s.starts_with('-') && s.len() > 1 => { + return args.usage_error(out, &format!("unknown option {a}"), USAGE); + } + _ if file.is_none() => file = Some(a), + _ => return args.usage_error(out, &format!("unexpected argument {a}"), USAGE), + } + } + let Some(file) = file else { + return args.usage_error(out, "missing FILE", USAGE); + }; + let h5 = match H5::open(std::path::Path::new(&file)) { + Ok(h) => h, + Err(e) => { + writeln!(out.e, "h5rs stat: {e}")?; + return Ok(2); + } + }; + let mut s = Stats::default(); + let mut errors: Vec = Vec::new(); + let walk = h5.walk(|it| { + if let Some(l) = it.link + && !matches!(l.kind, LinkKind::Hard(_)) + { + s.links += 1; + return; + } + if it.first_path.is_some() { + return; + } + let Some(Ok(h)) = it.header else { + if let Some(Err(e)) = it.header { + s.unreadable += 1; + errors.push(format!("{}: {e}", it.path)); + } + return; + }; + s.max_links_to_object = s.max_links_to_object.max(info::link_count(h)); + match h5.attributes(h) { + Ok((a, errs)) => { + let n = a.len() + errs.len(); + if n > 0 { + s.attr_objects += 1; + s.max_attrs = s.max_attrs.max(n as u64); + *s.attr_counts.entry(n).or_default() += 1; + } + for e in errs { + errors.push(format!("{}: attribute: {e}", it.path)); + } + } + Err(e) => errors.push(format!("{}: {e}", it.path)), + } + match Kind::of(h) { + Kind::Group => { + s.groups += 1; + match h5.links(h) { + Ok(l) => { + s.max_objects_in_group = s.max_objects_in_group.max(l.len() as u64); + *s.group_sizes.entry(l.len()).or_default() += 1; + } + Err(e) => { + s.unreadable += 1; + errors.push(format!("{}: {e}", it.path)); + } + } + } + Kind::Datatype => s.datatypes += 1, + Kind::Unknown => { + // The root group of a file with no links at all has no + // group messages; count it as a group. + if it.link.is_none() { + s.groups += 1; + *s.group_sizes.entry(0).or_default() += 1; + } else { + s.other += 1; + } + } + Kind::Dataset => { + s.datasets += 1; + let info = DsInfo::read(&h5, it.path, h); + if let Ok(ds) = &info.ds { + let rank = match ds.space_type { + DataspaceType::Simple => ds.dimensions.len(), + _ => 0, + }; + *s.ranks.entry(rank).or_default() += 1; + if rank == 1 { + s.max_1d = s.max_1d.max(ds.dimensions[0]); + } + } + match &info.layout { + Ok(l) => { + let i = match l { + DataLayout::Compact { .. } => 0, + DataLayout::Contiguous { .. } => 1, + DataLayout::Chunked { .. } => 2, + DataLayout::Virtual { .. } => 3, + }; + s.layout[i] += 1; + } + Err(e) => errors.push(format!("{}: {e}", it.path)), + } + if info.external { + s.external += 1; + } + match &info.filters { + Ok(Some(p)) if !p.filters.is_empty() => { + let mut kinds: Vec<&'static str> = + p.filters.iter().map(|f| filter_kind(f.filter_id)).collect(); + kinds.sort_unstable(); + kinds.dedup(); + for k in kinds { + *s.filter_counts.entry(k).or_default() += 1; + } + } + Ok(_) => s.no_filter += 1, + Err(e) => errors.push(format!("{}: {e}", it.path)), + } + match info::allocated_bytes(&h5, &info) { + Ok(b) => s.raw_data = s.raw_data.saturating_add(b), + Err(e) => { + s.raw_errors += 1; + errors.push(format!("{}: storage size: {e}", it.path)); + } + } + } + } + }); + if let Err(e) = walk { + errors.push(e.to_string()); + } + report(&h5, &file, &s, out)?; + for e in &errors { + writeln!(out.e, "h5rs stat: {e}")?; + } + Ok(if errors.is_empty() { 0 } else { 1 }) +} + +fn report(h5: &H5, file: &str, s: &Stats, out: &mut Out) -> std::io::Result<()> { + let o = &mut out.o; + writeln!(o, "Filename: {file}")?; + writeln!(o, "File information")?; + writeln!(o, "\t# of unique groups: {}", s.groups)?; + writeln!(o, "\t# of unique datasets: {}", s.datasets)?; + writeln!(o, "\t# of unique named datatypes: {}", s.datatypes)?; + writeln!(o, "\t# of unique links: {}", s.links)?; + writeln!(o, "\t# of unique other: {}", s.other)?; + if s.unreadable > 0 { + writeln!(o, "\t# of unreadable objects: {}", s.unreadable)?; + } + writeln!(o, "\tMax. # of links to object: {}", s.max_links_to_object)?; + writeln!( + o, + "\tMax. # of objects in group: {}", + s.max_objects_in_group + )?; + let sb = h5.sb(); + writeln!(o, "Superblock:")?; + writeln!(o, "\tVersion: {}", sb.version)?; + writeln!(o, "\tSize of offsets: {} bytes", sb.offset_size)?; + writeln!(o, "\tSize of lengths: {} bytes", sb.length_size)?; + writeln!(o, "\tUser block: {} bytes", h5.file.user_block_size())?; + writeln!(o, "Group bins:")?; + for (n, c) in &s.group_sizes { + writeln!(o, "\t# of groups with {n} link(s): {c}")?; + } + writeln!(o, "\tTotal # of groups: {}", s.groups)?; + writeln!(o, "Dataset dimension information:")?; + writeln!( + o, + "\tMax. rank of datasets: {}", + s.ranks.keys().next_back().copied().unwrap_or(0) + )?; + writeln!(o, "\tDataset ranks:")?; + for (r, c) in &s.ranks { + writeln!(o, "\t\t# of dataset with rank {r}: {c}")?; + } + writeln!(o, "1-D Dataset information:")?; + writeln!(o, "\tMax. dimension size of 1-D datasets: {}", s.max_1d)?; + writeln!(o, "Dataset storage information:")?; + writeln!(o, "\tTotal raw data size: {}", s.raw_data)?; + if s.raw_errors > 0 { + writeln!( + o, + "\tDatasets whose storage size could not be read: {}", + s.raw_errors + )?; + } + writeln!(o, "Dataset layout information:")?; + for (i, n) in ["COMPACT", "CONTIG", "CHUNKED", "VIRTUAL"] + .iter() + .enumerate() + { + writeln!(o, "\tDataset layout counts[{n}]: {}", s.layout[i])?; + } + writeln!(o, "\tDatasets with external raw data: {}", s.external)?; + writeln!(o, "Dataset filters information:")?; + writeln!(o, "\tNumber of datasets with:")?; + writeln!(o, "\t\tNO filter: {}", s.no_filter)?; + for k in FILTER_KINDS.iter().filter(|k| !k.is_empty()) { + writeln!( + o, + "\t\t{k} filter: {}", + s.filter_counts.get(k).copied().unwrap_or(0) + )?; + } + writeln!(o, "Attribute information:")?; + for (n, c) in &s.attr_counts { + writeln!(o, "\t# of objects with {n} attribute(s): {c}")?; + } + writeln!( + o, + "\tTotal # of objects with attributes: {}", + s.attr_objects + )?; + writeln!(o, "\tMax. # of attributes to objects: {}", s.max_attrs)?; + let total = std::fs::metadata(&h5.path).map(|m| m.len()).unwrap_or(0); + let ub = h5.file.user_block_size(); + writeln!(o, "Summary of file space information:")?; + writeln!(o, " User block: {ub} bytes")?; + writeln!(o, " Raw data: {} bytes", s.raw_data)?; + writeln!( + o, + " Metadata and free space: {} bytes", + total.saturating_sub(s.raw_data).saturating_sub(ub) + )?; + writeln!(o, "Total space: {total} bytes") +} diff --git a/crates/clawhdf5-tools/src/value.rs b/crates/clawhdf5-tools/src/value.rs new file mode 100644 index 0000000..8160574 --- /dev/null +++ b/crates/clawhdf5-tools/src/value.rs @@ -0,0 +1,483 @@ +//! Decoding one element of any datatype into a [`Value`], and printing it. +//! +//! Decoding never panics: a short buffer, an unknown byte order or a +//! dangling heap reference becomes [`Value::Error`]. + +use clawhdf5_format::datatype::{Datatype, DatatypeByteOrder, ReferenceType, StringPadding}; +use serde_json::Value as J; + +use crate::dtype; +use crate::h5::H5; + +#[derive(Debug, Clone, PartialEq)] +pub enum Value { + Int(i128), + /// A float and the width (in bits) it was stored with, so it prints at + /// its own precision. + Float(f64, u8), + Str(String), + /// Opaque, bitfield, time and oversized integers. + Bytes(Vec), + /// An enum member (name, when the value matches one) and its value. + Enum(Option, i128), + Compound(Vec<(String, Value)>), + Array(Vec), + /// A variable-length sequence. + Seq(Vec), + /// A reference: the referenced object's address (`None` = null). + Ref(Option), + /// A region or attribute reference, kept as its bytes. + OtherRef(Vec), + Error(String), +} + +pub fn hex(b: &[u8]) -> String { + let mut s = String::from("0x"); + for x in b { + s.push_str(&format!("{x:02x}")); + } + s +} + +/// Element bytes as an unsigned integer (at most 16 bytes). +fn bits(b: &[u8], order: &DatatypeByteOrder) -> Option { + if b.len() > 16 { + return None; + } + let mut v = 0u128; + match order { + DatatypeByteOrder::LittleEndian => { + for (i, x) in b.iter().enumerate() { + v |= u128::from(*x) << (8 * i); + } + } + DatatypeByteOrder::BigEndian => { + for x in b { + v = (v << 8) | u128::from(*x); + } + } + DatatypeByteOrder::Vax => return None, + } + Some(v) +} + +/// The value of an integer (fixed-point) element, honouring its bit offset +/// and precision. `None` when it cannot be represented (over 16 bytes) or +/// `dt` is not an integer. +pub fn decode_int(dt: &Datatype, b: &[u8]) -> Option { + let Datatype::FixedPoint { + size, + byte_order, + signed, + bit_offset, + bit_precision, + } = dt + else { + return None; + }; + let size = usize::try_from(*size).ok()?; + let v = bits(b.get(..size)?, byte_order)?; + let off = u32::from(*bit_offset); + let prec = u32::from(*bit_precision).min(128); + if off >= 128 || prec == 0 { + return Some(0); + } + let mut x = v >> off; + if prec < 128 { + x &= (1u128 << prec) - 1; + } + if *signed && prec < 128 && (x >> (prec - 1)) & 1 == 1 { + x |= !0u128 << prec; + return Some(x as i128); + } + if !*signed && prec == 128 && x > i128::MAX as u128 { + return None; + } + Some(x as i128) +} + +fn decode_float(dt: &Datatype, b: &[u8]) -> Value { + let Datatype::FloatingPoint { + size, byte_order, .. + } = dt + else { + return Value::Error("not a float".into()); + }; + let Ok(n) = usize::try_from(*size) else { + return Value::Error("float size".into()); + }; + let Some(b) = b.get(..n) else { + return Value::Error("short element".into()); + }; + if dtype::is_ieee(dt) + && let Some(v) = bits(b, byte_order) + { + return match n { + 2 => Value::Float( + f64::from(clawhdf5_format::float16::f16_bits_to_f32(v as u16)), + 16, + ), + 4 => Value::Float(f64::from(f32::from_bits(v as u32)), 32), + _ => Value::Float(f64::from_bits(v as u64), 64), + }; + } + // Non-IEEE layouts (N-Bit floats, VAX order): the library converts. + match clawhdf5_format::data_read::read_as_f64(b, dt) { + Ok(v) if v.len() == 1 => Value::Float(v[0], (n * 8).min(64) as u8), + Ok(_) => Value::Error("float conversion".into()), + Err(e) => Value::Error(e.to_string()), + } +} + +fn trim_string(b: &[u8], pad: Option<&StringPadding>) -> String { + let cut = b.iter().position(|&c| c == 0).unwrap_or(b.len()); + let mut s = &b[..cut]; + if matches!(pad, Some(StringPadding::SpacePad)) { + while let [rest @ .., b' '] = s { + s = rest; + } + } + String::from_utf8_lossy(s).into_owned() +} + +/// Little-endian unsigned integer of `b` (up to 8 bytes). +fn le(b: &[u8]) -> u64 { + b.iter() + .take(8) + .enumerate() + .fold(0u64, |a, (i, &x)| a | (u64::from(x) << (8 * i))) +} + +/// Decodes elements of one file. +pub struct Decoder<'a> { + pub h5: &'a H5, +} + +impl<'a> Decoder<'a> { + pub fn new(h5: &'a H5) -> Self { + Self { h5 } + } + + /// Decode element `i` of `raw`, an array of `dt` elements. + pub fn element(&self, dt: &Datatype, raw: &[u8], i: usize) -> Value { + let size = dt.type_size() as usize; + match i + .checked_mul(size) + .and_then(|s| raw.get(s..s.checked_add(size)?)) + { + Some(b) => self.decode(dt, b, 0), + None => Value::Error("element out of range".into()), + } + } + + pub fn decode(&self, dt: &Datatype, b: &[u8], depth: u32) -> Value { + if depth > 32 { + return Value::Error("datatype nesting too deep".into()); + } + let size = dt.type_size() as usize; + let Some(b) = b.get(..size) else { + return Value::Error("short element".into()); + }; + match dt { + Datatype::FixedPoint { .. } => match decode_int(dt, b) { + Some(v) => Value::Int(v), + None => Value::Bytes(b.to_vec()), + }, + Datatype::FloatingPoint { .. } => decode_float(dt, b), + Datatype::Time { .. } | Datatype::BitField { .. } | Datatype::Opaque { .. } => { + Value::Bytes(b.to_vec()) + } + Datatype::String { padding, .. } => Value::Str(trim_string(b, Some(padding))), + Datatype::Compound { members, .. } => { + let mut out = Vec::with_capacity(members.len()); + for m in members { + let off = usize::try_from(m.byte_offset).unwrap_or(usize::MAX); + let v = match b.get(off..) { + Some(mb) => self.decode(&m.datatype, mb, depth + 1), + None => Value::Error("member out of bounds".into()), + }; + out.push((m.name.clone(), v)); + } + Value::Compound(out) + } + Datatype::Reference { ref_type, .. } => match ref_type { + ReferenceType::Object | ReferenceType::Object2 => { + match clawhdf5_format::data_read::read_object_references(b, dt, self.h5.os()) { + Ok(r) if r.len() == 1 => { + let a = r[0].address; + let undef = a == u64::MAX + || (self.h5.os() < 8 && a == (1u64 << (8 * self.h5.os())) - 1); + Value::Ref(if undef || a == 0 { None } else { Some(a) }) + } + Ok(_) => Value::Error("reference".into()), + Err(e) => Value::Error(e.to_string()), + } + } + _ => Value::OtherRef(b.to_vec()), + }, + Datatype::Enumeration { + base_type, members, .. + } => { + let Some(v) = decode_int(base_type, b) else { + return Value::Bytes(b.to_vec()); + }; + let bs = base_type.type_size() as usize; + let name = members + .iter() + .find(|m| m.value.get(..bs) == b.get(..bs)) + .map(|m| m.name.clone()); + Value::Enum(name, v) + } + Datatype::Array { + base_type, + dimensions, + } => { + let n = dimensions + .iter() + .try_fold(1usize, |a, &d| a.checked_mul(d as usize)); + let bs = base_type.type_size() as usize; + let Some(n) = n.filter(|n| n.checked_mul(bs).is_some_and(|t| t <= b.len())) else { + return Value::Error("array larger than its element".into()); + }; + let mut out = Vec::with_capacity(n); + for k in 0..n { + out.push(self.decode(base_type, &b[k * bs..], depth + 1)); + } + Value::Array(out) + } + Datatype::VariableLength { + is_string, + padding, + base_type, + .. + } => self.decode_vlen(*is_string, padding.as_ref(), base_type, b, depth), + } + } + + fn decode_vlen( + &self, + is_string: bool, + padding: Option<&StringPadding>, + base: &Datatype, + b: &[u8], + depth: u32, + ) -> Value { + let os = usize::from(self.h5.os()); + let (Some(lenb), Some(addrb), Some(idxb)) = + (b.get(..4), b.get(4..4 + os), b.get(4 + os..8 + os)) + else { + return Value::Error("short VL element".into()); + }; + let len = le(lenb) as usize; + let addr = le(addrb); + let idx = le(idxb) as u32; + let undef = if os >= 8 { + u64::MAX + } else { + (1u64 << (8 * os)) - 1 + }; + let obj = if len == 0 || addr == 0 || addr == undef { + Vec::new() + } else { + match self.h5.heap_object(addr, idx) { + Ok(o) => o, + Err(e) => return Value::Error(e.to_string()), + } + }; + if is_string { + let l = len.min(obj.len()); + return Value::Str(trim_string(&obj[..l], padding)); + } + let bs = base.type_size() as usize; + if bs == 0 { + return Value::Error("VL base type of size 0".into()); + } + match len.checked_mul(bs) { + Some(need) if need <= obj.len() => {} + _ => return Value::Error("VL sequence longer than its heap object".into()), + } + let mut out = Vec::with_capacity(len); + for k in 0..len { + out.push(self.decode(base, &obj[k * bs..], depth + 1)); + } + Value::Seq(out) + } +} + +/// Format a float like C's `%g` would at full round-trip precision: plain +/// digits for moderate magnitudes, an exponent otherwise. +pub fn fmt_float(v: f64, width: u8) -> String { + if v.is_nan() { + return "NaN".into(); + } + if v.is_infinite() { + return if v > 0.0 { "Inf".into() } else { "-Inf".into() }; + } + let a = v.abs(); + let plain = a == 0.0 || (1e-5..1e16).contains(&a); + match (width, plain) { + (16 | 32, true) => format!("{}", v as f32), + (16 | 32, false) => format!("{:e}", v as f32), + (_, true) => format!("{v}"), + (_, false) => format!("{v:e}"), + } +} + +fn escape(s: &str) -> String { + let mut o = String::with_capacity(s.len() + 2); + for c in s.chars() { + match c { + '"' => o.push_str("\\\""), + '\\' => o.push_str("\\\\"), + '\n' => o.push_str("\\n"), + '\r' => o.push_str("\\r"), + '\t' => o.push_str("\\t"), + c if (c as u32) < 0x20 => o.push_str(&format!("\\{:03o}", c as u32)), + c => o.push(c), + } + } + o +} + +/// A fixed-length string element's bytes quoted as h5dump prints a +/// null-padded string: every byte, NULs as `\000`. +pub fn quote_bytes(b: &[u8]) -> String { + format!("\"{}\"", escape(&String::from_utf8_lossy(b))) +} + +/// Text form, as in an h5dump DATA block. +pub fn text(v: &Value, h5paths: &dyn Fn(u64) -> Option) -> String { + match v { + Value::Int(i) => i.to_string(), + Value::Float(f, w) => fmt_float(*f, *w), + Value::Str(s) => format!("\"{}\"", escape(s)), + Value::Bytes(b) => hex(b), + Value::Enum(Some(n), _) => n.clone(), + Value::Enum(None, i) => i.to_string(), + Value::Compound(ms) => format!( + "{{ {} }}", + ms.iter() + .map(|(_, v)| text(v, h5paths)) + .collect::>() + .join(", ") + ), + Value::Array(vs) => format!( + "[ {} ]", + vs.iter() + .map(|v| text(v, h5paths)) + .collect::>() + .join(", ") + ), + Value::Seq(vs) => format!( + "({})", + vs.iter() + .map(|v| text(v, h5paths)) + .collect::>() + .join(", ") + ), + Value::Ref(None) => "NULL".into(), + Value::Ref(Some(a)) => match h5paths(*a) { + Some(p) => format!("\"{p}\""), + None => format!("{a:#x}"), + }, + Value::OtherRef(b) => hex(b), + Value::Error(e) => format!(""), + } +} + +/// hdf5-json value form. +pub fn to_json(v: &Value, h5paths: &dyn Fn(u64) -> Option) -> J { + match v { + Value::Int(i) => { + if let Ok(x) = i64::try_from(*i) { + J::from(x) + } else if let Ok(x) = u64::try_from(*i) { + J::from(x) + } else { + J::from(i.to_string()) + } + } + Value::Float(f, _) => { + if f.is_finite() { + serde_json::Number::from_f64(*f) + .map(J::Number) + .unwrap_or(J::Null) + } else if f.is_nan() { + J::from("NaN") + } else if *f > 0.0 { + J::from("Infinity") + } else { + J::from("-Infinity") + } + } + Value::Str(s) => J::from(s.as_str()), + Value::Bytes(b) | Value::OtherRef(b) => J::from(hex(b)), + Value::Enum(_, i) => to_json(&Value::Int(*i), h5paths), + Value::Compound(ms) => J::Array(ms.iter().map(|(_, v)| to_json(v, h5paths)).collect()), + Value::Array(vs) | Value::Seq(vs) => { + J::Array(vs.iter().map(|v| to_json(v, h5paths)).collect()) + } + Value::Ref(None) => J::Null, + Value::Ref(Some(a)) => J::from(h5paths(*a).unwrap_or_else(|| format!("{a:#x}"))), + Value::Error(e) => serde_json::json!({ "error": e }), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn int(size: u32, signed: bool, order: DatatypeByteOrder, off: u16, prec: u16) -> Datatype { + Datatype::FixedPoint { + size, + byte_order: order, + signed, + bit_offset: off, + bit_precision: prec, + } + } + + #[test] + fn integers_decode_with_order_offset_and_sign() { + let le = DatatypeByteOrder::LittleEndian; + let be = DatatypeByteOrder::BigEndian; + assert_eq!( + decode_int(&int(2, true, le.clone(), 0, 16), &[0xff, 0xff]), + Some(-1) + ); + assert_eq!( + decode_int(&int(2, false, le, 0, 16), &[0xff, 0xff]), + Some(65535) + ); + assert_eq!( + decode_int(&int(2, true, be.clone(), 0, 16), &[0x80, 0x00]), + Some(-32768) + ); + // 17-bit signed field at offset 4: -5 + let stored = (((-5i32) as u32) & 0x1_FFFF) << 4; + assert_eq!( + decode_int(&int(4, true, be, 4, 17), &stored.to_be_bytes()), + Some(-5) + ); + assert_eq!( + decode_int(&int(4, true, DatatypeByteOrder::Vax, 0, 32), &[0; 4]), + None + ); + assert_eq!( + decode_int( + &int(4, true, DatatypeByteOrder::LittleEndian, 0, 32), + &[0; 2] + ), + None + ); + } + + #[test] + fn floats_print_at_their_own_precision() { + assert_eq!(fmt_float(f64::from(0.1f32), 32), "0.1"); + assert_eq!(fmt_float(0.1, 64), "0.1"); + assert_eq!(fmt_float(1e20, 64), "1e20"); + assert_eq!(fmt_float(2.0, 64), "2"); + assert_eq!(fmt_float(f64::NAN, 64), "NaN"); + } +} diff --git a/crates/clawhdf5-tools/tests/gen_files.py b/crates/clawhdf5-tools/tests/gen_files.py new file mode 100644 index 0000000..8bc4e1e --- /dev/null +++ b/crates/clawhdf5-tools/tests/gen_files.py @@ -0,0 +1,164 @@ +"""Write the HDF5 files the h5rs interop tests run on. + +usage: gen_files.py OUTDIR + +Writes OUTDIR/{earliest,latest}.h5 (the same content with the oldest and the +newest file-format structures: symbol tables and v1 B-trees vs. v2 object +headers, fractal heaps, v2 B-trees and the chunk indexes of HDF5 1.10+), the +pairs the diff tests compare, and a file with a user block. Prints one JSON +object with the values h5py reads back, for the dump tests. +""" + +import json +import os +import sys + +import h5py +import numpy as np + +out = sys.argv[1] + + +def content(f, dense): + f.attrs["title"] = "h5rs test" + f.attrs["version"] = np.int64(3) + f.attrs["scale"] = np.array([0.5, 1.5], dtype="f4") + f["contig"] = np.arange(12, dtype="f8").reshape(3, 4) + dcpl = h5py.h5p.create(h5py.h5p.DATASET_CREATE) + dcpl.set_layout(h5py.h5d.COMPACT) + space = h5py.h5s.create_simple((4,)) + h5py.h5d.create(f.id, b"compact", h5py.h5t.STD_I16LE, space, dcpl).write( + h5py.h5s.ALL, h5py.h5s.ALL, np.arange(4, dtype="f4")]) + f.create_dataset( + "enum", data=np.array([0, 1, 1], dtype="u1"), + dtype=h5py.enum_dtype({"RED": 0, "GREEN": 1}, basetype="u1"), + ) + f["be"] = np.arange(5, dtype=">i4") + f["arr"] = np.array([([1, 2, 3],), ([4, 5, 6],)], dtype=[("v", "3i4")]) + f["named_t"] = np.dtype("i8") + f["soft"] = h5py.SoftLink("/contig") + f["dangling"] = h5py.SoftLink("/nowhere") + f["external"] = h5py.ExternalLink("other.h5", "/x") + f["hard2"] = g["sub"] + many = f.create_group("many") + for i in range(12 if dense else 4): + many[f"d{i:02}"] = np.int32(i) + many.attrs[f"a{i:02}"] = i + + +values = {} +# "latest" under HDF5 2.0 writes datatype messages that libhdf5 1.14 tools +# cannot read, so the newest format is taken as 1.14's. +for libver in ("earliest", "latest"): + path = os.path.join(out, f"{libver}.h5") + bounds = ("earliest", "v114") if libver == "earliest" else ("v114", "v114") + with h5py.File(path, "w", libver=bounds) as f: + content(f, True) + with h5py.File(path, "r") as f: + vals = {} + + def grab(name, obj): + if isinstance(obj, h5py.Dataset) and obj.dtype.kind in "iuf" and obj.shape is not None: + vals["/" + name] = obj[()].tolist() + + f.visititems(grab) + values[libver] = vals + +# diff pairs +def small(path, data=None, extra=False, attr=False, shape=(3, 4), dtype="f8"): + with h5py.File(os.path.join(out, path), "w") as f: + d = np.arange(12, dtype=dtype).reshape(shape) if data is None else data + f["d"] = d + f["g/x"] = np.arange(3) + if extra: + f["only_here"] = 1 + if attr: + f["d"].attrs["u"] = 1 + + +base = np.arange(12, dtype="f8").reshape(3, 4) +small("base.h5") +small("same.h5") +changed = base.copy() +changed[0, 2] += 0.001 +changed[2, 3] += 0.001 +small("changed.h5", data=changed) +small("extra.h5", extra=True) +small("attr.h5", attr=True) +small("reshaped.h5", shape=(4, 3)) +small("int.h5", dtype="i4") + +# One object under two names (a hard link) against two separate copies. +with h5py.File(os.path.join(out, "hardlinked.h5"), "w") as f: + f["x"] = np.arange(5) + f["y"] = f["x"] + g = f.create_group("g") + g["d"] = np.arange(3) + g.create_group("s")["e"] = np.arange(2) + f["h"] = g +for name, last in (("copied.h5", 1), ("copied_changed.h5", 9)): + with h5py.File(os.path.join(out, name), "w") as f: + f["x"] = np.arange(5) + f["y"] = np.arange(5) + for gname in ("g", "h"): + g = f.create_group(gname) + g["d"] = np.arange(3) + g.create_group("s")["e"] = np.array([0, last if gname == "h" else 1]) + +# 64-bit integers one apart, beyond f64's 2^53 integer precision. +for name, d in (("big1.h5", 0), ("big2.h5", 1)): + with h5py.File(os.path.join(out, name), "w") as f: + f["i"] = np.array([2**60 + d, -(2**62) - d], dtype="i8") + f["u"] = np.array([2**64 - 1 - d], dtype="u8") + +# Soft links: the same link targets, whose target objects differ. +for name, v in (("soft1.h5", 0), ("soft2.h5", 1)): + with h5py.File(os.path.join(out, name), "w") as f: + f["z"] = np.arange(4) + v + f.create_group("g")["s"] = h5py.SoftLink("/z") + grp = f.create_group("grp") + grp["d"] = np.arange(3) + v + f["lnk"] = h5py.SoftLink("/grp") + f["dang"] = h5py.SoftLink("/nowhere") +# Soft links: different link targets, whose target objects are equal. +for name, t in (("target1.h5", "/a"), ("target2.h5", "/b")): + with h5py.File(os.path.join(out, name), "w") as f: + f["a"] = np.arange(4) + f["b"] = np.arange(4) + f["s"] = h5py.SoftLink(t) + f["rel"] = h5py.SoftLink("a") + +# Null-padded fixed strings with NULs, at the top level and nested in a +# compound and in an array member. +with h5py.File(os.path.join(out, "nulstrings.h5"), "w") as f: + f["top"] = np.array([b"", b"ab", b"a\x00b"], dtype="S3") + f["cmp"] = np.array( + [(b"", 1), (b"ab", 2), (b"a\x00b", 3)], dtype=[("s", "S3"), ("i", "i4")] + ) + f["arr"] = np.array([([b"", b"x"],)], dtype=[("v", "(2,)S2")]) + +with h5py.File(os.path.join(out, "userblock.h5"), "w", userblock_size=1024) as f: + f["d"] = np.arange(10) + +print(json.dumps(values)) diff --git a/crates/clawhdf5-tools/tests/h5rs_interop.rs b/crates/clawhdf5-tools/tests/h5rs_interop.rs new file mode 100644 index 0000000..7d97cc5 --- /dev/null +++ b/crates/clawhdf5-tools/tests/h5rs_interop.rs @@ -0,0 +1,775 @@ +//! `h5rs` against libhdf5's own tools, on files h5py writes +//! (`tests/gen_files.py`): `ls` against h5ls, `stat` against h5stat, `dump` +//! against h5dump, `diff` exit codes against h5diff, `dump --json` values +//! against h5py, and `check` on valid and deliberately corrupted files. +//! +//! Needs python3 with h5py (`CLAWHDF5_PYTHON`) and, for the comparisons, +//! the libhdf5 command-line tools (h5ls, h5stat, h5dump, h5diff) on `PATH`. +//! Each test skips when what it needs is missing, unless +//! `CLAWHDF5_REQUIRE_INTEROP=1`, which turns a skip into a failure. + +use std::collections::BTreeMap; +use std::path::{Path, PathBuf}; +use std::process::{Command, Output}; + +use clawhdf5_format::checksum::jenkins_lookup3; + +fn python() -> String { + std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string()) +} + +fn interop_required() -> bool { + std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1") +} + +fn python_available() -> bool { + Command::new(python()) + .args(["-c", "import h5py, numpy"]) + .output() + .map(|o| o.status.success()) + .unwrap_or(false) +} + +fn tool_available(name: &str) -> bool { + Command::new(name) + .arg("--version") + .output() + .map(|o| o.status.success()) + .unwrap_or(false) +} + +/// Skip (return true) when `what` is unavailable, or fail when interop is +/// required. +fn missing(ok: bool, what: &str) -> bool { + if ok { + return false; + } + assert!( + !interop_required(), + "CLAWHDF5_REQUIRE_INTEROP=1 but {what} is not available" + ); + eprintln!("SKIP: {what} not available"); + true +} + +fn h5rs(args: &[&str]) -> Output { + Command::new(env!("CARGO_BIN_EXE_h5rs")) + .args(args) + .output() + .expect("run h5rs") +} + +fn run(tool: &str, args: &[&str]) -> Output { + Command::new(tool).args(args).output().expect("run tool") +} + +fn stdout(o: &Output) -> String { + String::from_utf8_lossy(&o.stdout).into_owned() +} + +/// The generated files and what h5py read from them. +struct Files { + dir: tempfile::TempDir, + values: serde_json::Value, +} + +impl Files { + fn path(&self, name: &str) -> PathBuf { + self.dir.path().join(name) + } + + fn p(&self, name: &str) -> String { + self.path(name).to_string_lossy().into_owned() + } +} + +fn generate() -> Option { + if missing(python_available(), "python3 with h5py") { + return None; + } + let dir = tempfile::tempdir().unwrap(); + let script = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/gen_files.py"); + let out = Command::new(python()) + .arg(&script) + .arg(dir.path()) + .output() + .expect("run gen_files.py"); + assert!( + out.status.success(), + "gen_files.py failed:\n{}", + String::from_utf8_lossy(&out.stderr) + ); + let values = serde_json::from_slice(&out.stdout).expect("gen_files.py output"); + Some(Files { dir, values }) +} + +const LIBVERS: [&str; 2] = ["earliest.h5", "latest.h5"]; + +// --------------------------------------------------------------------------- +// ls +// --------------------------------------------------------------------------- + +/// Every h5ls line must be the start of the matching h5rs line (h5rs adds +/// the datatype after a dataset's shape). +fn assert_ls_like(ours: &str, reference: &str, what: &str) { + let ours: Vec<&str> = ours.lines().collect(); + let refs: Vec<&str> = reference.lines().collect(); + assert_eq!( + ours.len(), + refs.len(), + "{what}: line count\nh5rs:\n{}\nh5ls:\n{}", + ours.join("\n"), + refs.join("\n") + ); + for (o, r) in ours.iter().zip(&refs) { + let r = r.trim_end(); + assert!( + o.starts_with(r) && (o.len() == r.len() || r.contains("Dataset {")), + "{what}:\n h5rs: {o}\n h5ls: {r}" + ); + } +} + +#[test] +fn ls_matches_h5ls() { + let Some(f) = generate() else { return }; + if missing(tool_available("h5ls"), "h5ls") { + return; + } + for name in LIBVERS.iter().chain(&["userblock.h5"]) { + let p = f.p(name); + for (args, what) in [(vec!["-r"], "recursive"), (vec![], "root")] { + let mut a: Vec<&str> = args.clone(); + a.push(&p); + let ours = h5rs(&[&["ls"], a.as_slice()].concat()); + assert!(ours.status.success(), "{name} {what}: {:?}", ours); + let reference = run("h5ls", &a); + assert_ls_like( + &stdout(&ours), + &stdout(&reference), + &format!("{name} {what}"), + ); + } + } + // A group, and a dataset, named after the file. + for obj in ["/grp", "/grp/sub", "/contig", "/grp/ext2"] { + let arg = format!("{}{obj}", f.p("latest.h5")); + let ours = h5rs(&["ls", &arg]); + let reference = run("h5ls", &[&arg]); + assert_ls_like(&stdout(&ours), &stdout(&reference), &arg); + let ours = h5rs(&["ls", "-r", &arg]); + let reference = run("h5ls", &["-r", &arg]); + assert_ls_like(&stdout(&ours), &stdout(&reference), &format!("-r {arg}")); + } +} + +#[test] +fn ls_verbose_describes_layout_filters_and_attributes() { + let Some(f) = generate() else { return }; + let o = h5rs(&["ls", "-r", "-v", &f.p("latest.h5")]); + assert!(o.status.success()); + let s = stdout(&o); + for want in [ + "Layout: chunked (extensible array index)", + "Layout: chunked (v2 B-tree index)", + "Layout: chunked (fixed array index)", + "Layout: chunked (single chunk index)", + "Filter-0: shuffle-2", + "Filter-1: deflate-1", + "Filter-2: fletcher32-3", + "Chunks: {100} 400 bytes", + "Attribute: units scalar", + "Data: \"m\"", + ] { + assert!(s.contains(want), "missing {want:?} in\n{s}"); + } + let o = h5rs(&["ls", "-v", &format!("{}/grp", f.p("earliest.h5"))]); + assert!(stdout(&o).contains("Layout: chunked (v1 B-tree index)")); +} + +// --------------------------------------------------------------------------- +// stat +// --------------------------------------------------------------------------- + +/// `key: value` lines of an h5stat-style report. +fn stat_facts(s: &str) -> BTreeMap { + s.lines() + .filter_map(|l| { + let (k, v) = l.rsplit_once(':')?; + Some((k.trim().to_string(), v.trim().to_string())) + }) + .collect() +} + +#[test] +fn stat_matches_h5stat() { + let Some(f) = generate() else { return }; + if missing(tool_available("h5stat"), "h5stat") { + return; + } + let keys = [ + "# of unique groups", + "# of unique datasets", + "# of unique named datatypes", + "# of unique links", + "# of unique other", + "Max. # of links to object", + "Max. # of objects in group", + "Max. rank of datasets", + "Max. dimension size of 1-D datasets", + "Total raw data size", + "Dataset layout counts[COMPACT]", + "Dataset layout counts[CONTIG]", + "Dataset layout counts[CHUNKED]", + "Dataset layout counts[VIRTUAL]", + "NO filter", + "GZIP filter", + "SHUFFLE filter", + "FLETCHER32 filter", + "SZIP filter", + "NBIT filter", + "SCALEOFFSET filter", + "USER-DEFINED filter", + "Max. # of attributes to objects", + "Total space", + ]; + for name in LIBVERS.iter().chain(&["userblock.h5", "base.h5"]) { + let p = f.p(name); + let ours = h5rs(&["stat", &p]); + assert!(ours.status.success(), "{name}: {ours:?}"); + let reference = run("h5stat", &[&p]); + assert!(reference.status.success(), "h5stat {name} failed"); + let (o, r) = (stat_facts(&stdout(&ours)), stat_facts(&stdout(&reference))); + for k in keys { + // h5stat leaves out the attribute summary when nothing has one. + let Some(want) = r.get(k) else { continue }; + assert_eq!(o.get(k), Some(want), "{name}: {k}"); + } + } +} + +// --------------------------------------------------------------------------- +// dump +// --------------------------------------------------------------------------- + +#[test] +fn dump_matches_h5dump() { + let Some(f) = generate() else { return }; + if missing(tool_available("h5dump"), "h5dump") { + return; + } + for name in LIBVERS.iter().chain(&["userblock.h5", "changed.h5"]) { + let p = f.p(name); + for args in [vec![], vec!["-A"]] { + let mut a = args.clone(); + a.push(p.as_str()); + let ours = h5rs(&[&["dump"], a.as_slice()].concat()); + assert!(ours.status.success(), "{name} {args:?}: {ours:?}"); + let reference = run("h5dump", &a); + // h5dump names the file as given; h5rs by its file name. + let r = stdout(&reference).replacen(&p, name, 1); + assert_eq!(stdout(&ours), r, "{name} {args:?}"); + } + } + let p = f.p("latest.h5"); + let ours = h5rs(&["dump", "-d", "/grp/ext2", &p]); + let reference = run("h5dump", &["-d", "/grp/ext2", &p]); + assert_eq!( + stdout(&ours), + stdout(&reference).replacen(&p, "latest.h5", 1) + ); +} + +/// A null-padded fixed string prints every byte (NULs as `\000`), as +/// h5dump prints it, also inside a compound and an array member. +#[test] +fn dump_shows_nul_padding_in_nested_strings() { + let Some(f) = generate() else { return }; + let p = f.p("nulstrings.h5"); + let ours = stdout(&h5rs(&["dump", &p])); + for want in [ + r#"(0): "\000\000\000", "ab\000", "a\000b""#, + r#""\000\000\000","#, + r#""a\000b","#, + r#"[ "\000\000", "x\000" ]"#, + ] { + assert!(ours.contains(want), "no {want} in\n{ours}"); + } + if missing(tool_available("h5dump"), "h5dump") { + return; + } + let reference = run("h5dump", &[&p]); + assert_eq!(ours, stdout(&reference).replacen(&p, "nulstrings.h5", 1)); +} + +#[test] +fn dump_json_values_match_h5py() { + let Some(f) = generate() else { return }; + for name in LIBVERS { + let o = h5rs(&["dump", "--json", &f.p(name)]); + assert!(o.status.success(), "{name}: {o:?}"); + let doc: serde_json::Value = serde_json::from_slice(&o.stdout).unwrap(); + assert_eq!(doc["apiVersion"], "1.1.1"); + let root = doc["root"].as_str().unwrap(); + assert!(doc["groups"][root].is_object()); + let mut by_path = BTreeMap::new(); + for (_, d) in doc["datasets"].as_object().unwrap() { + for a in d["alias"].as_array().unwrap() { + by_path.insert(a.as_str().unwrap().to_string(), d.clone()); + } + } + let want = f.values[name.trim_end_matches(".h5")].as_object().unwrap(); + assert!(want.len() > 10); + for (path, v) in want { + let d = by_path + .get(path) + .unwrap_or_else(|| panic!("{name}: no {path}")); + let got = &d["value"]; + // JSON numbers: compare as f64 (h5py's ints and floats both + // round-trip exactly through JSON here). + assert_eq!(flatten(got), flatten(v), "{name}: {path}"); + } + let links = doc["groups"][root]["links"].as_array().unwrap(); + let soft = links.iter().find(|l| l["title"] == "soft").unwrap(); + assert_eq!(soft["class"], "H5L_TYPE_SOFT"); + assert_eq!(soft["h5path"], "/contig"); + let ext = links.iter().find(|l| l["title"] == "external").unwrap(); + assert_eq!(ext["class"], "H5L_TYPE_EXTERNAL"); + assert_eq!(ext["file"], "other.h5"); + } +} + +fn flatten(v: &serde_json::Value) -> Vec { + match v { + serde_json::Value::Array(a) => a.iter().flat_map(flatten).collect(), + serde_json::Value::Number(n) => vec![n.as_f64().unwrap()], + other => panic!("not numeric: {other}"), + } +} + +// --------------------------------------------------------------------------- +// diff +// --------------------------------------------------------------------------- + +fn code(o: &Output) -> i32 { + o.status.code().expect("exit code") +} + +#[test] +fn diff_exit_codes_match_h5diff() { + let Some(f) = generate() else { return }; + if missing(tool_available("h5diff"), "h5diff") { + return; + } + let (b, same, changed, extra, attr) = ( + f.p("base.h5"), + f.p("same.h5"), + f.p("changed.h5"), + f.p("extra.h5"), + f.p("attr.h5"), + ); + let missing_file = f.p("does-not-exist.h5"); + let cases: Vec<(Vec<&str>, i32)> = vec![ + (vec![&b, &same], 0), + (vec![&b, &b], 0), + (vec![&b, &changed], 1), + (vec![&b, &changed, "/d"], 1), + (vec![&b, &changed, "/g"], 0), + (vec!["-d", "0.01", &b, &changed], 0), + (vec!["-d", "0.0001", &b, &changed], 1), + (vec!["-p", "0.01", &b, &changed], 0), + (vec!["-p", "1e-9", &b, &changed], 1), + (vec![&b, &extra], 1), + (vec![&b, &attr], 1), + (vec![&b, &attr, "/g"], 0), + (vec![&b, &same, "/nope"], 2), + (vec![&b, &missing_file], 2), + (vec![&b, &same, "/d", "/d"], 0), + ]; + for (args, want) in cases { + let theirs = code(&run("h5diff", &args)); + assert_eq!(theirs, want, "h5diff {args:?} (test expectation)"); + let o = h5rs(&[&["diff"], args.as_slice()].concat()); + assert_eq!(code(&o), want, "h5rs diff {args:?}: {}", stdout(&o)); + let q = h5rs(&[&["diff", "-q"], args.as_slice()].concat()); + assert_eq!(code(&q), want, "h5rs diff -q {args:?}"); + assert!(q.stdout.is_empty(), "-q printed output"); + } + // The report lists the differing positions. + let o = h5rs(&["diff", "-r", &b, &changed, "/d"]); + let s = stdout(&o); + assert!(s.contains("[ 0 2 ]") && s.contains("[ 2 3 ]"), "{s}"); + assert!(s.contains("2 difference(s) found"), "{s}"); +} + +/// The option names are h5diff's: -c is --compare (a flag), the count is +/// -n/--count, and the --name=value forms are accepted. +#[test] +fn diff_options_are_named_like_h5diff() { + let Some(f) = generate() else { return }; + let (b, changed) = (f.p("base.h5"), f.p("changed.h5")); + let cases: Vec<(Vec<&str>, i32)> = vec![ + // "2" is then the first file name, which does not exist. + (vec!["-r", "-c", "2", &b, &changed], 2), + (vec!["-c", &b, &changed], 1), + (vec!["--compare", &b, &b], 0), + (vec!["-r", "-n", "1", &b, &changed], 1), + (vec!["-r", "--count=1", &b, &changed], 1), + (vec!["--delta=0.01", &b, &changed], 0), + (vec!["--relative=1e-9", &b, &changed], 1), + ]; + let have_h5diff = tool_available("h5diff"); + for (args, want) in cases { + if have_h5diff { + assert_eq!(code(&run("h5diff", &args)), want, "h5diff {args:?}"); + } + let o = h5rs(&[&["diff"], args.as_slice()].concat()); + assert_eq!(code(&o), want, "h5rs diff {args:?}: {}", stdout(&o)); + } + for count in [vec!["-n", "1"], vec!["--count=1"]] { + let o = h5rs(&[&["diff", "-r"], count.as_slice(), &[&b, &changed, "/d"]].concat()); + let s = stdout(&o); + assert!( + s.contains("[ 0 2 ]") && !s.contains("[ 2 3 ]"), + "{count:?}: {s}" + ); + assert!(s.contains("2 difference(s) found"), "{s}"); + } + assert_eq!(code(&h5rs(&["diff", "--bogus=1", &b, &b])), 2); +} + +/// Tolerances on 64-bit integers one apart, far beyond f64's integer +/// precision: compared exactly, as h5diff does. +#[test] +fn diff_tolerances_compare_large_integers_exactly() { + let Some(f) = generate() else { return }; + let (a, b) = (f.p("big1.h5"), f.p("big2.h5")); + let cases: Vec<(Vec<&str>, i32)> = vec![ + (vec![&a, &b], 1), + (vec!["-d", "0", &a, &b], 1), + (vec!["-d", "0.5", &a, &b], 1), + (vec!["-d", "1", &a, &b], 0), + (vec!["-d", "0", &a, &b, "/u"], 1), + (vec!["-p", "0", &a, &b], 1), + (vec!["-p", "1e-18", &a, &b, "/i"], 1), + (vec!["-p", "1e-15", &a, &b], 0), + ]; + let have_h5diff = tool_available("h5diff"); + for (args, want) in cases { + if have_h5diff { + assert_eq!(code(&run("h5diff", &args)), want, "h5diff {args:?}"); + } + let o = h5rs(&[&["diff"], args.as_slice()].concat()); + assert_eq!(code(&o), want, "h5rs diff {args:?}: {}", stdout(&o)); + } + // The reported difference is exact too. + let s = stdout(&h5rs(&["diff", "-r", "-d", "0", &a, &b, "/u"])); + assert!( + s.contains("18446744073709551615 18446744073709551614 1\n"), + "{s}" + ); +} + +/// A soft link is compared as a link (its target path), as h5diff does, +/// unless `--follow-symlinks`, which compares (and walks into) what it +/// leads to; two dangling links are then the same. +#[test] +fn diff_soft_links_like_h5diff() { + let Some(f) = generate() else { return }; + let (s1, s2, t1, t2) = ( + f.p("soft1.h5"), + f.p("soft2.h5"), + f.p("target1.h5"), + f.p("target2.h5"), + ); + let fl = "--follow-symlinks"; + let cases: Vec<(Vec<&str>, i32)> = vec![ + (vec![&s1, &s2, "/g/s"], 0), + (vec![&s1, &s2, "/g"], 0), + (vec![&s1, &s2, "/lnk"], 0), + (vec![&s1, &s2, "/dang"], 0), + (vec![&s1, &s2, "/lnk/d"], 1), + (vec![&t1, &t2, "/s"], 1), + (vec![&t1, &t2, "/s", "/rel"], 1), + (vec![fl, &s1, &s2, "/g/s"], 1), + (vec![fl, &s1, &s2, "/g"], 1), + (vec![fl, &s1, &s2, "/lnk"], 1), + (vec![fl, &s1, &s2, "/dang"], 0), + (vec![fl, &s1, &s1], 0), + (vec![fl, &t1, &t2, "/s"], 0), + (vec![fl, &t1, &t2, "/s", "/rel"], 0), + ]; + let have_h5diff = tool_available("h5diff"); + for (args, want) in cases { + if have_h5diff { + assert_eq!(code(&run("h5diff", &args)), want, "h5diff {args:?}"); + } + let o = h5rs(&[&["diff"], args.as_slice()].concat()); + assert_eq!(code(&o), want, "h5rs diff {args:?}: {}", stdout(&o)); + } +} + +/// An object hard-linked under two names is compared under both, with +/// everything below it, so a file that shares one object between two names +/// equals a file that stores two identical copies. +#[test] +fn diff_compares_every_hard_link_path() { + let Some(f) = generate() else { return }; + let (linked, copied, changed) = ( + f.p("hardlinked.h5"), + f.p("copied.h5"), + f.p("copied_changed.h5"), + ); + // A hard-linked dataset: h5diff agrees (exit 0). + for (a, b) in [(&linked, &copied), (&copied, &linked)] { + let o = h5rs(&["diff", a, b, "/y"]); + assert_eq!(code(&o), 0, "{a} {b} /y: {}", stdout(&o)); + if tool_available("h5diff") { + assert_eq!(code(&run("h5diff", &[a, b, "/y"])), 0, "h5diff /y"); + } + // The whole file, including a hard-linked group's members (h5diff + // does not descend a group's second name, so it reports those + // members as present in one file only and exits 1). + let o = h5rs(&["diff", a, b]); + assert_eq!(code(&o), 0, "{a} {b}: {}", stdout(&o)); + assert!(!stdout(&o).contains("exists only"), "{}", stdout(&o)); + } + // A value under the second name is compared, not skipped. + let o = h5rs(&["diff", "-r", &linked, &changed]); + assert_eq!(code(&o), 1, "{}", stdout(&o)); + assert!( + stdout(&o).contains("dataset: and "), + "{}", + stdout(&o) + ); + assert!(!stdout(&o).contains(""), "{}", stdout(&o)); +} + +/// Where h5rs deliberately differs from h5diff: objects that cannot be +/// compared (another shape, another datatype class) are a difference. +#[test] +fn diff_counts_incomparable_objects_as_different() { + let Some(f) = generate() else { return }; + let b = f.p("base.h5"); + for other in ["reshaped.h5", "int.h5"] { + let o = h5rs(&["diff", &b, &f.p(other)]); + assert_eq!(code(&o), 1, "{other}: {}", stdout(&o)); + assert!(stdout(&o).contains("Not comparable"), "{other}"); + } +} + +// --------------------------------------------------------------------------- +// check +// --------------------------------------------------------------------------- + +#[test] +fn check_accepts_valid_files() { + let Some(f) = generate() else { return }; + for name in LIBVERS.iter().chain(&[ + "userblock.h5", + "base.h5", + "changed.h5", + "extra.h5", + "attr.h5", + ]) { + let o = h5rs(&["check", "--data", &f.p(name)]); + assert_eq!(code(&o), 0, "{name}:\n{}", stdout(&o)); + assert!(stdout(&o).contains("no problems found")); + } + // The newest format's checksummed structures were all verified. + let s = stdout(&h5rs(&["check", &f.p("latest.h5")])); + assert!(s.contains("superblock 1, v2 object headers 33,"), "{s}"); + assert!(s.contains("chunk indexes 5"), "{s}"); +} + +/// Signatures of checksummed structures, and where each one's checksum +/// is: found by trying every end offset (the checksum is the Jenkins +/// lookup3 hash of the bytes before it). A fractal heap direct block's +/// checksum is instead stored near its start, computed over the whole block +/// with the field zeroed. +fn checksummed_structures(data: &[u8]) -> Vec<(&'static str, usize, usize)> { + const SIGS: [&str; 13] = [ + "OHDR", "OCHK", "BTHD", "BTIN", "BTLF", "FRHP", "FHIB", "EAHD", "EAIB", "EASB", "EADB", + "FAHD", "FADB", + ]; + let mut found = Vec::new(); + for start in 0..data.len().saturating_sub(4) { + let sig = &data[start..start + 4]; + if let Some(name) = SIGS.iter().find(|s| s.as_bytes() == sig) { + let limit = data.len().min(start + 65536); + for end in start + 8..limit.saturating_sub(4) { + let stored = u32::from_le_bytes(data[end..end + 4].try_into().unwrap()); + if jenkins_lookup3(&data[start..end]) == stored { + found.push((*name, start, end)); + break; + } + } + } else if sig == b"FHDB" { + // version(1) heap address(8) block offset(1..8) checksum(4) + 'found: for boff in 1..=8usize { + let at = start + 5 + 8 + boff; + let Some(stored) = data.get(at..at + 4) else { + break; + }; + let stored = u32::from_le_bytes(stored.try_into().unwrap()); + for size in [512usize, 1024, 2048, 4096, 8192, 16384, 32768, 65536] { + let Some(block) = data.get(start..start + size) else { + break; + }; + let mut b = block.to_vec(); + b[at - start..at - start + 4].fill(0); + if jenkins_lookup3(&b) == stored { + found.push(("FHDB", start, at)); + break 'found; + } + } + } + } + } + found +} + +#[test] +fn check_flags_every_corrupted_checksum() { + let Some(f) = generate() else { return }; + let path = f.path("latest.h5"); + let data = std::fs::read(&path).unwrap(); + let structures = checksummed_structures(&data); + let kinds: std::collections::BTreeSet<&str> = structures.iter().map(|s| s.0).collect(); + for want in [ + "OHDR", "OCHK", "BTHD", "BTLF", "FRHP", "FHDB", "EAHD", "EAIB", "EADB", "FAHD", "FADB", + ] { + assert!(kinds.contains(want), "test file has no {want}: {kinds:?}"); + } + let bad = f.path("bad.h5"); + for (name, start, at) in structures { + let mut d = data.clone(); + d[at] ^= 0x01; + std::fs::write(&bad, &d).unwrap(); + let o = h5rs(&["check", &bad.to_string_lossy()]); + let s = stdout(&o); + assert_eq!( + code(&o), + 1, + "{name} at {start:#x}: checksum flip not flagged\n{s}" + ); + assert!( + s.lines() + .any(|l| l.starts_with("problem: 0x") && l.contains("checksum")), + "{name} at {start:#x}: no checksum problem reported\n{s}" + ); + } +} + +/// `check --data` follows variable-length elements into the global heap: +/// a damaged collection is reported at its address, for the dataset and +/// the attribute that point into it. Without --data it is not read. +#[test] +fn check_data_follows_vl_data_into_the_global_heap() { + let Some(f) = generate() else { return }; + for name in LIBVERS { + let mut data = std::fs::read(f.path(name)).unwrap(); + let gcols: Vec = (0..data.len() - 4) + .filter(|&i| &data[i..i + 4] == b"GCOL") + .collect(); + assert!(!gcols.is_empty(), "{name}: no global heap"); + // "GCOL" version(1) reserved(3) size(8), then heap object 1: + // index(2) refcount(2) reserved(4) size(8) — claim 4 GiB. + let at = gcols[0]; + data[at + 24..at + 28].fill(0xff); + let bad = f.path(&format!("bad-gcol-{name}")); + std::fs::write(&bad, &data).unwrap(); + let bad = bad.to_string_lossy().into_owned(); + let o = h5rs(&["check", "--data", &bad]); + let s = stdout(&o); + assert_eq!(code(&o), 1, "{name}: {s}"); + let want = format!("problem: {at:#x} /vlstr: variable-length data: global heap"); + assert!(s.contains(&want), "{name}: no {want:?} in\n{s}"); + // h5dump refuses the file too. + if tool_available("h5dump") { + assert_ne!(code(&run("h5dump", &[&bad])), 0, "{name}: h5dump read it"); + } + assert_eq!(code(&h5rs(&["check", &bad])), 0, "{name}: without --data"); + let ok = h5rs(&["check", "--data", &f.p(name)]); + assert!( + stdout(&ok).contains(&format!("global heap collections read: {}", gcols.len())), + "{name}: {}", + stdout(&ok) + ); + } +} + +#[test] +fn check_flags_a_truncated_file() { + let Some(f) = generate() else { return }; + let data = std::fs::read(f.path("latest.h5")).unwrap(); + let bad = f.path("truncated.h5"); + std::fs::write(&bad, &data[..data.len() * 3 / 4]).unwrap(); + let o = h5rs(&["check", &bad.to_string_lossy()]); + assert_eq!(code(&o), 1); + assert!(stdout(&o).contains("file is truncated"), "{}", stdout(&o)); +} + +/// A v1 B-tree chunk index has no checksum; its keys are checked against +/// the dataset: a chunk offset that is not a multiple of the chunk size. +#[test] +fn check_flags_a_misaligned_chunk() { + let Some(f) = generate() else { return }; + let path = f.path("earliest.h5"); + let mut data = std::fs::read(&path).unwrap(); + // The leaf chunk B-tree of /grp/gz (10 deflated chunks of 100 i32, so + // each stored in under 400 bytes): "TREE" type(1)=1 level(1)=0 + // entries(2) left(8) right(8), then key, child, key, child, ... where a + // rank-1 chunk key is size(4) mask(4) offsets(2 x 8) and a child is 8. + // Key 1 starts at 24 + 24 + 8 = 56; its dimension-0 offset (100) at 64. + let mut patched = false; + for start in 0..data.len() - 72 { + if &data[start..start + 4] == b"TREE" && data[start + 4] == 1 && data[start + 5] == 0 { + let entries = u16::from_le_bytes(data[start + 6..start + 8].try_into().unwrap()); + let size0 = u32::from_le_bytes(data[start + 24..start + 28].try_into().unwrap()); + let off1 = u64::from_le_bytes(data[start + 64..start + 72].try_into().unwrap()); + if entries == 10 && size0 < 400 && off1 == 100 { + data[start + 64..start + 72].copy_from_slice(&103u64.to_le_bytes()); + patched = true; + break; + } + } + } + assert!(patched, "no chunk B-tree found for /grp/gz"); + let bad = f.path("misaligned.h5"); + std::fs::write(&bad, &data).unwrap(); + let o = h5rs(&["check", &bad.to_string_lossy()]); + let s = stdout(&o); + assert_eq!(code(&o), 1, "{s}"); + // The library's chunk-index reader refuses the key first (as libhdf5 + // does); either way the problem is reported against the dataset. + assert!( + s.contains("/grp/gz: chunk at [103, 0] offset 103 in dimension 0 is not a multiple of the chunk size 100") + || s.contains("/grp/gz: chunk index (v1 B-tree): chunked read error: bad coordinate offset [103, 0] for chunk dimensions [100, 4]"), + "{s}" + ); +} + +#[test] +fn every_subcommand_rejects_a_non_hdf5_file_cleanly() { + let dir = tempfile::tempdir().unwrap(); + let p = dir.path().join("junk.h5"); + std::fs::write(&p, b"this is not an HDF5 file at all").unwrap(); + let p = p.to_string_lossy().into_owned(); + for args in [ + vec!["ls", p.as_str()], + vec!["dump", p.as_str()], + vec!["stat", p.as_str()], + vec!["diff", p.as_str(), p.as_str()], + ] { + let o = h5rs(&args); + assert_eq!(code(&o), 2, "{args:?}"); + assert!(String::from_utf8_lossy(&o.stderr).contains("not an HDF5 file")); + } + let o = h5rs(&["check", &p]); + assert_eq!(code(&o), 1); + assert!(stdout(&o).contains("no HDF5 signature")); + assert_eq!(code(&h5rs(&["nonsense"])), 2); + assert_eq!(code(&h5rs(&["ls"])), 2); + assert_eq!(code(&h5rs(&["--help"])), 0); +} diff --git a/crates/clawhdf5-wasm/Cargo.toml b/crates/clawhdf5-wasm/Cargo.toml new file mode 100644 index 0000000..303837a --- /dev/null +++ b/crates/clawhdf5-wasm/Cargo.toml @@ -0,0 +1,30 @@ +[package] +name = "clawhdf5-wasm" +version.workspace = true +edition.workspace = true +rust-version.workspace = true +description = "Read HDF5 and NetCDF-4 files in the browser: clawhdf5's reader compiled to WebAssembly" +license.workspace = true +repository.workspace = true +keywords = ["hdf5", "netcdf", "wasm", "browser"] +categories = ["wasm", "parser-implementations", "science"] +# Distributed as the wasm-bindgen package built by examples/wasm-viewer/build.sh. +publish = false + +[lib] +# cdylib for wasm-bindgen; rlib so the pure-Rust core is tested natively. +crate-type = ["cdylib", "rlib"] + +[dependencies] +# No mmap (there is no file system) and no `parallel` (no threads on +# wasm32-unknown-unknown). lz4 is pure Rust; zstd and szip link C and are +# left out, so such datasets fail with a clear filter error. +clawhdf5 = { path = "../clawhdf5", version = "2.7.0", default-features = false, features = ["lz4"] } +clawhdf5-format = { path = "../clawhdf5-format", version = "2.7.0" } +# Must match the wasm-bindgen CLI exactly; build.sh checks. +wasm-bindgen = "0.2.129" +js-sys = "0.3.106" + +[dev-dependencies] +serde_json = "1" +tempfile = { workspace = true } diff --git a/crates/clawhdf5-wasm/src/core.rs b/crates/clawhdf5-wasm/src/core.rs new file mode 100644 index 0000000..baadcda --- /dev/null +++ b/crates/clawhdf5-wasm/src/core.rs @@ -0,0 +1,603 @@ +//! The reader behind the JavaScript API, in plain Rust so it is tested +//! natively. The `wasm_bindgen` layer in `lib.rs` only converts these types +//! to JavaScript values. +//! +//! Every read either returns the dataset's values or an error: a datatype +//! with no typed-array mapping (compound, reference, opaque, ...) is refused +//! with a message naming it, never returned as reinterpreted bytes. + +use clawhdf5::{AttrValue, File, Selection}; +use clawhdf5_format::data_read; +use clawhdf5_format::datatype::{Datatype, DatatypeByteOrder}; + +/// Errors are reported to JavaScript as messages. +pub type Result = std::result::Result; + +fn err(e: impl std::fmt::Display) -> String { + e.to_string() +} + +/// What a path names. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Kind { + Group, + Dataset, +} + +impl Kind { + pub fn as_str(self) -> &'static str { + match self { + Kind::Group => "group", + Kind::Dataset => "dataset", + } + } +} + +/// One entry of a group listing. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Child { + pub name: String, + pub kind: Kind, +} + +/// A dataset's metadata. +#[derive(Debug, Clone, PartialEq)] +pub struct DatasetInfo { + /// Dataspace dimensions (empty for a scalar). + pub shape: Vec, + /// Maximum dimensions, `None` per unlimited dimension; `None` overall + /// when the dataspace records none. + pub maxshape: Option>>, + /// Human-readable datatype, e.g. `f64`, `i16 (big-endian)`, `string[8]`. + pub dtype: String, + /// Dimensions of an array datatype's elements, appended to the shape of + /// what [`Reader::read`] returns (empty otherwise). + pub element_shape: Vec, +} + +/// Decoded values, one variant per JavaScript typed array. +#[derive(Debug, Clone, PartialEq)] +pub enum Data { + F32(Vec), + F64(Vec), + I8(Vec), + I16(Vec), + I32(Vec), + I64(Vec), + U8(Vec), + U16(Vec), + U32(Vec), + U64(Vec), + /// Fixed- and variable-length strings, and enumeration member names. + Strings(Vec), +} + +impl Data { + pub fn len(&self) -> usize { + match self { + Data::F32(v) => v.len(), + Data::F64(v) => v.len(), + Data::I8(v) => v.len(), + Data::I16(v) => v.len(), + Data::I32(v) => v.len(), + Data::I64(v) => v.len(), + Data::U8(v) => v.len(), + Data::U16(v) => v.len(), + Data::U32(v) => v.len(), + Data::U64(v) => v.len(), + Data::Strings(v) => v.len(), + } + } + + pub fn is_empty(&self) -> bool { + self.len() == 0 + } +} + +/// Values in row-major order with their shape. +#[derive(Debug, Clone, PartialEq)] +pub struct Array { + pub shape: Vec, + pub data: Data, +} + +/// A regular hyperslab, as in `H5Sselect_hyperslab`. `stride` and `block` +/// default to 1 in every dimension. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Hyperslab { + pub start: Vec, + pub count: Vec, + pub stride: Option>, + pub block: Option>, +} + +/// An attribute: its value, or why it has none. +#[derive(Debug, Clone)] +pub struct Attr { + pub name: String, + pub value: AttrValue, +} + +/// An open file, held in memory. +pub struct Reader { + file: File, +} + +impl Reader { + /// Parse a file from its bytes (the browser hands over the whole file). + pub fn open(bytes: Vec) -> Result { + Ok(Self { + file: File::from_bytes(bytes).map_err(err)?, + }) + } + + /// Whether `path` names a group or a dataset. + pub fn kind(&self, path: &str) -> Result { + match self.file.dataset(path) { + Ok(_) => Ok(Kind::Dataset), + Err(clawhdf5::Error::NotADataset(_)) => Ok(Kind::Group), + Err(e) => Err(err(e)), + } + } + + /// The groups, then the datasets, in the group at `path` (`/` is the + /// root). Soft links are listed as their targets; external and dangling + /// links, and named datatypes, are left out. + pub fn list(&self, path: &str) -> Result> { + if self.kind(path)? != Kind::Group { + return Err(format!("not a group: {path}")); + } + let group = self.file.group(path).map_err(err)?; + let mut out: Vec = group + .groups() + .map_err(err)? + .into_iter() + .map(|name| Child { + name, + kind: Kind::Group, + }) + .collect(); + out.extend( + group + .datasets() + .map_err(err)? + .into_iter() + .map(|name| Child { + name, + kind: Kind::Dataset, + }), + ); + Ok(out) + } + + /// Shape, max shape and datatype of the dataset at `path`. + pub fn info(&self, path: &str) -> Result { + let ds = self.file.dataset(path).map_err(err)?; + let dt = ds.raw_datatype().map_err(err)?; + let maxshape = ds.max_dimensions().map_err(err)?.map(|dims| { + dims.into_iter() + .map(|d| (d != u64::MAX).then_some(d)) + .collect() + }); + Ok(DatasetInfo { + shape: ds.shape().map_err(err)?, + maxshape, + dtype: describe(&dt), + element_shape: element_shape(&dt), + }) + } + + /// The attributes of the group or dataset at `path`, sorted by name, and + /// one message per attribute that could not be read at all. An attribute + /// whose type has no plain JavaScript form is returned as + /// [`AttrValue::Raw`]. + pub fn attrs(&self, path: &str) -> Result<(Vec, Vec)> { + let (map, errors) = match self.kind(path)? { + Kind::Dataset => self + .file + .dataset(path) + .and_then(|d| d.attrs_with_errors()) + .map_err(err)?, + Kind::Group => self + .file + .group(path) + .and_then(|g| g.attrs_with_errors()) + .map_err(err)?, + }; + let mut attrs: Vec = map + .into_iter() + .map(|(name, value)| Attr { name, value }) + .collect(); + attrs.sort_by(|a, b| a.name.cmp(&b.name)); + Ok((attrs, errors.into_iter().map(err).collect())) + } + + /// Read the dataset at `path`, whole or a hyperslab of it. + pub fn read(&self, path: &str, slab: Option<&Hyperslab>) -> Result { + let ds = self.file.dataset(path).map_err(err)?; + let dt = ds.raw_datatype().map_err(err)?; + let shape = ds.shape().map_err(err)?; + let (selection, mut out_shape) = match slab { + None => (Selection::All, shape.clone()), + Some(h) => hyperslab_selection(h, &shape)?, + }; + let raw = ds.read_selection(&selection).map_err(err)?; + let data = self.decode(&raw, &dt)?; + out_shape.extend(element_shape(&dt)); + let expected = out_shape + .iter() + .try_fold(1u64, |acc, &d| acc.checked_mul(d)) + .ok_or("selection size overflows")?; + if data.len() as u64 != expected { + return Err(format!( + "read {} values for shape {out_shape:?} ({expected} expected)", + data.len() + )); + } + Ok(Array { + shape: out_shape, + data, + }) + } + + fn decode(&self, raw: &[u8], dt: &Datatype) -> Result { + let base = array_base(dt); + let is_array = !std::ptr::eq(base, dt); + Ok(match base { + Datatype::FloatingPoint { size, .. } if *size <= 4 => { + Data::F32(data_read::read_as_f32(raw, dt).map_err(err)?) + } + Datatype::FloatingPoint { .. } => { + Data::F64(data_read::read_as_f64(raw, dt).map_err(err)?) + } + Datatype::FixedPoint { size, signed, .. } => { + let signed_ints = || data_read::read_as_i64(raw, dt).map_err(err); + let unsigned_ints = || data_read::read_as_u64(raw, dt).map_err(err); + match (size, signed) { + (1, true) => Data::I8(narrow(signed_ints()?)?), + (2, true) => Data::I16(narrow(signed_ints()?)?), + (4, true) => Data::I32(narrow(signed_ints()?)?), + (_, true) => Data::I64(signed_ints()?), + (1, false) => Data::U8(narrow(unsigned_ints()?)?), + (2, false) => Data::U16(narrow(unsigned_ints()?)?), + (4, false) => Data::U32(narrow(unsigned_ints()?)?), + (_, false) => Data::U64(unsigned_ints()?), + } + } + Datatype::String { .. } if !is_array => { + Data::Strings(data_read::read_as_strings(raw, dt).map_err(err)?) + } + Datatype::VariableLength { + is_string: true, .. + } if !is_array => { + let size = dt.type_size() as usize; + if size == 0 || !raw.len().is_multiple_of(size) { + return Err(format!( + "{} bytes is not a whole number of {size}-byte string references", + raw.len() + )); + } + let sb = self.file.superblock(); + Data::Strings( + clawhdf5_format::vl_data::read_vl_strings( + self.file.as_bytes(), + raw, + (raw.len() / size) as u64, + sb.offset_size, + sb.length_size, + ) + .map_err(err)?, + ) + } + Datatype::Enumeration { .. } if !is_array => { + Data::Strings(data_read::read_enum_names(raw, dt).map_err(err)?) + } + _ => { + return Err(format!( + "reading {} datasets is not supported", + describe(dt) + )); + } + }) + } +} + +/// Narrow integers read at 64 bits to the dataset's own width. The source is +/// that width, so this cannot fail on correct input; it is checked anyway. +fn narrow>(v: Vec) -> Result> { + v.into_iter() + .map(|x| T::try_from(x).map_err(|_| format!("value {x} out of range"))) + .collect() +} + +/// Innermost element type of (possibly nested) array datatypes. +fn array_base(dt: &Datatype) -> &Datatype { + match dt { + Datatype::Array { base_type, .. } => array_base(base_type), + _ => dt, + } +} + +fn element_shape(dt: &Datatype) -> Vec { + match dt { + Datatype::Array { + base_type, + dimensions, + } => { + let mut dims: Vec = dimensions.iter().map(|&d| u64::from(d)).collect(); + dims.extend(element_shape(base_type)); + dims + } + _ => Vec::new(), + } +} + +fn hyperslab_selection(h: &Hyperslab, shape: &[u64]) -> Result<(Selection, Vec)> { + let rank = shape.len(); + let ones = vec![1u64; rank]; + let stride = h.stride.clone().unwrap_or_else(|| ones.clone()); + let block = h.block.clone().unwrap_or(ones); + for (what, v) in [ + ("start", &h.start), + ("count", &h.count), + ("stride", &stride), + ("block", &block), + ] { + if v.len() != rank { + return Err(format!( + "hyperslab {what} has {} dimensions, the dataset has {rank}", + v.len() + )); + } + } + let mut out = Vec::with_capacity(rank); + for d in 0..rank { + if stride[d] == 0 || block[d] == 0 { + return Err(format!("hyperslab stride and block must be >= 1 (dim {d})")); + } + if h.count[d] > 1 && block[d] > stride[d] { + return Err(format!( + "hyperslab blocks overlap in dim {d}: block {} > stride {}", + block[d], stride[d] + )); + } + // Last element selected: start + (count-1)*stride + block - 1. + if h.count[d] > 0 { + let last = (h.count[d] - 1) + .checked_mul(stride[d]) + .and_then(|x| x.checked_add(h.start[d])) + .and_then(|x| x.checked_add(block[d] - 1)); + match last { + Some(l) if l < shape[d] => {} + _ => { + return Err(format!( + "hyperslab exceeds dimension {d} (extent {})", + shape[d] + )); + } + } + } + out.push( + h.count[d] + .checked_mul(block[d]) + .ok_or("selection size overflows")?, + ); + } + Ok(( + Selection::Hyperslab { + start: h.start.clone(), + stride, + count: h.count.clone(), + block, + }, + out, + )) +} + +/// A short, human-readable datatype name. +pub fn describe(dt: &Datatype) -> String { + fn endian(order: &DatatypeByteOrder) -> &'static str { + match order { + DatatypeByteOrder::BigEndian => " (big-endian)", + DatatypeByteOrder::Vax => " (VAX)", + _ => "", + } + } + match dt { + Datatype::FixedPoint { + size, + signed, + byte_order, + .. + } => format!( + "{}{}{}", + if *signed { "i" } else { "u" }, + size * 8, + endian(byte_order) + ), + Datatype::FloatingPoint { + size, byte_order, .. + } => format!("f{}{}", size * 8, endian(byte_order)), + Datatype::Time { size, .. } => format!("time{}", size * 8), + Datatype::String { size, .. } => format!("string[{size}]"), + Datatype::BitField { size, .. } => format!("bitfield{}", size * 8), + Datatype::Opaque { size, .. } => format!("opaque[{size}]"), + Datatype::Compound { members, .. } => { + let fields: Vec = members + .iter() + .map(|m| format!("{}: {}", m.name, describe(&m.datatype))) + .collect(); + format!("compound{{{}}}", fields.join(", ")) + } + Datatype::Reference { .. } => "reference".to_string(), + Datatype::Enumeration { + base_type, members, .. + } => { + let names: Vec<&str> = members.iter().map(|m| m.name.as_str()).collect(); + format!("enum<{}>{{{}}}", describe(base_type), names.join(", ")) + } + Datatype::VariableLength { + is_string: true, .. + } => "vlen string".to_string(), + Datatype::VariableLength { base_type, .. } => { + format!("vlen<{}>", describe(base_type)) + } + Datatype::Array { + base_type, + dimensions, + } => format!("array{dimensions:?}<{}>", describe(base_type)), + } +} + +#[cfg(test)] +mod tests { + use super::*; + use clawhdf5::FileBuilder; + + fn sample() -> Reader { + let mut b = FileBuilder::new(); + b.create_dataset("grid") + .with_f64_data(&(0..12).map(f64::from).collect::>()) + .with_shape(&[3, 4]) + .with_chunks(&[2, 2]) + .with_deflate(4); + b.create_dataset("bytes").with_u8_data(&[1, 2, 250]); + let mut g = b.create_group("sensors"); + g.create_dataset("temp").with_f32_data(&[1.5, -2.25]); + g.set_attr("location", AttrValue::String("lab".into())); + b.add_group(g.finish()); + b.set_attr("version", AttrValue::I64(3)); + b.set_attr("scale", AttrValue::F64Array(vec![0.5, 2.0])); + Reader::open(b.finish().unwrap()).unwrap() + } + + #[test] + fn lists_groups_then_datasets() { + let r = sample(); + let names: Vec<(String, Kind)> = r + .list("/") + .unwrap() + .into_iter() + .map(|c| (c.name, c.kind)) + .collect(); + assert_eq!(names[0], ("sensors".to_string(), Kind::Group)); + let mut ds: Vec<&str> = names[1..].iter().map(|(n, _)| n.as_str()).collect(); + ds.sort(); + assert_eq!(ds, ["bytes", "grid"]); + assert_eq!( + r.list("sensors").unwrap(), + vec![Child { + name: "temp".into(), + kind: Kind::Dataset + }] + ); + assert!(r.list("grid").unwrap_err().contains("not a group")); + assert!(r.list("missing").is_err()); + } + + #[test] + fn info_reports_shape_and_dtype() { + let r = sample(); + let i = r.info("grid").unwrap(); + assert_eq!(i.shape, vec![3, 4]); + assert_eq!(i.dtype, "f64"); + assert!(i.element_shape.is_empty()); + assert_eq!(r.info("sensors/temp").unwrap().dtype, "f32"); + assert_eq!(r.kind("/sensors").unwrap(), Kind::Group); + assert_eq!(r.kind("/sensors/temp").unwrap(), Kind::Dataset); + } + + #[test] + fn reads_whole_and_hyperslab() { + let r = sample(); + let all = r.read("grid", None).unwrap(); + assert_eq!(all.shape, vec![3, 4]); + assert_eq!(all.data, Data::F64((0..12).map(f64::from).collect())); + + let slab = Hyperslab { + start: vec![1, 0], + count: vec![2, 2], + stride: Some(vec![1, 2]), + block: None, + }; + let part = r.read("grid", Some(&slab)).unwrap(); + assert_eq!(part.shape, vec![2, 2]); + assert_eq!(part.data, Data::F64(vec![4.0, 6.0, 8.0, 10.0])); + + assert_eq!( + r.read("bytes", None).unwrap().data, + Data::U8(vec![1, 2, 250]) + ); + assert_eq!( + r.read("sensors/temp", None).unwrap().data, + Data::F32(vec![1.5, -2.25]) + ); + } + + #[test] + fn bad_hyperslabs_are_refused() { + let r = sample(); + let mk = |start: Vec, count: Vec| Hyperslab { + start, + count, + stride: None, + block: None, + }; + assert!( + r.read("grid", Some(&mk(vec![0], vec![1]))) + .unwrap_err() + .contains("dimensions") + ); + assert!( + r.read("grid", Some(&mk(vec![2, 0], vec![2, 1]))) + .unwrap_err() + .contains("exceeds") + ); + let overlap = Hyperslab { + start: vec![0, 0], + count: vec![2, 1], + stride: Some(vec![1, 1]), + block: Some(vec![2, 1]), + }; + assert!( + r.read("grid", Some(&overlap)) + .unwrap_err() + .contains("overlap") + ); + } + + #[test] + fn attrs_are_sorted() { + let r = sample(); + let (attrs, errors) = r.attrs("/").unwrap(); + assert!(errors.is_empty()); + let names: Vec<&str> = attrs.iter().map(|a| a.name.as_str()).collect(); + assert_eq!(names, ["scale", "version"]); + let (g, _) = r.attrs("sensors").unwrap(); + assert!(matches!(&g[0].value, AttrValue::String(s) if s == "lab")); + } + + #[test] + fn compound_is_refused_not_reinterpreted() { + use clawhdf5::CompoundTypeBuilder; + let ct = CompoundTypeBuilder::new() + .f64_field("x") + .i32_field("n") + .build(); + let mut rec = Vec::new(); + rec.extend_from_slice(&1.0f64.to_le_bytes()); + rec.extend_from_slice(&7i32.to_le_bytes()); + let mut b = FileBuilder::new(); + b.create_dataset("table").with_compound_data(ct, rec, 1); + let r = Reader::open(b.finish().unwrap()).unwrap(); + let e = r.read("table", None).unwrap_err(); + assert!(e.contains("compound{x: f64, n: i32}"), "{e}"); + assert!(e.contains("not supported"), "{e}"); + } + + #[test] + fn garbage_is_an_error() { + assert!(Reader::open(vec![0u8; 64]).is_err()); + assert!(Reader::open(Vec::new()).is_err()); + } +} diff --git a/crates/clawhdf5-wasm/src/lib.rs b/crates/clawhdf5-wasm/src/lib.rs new file mode 100644 index 0000000..7c6b4a9 --- /dev/null +++ b/crates/clawhdf5-wasm/src/lib.rs @@ -0,0 +1,244 @@ +//! clawhdf5's HDF5 reader for JavaScript, via `wasm-bindgen`. +//! +//! ```js +//! import init, { open } from "./pkg/clawhdf5_wasm.js"; +//! await init(); +//! const file = open(new Uint8Array(await blob.arrayBuffer())); +//! file.list("/"); // [{ name, kind: "group" | "dataset" }] +//! file.info("/x"); // { shape, maxshape, dtype, elementShape } +//! file.attrs("/x"); // [{ name, value, dtype }] +//! file.read("/x"); // { shape, dtype, data: Float64Array | ... | string[] } +//! file.readHyperslab("/x", [0, 0], [10, 10]); // stride, block optional +//! file.free(); +//! ``` +//! +//! Numeric data comes back in the typed array of the stored width +//! (`Int16Array` for `i16`, `BigInt64Array` for `i64`, `Float32Array` for +//! `f32` and `f16`, ...); strings and enumeration names as arrays of strings. +//! Anything else is a thrown `Error` naming the datatype. Only the reader is +//! exposed: nothing here writes files. +//! +//! The logic lives in [`core`], which is plain Rust and tested natively. + +pub mod core; + +use clawhdf5::AttrValue; +use js_sys::{Array, Object, Reflect}; +use wasm_bindgen::prelude::*; + +use crate::core::{Data, Hyperslab, Reader}; + +/// JavaScript numbers are exact up to 2^53. +const MAX_SAFE_INTEGER: f64 = 9_007_199_254_740_991.0; + +fn js_err(msg: String) -> JsError { + JsError::new(&msg) +} + +fn set(obj: &Object, key: &str, value: impl Into) { + // Defining a property on a fresh plain object cannot fail. + Reflect::set(obj, &JsValue::from_str(key), &value.into()).unwrap_throw(); +} + +fn shape_to_js(shape: &[u64]) -> Array { + shape.iter().map(|&d| JsValue::from_f64(d as f64)).collect() +} + +fn indices_from_js(what: &str, v: &[f64]) -> Result, JsError> { + v.iter() + .map(|&x| { + if x.is_finite() && x >= 0.0 && x.fract() == 0.0 && x <= MAX_SAFE_INTEGER { + Ok(x as u64) + } else { + Err(js_err(format!( + "{what} must hold non-negative integers, got {x}" + ))) + } + }) + .collect() +} + +fn data_to_js(data: Data) -> JsValue { + match data { + Data::F32(v) => js_sys::Float32Array::from(&v[..]).into(), + Data::F64(v) => js_sys::Float64Array::from(&v[..]).into(), + Data::I8(v) => js_sys::Int8Array::from(&v[..]).into(), + Data::I16(v) => js_sys::Int16Array::from(&v[..]).into(), + Data::I32(v) => js_sys::Int32Array::from(&v[..]).into(), + Data::I64(v) => js_sys::BigInt64Array::from(&v[..]).into(), + Data::U8(v) => js_sys::Uint8Array::from(&v[..]).into(), + Data::U16(v) => js_sys::Uint16Array::from(&v[..]).into(), + Data::U32(v) => js_sys::Uint32Array::from(&v[..]).into(), + Data::U64(v) => js_sys::BigUint64Array::from(&v[..]).into(), + Data::Strings(v) => v + .into_iter() + .map(|s| JsValue::from_str(&s)) + .collect::() + .into(), + } +} + +/// A scalar integer as a `number` when exact, else a `bigint`. +fn int_to_js(x: i128) -> JsValue { + if (x as f64).abs() <= MAX_SAFE_INTEGER { + JsValue::from_f64(x as f64) + } else if let Ok(v) = i64::try_from(x) { + JsValue::from(v) + } else { + JsValue::from(x as u64) + } +} + +/// An attribute value, and the datatype of one that has no JavaScript form. +fn attr_to_js(value: AttrValue) -> (JsValue, Option) { + match value { + AttrValue::F64(x) => (JsValue::from_f64(x), None), + AttrValue::F64Array(v) => (js_sys::Float64Array::from(&v[..]).into(), None), + AttrValue::I64(x) => (int_to_js(i128::from(x)), None), + AttrValue::I64Array(v) => (js_sys::BigInt64Array::from(&v[..]).into(), None), + AttrValue::U64(x) => (int_to_js(i128::from(x)), None), + AttrValue::U64Array(v) => (js_sys::BigUint64Array::from(&v[..]).into(), None), + AttrValue::String(s) => (JsValue::from_str(&s), None), + AttrValue::StringArray(v) => ( + v.iter() + .map(|s| JsValue::from_str(s)) + .collect::() + .into(), + None, + ), + AttrValue::Raw { datatype, .. } => (JsValue::NULL, Some(core::describe(&datatype))), + } +} + +/// An open HDF5 (or NetCDF-4) file. +#[wasm_bindgen] +pub struct H5File { + inner: Reader, +} + +/// Open a file from its bytes. Throws if they are not an HDF5 file. +#[wasm_bindgen] +pub fn open(bytes: Vec) -> Result { + H5File::new(bytes) +} + +/// The clawhdf5 version this module was built from. +#[wasm_bindgen] +pub fn version() -> String { + env!("CARGO_PKG_VERSION").to_string() +} + +#[wasm_bindgen] +impl H5File { + /// Same as [`open`]. + #[wasm_bindgen(constructor)] + pub fn new(bytes: Vec) -> Result { + Ok(H5File { + inner: Reader::open(bytes).map_err(js_err)?, + }) + } + + /// `"group"` or `"dataset"`. + pub fn kind(&self, path: &str) -> Result { + Ok(self.inner.kind(path).map_err(js_err)?.as_str().to_string()) + } + + /// The group's members: `[{ name, kind }]`, groups first. + pub fn list(&self, path: &str) -> Result { + Ok(self + .inner + .list(path) + .map_err(js_err)? + .into_iter() + .map(|c| { + let o = Object::new(); + set(&o, "name", c.name); + set(&o, "kind", c.kind.as_str()); + JsValue::from(o) + }) + .collect()) + } + + /// `{ shape, maxshape, dtype, elementShape }`. `maxshape` is `null` + /// when not recorded, with `null` for each unlimited dimension. + pub fn info(&self, path: &str) -> Result { + let i = self.inner.info(path).map_err(js_err)?; + let o = Object::new(); + set(&o, "shape", shape_to_js(&i.shape)); + let max: JsValue = match i.maxshape { + None => JsValue::NULL, + Some(dims) => dims + .into_iter() + .map(|d| d.map_or(JsValue::NULL, |d| JsValue::from_f64(d as f64))) + .collect::() + .into(), + }; + set(&o, "maxshape", max); + set(&o, "dtype", i.dtype); + set(&o, "elementShape", shape_to_js(&i.element_shape)); + Ok(o) + } + + /// `[{ name, value, dtype }]`, sorted by name. Scalars are `number` + /// (`bigint` beyond 2^53) or `string`; arrays are typed arrays or + /// `string[]`. An attribute with no JavaScript form has `value: null` + /// and its `dtype`; one that could not be read at all is reported by + /// [`attrErrors`](Self::attr_errors). + pub fn attrs(&self, path: &str) -> Result { + let (attrs, _) = self.inner.attrs(path).map_err(js_err)?; + Ok(attrs + .into_iter() + .map(|a| { + let o = Object::new(); + set(&o, "name", a.name); + let (value, dtype) = attr_to_js(a.value); + set(&o, "value", value); + set(&o, "dtype", dtype.map_or(JsValue::NULL, JsValue::from)); + JsValue::from(o) + }) + .collect()) + } + + /// Messages for attributes that could not be read. + #[wasm_bindgen(js_name = attrErrors)] + pub fn attr_errors(&self, path: &str) -> Result { + let (_, errors) = self.inner.attrs(path).map_err(js_err)?; + Ok(errors.into_iter().map(JsValue::from).collect()) + } + + /// The whole dataset: `{ shape, dtype, data }`, `data` in row-major + /// order. + pub fn read(&self, path: &str) -> Result { + self.read_impl(path, None) + } + + /// A regular hyperslab (`H5Sselect_hyperslab`): `stride` and `block` + /// default to 1. The result's shape is `count * block` per dimension. + #[wasm_bindgen(js_name = readHyperslab)] + pub fn read_hyperslab( + &self, + path: &str, + start: Vec, + count: Vec, + stride: Option>, + block: Option>, + ) -> Result { + let slab = Hyperslab { + start: indices_from_js("start", &start)?, + count: indices_from_js("count", &count)?, + stride: stride.map(|s| indices_from_js("stride", &s)).transpose()?, + block: block.map(|b| indices_from_js("block", &b)).transpose()?, + }; + self.read_impl(path, Some(&slab)) + } + + fn read_impl(&self, path: &str, slab: Option<&Hyperslab>) -> Result { + let dtype = self.inner.info(path).map_err(js_err)?.dtype; + let a = self.inner.read(path, slab).map_err(js_err)?; + let o = Object::new(); + set(&o, "shape", shape_to_js(&a.shape)); + set(&o, "dtype", dtype); + set(&o, "data", data_to_js(a.data)); + Ok(o) + } +} diff --git a/crates/clawhdf5-wasm/tests/h5py_interop.rs b/crates/clawhdf5-wasm/tests/h5py_interop.rs new file mode 100644 index 0000000..7ef60ab --- /dev/null +++ b/crates/clawhdf5-wasm/tests/h5py_interop.rs @@ -0,0 +1,265 @@ +//! The wasm reader's core against files written by h5py and netCDF4, with +//! the values libhdf5 reads back as the reference. The generator, +//! `examples/wasm-viewer/test/make_fixture.py`, is shared with the Node test +//! of the built wasm package, so both compare against the same expectations. +//! +//! Skipped when python3 with h5py/netCDF4 is missing, unless +//! `CLAWHDF5_REQUIRE_INTEROP=1`. `CLAWHDF5_PYTHON` names the interpreter. + +use std::path::{Path, PathBuf}; +use std::process::Command; + +use clawhdf5::AttrValue; +use clawhdf5_wasm::core::{Data, Hyperslab, Kind, Reader}; +use serde_json::Value; + +fn python() -> String { + std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string()) +} + +fn interop_required() -> bool { + std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1") +} + +fn python_available() -> bool { + Command::new(python()) + .args(["-c", "import h5py, netCDF4, numpy"]) + .output() + .map(|o| o.status.success()) + .unwrap_or(false) +} + +fn generator() -> PathBuf { + Path::new(env!("CARGO_MANIFEST_DIR")).join("../../examples/wasm-viewer/test/make_fixture.py") +} + +/// Values as comparable strings: integers exactly, floats by their f64 +/// value (an f32 widens exactly), strings as themselves. +fn data_strings(d: &Data) -> (&'static str, Vec) { + fn s(v: &[T]) -> Vec { + v.iter().map(ToString::to_string).collect() + } + fn f>(v: &[T]) -> Vec { + v.iter().map(|&x| format!("{:?}", x.into())).collect() + } + match d { + Data::F32(v) => ("f32", f(v)), + Data::F64(v) => ("f64", f(v)), + Data::I8(v) => ("i8", s(v)), + Data::I16(v) => ("i16", s(v)), + Data::I32(v) => ("i32", s(v)), + Data::I64(v) => ("i64", s(v)), + Data::U8(v) => ("u8", s(v)), + Data::U16(v) => ("u16", s(v)), + Data::U32(v) => ("u32", s(v)), + Data::U64(v) => ("u64", s(v)), + Data::Strings(v) => ("strings", v.clone()), + } +} + +fn expected_strings(kind: &str, values: &Value) -> Vec { + values + .as_array() + .unwrap() + .iter() + .map(|v| match (kind, v) { + ("f32" | "f64", Value::Number(n)) => format!("{:?}", n.as_f64().unwrap()), + (_, Value::String(s)) => s.clone(), + other => panic!("unexpected expected value {other:?}"), + }) + .collect() +} + +fn shape(v: &Value) -> Vec { + v.as_array() + .unwrap() + .iter() + .map(|x| x.as_u64().unwrap()) + .collect() +} + +fn check_attr(file: &str, path: &str, name: &str, got: &AttrValue, want: &Value) { + let ctx = format!("{file}:{path}@{name}"); + let ints = |v: &Value| -> Vec { + v.as_array() + .unwrap() + .iter() + .map(|x| x.as_str().unwrap().to_string()) + .collect() + }; + let scalar = want["scalar"].as_bool().unwrap_or(false); + match got { + AttrValue::String(s) => assert_eq!(want["string"].as_str(), Some(s.as_str()), "{ctx}"), + AttrValue::StringArray(v) => { + let w: Vec<&str> = want["strings"] + .as_array() + .unwrap_or_else(|| panic!("{ctx}: got {v:?}")) + .iter() + .map(|x| x.as_str().unwrap()) + .collect(); + assert_eq!(v, &w, "{ctx}"); + } + AttrValue::I64(x) => { + assert!(scalar, "{ctx}"); + assert_eq!(vec![x.to_string()], ints(&want["int"]), "{ctx}"); + } + AttrValue::U64(x) => { + assert!(scalar, "{ctx}"); + assert_eq!(vec![x.to_string()], ints(&want["int"]), "{ctx}"); + } + AttrValue::I64Array(v) => assert_eq!( + v.iter().map(ToString::to_string).collect::>(), + ints(&want["int"]), + "{ctx}" + ), + AttrValue::U64Array(v) => assert_eq!( + v.iter().map(ToString::to_string).collect::>(), + ints(&want["int"]), + "{ctx}" + ), + AttrValue::F64(x) => { + assert!(scalar, "{ctx}"); + assert_eq!(Some(*x), want["float"][0].as_f64(), "{ctx}"); + } + AttrValue::F64Array(v) => { + let w: Vec = want["float"] + .as_array() + .unwrap() + .iter() + .map(|x| x.as_f64().unwrap()) + .collect(); + assert_eq!(v, &w, "{ctx}"); + } + AttrValue::Raw { datatype, .. } => { + let want = want["raw"] + .as_str() + .unwrap_or_else(|| panic!("{ctx}: undecoded {datatype:?}")); + let d = clawhdf5_wasm::core::describe(datatype); + assert!(d.contains(want), "{ctx}: {d}"); + } + } +} + +fn check_file(dir: &Path, file: &str, exp: &Value) { + let r = Reader::open(std::fs::read(dir.join(file)).unwrap()).unwrap(); + + for (path, want) in exp["lists"].as_object().unwrap() { + let list = r + .list(path) + .unwrap_or_else(|e| panic!("{file}:{path}: {e}")); + for (kind, key) in [(Kind::Group, "groups"), (Kind::Dataset, "datasets")] { + let mut got: Vec<&str> = list + .iter() + .filter(|c| c.kind == kind) + .map(|c| c.name.as_str()) + .collect(); + got.sort(); + let w: Vec<&str> = want[key] + .as_array() + .unwrap() + .iter() + .map(|x| x.as_str().unwrap()) + .collect(); + assert_eq!(got, w, "{file}:{path} {key}"); + } + } + + for (path, want) in exp["datasets"].as_object().unwrap() { + let kind = want["kind"].as_str().unwrap(); + let a = match (r.read(path, None), want["unavailable"].as_str()) { + (Ok(a), _) => a, + // A filter this build may lack (zstd: the wasm build has it + // off, a workspace build may unify it on) must fail clearly. + (Err(e), Some(why)) => { + assert!(e.contains(why), "{file}:{path}: {e}"); + continue; + } + (Err(e), None) => panic!("{file}:{path}: {e}"), + }; + assert_eq!(a.shape, shape(&want["shape"]), "{file}:{path} shape"); + let (got_kind, got) = data_strings(&a.data); + assert_eq!(got_kind, kind, "{file}:{path} kind"); + assert_eq!( + got, + expected_strings(kind, &want["values"]), + "{file}:{path}" + ); + + if let Some(slab) = want.get("slab") { + let h = Hyperslab { + start: shape(&slab["start"]), + count: shape(&slab["count"]), + stride: Some(shape(&slab["stride"])), + block: None, + }; + let a = r + .read(path, Some(&h)) + .unwrap_or_else(|e| panic!("{file}:{path} {h:?}: {e}")); + assert_eq!(a.shape, shape(&slab["shape"]), "{file}:{path} slab shape"); + assert_eq!( + data_strings(&a.data).1, + expected_strings(kind, &slab["values"]), + "{file}:{path} slab" + ); + } + } + + for (path, what) in exp["errors"].as_object().unwrap() { + let e = r.read(path, None).expect_err(path); + assert!(e.contains(what.as_str().unwrap()), "{file}:{path}: {e}"); + } + + let skip: Vec<&str> = exp["skip_attrs"] + .as_array() + .unwrap() + .iter() + .map(|x| x.as_str().unwrap()) + .collect(); + for (path, want) in exp["attrs"].as_object().unwrap() { + let (attrs, errors) = r + .attrs(path) + .unwrap_or_else(|e| panic!("{file}:{path}: {e}")); + assert!(errors.is_empty(), "{file}:{path}: {errors:?}"); + let want = want.as_object().unwrap(); + let mut compared = 0; + for a in &attrs { + if a.name.starts_with('_') || skip.contains(&a.name.as_str()) { + continue; + } + let w = want + .get(&a.name) + .unwrap_or_else(|| panic!("{file}:{path}: unexpected attribute {}", a.name)); + check_attr(file, path, &a.name, &a.value, w); + compared += 1; + } + assert_eq!(compared, want.len(), "{file}:{path}: attributes missing"); + } +} + +#[test] +fn reads_what_h5py_and_netcdf4_wrote() { + if !python_available() { + assert!( + !interop_required(), + "CLAWHDF5_REQUIRE_INTEROP=1 but python with h5py, netCDF4 and numpy is not available" + ); + eprintln!("SKIP: python with h5py/netCDF4 not available"); + return; + } + let dir = tempfile::tempdir().unwrap(); + let out = Command::new(python()) + .arg(generator()) + .arg(dir.path()) + .output() + .expect("run python"); + assert!( + out.status.success(), + "fixture generator failed:\n{}", + String::from_utf8_lossy(&out.stderr) + ); + let exp: Value = + serde_json::from_slice(&std::fs::read(dir.path().join("expected.json")).unwrap()).unwrap(); + for file in ["fixture.h5", "fixture.nc"] { + check_file(dir.path(), file, &exp[file]); + } +} diff --git a/crates/clawhdf5/Cargo.toml b/crates/clawhdf5/Cargo.toml index 2e2b4b4..a6401fe 100644 --- a/crates/clawhdf5/Cargo.toml +++ b/crates/clawhdf5/Cargo.toml @@ -31,7 +31,7 @@ name = "parallel_bench" harness = false [features] -default = ["mmap", "provenance"] +default = ["mmap", "provenance", "lzf"] mmap = ["clawhdf5-io/mmap"] parallel = ["clawhdf5-format/parallel", "rayon"] # zlib-ng (C, needs cmake) instead of the default pure-Rust zlib-rs. @@ -41,6 +41,14 @@ zstd = ["clawhdf5-format/zstd"] blake3_hash = ["clawhdf5-format/blake3_hash"] lz4 = ["clawhdf5-format/lz4"] pcodec = ["clawhdf5-format/pcodec"] +# Plugin filters, pure Rust (no C). LZF (32000) is h5py's built-in +# compression; it has no dependencies, so it is on by default. +lzf = ["clawhdf5-format/lzf"] +bitshuffle = ["clawhdf5-format/bitshuffle"] +bzip2 = ["clawhdf5-format/bzip2"] +blosc = ["clawhdf5-format/blosc"] +# Every plugin filter. +plugin-filters = ["lzf", "bitshuffle", "bzip2", "blosc"] # Dataset::verify_provenance() — recompute a dataset's SHA-256 and compare # against its stored _provenance_sha256 attribute. On by default, matching # clawhdf5-format's own default-on `provenance` feature. diff --git a/crates/clawhdf5/src/lazy.rs b/crates/clawhdf5/src/lazy.rs index 75ff609..d893485 100644 --- a/crates/clawhdf5/src/lazy.rs +++ b/crates/clawhdf5/src/lazy.rs @@ -43,6 +43,8 @@ pub struct LazyFile { /// Offset of the superblock in the file (the user-block size); every /// HDF5 address is relative to it. base: usize, + /// End of the HDF5 data (`Superblock::data_end`, absolute). + end: usize, superblock: Superblock, root_header: ObjectHeader, /// Cache of parsed object headers, keyed by address. @@ -74,9 +76,13 @@ impl LazyFile { /// /// Parses only the superblock and root group object header. pub fn open(reader: R) -> Result { + let whole_len = reader.as_bytes().len() as u64; let (user_block, data) = signature::split_user_block(reader.as_bytes())?; let base = user_block.len(); let superblock = Superblock::parse(data, 0)?; + // Refuse a truncated file; read nothing past the recorded end of file. + let end = base + superblock.data_end(base as u64, whole_len)? as usize; + let data = &reader.as_bytes()[base..end]; let root_header = ObjectHeader::parse( data, superblock.root_group_address as usize, @@ -86,6 +92,7 @@ impl LazyFile { Ok(Self { reader, base, + end, superblock, root_header, header_cache: RefCell::new(HashMap::new()), @@ -104,7 +111,7 @@ impl LazyFile { } fn hdf5_bytes(&self) -> &[u8] { - &self.reader.as_bytes()[self.base..] + &self.reader.as_bytes()[self.base..self.end] } /// Returns a reference to the parsed superblock. @@ -479,7 +486,7 @@ impl<'f, R: HDF5Read> LazyDataset<'f, R> { fn datatype(&self) -> Result { let data = self.required_payload(MessageType::Datatype)?; - let (dt, _) = Datatype::parse(&data)?; + let (dt, _) = Datatype::parse_in_header(&data, self.header.version)?; Ok(dt) } diff --git a/crates/clawhdf5/src/lib.rs b/crates/clawhdf5/src/lib.rs index 40ae557..12e2f69 100644 --- a/crates/clawhdf5/src/lib.rs +++ b/crates/clawhdf5/src/lib.rs @@ -144,6 +144,35 @@ mod tests { assert_eq!(ds.dtype().unwrap(), DType::I32); } + #[test] + fn dataset_raw_datatype() { + use clawhdf5_format::datatype::Datatype; + let bytes = make_simple_file(); + let file = File::from_bytes(bytes).unwrap(); + let dt = file.dataset("counts").unwrap().raw_datatype().unwrap(); + assert!( + matches!( + dt, + Datatype::FixedPoint { + size: 4, + signed: true, + .. + } + ), + "{dt:?}" + ); + // The full type pairs with read_selection's bytes. + let raw = file + .dataset("counts") + .unwrap() + .read_selection(&Selection::All) + .unwrap(); + assert_eq!( + clawhdf5_format::data_read::read_as_i64(&raw, &dt).unwrap(), + vec![10, 20, 30] + ); + } + #[test] fn root_group_datasets() { let bytes = make_simple_file(); diff --git a/crates/clawhdf5/src/mmap_file.rs b/crates/clawhdf5/src/mmap_file.rs index eeb1efb..7f544ca 100644 --- a/crates/clawhdf5/src/mmap_file.rs +++ b/crates/clawhdf5/src/mmap_file.rs @@ -35,6 +35,8 @@ pub struct MmapFile { /// Offset of the superblock in the mapped file (the user-block size); /// every HDF5 address is relative to it. base: usize, + /// End of the HDF5 data (`Superblock::data_end`, absolute). + end: usize, superblock: Superblock, } @@ -42,12 +44,16 @@ impl MmapFile { /// Open an HDF5 file using memory-mapped I/O. pub fn open>(path: P) -> Result { let reader = MmapReader::open(path).map_err(Error::Io)?; + let whole_len = reader.as_bytes().len() as u64; let (user_block, data) = signature::split_user_block(reader.as_bytes())?; let base = user_block.len(); let superblock = Superblock::parse(data, 0)?; + // Refuse a truncated file; read nothing past the recorded end of file. + let end = base + superblock.data_end(base as u64, whole_len)? as usize; Ok(Self { reader, base, + end, superblock, }) } @@ -55,7 +61,7 @@ impl MmapFile { /// The file's bytes from the superblock on — the space HDF5 addresses /// index into. fn hdf5_bytes(&self) -> &[u8] { - &self.reader.as_bytes()[self.base..] + &self.reader.as_bytes()[self.base..self.end] } /// Size of the user block before the superblock (0 for most files). @@ -426,7 +432,7 @@ impl<'f> MmapDataset<'f> { fn datatype(&self) -> Result { let data = self.required_payload(MessageType::Datatype)?; - let (dt, _) = Datatype::parse(&data)?; + let (dt, _) = Datatype::parse_in_header(&data, self.header.version)?; Ok(dt) } diff --git a/crates/clawhdf5/src/reader.rs b/crates/clawhdf5/src/reader.rs index 65c0292..83cbfa8 100644 --- a/crates/clawhdf5/src/reader.rs +++ b/crates/clawhdf5/src/reader.rs @@ -45,26 +45,34 @@ impl Backing { } } -/// The file's bytes, viewed from the superblock on. A file may start with a -/// user block (the superblock at 512, 1024, …); every HDF5 address is -/// relative to the superblock, so all parsing goes through [`Self::as_bytes`]. +/// The file's bytes, viewed from the superblock on and up to the end of +/// file the superblock records. A file may start with a user block (the +/// superblock at 512, 1024, …); every HDF5 address is relative to the +/// superblock, so all parsing goes through [`Self::as_bytes`]. struct FileData { backing: Backing, /// Offset of the superblock in the file (the user-block size). base: usize, + /// End of the HDF5 data in the file (`Superblock::data_end`, absolute). + end: usize, } impl FileData { - /// Locate the superblock and parse it. + /// Locate the superblock and parse it. A truncated file is refused, and + /// bytes past the recorded end of file are not read, as in libhdf5. fn new(backing: Backing) -> Result<(Self, Superblock), Error> { - let (user_block, hdf5) = signature::split_user_block(backing.whole_file())?; + let whole = backing.whole_file(); + let (user_block, hdf5) = signature::split_user_block(whole)?; let base = user_block.len(); let superblock = Superblock::parse(hdf5, 0)?; - Ok((Self { backing, base }, superblock)) + let end = superblock.data_end(base as u64, whole.len() as u64)?; + // data_end is at most the file length (less the user block). + let end = base + end as usize; + Ok((Self { backing, base, end }, superblock)) } fn as_bytes(&self) -> &[u8] { - &self.backing.whole_file()[self.base..] + &self.backing.whole_file()[self.base..self.end] } fn len(&self) -> usize { @@ -416,6 +424,15 @@ impl<'f> Dataset<'f> { Ok(classify_datatype(&dt)) } + /// Returns the dataset's full datatype as stored in the file (byte order, + /// string padding, compound layout, ...), with a committed datatype + /// resolved. Use it with the `clawhdf5_format::data_read` converters on + /// the bytes [`read_selection`](Self::read_selection) returns, for types + /// the typed `read_*` methods do not cover. + pub fn raw_datatype(&self) -> Result { + self.datatype() + } + /// Read all data as `f64` values. pub fn read_f64(&self) -> Result, Error> { let dt = self.datatype()?; @@ -850,7 +867,7 @@ impl<'f> Dataset<'f> { fn datatype(&self) -> Result { let data = self.required_payload(MessageType::Datatype)?; - let (dt, _) = Datatype::parse(&data)?; + let (dt, _) = Datatype::parse_in_header(&data, self.header.version)?; Ok(dt) } diff --git a/crates/clawhdf5/tests/chunk_index_interop.rs b/crates/clawhdf5/tests/chunk_index_interop.rs index b72b7e2..3ff0420 100644 --- a/crates/clawhdf5/tests/chunk_index_interop.rs +++ b/crates/clawhdf5/tests/chunk_index_interop.rs @@ -515,6 +515,31 @@ fn we_write_maxshape_larger_than_shape() { check_we_write(&cases); } +/// A version-4 layout must encode every chunk dimension in the fewest bytes +/// that hold the largest one, as libhdf5 does: HDF5 2.0.0 (h5py 3.16) +/// refuses a wider encoding ("stored chunk dimension encoding length does +/// not match value calculated from chunk dimensions"). We rounded 3 bytes +/// up to 4, so h5py could not open any dataset we wrote with a chunk +/// dimension from 65 536 to 16 777 215. +#[test] +fn we_write_chunk_dimensions_in_the_fewest_bytes() { + const U: u64 = u64::MAX; + let mut cases = vec![ + // Single chunk, Fixed Array, Extensible Array, v2 B-tree. + wcase("single_70000", &[70_000], &[70_000], None), + wcase("fa_70000", &[140_000], &[70_000], None), + wcase("ea_70000", &[140_000], &[70_000], Some(&[U])), + wcase("bt2_70000", &[2, 70_000], &[1, 70_000], Some(&[U, U])), + // 2 bytes and 1 byte still, with the element size (4) the largest. + wcase("fa_300", &[600], &[300], None), + wcase("fa_3", &[6], &[3], None), + ]; + let mut filtered = wcase("single_70000_deflate", &[70_000], &[70_000], None); + filtered.deflate = true; + cases.push(filtered); + check_we_write(&cases); +} + /// More than one unlimited dimension needs a version-2 B-tree chunk index, /// as the library uses; an Extensible Array for `(None, None)` made libhdf5 /// refuse the whole file ("already found unlimited dimension"). diff --git a/crates/clawhdf5/tests/fixtures/written_by_v2_7_0.h5 b/crates/clawhdf5/tests/fixtures/written_by_v2_7_0.h5 new file mode 100644 index 0000000..ebb7cf1 Binary files /dev/null and b/crates/clawhdf5/tests/fixtures/written_by_v2_7_0.h5 differ diff --git a/crates/clawhdf5/tests/fixtures/written_by_v2_7_0_paged.h5 b/crates/clawhdf5/tests/fixtures/written_by_v2_7_0_paged.h5 new file mode 100644 index 0000000..8f83714 Binary files /dev/null and b/crates/clawhdf5/tests/fixtures/written_by_v2_7_0_paged.h5 differ diff --git a/crates/clawhdf5/tests/h5py_chunked_read_tests.rs b/crates/clawhdf5/tests/h5py_chunked_read_tests.rs index 2b9e902..467931e 100644 --- a/crates/clawhdf5/tests/h5py_chunked_read_tests.rs +++ b/crates/clawhdf5/tests/h5py_chunked_read_tests.rs @@ -343,3 +343,115 @@ print("OK") let want: Vec = line[990..].iter().flat_map(|v| v.to_le_bytes()).collect(); assert_eq!(tail, want); } + +// --------------------------------------------------------------------------- +// Chunk dimension widths in layout version 4 +// --------------------------------------------------------------------------- + +/// A version-4 layout stores every chunk dimension in the fewest bytes that +/// hold the largest one (the element size included). A chunk dimension of +/// 70 000 takes 3 bytes and 2^32 + 1 elements would take 5; widths other +/// than 1, 2, 4 and 8 were refused, so these h5py files did not open. +#[test] +fn h5py_layout_v4_chunk_dimensions_of_3_bytes_read() { + skip_if_no_python!(); + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("enc3.h5"); + let p = path.display().to_string(); + run_python(&format!( + r#" +import h5py, numpy as np +with h5py.File("{p}", "w", libver="latest") as f: + f.create_dataset("single", data=np.arange(70000, dtype=" 0 or raw.find(bytes([4, 2, 1, 2, 3])) > 0 +"# + )); + let file = File::open(&path).unwrap(); + let single = file.dataset("single").unwrap().read_u64().unwrap(); + assert!(single.iter().copied().eq(0..70000), "single"); + let ea = file.dataset("ea").unwrap().read_f64().unwrap(); + assert_eq!(ea, (0..10).map(f64::from).collect::>()); + let fa = file.dataset("fa").unwrap().read_i32().unwrap(); + assert!(fa.iter().copied().eq(0..3 * 70000), "fa"); +} + +// --------------------------------------------------------------------------- +// Chunks that decode short +// --------------------------------------------------------------------------- + +/// HDF5 stores every chunk at the full chunk size, so a chunk whose filters +/// decode to fewer bytes is corrupt. libhdf5 returns the rest of such a +/// chunk uninitialised (or, for filters that check, fails); clawhdf5 padded +/// it with zeros and returned it as data. Every read path must fail, +/// naming the chunk, and chunks that decode fully must still read. +#[test] +fn h5py_short_decoded_chunk_is_an_error() { + skip_if_no_python!(); + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("short.h5"); + let p = path.display().to_string(); + run_python(&format!( + r#" +import h5py, numpy as np, zlib +with h5py.File("{p}", "w") as f: + # 1-D, 4 gzip chunks of 8 i4; chunk 2 (offset 16) inflates to 16 bytes. + ds = f.create_dataset("line", shape=(32,), chunks=(8,), dtype=" = (1..=16i32).flat_map(i32::to_le_bytes).collect(); + assert_eq!(line.read_selection(&hyperslab(0, 16)).unwrap(), want); + let many = file.dataset("many").unwrap(); + assert!(many.read_selection(&hyperslab(190, 20)).is_err()); + + // The memory-mapped and lazy readers. + let mm = clawhdf5::MmapFile::open(&path).unwrap(); + assert!(mm.dataset("line").unwrap().read_i32().is_err()); + assert!(mm.dataset("many").unwrap().read_i32().is_err()); + let lazy = clawhdf5::LazyFile::open_mmap(&path).unwrap(); + assert!(lazy.dataset("line").unwrap().read_i32().is_err()); + assert!(lazy.dataset("grid").unwrap().read_f64().is_err()); +} diff --git a/crates/clawhdf5/tests/header_validation_interop.rs b/crates/clawhdf5/tests/header_validation_interop.rs new file mode 100644 index 0000000..2f52d27 --- /dev/null +++ b/crates/clawhdf5/tests/header_validation_interop.rs @@ -0,0 +1,510 @@ +//! Corrupt files that libhdf5 refuses must be refused here too, not read. +//! +//! h5py writes a valid file, the script corrupts a copy the way a damaged or +//! malicious file would be, and records whether h5py (libhdf5) still opens +//! and reads the object. clawhdf5 must agree: read the valid file, refuse +//! each corrupt one. Skipped when python3 with h5py is unavailable, unless +//! `CLAWHDF5_REQUIRE_INTEROP=1`. + +use std::path::Path; +use std::process::Command; + +use clawhdf5::File; + +fn python() -> String { + std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string()) +} + +fn interop_required() -> bool { + std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1") +} + +fn python_available() -> bool { + Command::new(python()) + .args(["-c", "import h5py, numpy"]) + .output() + .map(|o| o.status.success()) + .unwrap_or(false) +} + +macro_rules! skip_if_no_python { + () => { + if !python_available() { + assert!( + !interop_required(), + "CLAWHDF5_REQUIRE_INTEROP=1 but python3 with h5py is not available" + ); + eprintln!("SKIP: python3 with h5py not available"); + return; + } + }; +} + +/// Runs `body` (Python, with `h5py`, `numpy as np`, `struct` imported and +/// `d` the output directory) and then, for every `NAME.h5` it wrote, +/// prints `NAME ok` when h5py opens and reads dataset `d` and `NAME ERROR` +/// otherwise. Returns those lines, sorted. +fn h5py_verdicts(dir: &Path, body: &str) -> Vec { + let script = format!( + r#" +import h5py, numpy as np, struct, os, glob +d = "{dir}" +{body} +for path in sorted(glob.glob(os.path.join(d, "*.h5"))): + name = os.path.basename(path)[:-3] + try: + with h5py.File(path, "r") as f: + f["d"][()] + print(name, "ok") + except Exception: + print(name, "ERROR") +"#, + dir = dir.display() + ); + let out = Command::new(python()) + .args(["-c", &script]) + .output() + .expect("failed to run python"); + assert!( + out.status.success(), + "python failed:\n{}", + String::from_utf8_lossy(&out.stderr) + ); + let mut lines: Vec = String::from_utf8_lossy(&out.stdout) + .lines() + .map(str::to_owned) + .collect(); + lines.sort(); + lines +} + +/// Whether clawhdf5 opens and reads dataset `d` of `path` (as raw bytes of +/// whatever type it has). +fn clawhdf5_reads(path: &Path) -> Result<(), String> { + let file = File::open(path).map_err(|e| format!("open: {e}"))?; + let ds = file.dataset("d").map_err(|e| format!("dataset: {e}"))?; + ds.dtype().map_err(|e| format!("dtype: {e}"))?; + ds.shape().map_err(|e| format!("shape: {e}"))?; + file.read_multi(&["d"]) + .map(|_| ()) + .map_err(|e| format!("read: {e}")) +} + +/// h5py's verdict for each file must be `expected`, and clawhdf5 must read +/// exactly the files h5py reads. +fn assert_agrees_with_h5py(dir: &Path, verdicts: &[String], expected: &[&str]) { + assert_eq!(verdicts, expected, "h5py's view changed"); + for line in verdicts { + let (name, verdict) = line.split_once(' ').unwrap(); + let ours = clawhdf5_reads(&dir.join(format!("{name}.h5"))); + match verdict { + "ok" => assert!(ours.is_ok(), "{name}: h5py reads it, we fail: {ours:?}"), + _ => assert!(ours.is_err(), "{name}: h5py refuses it, we read it"), + } + } +} + +#[test] +fn chunk_dimensions_libhdf5_refuses_are_refused() { + skip_if_no_python!(); + let dir = tempfile::tempdir().unwrap(); + // A chunked int32 dataset (chunk 37, element size 4 after it in the + // layout message), with libver earliest (layout v3) and latest (v4). + // The corrupt copies set the chunk dimension to 0 (which read as all + // fill values), to 0x80000000 (an 8 GiB chunk) and to 38, which the + // chunk index's offsets 37 and 74 are not multiples of (libhdf5: "bad + // coordinate offset"; the chunks were read at the wrong place). A v4 + // layout indexes chunks by position, not offset, but both libraries + // refuse the changed chunk grid there too. + let verdicts = h5py_verdicts( + dir.path(), + r#" +for libver in ("earliest", "latest"): + good = os.path.join(d, f"{libver}_good.h5") + with h5py.File(good, "w", libver=libver) as f: + f.create_dataset("d", data=np.arange(100, dtype=" 0 + for name, value in (("zero", 0), ("huge", 0x80000000), ("offgrid", 38)): + bad = bytearray(data) + if value >= 1 << (8 * width): + continue + bad[at:at + width] = value.to_bytes(width, "little") + open(os.path.join(d, f"{libver}_{name}.h5"), "wb").write(bad) +"#, + ); + assert_agrees_with_h5py( + dir.path(), + &verdicts, + &[ + "earliest_good ok", + "earliest_huge ERROR", + "earliest_offgrid ERROR", + "earliest_zero ERROR", + "latest_good ok", + "latest_offgrid ERROR", + "latest_zero ERROR", + ], + ); +} + +/// Python: `fix_ohdr(buf, off)` recomputes the Jenkins lookup3 checksum of +/// the version-2 object header chunk 0 at `off`, so a field can be changed +/// in a `libver="latest"` file without the checksum failing first. +const FIX_OHDR_PY: &str = r#" +def _rot(x, k): + return ((x << k) | (x >> (32 - k))) & 0xFFFFFFFF +def lookup3(data): + M = 0xFFFFFFFF + n = len(data); a = b = c = (0xDEADBEEF + n) & M; i = 0 + w = lambda j: int.from_bytes(data[j:j + 4], "little") + while n > 12: + a = (a + w(i)) & M; b = (b + w(i + 4)) & M; c = (c + w(i + 8)) & M + a = (a - c) & M; a ^= _rot(c, 4); c = (c + b) & M + b = (b - a) & M; b ^= _rot(a, 6); a = (a + c) & M + c = (c - b) & M; c ^= _rot(b, 8); b = (b + a) & M + a = (a - c) & M; a ^= _rot(c, 16); c = (c + b) & M + b = (b - a) & M; b ^= _rot(a, 19); a = (a + c) & M + c = (c - b) & M; c ^= _rot(b, 4); b = (b + a) & M + n -= 12; i += 12 + if n == 0: + return c + t = bytes(data[i:]) + bytes(12) + w = lambda j: int.from_bytes(t[j:j + 4], "little") + a = (a + w(0)) & M; b = (b + w(4)) & M; c = (c + w(8)) & M + c ^= b; c = (c - _rot(b, 14)) & M + a ^= c; a = (a - _rot(c, 11)) & M + b ^= a; b = (b - _rot(a, 25)) & M + c ^= b; c = (c - _rot(b, 16)) & M + a ^= c; a = (a - _rot(c, 4)) & M + b ^= a; b = (b - _rot(a, 14)) & M + c ^= b; c = (c - _rot(b, 24)) & M + return c +def fix_ohdr(buf, off): + assert buf[off:off + 4] == b"OHDR" + flags = buf[off + 5]; p = off + 6 + if flags & 0x20: p += 16 + if flags & 0x10: p += 4 + width = 1 << (flags & 3) + end = p + width + int.from_bytes(buf[p:p + width], "little") + buf[end:end + 4] = lookup3(bytes(buf[off:end])).to_bytes(4, "little") +"#; + +/// A chunked layout records the element size as its last dimension; libhdf5 +/// refuses a dataset whose datatype has another size ("stored datatype size +/// in chunk layout does not match datatype description"). This read the +/// chunks laid out with the wrong element size. +#[test] +fn chunk_layout_element_size_must_match_the_datatype() { + skip_if_no_python!(); + let dir = tempfile::tempdir().unwrap(); + let body = format!( + "{FIX_OHDR_PY}{}", + r#" +for libver in ("earliest", "latest"): + good = os.path.join(d, f"{libver}_good.h5") + with h5py.File(good, "w", libver=libver) as f: + f.create_dataset("d", data=np.arange(100, dtype=" 6 + for size in (2, 8): + bad = bytearray(data) + bad[at:at + width] = size.to_bytes(width, "little") + if libver == "latest": + fix_ohdr(bad, bad.rfind(b"OHDR", 0, at)) + open(os.path.join(d, f"{libver}_size{size}.h5"), "wb").write(bad) +"# + ); + let verdicts = h5py_verdicts(dir.path(), &body); + assert_agrees_with_h5py( + dir.path(), + &verdicts, + &[ + "earliest_good ok", + "earliest_size2 ERROR", + "earliest_size8 ERROR", + "latest_good ok", + "latest_size2 ERROR", + "latest_size8 ERROR", + ], + ); + for name in ["earliest_size2", "latest_size8"] { + let err = clawhdf5_reads(&dir.path().join(format!("{name}.h5"))).unwrap_err(); + assert!( + err.contains("stored datatype size in chunk layout"), + "{name}: {err}" + ); + } +} + +/// A v2 object header message whose size runs past the messages into the +/// chunk's checksum is refused by libhdf5 whether it runs 1 byte or more +/// into the checksum (its loop stops at the checksum, and then the checksum +/// read and the size check fail), so it is refused here too. +#[test] +fn v2_header_message_running_into_the_checksum_is_refused() { + skip_if_no_python!(); + let dir = tempfile::tempdir().unwrap(); + let body = format!( + "{FIX_OHDR_PY}{}", + r#" +good = os.path.join(d, "good.h5") +with h5py.File(good, "w", libver="latest") as f: + f.create_dataset("d", data=np.arange(4, dtype=" 0 +bad = bytearray(data); bad[at + 4] |= 0x40 +save("layout_shareable", bad) +# A message size that is not a multiple of 8 in a v1 header. +bad = bytearray(data); bad[at + 2] = 23 +save("layout_unaligned", bad) + +# A compound whose second field repeats the first's name, or overlaps it. +dt = np.dtype([("aa", ">()); +} + +/// A variable-length compound member takes 4 + offset size + 4 bytes, which +/// is 12 in a file with 4-byte offsets, and libhdf5 checks for overlapping +/// members with that stored size. A member right after one was refused as +/// "member overlaps with previous member" (the check took the 16 bytes of +/// an 8-byte-offset file), and with it every attribute of the object. +#[test] +fn variable_length_compound_members_in_files_with_4_byte_offsets() { + skip_if_no_python!(); + let dir = tempfile::tempdir().unwrap(); + run_python( + dir.path(), + r#" +from h5py import h5f, h5p +dt = np.dtype([("s", h5py.string_dtype()), ("i", " PathBuf { + PathBuf::from(env!("CARGO_MANIFEST_DIR")) + .join("tests/fixtures") + .join(name) +} + +#[test] +fn every_object_of_a_v2_7_0_file_reads() { + let file = File::open(fixture("written_by_v2_7_0.h5")).unwrap(); + + let (attrs, errors) = file.root().attrs_with_errors().unwrap(); + assert!(errors.is_empty(), "{errors:?}"); + // (It reads as no strings, as it did before.) + assert!( + matches!(&attrs["empty"], AttrValue::StringArray(v) if v.iter().all(String::is_empty)), + "{:?}", + attrs["empty"] + ); + assert!(matches!(&attrs["title"], AttrValue::String(s) if s == "old")); + + let f32s = |name: &str| file.dataset(name).unwrap().read_f32().unwrap(); + assert_eq!(f32s("f32"), [1.0, 2.0, 3.0]); + assert_eq!( + f32s("f32_2d"), + (0..60).map(|x| x as f32).collect::>() + ); + assert_eq!( + f32s("chunked"), + (0..1000).map(|x| x as f32).collect::>() + ); + assert!(f32s("empty").is_empty()); + assert_eq!( + file.dataset("f64").unwrap().read_f64().unwrap(), + [1.0, 2.0, 3.0] + ); + assert_eq!(file.dataset("i32").unwrap().read_i32().unwrap(), [1, -2, 3]); + assert_eq!(file.dataset("i64").unwrap().read_i64().unwrap(), [1, -2, 3]); + assert_eq!(file.dataset("u64").unwrap().read_u64().unwrap(), [1, 2, 3]); + assert_eq!( + file.dataset("chunked_2d").unwrap().read_i32().unwrap(), + (0..600).collect::>() + ); + for name in ["unlimited", "maxshape"] { + assert_eq!( + file.dataset(name).unwrap().read_f64().unwrap(), + (0..100).map(|x| x as f64).collect::>(), + "{name}" + ); + } + assert_eq!( + file.dataset("compact").unwrap().read_i32().unwrap(), + [7, 8, 9] + ); + for name in ["u8", "compound", "enum", "enum8"] { + file.dataset(name) + .unwrap() + .dtype() + .unwrap_or_else(|e| panic!("{name}: {e}")); + } + + let grp = file.group("grp").unwrap(); + let (attrs, errors) = grp.attrs_with_errors().unwrap(); + assert!(errors.is_empty(), "{errors:?}"); + assert_eq!(attrs.len(), 21); + assert_eq!(grp.dataset("d").unwrap().read_f32().unwrap(), [4.0, 5.0]); + + let paged = File::open(fixture("written_by_v2_7_0_paged.h5")).unwrap(); + assert_eq!(paged.dataset("d").unwrap().read_f32().unwrap(), [1.0, 2.0]); +} + +/// Datatypes libhdf5 (and this reader) refuse — a compound with a repeated +/// field name or no fields, an enum member with an empty name — must not be +/// written: they made files that did not read back. Valid neighbours of each +/// still write and read back. +#[test] +fn datatypes_the_reader_refuses_are_not_written() { + use clawhdf5::{CompoundTypeBuilder, EnumTypeBuilder, FileBuilder}; + + let write_compound = |dt, raw: Vec| { + let mut b = FileBuilder::new(); + b.create_dataset("d").with_compound_data(dt, raw, 1); + b.finish() + }; + let write_enum = |dt| { + let mut b = FileBuilder::new(); + b.create_dataset("d").with_enum_u8_data(dt, &[0, 1]); + b.finish() + }; + let refused = |r: Result, clawhdf5::Error>, what: &str| { + let err = r.expect_err("written"); + let msg = err.to_string(); + assert!(msg.contains("datatype cannot be written"), "{msg}"); + assert!(msg.contains(what), "{msg}"); + }; + + let dup = CompoundTypeBuilder::new() + .f64_field("x") + .f64_field("x") + .build(); + refused( + write_compound(dup, vec![0; 16]), + "duplicated compound field name 'x'", + ); + refused( + write_compound(CompoundTypeBuilder::new().build(), vec![]), + "invalid", + ); + let empty_name = EnumTypeBuilder::u8_based() + .u8_value("A", 0) + .u8_value("", 1) + .build(); + refused(write_enum(empty_name), "0 length enum name"); + + let ok = CompoundTypeBuilder::new() + .f64_field("x") + .f64_field("y") + .build(); + let bytes = write_compound(ok, vec![0; 16]).unwrap(); + let file = File::from_bytes(bytes).unwrap(); + file.dataset("d").unwrap().dtype().unwrap(); + let ok = EnumTypeBuilder::u8_based() + .u8_value("A", 0) + .u8_value("B", 1) + .build(); + let file = File::from_bytes(write_enum(ok).unwrap()).unwrap(); + file.dataset("d").unwrap().dtype().unwrap(); +} diff --git a/crates/clawhdf5/tests/parallel_integration.rs b/crates/clawhdf5/tests/parallel_integration.rs index f950844..bdcf4f0 100644 --- a/crates/clawhdf5/tests/parallel_integration.rs +++ b/crates/clawhdf5/tests/parallel_integration.rs @@ -292,7 +292,8 @@ mod parallel_tests { } // Sequential decompression - let _sequential: Vec> = chunk_infos + #[cfg_attr(not(feature = "parallel"), allow(unused_variables))] + let sequential: Vec> = chunk_infos .iter() .map(|ci| { let addr = ci.address as usize; diff --git a/crates/clawhdf5/tests/plugin_filters_interop.rs b/crates/clawhdf5/tests/plugin_filters_interop.rs new file mode 100644 index 0000000..81cc87c --- /dev/null +++ b/crates/clawhdf5/tests/plugin_filters_interop.rs @@ -0,0 +1,469 @@ +//! Plugin filters (LZF, bitshuffle, bzip2, blosc) against libhdf5. +//! +//! Read direction: h5py (with hdf5plugin for everything but LZF, which h5py +//! ships) writes each filter over a matrix of dtypes (1-8 bytes, both byte +//! orders), 1-3 dimensional shapes whose chunks do not divide them (so the +//! edge chunks are partial), compressible and incompressible data, and the +//! filter's own options; every dataset has an unfiltered twin, and clawhdf5 +//! must read the filtered one byte for byte equal to it. +//! +//! Write direction: clawhdf5 writes with its encoder, and h5py must read the +//! values back. +//! +//! Skipped when python3 with h5py (and hdf5plugin) is unavailable, unless +//! `CLAWHDF5_REQUIRE_INTEROP=1`. +#![allow(dead_code)] + +use std::process::Command; + +use clawhdf5::File; +use clawhdf5_format::selection::Selection; + +fn python() -> String { + std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string()) +} + +fn interop_required() -> bool { + std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1") +} + +fn python_has(modules: &str) -> bool { + Command::new(python()) + .args(["-c", &format!("import {modules}")]) + .output() + .map(|o| o.status.success()) + .unwrap_or(false) +} + +/// Whether the interop test can run; panics instead of skipping when +/// `CLAWHDF5_REQUIRE_INTEROP=1`. +fn have_python(modules: &str) -> bool { + if python_has(modules) { + return true; + } + assert!( + !interop_required(), + "CLAWHDF5_REQUIRE_INTEROP=1 but python3 with {modules} is not available" + ); + eprintln!("SKIP: python3 with {modules} not available"); + false +} + +fn run_python(script: &str, args: &[&str]) -> String { + let output = Command::new(python()) + .arg("-c") + .arg(script) + .args(args) + .output() + .expect("failed to run python"); + if !output.status.success() { + panic!( + "Python script failed:\nSTDOUT: {}\nSTDERR: {}", + String::from_utf8_lossy(&output.stdout), + String::from_utf8_lossy(&output.stderr) + ); + } + String::from_utf8_lossy(&output.stdout).trim().to_string() +} + +/// Writes `f{i}` (filtered) and `r{i}` (unfiltered twin) for every filter +/// setting in `FILTERS` × every case; prints the number of pairs. +const GENERATE: &str = r#" +import sys +import numpy as np, h5py +try: + import hdf5plugin +except ImportError: + hdf5plugin = None +path = sys.argv[1] +FILTERS = eval('(' + sys.argv[2] + ')') +cases = [ + ('i4', (37, 53), (8, 8), 'ramp'), + (' 0); + let file = File::open(&path).unwrap(); + for i in 0..n { + let filtered = file.dataset(&format!("f{i}")).unwrap(); + let case = format!("{:?}", filtered.attrs().unwrap().get("case")); + let got = filtered + .read_selection(&Selection::All) + .unwrap_or_else(|e| panic!("{tag} f{i} {case}: {e}")); + let want = file + .dataset(&format!("r{i}")) + .unwrap() + .read_selection(&Selection::All) + .unwrap(); + assert!(got == want, "{tag} f{i} {case}: data differs"); + } +} + +/// Values the write-direction tests store: row-major, compressible with some +/// variation. +fn ramp_i32(n: usize) -> Vec { + (0..n).map(|i| ((i * 3) % 251) as i32 - 60).collect() +} +fn ramp_f64(n: usize) -> Vec { + (0..n).map(|i| ((i * 3) % 251) as f64 / 7.0).collect() +} +fn ramp_u8(n: usize) -> Vec { + (0..n).map(|i| ((i * 7) % 256) as u8).collect() +} + +/// h5py checks the datasets `write_ours` wrote against the same ramps. +const VERIFY: &str = r#" +import sys +import numpy as np, h5py +try: + import hdf5plugin +except ImportError: + pass +bad = [] +with h5py.File(sys.argv[1], 'r') as f: + for name in f: + ds = f[name] + n = ds.size + k = np.arange(n) + if ds.dtype == np.int32: + want = ((k * 3) % 251 - 60).astype(np.int32) + elif ds.dtype == np.float64: + want = ((k * 3) % 251) / 7.0 + else: + want = ((k * 7) % 256).astype(np.uint8) + got = ds[()].reshape(-1) + if not np.array_equal(got, want): + bad.append(name) + # The dataset really is filtered by the plugin under test. + if int(sys.argv[2]) not in [int(x) for x in ds._filters.keys() if x.isdigit()] \ + and sys.argv[3] not in ds._filters: + bad.append(name + ':filter-missing:' + repr(ds._filters)) +print('OK' if not bad else 'BAD ' + ' '.join(bad)) +"#; + +/// Write i32/f64/u8 datasets (1-D and 2-D, partial edge chunks) with +/// `configure` applying the filter, then have h5py read them back. +fn check_ours_read_by_h5py( + tag: &str, + filter_id: u16, + filter_name: &str, + configure: impl Fn(&mut clawhdf5_format::type_builders::DatasetBuilder), +) { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join(format!("{tag}_ours.h5")); + let mut fb = clawhdf5::FileBuilder::new(); + { + let ds = fb.create_dataset("i32_1d"); + ds.with_i32_data(&ramp_i32(10_000)).with_chunks(&[3000]); + configure(ds); + } + { + let ds = fb.create_dataset("f64_2d"); + ds.with_f64_data(&ramp_f64(37 * 53)) + .with_shape(&[37, 53]) + .with_chunks(&[10, 16]); + configure(ds); + } + { + let ds = fb.create_dataset("u8_1d"); + ds.with_u8_data(&ramp_u8(5000)).with_chunks(&[777]); + configure(ds); + } + { + let ds = fb.create_dataset("f64_big"); + ds.with_f64_data(&ramp_f64(100_000)).with_chunks(&[40_000]); + configure(ds); + } + fb.write(&path).unwrap(); + + // clawhdf5 reads its own output. + let file = File::open(&path).unwrap(); + assert_eq!( + file.dataset("i32_1d").unwrap().read_i32().unwrap(), + ramp_i32(10_000) + ); + assert_eq!( + file.dataset("f64_2d").unwrap().read_f64().unwrap(), + ramp_f64(37 * 53) + ); + + let out = run_python( + VERIFY, + &[path.to_str().unwrap(), &filter_id.to_string(), filter_name], + ); + assert_eq!(out, "OK", "{tag}: h5py could not read our output"); +} + +#[cfg(feature = "bitshuffle")] +#[test] +fn bitshuffle_written_by_hdf5plugin_reads_exactly() { + if !have_python("h5py, hdf5plugin") { + return; + } + check_h5py_written( + "bitshuffle", + r#"[('none', hdf5plugin.Bitshuffle(cname='none')), + ('lz4', hdf5plugin.Bitshuffle(cname='lz4')), + ('lz4 nelems=16', hdf5plugin.Bitshuffle(nelems=16, cname='lz4')), + ('none nelems=64', hdf5plugin.Bitshuffle(nelems=64, cname='none')), + ('zstd', hdf5plugin.Bitshuffle(cname='zstd')), + ('zstd clevel=19 nelems=2048', hdf5plugin.Bitshuffle(nelems=2048, cname='zstd', clevel=19))]"#, + ); +} + +#[cfg(feature = "bitshuffle")] +#[test] +fn bitshuffle_written_by_clawhdf5_reads_in_hdf5plugin() { + use clawhdf5_format::chunked_write::{BitshuffleCompression, PluginFilter}; + if !have_python("h5py, hdf5plugin") { + return; + } + for (tag, compression) in [ + ("bshuf_none", BitshuffleCompression::None), + ("bshuf_lz4", BitshuffleCompression::Lz4), + ("bshuf_zstd", BitshuffleCompression::Zstd { level: 3 }), + ] { + check_ours_read_by_h5py(tag, 32008, "bitshuffle", |ds| { + ds.with_bitshuffle(compression); + }); + check_ours_read_by_h5py(tag, 32008, "bitshuffle", |ds| { + ds.with_plugin_filter(PluginFilter::Bitshuffle { + block_size: 40, + compression, + }); + }); + } +} + +#[cfg(feature = "bzip2")] +#[test] +fn bzip2_written_by_hdf5plugin_reads_exactly() { + if !have_python("h5py, hdf5plugin") { + return; + } + check_h5py_written( + "bzip2", + r#"[('bzip2 9', hdf5plugin.BZip2()), + ('bzip2 1', hdf5plugin.BZip2(blocksize=1)), + ('bzip2 5 + shuffle', dict(**hdf5plugin.BZip2(blocksize=5), shuffle=True))]"#, + ); +} + +#[cfg(feature = "bzip2")] +#[test] +fn bzip2_written_by_clawhdf5_reads_in_hdf5plugin() { + if !have_python("h5py, hdf5plugin") { + return; + } + check_ours_read_by_h5py("bzip2", 307, "bzip2", |ds| { + ds.with_bzip2(9); + }); + check_ours_read_by_h5py("bzip2_1", 307, "bzip2", |ds| { + ds.with_bzip2(1).without_shuffle(); + }); +} + +#[cfg(feature = "blosc")] +#[test] +fn blosc_written_by_hdf5plugin_reads_exactly() { + if !have_python("h5py, hdf5plugin") { + return; + } + // Every codec hdf5plugin's Blosc offers, each shuffle mode, and levels + // from "store" to maximum. + check_h5py_written( + "blosc", + r#"[(f'{c} {s} {l}', hdf5plugin.Blosc(cname=c, clevel=l, shuffle=s)) + for c in ['blosclz', 'lz4', 'lz4hc', 'snappy', 'zlib', 'zstd'] + for s, l in [(hdf5plugin.Blosc.NOSHUFFLE, 5), + (hdf5plugin.Blosc.SHUFFLE, 9), + (hdf5plugin.Blosc.BITSHUFFLE, 1)]] + + [('blosclz level 0', hdf5plugin.Blosc(cname='blosclz', clevel=0))]"#, + ); +} + +#[cfg(feature = "blosc")] +#[test] +fn blosc_written_by_clawhdf5_reads_in_hdf5plugin() { + use clawhdf5_format::chunked_write::{BloscCodec, BloscShuffle}; + if !have_python("h5py, hdf5plugin") { + return; + } + for codec in [ + BloscCodec::Lz4, + BloscCodec::Snappy, + BloscCodec::Zlib, + BloscCodec::Zstd, + ] { + for (shuffle, level) in [ + (BloscShuffle::None, 5), + (BloscShuffle::Byte, 9), + (BloscShuffle::Bit, 1), + (BloscShuffle::Byte, 0), + ] { + check_ours_read_by_h5py( + &format!("blosc_{codec:?}_{shuffle:?}_{level}"), + 32001, + "blosc", + |ds| { + ds.with_blosc(codec, level, shuffle); + }, + ); + } + } +} + +#[cfg(feature = "lzf")] +#[test] +fn lzf_written_by_h5py_reads_exactly() { + if !have_python("h5py") { + return; + } + check_h5py_written( + "lzf", + r#"[('lzf', dict(compression='lzf')), + ('lzf+shuffle', dict(compression='lzf', shuffle=True)), + ('lzf+shuffle+fletcher32', dict(compression='lzf', shuffle=True, fletcher32=True))]"#, + ); +} + +#[cfg(feature = "lzf")] +#[test] +fn lzf_written_by_clawhdf5_reads_in_h5py() { + if !have_python("h5py") { + return; + } + check_ours_read_by_h5py("lzf", 32000, "lzf", |ds| { + ds.with_lzf(); + }); + check_ours_read_by_h5py("lzf_noshuffle", 32000, "lzf", |ds| { + ds.with_lzf().without_shuffle(); + }); +} + +/// Blosc2 and ZFP are not implemented: reading them must be a clear error +/// naming the filter, never data. +#[test] +fn unimplemented_filters_are_a_clear_error() { + if !have_python("h5py, hdf5plugin") { + return; + } + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("unimplemented.h5"); + run_python( + r#" +import sys +import numpy as np, h5py, hdf5plugin +with h5py.File(sys.argv[1], 'w') as f: + d = np.arange(4096, dtype=' = (0..16).collect(); + assert_eq!(file.dataset("lzf_ok").unwrap().read_i32().unwrap(), want); +} diff --git a/docs/known-issues.md b/docs/known-issues.md index 4ea7518..bf8f77a 100644 --- a/docs/known-issues.md +++ b/docs/known-issues.md @@ -7,6 +7,18 @@ deleting it. --- +## Concurrent and contiguous read performance (measured 2026-09-26) + +**Status:** open. Measured on tank with `concurrent_read` against h5py +3.16 / HDF5 2.0 (`BENCHMARKS.md`, "Concurrent reads"): +- Full reads of chunked datasets from several threads through one `File` + stop scaling at about 4 threads (880 MB/s on deflate data vs 4424 MB/s + for 16 h5py processes). Hyperslab reads, which skip the chunk cache, + scale to 1244 MB/s, so the `File`'s shared chunk cache is the suspect. +- Contiguous datasets read 4x slower than h5py on one thread (2.5 vs + 9.8 GB/s full, 0.12x for 256 x 256 hyperslabs). +Values are correct; this is speed only. + ## Silent wrong data found by the 2026-09-25 HDF5 audit **Status:** fixed after v2.7.0 (2026-09-25). **Every release up @@ -129,13 +141,66 @@ fill-value item that did is fixed). failing the others. - **Other readers:** - VL-string datasets are not readable through `File`. + - Variable-length values inside a compound (and VL-string attributes) in + a file with 4-byte offsets (`sizeof_addr = 4`) fail with + `GlobalHeapObjectNotFound` or come back as `Raw`: these paths assume + the 16-byte element of an 8-byte-offset file. The datatype itself reads + (it was refused as "member overlaps with previous member" until + 2026-09-26). - Metadata cache images are not supported. - x87 long double and binary128 are refused. - N-Bit on 64-bit scale-offset data and some N-Bit parameter layouts fail. - **Filters:** blosc, blosc2, bitshuffle, bzip2, LZF and zfp are not - implemented. -- **Header checks:** on 12 CVE datasets libhdf5 rejects a corrupt header and - we read data anyway. We need stricter header checks. + implemented. **Fixed 2026-09-26** for LZF (default-on `lzf` feature), + bitshuffle, bzip2 and Blosc 1 (`bitshuffle`, `bzip2`, `blosc`, or + `plugin-filters` for all), read and write, pure Rust; h5ex_d_lzf, + h5ex_d_bshuf, h5ex_d_bzip2 and h5ex_d_blosc now read (conformance 573 of + 697 ok). **Still open:** Blosc2 (32026 — hdf5plugin stores each chunk as a + Blosc2 super-chunk frame, and n-D chunks as B2ND arrays) and ZFP (32013); + both fail with an `UnsupportedFilter` error that names the filter, and + either can be plugged in with `filter_registry::register_filter` (32023, + Granular BitRound, too, since 2026-09-26 even with the `pcodec` feature). +- **Wrong data: a chunk whose filters decode to fewer bytes than the chunk + read with zeros for the missing bytes** (any filter; found reviewing the plugin + filters). **Fixed 2026-09-26:** it is an error naming the chunk. A corrupt + chunk must never read as zeros. Unfiltered chunks are read at their stored + size and are not checked this way. +- **Crash:** a hostile Blosc chunk (frame size below its header) panicked in + builds with overflow checks. **Fixed 2026-09-26**; the new decoders are + fuzzed in the unit tests. +- ~~**Header checks:** on 12 CVE datasets libhdf5 rejects a corrupt header and + we read data anyway. We need stricter header checks.~~ **Fixed + 2026-09-26** (counted again: 18 objects on the CVE corpus that libhdf5 + refuses; some read as wrong data, e.g. a zero chunk dimension read as all + fill values): object headers, datatypes, chunk dimensions and chunk-index + offsets are checked as libhdf5 checks them, and truncated files are + refused. 17 of the 18 now + fail as in libhdf5 (conformance on tank, `conformance/run.sh --no-fetch`, + 2026-09-26: 571 of 697 ok). Still read where libhdf5 refuses: + - `cve-2024-32624.h5` `/Dset_OBJREF`: a dataspace whose storage size + overflows 64 bits. `File::dataset` and `shape()` succeed (libhdf5 + refuses at open); reading the values fails. + - `cve-2020-10810.h5`, `cve-2020-10812.h5` (whole files libhdf5 cannot + open, not among the 18): libhdf5 decodes the superblock extension's File + Space Info and metadata-cache-image messages at open and refuses these + files; we do not decode those messages at open. + - Deliberately not refused, because clawhdf5 up to v2.7.0 wrote them: a + float sign bit position outside the type, and a size-0 string type. + - Not refused because current libhdf5 reads it though HDF5 2.0.0 + (h5py 3.16) refuses it: a v4 chunked layout whose dimensions are + encoded in more bytes than they need (HDFGroup/hdf5@e124c36, + 2026-06-05, relaxed that check; clawhdf5 wrote such layouts until + 2026-09-26). + - Not refused because HDF5 2.0 (h5py 3.16) reads them though newer + libhdf5 refuses them: bit-field offset/precision outside the type, an + unknown variable-length kind, an array type whose stored size is not + its element count times its base size. + - (`cve-2024-32616` `/group1/dset3` and `cve-2025-2309`'s `Comp_OBJREF` + attribute are h5py/numpy type-mapping failures, not libhdf5 refusals.) + - `h5rs check` validates with the library's parsers, so it inherits what + they accept: of the 150 CVE and fuzzer files, `check --data` passes 16, + and h5dump 1.14.6 rejects 9 of those (tank, 2026-09-26; 28 and 21 + before these checks). - **Writer:** - Nested groups beyond one level: path-like names are now refused, not created. @@ -422,6 +487,23 @@ the same agent-store interop test. **Fix:** an empty contiguous dataset gets the undefined address (all `0xff`), which is what libhdf5 itself writes. +## `clawhdf5-wasm` (browser) limits + +**Status:** open (by design for now; added 2026-09-26). + +- The whole file is held in memory: `open()` takes its bytes. There are no + HTTP range reads, so a multi-GB file does not fit a browser tab. +- Compound, reference, opaque, bitfield, time and VL-sequence datasets are + refused with an error naming the type; attributes of those types come back + as `value: null` with their `dtype`. +- No Zstd or SZIP (both link C): such datasets fail with + `unsupported filter: 32015` / `: 4`. pcodec is not enabled either. +- External links and virtual-dataset sources in other files cannot be + followed (no file system). +- Variable-length string datasets are read by decoding `read_selection`'s + bytes with `clawhdf5_format::vl_data` in the wasm crate; `File` itself still + cannot (see the audit gaps above). + ## The Node.js package (`packages/clawhdf5-node`) does not work **Status:** open (found 2026-09-25). Unpublished; not built or tested in CI. diff --git a/examples/wasm-viewer/.gitignore b/examples/wasm-viewer/.gitignore new file mode 100644 index 0000000..f57bf2b --- /dev/null +++ b/examples/wasm-viewer/.gitignore @@ -0,0 +1,2 @@ +# Generated by build.sh. +/pkg/ diff --git a/examples/wasm-viewer/README.md b/examples/wasm-viewer/README.md new file mode 100644 index 0000000..1a5c47d --- /dev/null +++ b/examples/wasm-viewer/README.md @@ -0,0 +1,97 @@ +# HDF5 viewer in the browser + +A single page that opens an HDF5 or NetCDF-4 file entirely in the browser +with `clawhdf5-wasm` (clawhdf5's reader compiled to WebAssembly): drop a +file, browse its groups, and look at a dataset's type, shape, attributes +and values (a 50 x 12 window at a time, read as a hyperslab, with the +leading dimensions of a 3-D+ dataset held at chosen indices). The file never +leaves the page. + +## Build and open + +```bash +rustup target add wasm32-unknown-unknown +cargo install wasm-bindgen-cli --version 0.2.129 # must equal the crate version; build.sh checks +bash examples/wasm-viewer/build.sh # writes examples/wasm-viewer/pkg/ (not committed) +python3 -m http.server -d examples/wasm-viewer 8000 # wasm cannot load from file:// +``` + +Then open . `?file=&path=` opens a +file from a URL (same origin, or one serving CORS headers) and selects an +object in it, e.g. `?file=data/run1.h5&path=/results/energy`. + +## JavaScript API + +```js +import init, { open } from "./pkg/clawhdf5_wasm.js"; +await init(); +const f = open(new Uint8Array(await blob.arrayBuffer())); +f.list("/"); // [{ name, kind: "group" | "dataset" }], groups first +f.info("/grid"); // { shape, maxshape, dtype, elementShape } +f.attrs("/grid"); // [{ name, value, dtype }] +f.read("/grid"); // { shape, dtype, data } +f.readHyperslab("/grid", [0, 0], [10, 5], [2, 1]); // start, count, stride?, block? +f.free(); +``` + +`data` is the typed array of the stored width (`Float64Array`, +`Float32Array` also for `f16`, `Int8Array` ... `BigInt64Array`, +`BigUint64Array`), or an array of strings for fixed- and variable-length +strings and enumerations (h5py booleans read as `"TRUE"`/`"FALSE"`). Array +datatypes are flattened, their dimensions appended to `shape`. Anything +else throws an `Error` naming the type. + +## Limits + +- Read-only, and the whole file is held in memory (no range requests). +- Compound, reference, opaque and variable-length-sequence datasets are + refused with an error. Attributes of those types are listed with + `value: null` and their `dtype`. +- No Zstd or SZIP filters (they link C): such a dataset fails with + `unsupported filter`. Deflate, shuffle, Fletcher-32, LZ4, N-Bit and + scale-offset are read (within the limits in `docs/known-issues.md`). +- Virtual datasets whose sources are in other files, and external links, + cannot be followed: there is no file system. + +## Tests + +`test/run.sh` builds the package, writes `fixture.h5` (h5py) and +`fixture.nc` (netCDF4) with `test/make_fixture.py`, then: + +- runs `test/test.mjs` under Node: every dataset (whole and a strided + hyperslab), listing and attribute is compared with what libhdf5 reads + back, error paths are checked, and so are the page's DOM-free helpers + (`viewer-lib.js`); +- runs `test/browser.sh`: loads the page in headless Chromium with + `?file=fixture.h5&path=...` for eight objects and checks the rendered tree, + types, shapes, attribute and value cells, and the error shown for an + unsupported type. Skipped when no Chromium is found (`CHROME` names one; + a Playwright download under `~/.cache/ms-playwright` is picked up). + Drag-and-drop and the file picker are not driven by it; they share + `load()` with the `?file=` path. + +The same expectations are checked natively, without Node, by +`crates/clawhdf5-wasm/tests/h5py_interop.rs`, which is what CI runs (the CI +container has no Node or browser). + +## Size + +Measured 2026-09-26 on tank (rustc 1.98.1, wasm-bindgen 0.2.129, gzip 1.14, +`gzip -9 -n`), after `bash examples/wasm-viewer/build.sh`: + +| | raw | gzip -9 | +|---|---:|---:| +| `pkg/clawhdf5_wasm_bg.wasm` (profile `wasm-release`, opt-level `s`) | 627,501 B | 191,639 B | +| `pkg/clawhdf5_wasm.js` (wasm-bindgen glue) | 21,826 B | 4,487 B | +| same wasm at opt-level `z` | 693,068 B | 192,550 B | +| same wasm at opt-level `3` | 544,035 B | 198,803 B | +| h5wasm 0.10.3: wasm embedded in `dist/esm/hdf5_util.js` | 3,544,184 B | 907,096 B | +| h5wasm 0.10.3: `dist/esm/hdf5_util.js` as shipped | 4,150,134 B | 986,699 B | + +h5wasm figures: `npm pack h5wasm@0.10.3` (npm reports +`dist.unpackedSize` 14,731,385 B for the whole package), wasm extracted from +the `binaryDecode` literal in `hdf5_util.js`. h5wasm is the whole of libhdf5 +(writing, every datatype, plugins), so this compares download size, not +equal functionality. No `wasm-opt` pass was applied (binaryen is not +installed on tank). opt-level `s` is used because it is the smallest +compressed. diff --git a/examples/wasm-viewer/build.sh b/examples/wasm-viewer/build.sh new file mode 100755 index 0000000..6af81bf --- /dev/null +++ b/examples/wasm-viewer/build.sh @@ -0,0 +1,39 @@ +#!/usr/bin/env bash +# Build the clawhdf5-wasm package the viewer loads, into examples/wasm-viewer/pkg/. +# +# Needs the wasm32-unknown-unknown target and the wasm-bindgen CLI at the +# exact version cargo resolves for the wasm-bindgen crate: +# rustup target add wasm32-unknown-unknown +# cargo install wasm-bindgen-cli --version +# +# Then serve this directory over HTTP (browsers do not load wasm modules from +# file://) and open it: +# python3 -m http.server -d examples/wasm-viewer 8000 +set -euo pipefail + +HERE="$(cd "$(dirname "$0")" && pwd)" +ROOT="$(cd "$HERE/../.." && pwd)" +cd "$ROOT" + +# The resolved crate version (Cargo.lock is not committed, so ask cargo). +want=$(cargo pkgid wasm-bindgen | sed 's/.*[@#]//') +if ! command -v wasm-bindgen >/dev/null; then + echo "wasm-bindgen CLI not found: cargo install wasm-bindgen-cli --version $want" >&2 + exit 1 +fi +have=$(wasm-bindgen --version | awk '{print $2}') +if [ "$want" != "$have" ]; then + echo "wasm-bindgen CLI is $have but the crate is $want:" >&2 + echo " cargo install wasm-bindgen-cli --version $want" >&2 + exit 1 +fi + +cargo build -p clawhdf5-wasm --target wasm32-unknown-unknown --profile wasm-release + +target_dir=$(cargo metadata --format-version 1 --no-deps \ + | sed -n 's/.*"target_directory":"\([^"]*\)".*/\1/p') +wasm="$target_dir/wasm32-unknown-unknown/wasm-release/clawhdf5_wasm.wasm" + +rm -rf "$HERE/pkg" +wasm-bindgen --target web --out-dir "$HERE/pkg" "$wasm" +echo "built $HERE/pkg ($(wc -c < "$HERE/pkg/clawhdf5_wasm_bg.wasm") bytes of wasm)" diff --git a/examples/wasm-viewer/index.html b/examples/wasm-viewer/index.html new file mode 100644 index 0000000..d477e49 --- /dev/null +++ b/examples/wasm-viewer/index.html @@ -0,0 +1,278 @@ + + + + + +HDF5 Viewer + + + +
+

HDF5 Viewer

+ no file + +
+
+ +
+
+

Drop an HDF5 or NetCDF-4 file here, or use “Open file…”.

+

The file is read in this page by clawhdf5 compiled to WebAssembly; it is not uploaded anywhere.

+

+
+
+
+ + + diff --git a/examples/wasm-viewer/test/browser.sh b/examples/wasm-viewer/test/browser.sh new file mode 100644 index 0000000..a9ea5aa --- /dev/null +++ b/examples/wasm-viewer/test/browser.sh @@ -0,0 +1,96 @@ +#!/usr/bin/env bash +# Load the viewer page in headless Chromium and check what it renders. +# +# browser.sh FIXTURE_DIR +# +# FIXTURE_DIR holds fixture.h5 from make_fixture.py; ../pkg must be built. +# The page is opened with ?file=fixture.h5&path=, which fetches the +# file, builds the tree down to and shows it; the rendered DOM is +# dumped and checked for the values libhdf5 reads. +# +# Browser: $CHROME, else chromium/google-chrome on PATH, else a Playwright +# download under ~/.cache/ms-playwright. Exit 3 when none is found. +# BROWSER_DEBUG=/some/prefix saves each rendered page as prefix..html. +set -euo pipefail + +HERE="$(cd "$(dirname "$0")" && pwd)" +FIX="$(cd "$1" && pwd)" +PY="${CLAWHDF5_PYTHON:-python3}" + +chrome="${CHROME:-}" +if [ -z "$chrome" ]; then + for c in chromium chromium-browser google-chrome chrome-headless-shell; do + if command -v "$c" >/dev/null; then chrome="$(command -v "$c")"; break; fi + done +fi +if [ -z "$chrome" ]; then + chrome="$(ls -d "$HOME"/.cache/ms-playwright/chromium_headless_shell-*/chrome-headless-shell-linux64/chrome-headless-shell 2>/dev/null | tail -1 || true)" +fi +if [ -z "$chrome" ] || [ ! -x "$chrome" ]; then + echo "no Chromium found (set CHROME)" >&2 + exit 3 +fi + +root="$(mktemp -d)" +server="" +cleanup() { + [ -n "$server" ] && kill "$server" 2>/dev/null || true + rm -rf "$root" +} +trap cleanup EXIT +ln -s "$HERE/../index.html" "$HERE/../viewer-lib.js" "$HERE/../pkg" "$FIX/fixture.h5" "$root/" + +port=$("$PY" -c 'import socket; s = socket.socket(); s.bind(("127.0.0.1", 0)); print(s.getsockname()[1])') +"$PY" -m http.server --bind 127.0.0.1 --directory "$root" "$port" >/dev/null 2>&1 & +server=$! +for _ in $(seq 50); do + "$PY" -c "import urllib.request; urllib.request.urlopen('http://127.0.0.1:$port/index.html')" 2>/dev/null && break + sleep 0.1 +done + +fails=0 +# A fresh profile per page: a second instance on the same profile fails. +render() { + local profile + profile="$(mktemp -d "$root/profile.XXXXXX")" + "$chrome" --headless --no-sandbox --disable-gpu --user-data-dir="$profile" \ + --virtual-time-budget=20000 \ + --dump-dom "http://127.0.0.1:$port/index.html?file=fixture.h5&path=$1" 2>/dev/null +} +# expect PATH TEXT...: every TEXT appears in the page rendered for PATH. +expect() { + local path="$1" dom + shift + dom="$(render "$path")" + [ -n "${BROWSER_DEBUG:-}" ] && printf "%s\n" "$dom" > "$BROWSER_DEBUG.$(echo "$path" | tr / _).html" + for text in "$@"; do + if ! grep -qF -- "$text" <<<"$dom"; then + echo "FAIL: page for $path lacks: $text" >&2 + fails=$((fails + 1)) + fi + done + echo "rendered $path" +} + +# Tree (root expanded; the group row carries its path) and root attributes. +expect "/" 'data-path="/sensors"' 'data-path="/grid"' 'title"wasm fixture"' \ + 'big9223372036854775813' '(compound{x: f64, n: i32})' +# A chunked, deflated 2-D dataset: type, shape, the first window of values. +expect "/grid" '
f64
' '
(6, 10)
' '9' '0.25' '14.75' \ + 'showing rows 0–5, columns 0–9 of 60 values' +# Nested path revealed through the tree; big-endian float32. +expect "/sensors/temp" 'data-path="/sensors/temp"' '
f32
' '21.5' '22.25' +# 64-bit integers stay exact; strings; array datatype cells. +expect "/u64" '18446744073709551615' +expect "/vlen_str" '"двa"' '
vlen string
' +expect "/pairs" '[2, 3]' '
array[2]<i32>
' +# 3-D: leading dimension held at 0, window over the last two. +expect "/cube" '
(2, 5, 6)
' '29' 'dim 0' +# Unsupported type: an error, not values. +expect "/table" 'class="error"' 'reading compound{x: f64, n: i32} datasets is not supported' + +if [ "$fails" -gt 0 ]; then + echo "browser: $fails checks failed" >&2 + exit 1 +fi +echo "browser: all checks passed ($chrome)" diff --git a/examples/wasm-viewer/test/make_fixture.py b/examples/wasm-viewer/test/make_fixture.py new file mode 100644 index 0000000..8c7722c --- /dev/null +++ b/examples/wasm-viewer/test/make_fixture.py @@ -0,0 +1,203 @@ +"""Write HDF5 and NetCDF-4 test files with h5py/netCDF4, and what libhdf5 +reads back from them, for the clawhdf5-wasm tests. + + python make_fixture.py OUT_DIR + +writes OUT_DIR/fixture.h5, OUT_DIR/fixture.nc and OUT_DIR/expected.json. +Both the Rust test (crates/clawhdf5-wasm/tests/h5py_interop.rs, native) and +the Node test (test.mjs, the built wasm package) compare against the same +expected.json, so the two check the same values. + +Every expected value comes from h5py reading the file back (numpy slicing +for hyperslabs), never from the arrays that were written. Integers are +encoded as strings so JSON.parse keeps 64-bit values exact. +""" + +import json +import sys +import warnings +from pathlib import Path + +import h5py +import netCDF4 +import numpy as np + +try: # registers the LZ4/Zstd filters with libhdf5; optional + import hdf5plugin +except ImportError: + hdf5plugin = None + +# netCDF4 1.7 trips numpy 2.5's shape-setting deprecation on assignment. +warnings.filterwarnings("ignore", category=DeprecationWarning) + +out = Path(sys.argv[1]) +out.mkdir(parents=True, exist_ok=True) +h5 = out / "fixture.h5" +nc = out / "fixture.nc" + +rng = np.random.default_rng(7) +with h5py.File(h5, "w") as f: + f.attrs["title"] = "wasm fixture" + f.attrs["version"] = np.int64(3) + f.attrs["scale"] = np.array([0.5, 2.0]) + f.attrs["big"] = np.uint64(2**63 + 5) + f.attrs.create("vlen_note", "héllo", dtype=h5py.string_dtype()) + # No plain JavaScript form: listed with value null and its type. + f.attrs["origin"] = np.array((1.5, 2), dtype=[("x", "f4")) + f.create_dataset("f16", data=np.array([0.5, -1.25, 65504], dtype=" 2 else 1] + [2] * (obj.ndim - 1) + count = [min(c, (n - s - 1) // st + 1) + for c, s, st, n in zip(count, start, stride, obj.shape)] + return (start, count, stride) + + +json.dump({"fixture.h5": describe(h5), "fixture.nc": describe(nc)}, + open(out / "expected.json", "w"), indent=1, ensure_ascii=False) diff --git a/examples/wasm-viewer/test/run.sh b/examples/wasm-viewer/test/run.sh new file mode 100755 index 0000000..c1cdae0 --- /dev/null +++ b/examples/wasm-viewer/test/run.sh @@ -0,0 +1,37 @@ +#!/usr/bin/env bash +# Build the wasm package (../build.sh), test it under Node against files +# written by h5py and netCDF4 (make_fixture.py), then load the viewer page +# in headless Chromium if one is found (browser.sh). +# +# Needs node, the wasm-bindgen CLI (see ../build.sh) and a Python with h5py, +# netCDF4 and numpy: CLAWHDF5_PYTHON names it (default python3). Without that +# Python the test is skipped, unless CLAWHDF5_REQUIRE_INTEROP=1. +set -euo pipefail + +HERE="$(cd "$(dirname "$0")" && pwd)" +PY="${CLAWHDF5_PYTHON:-python3}" + +command -v node >/dev/null || { echo "node not found" >&2; exit 1; } +if ! "$PY" -c "import h5py, netCDF4, numpy" >/dev/null 2>&1; then + if [ "${CLAWHDF5_REQUIRE_INTEROP:-0}" = "1" ]; then + echo "CLAWHDF5_REQUIRE_INTEROP=1 but $PY lacks h5py/netCDF4/numpy" >&2 + exit 1 + fi + echo "SKIP: $PY lacks h5py/netCDF4/numpy" + exit 0 +fi + +bash "$HERE/../build.sh" +fix="$(mktemp -d)" +trap 'rm -rf "$fix"' EXIT +"$PY" "$HERE/make_fixture.py" "$fix" +node "$HERE/test.mjs" "$HERE/../pkg" "$fix" + +# The page itself, in headless Chromium when one is available. +status=0 +bash "$HERE/browser.sh" "$fix" || status=$? +if [ "$status" = 3 ]; then + echo "SKIP: viewer page in a browser (no Chromium; set CHROME)" +elif [ "$status" != 0 ]; then + exit "$status" +fi diff --git a/examples/wasm-viewer/test/test.mjs b/examples/wasm-viewer/test/test.mjs new file mode 100644 index 0000000..80ca30a --- /dev/null +++ b/examples/wasm-viewer/test/test.mjs @@ -0,0 +1,143 @@ +// Node test of the built wasm package (the exact pkg/ the viewer page loads) +// and the viewer's DOM-free helpers. Run by test/run.sh: +// node test.mjs PKG_DIR FIXTURE_DIR +// FIXTURE_DIR holds fixture.h5, fixture.nc and expected.json from +// make_fixture.py (values as libhdf5 reads them back). +import assert from "node:assert/strict"; +import { readFileSync } from "node:fs"; +import { join } from "node:path"; +import { pathToFileURL } from "node:url"; + +const [pkgDir, fixDir] = process.argv.slice(2); +const pkg = await import(pathToFileURL(join(pkgDir, "clawhdf5_wasm.js"))); +pkg.initSync({ module: readFileSync(join(pkgDir, "clawhdf5_wasm_bg.wasm")) }); +const lib = await import(pathToFileURL(join(import.meta.dirname, "..", "viewer-lib.js"))); + +let checks = 0; +const eq = (a, b, msg) => { assert.deepEqual(a, b, msg); checks++; }; + +const ARRAY_TYPES = { + f32: Float32Array, f64: Float64Array, i8: Int8Array, i16: Int16Array, i32: Int32Array, + i64: BigInt64Array, u8: Uint8Array, u16: Uint16Array, u32: Uint32Array, u64: BigUint64Array, + strings: Array, +}; + +function values(kind, data) { + const arr = Array.from(data); + if (kind === "f32" || kind === "f64" || kind === "strings") return arr; + return arr.map(String); +} + +function checkAttr(ctx, a, want) { + const v = a.value; + if ("raw" in want) { + eq(v, null, ctx); + assert.ok(a.dtype.includes(want.raw), `${ctx}: ${a.dtype}`); + return; + } + eq(a.dtype, null, `${ctx} dtype`); + if ("string" in want) return eq(v, want.string, ctx); + if ("strings" in want) return eq(v, want.strings, ctx); + if ("int" in want) { + if (want.scalar) { + assert.ok(typeof v === "number" || typeof v === "bigint", ctx); + return eq([String(v)], want.int, ctx); + } + assert.ok(v instanceof BigInt64Array || v instanceof BigUint64Array, ctx); + return eq(Array.from(v, String), want.int, ctx); + } + if ("float" in want) { + if (want.scalar) return eq([v], want.float, ctx); + assert.ok(v instanceof Float64Array, ctx); + return eq(Array.from(v), want.float, ctx); + } + assert.fail(`${ctx}: unknown expectation ${JSON.stringify(want)}`); +} + +const expected = JSON.parse(readFileSync(join(fixDir, "expected.json"), "utf8")); +for (const [name, exp] of Object.entries(expected)) { + const file = pkg.open(new Uint8Array(readFileSync(join(fixDir, name)))); + + for (const [path, want] of Object.entries(exp.lists)) { + eq(file.kind(path), "group", `${name}:${path} kind`); + const list = file.list(path); + for (const [kind, key] of [["group", "groups"], ["dataset", "datasets"]]) { + eq(list.filter((c) => c.kind === kind).map((c) => c.name).sort(), want[key], `${name}:${path} ${key}`); + } + } + + for (const [path, want] of Object.entries(exp.datasets)) { + const ctx = `${name}:${path}`; + eq(file.kind(path), "dataset", `${ctx} kind`); + if (want.unavailable) { + // The wasm build has no zstd (it links C): a clear error, no data. + assert.throws(() => file.read(path), (e) => e.message.includes(want.unavailable), ctx); + checks++; + continue; + } + const info = file.info(path); + eq([...info.shape, ...info.elementShape], want.shape, `${ctx} info shape`); + const r = file.read(path); + eq(r.shape, want.shape, `${ctx} shape`); + eq(r.dtype, info.dtype, `${ctx} dtype`); + assert.ok(r.data instanceof ARRAY_TYPES[want.kind], `${ctx}: ${r.data.constructor.name} for ${want.kind}`); + eq(values(want.kind, r.data), want.values, ctx); + if (want.slab) { + const s = want.slab; + const part = file.readHyperslab(path, s.start, s.count, s.stride); + eq(part.shape, s.shape, `${ctx} slab shape`); + eq(values(want.kind, part.data), s.values, `${ctx} slab`); + } + } + + for (const [path, what] of Object.entries(exp.errors)) { + assert.throws(() => file.read(path), (e) => e instanceof Error && e.message.includes(what), `${name}:${path}`); + checks++; + } + + for (const [path, want] of Object.entries(exp.attrs)) { + const attrs = file.attrs(path); + eq(file.attrErrors(path), [], `${name}:${path} attr errors`); + const seen = attrs.filter((a) => !a.name.startsWith("_") && !exp.skip_attrs.includes(a.name)); + eq(seen.map((a) => a.name).sort(), Object.keys(want).sort(), `${name}:${path} attr names`); + for (const a of seen) checkAttr(`${name}:${path}@${a.name}`, a, want[a.name]); + } + file.free(); +} + +// Errors reach JavaScript as thrown Errors, never as data. +const h5 = pkg.open(new Uint8Array(readFileSync(join(fixDir, "fixture.h5")))); +const throwsMsg = (fn, re) => { assert.throws(fn, (e) => e instanceof Error && re.test(e.message)); checks++; }; +throwsMsg(() => pkg.open(new Uint8Array(64)), /./); +throwsMsg(() => h5.read("/nope"), /./); +throwsMsg(() => h5.list("/grid"), /not a group/); +throwsMsg(() => h5.readHyperslab("/grid", [0], [1]), /dimensions/); +throwsMsg(() => h5.readHyperslab("/grid", [5, 0], [2, 1]), /exceeds/); +throwsMsg(() => h5.readHyperslab("/grid", [-1, 0], [1, 1]), /non-negative integers/); +throwsMsg(() => h5.readHyperslab("/grid", [0.5, 0], [1, 1]), /non-negative integers/); +// Info for a dataset with an unlimited dimension (netCDF "time"). +const nc = pkg.open(new Uint8Array(readFileSync(join(fixDir, "fixture.nc")))); +eq(nc.info("/time").maxshape, [null], "unlimited dimension is null"); +// Big integers stay exact. +eq(h5.read("/u64").data[0], 18446744073709551615n, "u64 max"); +eq(typeof pkg.version(), "string", "version"); + +// Viewer helpers. +eq(lib.joinPath("/", "a"), "/a", "joinPath root"); +eq(lib.joinPath("/a", "b"), "/a/b", "joinPath nested"); +eq(lib.viewWindow([], {}), null, "scalar window"); +eq(lib.viewWindow([7], { row: 5, rows: 50 }), { start: [5], count: [2], rows: 2, cols: 1, row: 5, col: 0 }, "1-D window"); +const w = lib.viewWindow([2, 5, 6], { row: 1, col: 4, rows: 3, cols: 5, fixed: [1] }); +eq(w, { start: [1, 1, 4], count: [1, 3, 2], rows: 3, cols: 2, row: 1, col: 4 }, "3-D window"); +// The window the page would request reads the same values as a direct slab. +const cube = h5.readHyperslab("/cube", w.start, w.count); +eq(lib.toRows(cube.data, w.rows, w.cols), [["40", "41"], ["46", "47"], ["52", "53"]], "cube window cells"); +const pairs = h5.readHyperslab("/pairs", [1], [2]); +eq(lib.toRows(pairs.data, 2, 1, lib.perElement([2])), [["[2, 3]"], ["[4, 5]"]], "array-type cells"); +eq(lib.formatValue(0.1 + 0.2), "0.3", "float formatting"); +eq(lib.formatValue(2n ** 64n - 1n), "18446744073709551615", "bigint formatting"); +eq(lib.formatValue("x"), '"x"', "string formatting"); +h5.free(); +nc.free(); + +console.log(`wasm package: ${checks} checks passed`); diff --git a/examples/wasm-viewer/viewer-lib.js b/examples/wasm-viewer/viewer-lib.js new file mode 100644 index 0000000..72c5bf3 --- /dev/null +++ b/examples/wasm-viewer/viewer-lib.js @@ -0,0 +1,77 @@ +// DOM-free helpers for the viewer, so Node can test them (test/test.mjs). + +/** Child path of `name` in the group at `parent`. */ +export function joinPath(parent, name) { + return parent === "/" ? `/${name}` : `${parent}/${name}`; +} + +/** One value as display text. */ +export function formatValue(v) { + if (v === null || v === undefined) return ""; + if (typeof v === "bigint") return v.toString(); + if (typeof v === "number") { + if (Number.isInteger(v)) return String(v); + return String(Number(v.toPrecision(7))); + } + if (typeof v === "string") return JSON.stringify(v); + if (ArrayBuffer.isView(v) || Array.isArray(v)) { + const items = Array.from(v.slice(0, 16), formatValue); + if (v.length > 16) items.push(`… (${v.length} values)`); + return `[${items.join(", ")}]`; + } + return String(v); +} + +/** + * The window of a dataset to show: a hyperslab over its `shape` with the + * last dimension as columns, the one before as rows, and any leading + * dimensions held at `fixed` indices. `row`/`col` are the window's top-left + * corner. Returns null for a scalar (read it whole). + */ +export function viewWindow(shape, { row = 0, col = 0, rows = 50, cols = 12, fixed = [] } = {}) { + const rank = shape.length; + if (rank === 0) return null; + const clamp = (x, n) => Math.max(0, Math.min(x, Math.max(0, n - 1))); + if (rank === 1) { + const r0 = clamp(row, shape[0]); + const n = Math.max(0, Math.min(rows, shape[0] - r0)); + return { start: [r0], count: [n], rows: n, cols: 1, row: r0, col: 0 }; + } + const lead = shape.slice(0, rank - 2).map((n, i) => clamp(fixed[i] ?? 0, n)); + const nr = shape[rank - 2]; + const nc = shape[rank - 1]; + const r0 = clamp(row, nr); + const c0 = clamp(col, nc); + const r = Math.max(0, Math.min(rows, nr - r0)); + const c = Math.max(0, Math.min(cols, nc - c0)); + return { + start: [...lead, r0, c0], + count: [...lead.map(() => 1), r, c], + rows: r, + cols: c, + row: r0, + col: c0, + }; +} + +/** + * Split row-major `data` into `rows` x `cols` cells of display text; each + * cell holds `per` consecutive values (the elements of an array datatype). + */ +export function toRows(data, rows, cols, per = 1) { + const out = []; + for (let r = 0; r < rows; r++) { + const row = []; + for (let c = 0; c < cols; c++) { + const i = (r * cols + c) * per; + row.push(per === 1 ? formatValue(data[i]) : formatValue(data.slice(i, i + per))); + } + out.push(row); + } + return out; +} + +/** Number of values an array datatype packs into each element. */ +export function perElement(elementShape) { + return elementShape.reduce((a, b) => a * b, 1); +} diff --git a/scripts/ci-test.sh b/scripts/ci-test.sh index 1d7a7f9..e99cbf1 100755 --- a/scripts/ci-test.sh +++ b/scripts/ci-test.sh @@ -64,10 +64,31 @@ run_step "cargo clippy --all-targets" cargo clippy \ # 3. Clippy over clawhdf5-format's optional features, which the default # workspace build never compiles (szip is left out: it needs libaec). +# plugin-filters = bitshuffle, bzip2, blosc (and the default-on lzf). run_step "cargo clippy (format feature matrix)" cargo clippy \ -p clawhdf5-format \ --all-targets \ - --features parallel,lz4,zstd,pcodec,fast-checksum \ + --features parallel,lz4,zstd,pcodec,fast-checksum,plugin-filters \ + -- -D warnings + +# Each plugin filter alone, so none of them leans on another's +# dependencies (bitshuffle and blosc share code). +plugin_filters_alone() { + local f + for f in bitshuffle bzip2 blosc; do + echo "--- $f" + cargo clippy -p clawhdf5-format --all-targets --features "$f" -- -D warnings || return 1 + done +} +run_step "cargo clippy (each plugin filter alone)" plugin_filters_alone + +# The facade's plugin-filter interop tests and its parallel decoding tests +# only build with those features on (parallel_integration.rs stopped +# compiling unnoticed while nothing built it). +run_step "cargo clippy (facade plugin filters + parallel)" cargo clippy \ + -p clawhdf5 \ + --all-targets \ + --features plugin-filters,parallel \ -- -D warnings # The HNSW index's parallel bulk build is feature-gated too. @@ -89,13 +110,17 @@ run_step "cargo clippy (fast-deflate / zlib-ng)" cargo clippy \ # that: fail if a crate that compiles C (a *-sys crate, cc or cmake) enters the # default dependency tree of any of them. clawhdf5-migrate (bundled SQLite), # clawhdf5-napi (Node) and clawhdf5-gpu (graphics drivers) are exempt. +# js-sys (clawhdf5-wasm's bindings to JavaScript) builds no C. no_c_in_default_build() { local crate found=0 for crate in clawhdf5-format clawhdf5-io clawhdf5-filters clawhdf5 \ - clawhdf5-agent clawhdf5-ann clawhdf5-accel clawhdf5-netcdf4 clawhdf5-cli; do + clawhdf5-agent clawhdf5-ann clawhdf5-accel clawhdf5-netcdf4 clawhdf5-cli \ + clawhdf5-tools \ + clawhdf5-wasm; do local c_deps c_deps=$(cargo tree -q -p "$crate" -e normal,build --prefix none \ - | grep -E '^([a-z0-9_-]+-sys|cc|cmake) v' | sort -u) + | grep -E '^([a-z0-9_-]+-sys|cc|cmake) v' \ + | grep -v '^js-sys v' | sort -u) if [ -n "$c_deps" ]; then echo "$crate pulls in C by default:" echo "$c_deps" | sed 's/^/ /' @@ -106,6 +131,33 @@ no_c_in_default_build() { } run_step "no C in the default build (core crates)" no_c_in_default_build +# The reader in the browser: the facade's read path must build for +# wasm32-unknown-unknown (no mmap, no threads, no file system), and the +# wasm-bindgen crate must build and lint there. Needs +# `rustup target add wasm32-unknown-unknown`. +run_step "wasm32 build (clawhdf5, no default features)" cargo build \ + -p clawhdf5 \ + --target wasm32-unknown-unknown \ + --no-default-features +run_step "wasm32 clippy (clawhdf5-wasm)" cargo clippy \ + -p clawhdf5-wasm \ + --target wasm32-unknown-unknown \ + --all-targets \ + -- -D warnings + +# The built wasm package, run under Node against h5py/netCDF4-written files, +# and the viewer page in headless Chromium when one is found. +# Needs node and the wasm-bindgen CLI, which the CI container does not have; +# the same expectations are checked natively by clawhdf5-wasm's h5py_interop +# test in the cargo test step. +if command -v node >/dev/null && command -v wasm-bindgen >/dev/null; then + run_step "wasm package under Node (+ browser)" bash "$SCRIPT_DIR/../examples/wasm-viewer/test/run.sh" +else + echo "" + echo "==> [wasm package under Node] SKIPPED: needs node and wasm-bindgen" + STEPS+=("SKIP: wasm package under Node") +fi + # The workspace declares a minimum Rust version (rust-version in Cargo.toml); # check that it really builds there, so the README badge and the manifests # cannot drift from the truth. Separate target dir: a different toolchain @@ -128,7 +180,11 @@ run_step "cargo test" cargo test \ run_step "cargo test (format feature matrix)" cargo test \ -p clawhdf5-format \ - --features parallel,lz4,zstd,pcodec,fast-checksum + --features parallel,lz4,zstd,pcodec,fast-checksum,plugin-filters + +run_step "cargo test (facade parallel)" cargo test \ + -p clawhdf5 \ + --features parallel run_step "cargo test (ann parallel)" cargo test \ -p clawhdf5-ann \ @@ -149,6 +205,9 @@ if "$PYTHON" -c "import h5py" >/dev/null 2>&1 || [ "${CLAWHDF5_REQUIRE_INTEROP:- # libhdf5's registered plugins) compile and run too. run_step "h5py interop (format, ignored tests)" cargo test \ -p clawhdf5-format --features lz4,zstd --test writer_h5py_tests -- --include-ignored + # LZF, bitshuffle, bzip2 and Blosc both ways against h5py + hdf5plugin. + run_step "h5py interop (plugin filters)" cargo test \ + -p clawhdf5 --features plugin-filters --test plugin_filters_interop else echo "" echo "==> [h5py interop] SKIPPED: no h5py in $PYTHON" diff --git a/scripts/h5rs-check-ok-files.sh b/scripts/h5rs-check-ok-files.sh new file mode 100755 index 0000000..eee3fa4 --- /dev/null +++ b/scripts/h5rs-check-ok-files.sh @@ -0,0 +1,79 @@ +#!/usr/bin/env bash +# scripts/h5rs-check-ok-files.sh — run `h5rs check` over the files the +# conformance sweep reads correctly and completely: the `ok_files` of +# conformance/baseline.json (clawhdf5 and h5py agree on every object) whose +# probe results record no error at all (some ok files are deliberately +# corrupt test files that both readers refuse in the same way; those are +# left out). A validator that flags a file libhdf5 and clawhdf5 both read +# in full has a false positive — unless the file really is damaged in a way +# readers tolerate, which the report must then show. +# +# Usage: scripts/h5rs-check-ok-files.sh [--data] [SAMPLE] +# --data pass --data to check (decode every chunk) +# SAMPLE check only every SAMPLE-th file (default 1: all of them) +# +# Environment: H5RS (binary; default builds release), CONFORMANCE_CACHE +# (corpus cache, default conformance/.cache), CONFORMANCE_OUT (the sweep's +# results, default $CONFORMANCE_CACHE/results; run conformance/run.sh first), +# TMO (timeout, default 60). +# +# Exit status: 0 = no file flagged; 1 = some file flagged or failed (listed); +# 2 = setup error. +set -uo pipefail +HERE="$(cd "$(dirname "$0")" && pwd)" +ROOT="$(cd "$HERE/.." && pwd)" +export PATH="$HOME/.cargo/bin:$PATH" +DATA=() +if [ "${1:-}" = "--data" ]; then DATA=(--data); shift; fi +SAMPLE="${1:-1}" +CACHE="${CONFORMANCE_CACHE:-$ROOT/conformance/.cache}" +C="$CACHE/corpus" +R="${CONFORMANCE_OUT:-$CACHE/results}" +[ -d "$C" ] || { echo "error: no corpus in $C (run conformance/fetch-corpus.sh)" >&2; exit 2; } +[ -d "$R/runs" ] || { echo "error: no sweep results in $R (run conformance/run.sh)" >&2; exit 2; } +if [ -z "${H5RS:-}" ]; then + cargo build -q --release -p clawhdf5-tools --manifest-path "$ROOT/Cargo.toml" || exit 2 + H5RS="${CARGO_TARGET_DIR:-$ROOT/target}/release/h5rs" +fi +TMO="${TMO:-60}" + +mapfile -t FILES < <(python3 -c ' +import json, os, sys +b = json.load(open(sys.argv[1])) +runs, step = sys.argv[3], int(sys.argv[2]) +def clean(f): + """Both readers read every object, attribute and group listing.""" + for side in ("ours.json", "ref.json"): + try: + d = json.load(open(os.path.join(runs, f.replace("/", "__"), side))) + except (OSError, ValueError): + return False + if "open_error" in d or d.get("truncated"): + return False + for o in d.get("objects", []): + if "error" in o or "attrs_error" in o or "list_error" in o: + return False + if any(isinstance(a, dict) and "error" in a for a in o.get("attrs", {}).values()): + return False + return True +files = [f for f in b["ok_files"] if clean(f)] +total = len(b["ok_files"]) +print(f"{len(files)} of {total} ok files are read in full by both readers", file=sys.stderr) +for i, f in enumerate(files): + if i % step == 0: + print(f) +' "$ROOT/conformance/baseline.json" "$SAMPLE" "$R/runs") + +flagged=0 checked=0 +for f in "${FILES[@]}"; do + checked=$((checked + 1)) + out=$(timeout "$TMO" "$H5RS" check -q "${DATA[@]}" "$C/$f" 2>&1) + rc=$? + if [ $rc -ne 0 ]; then + flagged=$((flagged + 1)) + echo "== rc=$rc $f" + echo "$out" | head -5 | sed 's/^/ /' + fi +done +echo "== checked $checked ok files: $flagged flagged" +[ $flagged -eq 0 ] diff --git a/scripts/h5rs-fuzz.sh b/scripts/h5rs-fuzz.sh new file mode 100755 index 0000000..0edf97e --- /dev/null +++ b/scripts/h5rs-fuzz.sh @@ -0,0 +1,115 @@ +#!/usr/bin/env bash +# scripts/h5rs-fuzz.sh — run every `h5rs` subcommand over every file in a +# corpus of hostile HDF5 files, each under a timeout and a memory limit, and +# fail on any panic, crash or hang. +# +# Usage: scripts/h5rs-fuzz.sh [CORPUS_DIR ...] +# default corpus: conformance/.cache/corpus/cve_hdf5 (fetch it with +# conformance/fetch-corpus.sh; the HDF Group's CVE reproducers) +# +# Environment: +# H5RS h5rs binary to test (default: a debug build, for overflow checks) +# TMO per-run timeout in seconds (default 60) +# MEM_KB per-run address-space limit in KiB (default 4 GiB) +# JOBS files in parallel (default nproc) +# MUTATE also run on N byte-flipped copies of each file (default 0) +# MAX_BYTES --max-bytes for the commands that read values (default 16 MiB) +# +# A run may exit 0 (fine), 1 (problems found / could not read something) or +# 2 (error). Anything else fails the sweep: 3 is a caught panic (h5rs prints +# "internal error"), 124 a timeout, 128+N a signal (crash, abort, OOM kill). +# +# Exit status: 0 = every run ended cleanly; 1 = at least one did not (listed); +# 2 = setup error. +set -uo pipefail +HERE="$(cd "$(dirname "$0")" && pwd)" +ROOT="$(cd "$HERE/.." && pwd)" +export PATH="$HOME/.cargo/bin:$PATH" + +CORPORA=("$@") +[ ${#CORPORA[@]} -eq 0 ] && CORPORA=("$ROOT/conformance/.cache/corpus/cve_hdf5") +for c in "${CORPORA[@]}"; do + [ -d "$c" ] || { echo "error: no corpus at $c (run conformance/fetch-corpus.sh)" >&2; exit 2; } +done + +if [ -z "${H5RS:-}" ]; then + # A debug build: overflow checks turn silent wraparound on hostile sizes + # into a caught panic this sweep reports. + echo "== building h5rs (debug, with overflow checks)" + cargo build -q -p clawhdf5-tools --manifest-path "$ROOT/Cargo.toml" || exit 2 + TD="${CARGO_TARGET_DIR:-$ROOT/target}" + H5RS="$TD/debug/h5rs" +fi +[ -x "$H5RS" ] || { echo "error: $H5RS is not executable" >&2; exit 2; } +export H5RS TMO="${TMO:-60}" MEM_KB="${MEM_KB:-4194304}" MUTATE="${MUTATE:-0}" +export MAX_BYTES="${MAX_BYTES:-16777216}" +JOBS="${JOBS:-$(nproc 2>/dev/null || echo 4)}" + +WORK="$(mktemp -d)" +trap 'rm -rf "$WORK"' EXIT +export WORK + +# Every file in the corpora, whatever its extension (the reproducers often +# have none), except the corpus's own scripts and docs. +find -L "${CORPORA[@]}" -type f ! -name '*.md' ! -name '*.yml' ! -name '*.sh' \ + ! -name '*.py' ! -name 'COPYING' ! -name '*.gif' | sort > "$WORK/files.txt" +N=$(wc -l < "$WORK/files.txt") +[ "$N" -gt 0 ] || { echo "error: no files in ${CORPORA[*]}" >&2; exit 2; } +echo "== $N files, $JOBS at a time (timeout ${TMO}s, limit $((MEM_KB / 1024)) MiB, $MUTATE mutations each)" + +one() { + local f="$1" out="$WORK/fail.$$.$RANDOM" + local inputs=("$f") + if [ "$MUTATE" -gt 0 ]; then + local i + for ((i = 0; i < MUTATE; i++)); do + local m="$WORK/mut.$$.$i" + python3 - "$f" "$m" "$i" <<'PY' || continue +import random, sys +src, dst, seed = sys.argv[1], sys.argv[2], int(sys.argv[3]) +data = bytearray(open(src, "rb").read()) +if not data: + sys.exit(1) +rng = random.Random(f"{src}:{seed}") +for _ in range(rng.randint(1, 8)): + data[rng.randrange(len(data))] ^= 1 << rng.randrange(8) +open(dst, "wb").write(data) +PY + inputs+=("$m") + done + fi + local x + for x in "${inputs[@]}"; do + local cmd + # --max-bytes bounds the work per run: a valid 1 GiB dataset (the libhdf5 + # test files have some) is not what this sweep is looking for. + local mb="--max-bytes $MAX_BYTES" + for cmd in "ls -r -v $mb" "dump $mb" "dump --json $mb" "dump -p -A" "stat" "check --data $mb" "diff SELF"; do + local argv + read -r -a argv <<< "$cmd" + if [ "${argv[0]}" = diff ]; then argv=(diff --max-bytes "$MAX_BYTES" "$x"); fi + ( ulimit -v "$MEM_KB"; exec timeout -k 2 "$TMO" "$H5RS" "${argv[@]}" "$x" ) \ + >/dev/null 2>"$out.err" + local rc=$? + if [ $rc -gt 2 ]; then + { + echo "rc=$rc: h5rs ${argv[*]} $x" + [ "$x" != "$f" ] && echo " (a mutation of $f)" + grep -m3 'internal error' "$out.err" | sed 's/^/ /' + } >> "$WORK/failures.txt.$$" + fi + done + [ "$x" != "$f" ] && rm -f "$x" + done + rm -f "$out.err" +} +export -f one +xargs -a "$WORK/files.txt" -d '\n' -P "$JOBS" -I{} bash -c 'one "$1"' _ {} + +cat "$WORK"/failures.txt.* > "$WORK/failures.txt" 2>/dev/null || true +if [ -s "$WORK/failures.txt" ]; then + echo "== FAILED: $(grep -c '^rc=' "$WORK/failures.txt") run(s) panicked, crashed or hung:" + cat "$WORK/failures.txt" + exit 1 +fi +echo "== ok: every subcommand ended cleanly on all $N files"