Compare commits
85
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
bb78d70b99 | ||
|
|
a7de15534c | ||
|
|
10d1029ead | ||
|
|
883980f2bd | ||
|
|
d6e426e6d5 | ||
|
|
256e7b89e4 | ||
|
|
f2e704abf3 | ||
|
|
61f36516d7 | ||
|
|
45720fe5a6 | ||
|
|
adf961c883 | ||
|
|
b4a44a2e66 | ||
|
|
17fa783dce | ||
|
|
90e050944f | ||
|
|
a6e90f3ee3 | ||
|
|
0555794850 | ||
|
|
efc2dc53c9 | ||
|
|
5c2f656fe7 | ||
|
|
945b13a1f1 | ||
|
|
9179aa356e | ||
|
|
e94a52a88b | ||
|
|
d54a0f4737 | ||
|
|
aadfd18d4c | ||
|
|
2c6c6c176e | ||
|
|
190918a478 | ||
|
|
38d0d4de02 | ||
|
|
1c85986079 | ||
|
|
c7092722aa | ||
|
|
36356ba8a1 | ||
|
|
85eb7f5ce2 | ||
|
|
8196fab72a | ||
|
|
8ebd488d9e | ||
|
|
42b81d9f1c | ||
|
|
72b9cfb1e1 | ||
|
|
650f355219 | ||
|
|
7f5cfee281 | ||
|
|
c5302e587e | ||
|
|
e1115bc92a | ||
|
|
36d7a6f234 | ||
|
|
4b23ad697c | ||
|
|
e7f2d8575d | ||
|
|
7c1968a34a | ||
|
|
6db13c60b8 | ||
|
|
d99426be94 | ||
|
|
f5505fb03d | ||
|
|
95dcb04454 | ||
|
|
1dba7b465a | ||
|
|
57e938c438 | ||
|
|
3000b40cf3 | ||
|
|
bc820fbd8c | ||
|
|
9066d34eaa | ||
|
|
540fa08907 | ||
|
|
14876b8ae5 | ||
|
|
5935e13866 | ||
|
|
8c3ef996ea | ||
|
|
74fdf0582b | ||
|
|
44f5f8b5c5 | ||
|
|
2f252df084 | ||
|
|
c8c2930fc0 | ||
|
|
417c9516ca | ||
|
|
53dbddb07b | ||
|
|
081341b433 | ||
|
|
d074385944 | ||
|
|
4a1876faf2 | ||
|
|
be88e3fec7 | ||
|
|
e162c013fd | ||
|
|
06dda26d85 | ||
|
|
585e14d5e2 | ||
|
|
183d96ee26 | ||
|
|
bba1560416 | ||
|
|
b36998ef01 | ||
|
|
aef8e766ae | ||
|
|
9ea44d473d | ||
|
|
46203ea761 | ||
|
|
75bdb53342 | ||
|
|
dd5b3f6633 | ||
|
|
79dfa78e8f | ||
|
|
87d64588e5 | ||
|
|
bdadf3447c | ||
|
|
0c65a27b00 | ||
|
|
db9af7972c | ||
|
|
7706697feb | ||
|
|
c0f704c381 | ||
|
|
a7920bd4b3 | ||
|
|
7e43b5366c | ||
|
|
73bb068264 |
@@ -33,10 +33,10 @@ jobs:
|
||||
# build (pure-Rust zlib-rs) does not need it.
|
||||
apt-get install -y --no-install-recommends python3 python3-venv cmake
|
||||
python3 -m venv /opt/interop
|
||||
/opt/interop/bin/pip install --no-cache-dir h5py numpy netCDF4 xarray
|
||||
/opt/interop/bin/pip install --no-cache-dir h5py numpy netCDF4 xarray hdf5plugin
|
||||
echo "/opt/interop/bin" >> "$GITHUB_PATH"
|
||||
- name: Show interop library versions
|
||||
run: /opt/interop/bin/python -c "import h5py, netCDF4; print('h5py', h5py.__version__, 'HDF5', h5py.version.hdf5_version, 'netCDF4', netCDF4.__version__)"
|
||||
run: /opt/interop/bin/python -c "import h5py, netCDF4, hdf5plugin; print('h5py', h5py.__version__, 'HDF5', h5py.version.hdf5_version, 'netCDF4', netCDF4.__version__, 'hdf5plugin', hdf5plugin.version)"
|
||||
- name: Run CI script
|
||||
env:
|
||||
# Name the interpreter outright rather than relying on $GITHUB_PATH
|
||||
|
||||
@@ -0,0 +1,56 @@
|
||||
name: Conformance
|
||||
# Nightly: read every file of the pinned public HDF5 corpora with clawhdf5 and
|
||||
# with h5py/libhdf5 and compare (conformance/run.sh; CONFORMANCE.md explains
|
||||
# the method). Fails on any panic, hang, crash or out-of-memory in clawhdf5,
|
||||
# and when the ok count drops below conformance/baseline.json or a file the
|
||||
# baseline lists as ok stops being ok. The report is printed into the job log;
|
||||
# nothing is uploaded (artifact actions are JavaScript, which rust:latest
|
||||
# cannot run — see CLAUDE.md).
|
||||
on:
|
||||
schedule:
|
||||
- cron: "17 3 * * *"
|
||||
workflow_dispatch:
|
||||
jobs:
|
||||
conformance:
|
||||
runs-on: ubuntu-latest
|
||||
container: rust:latest
|
||||
timeout-minutes: 60
|
||||
env:
|
||||
CARGO_NET_RETRY: "10"
|
||||
steps:
|
||||
# Plain git, not actions/checkout (a JavaScript action; see ci.yml).
|
||||
- name: Check out
|
||||
run: |
|
||||
git init -q .
|
||||
git remote add origin "${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}.git"
|
||||
for i in 1 2 3; do git fetch -q --depth 1 origin "${GITHUB_SHA}" && break; sleep 5; done
|
||||
git checkout -q FETCH_HEAD
|
||||
- name: Install h5py, h5dump and the probe's codec libraries
|
||||
# hdf5-tools: h5dump for the CVE-corpus comparison. libaec-dev and
|
||||
# pkg-config: the probe builds clawhdf5-format with `szip` (the core
|
||||
# crates' default build needs neither).
|
||||
run: |
|
||||
apt-get update
|
||||
apt-get install -y --no-install-recommends python3 python3-venv hdf5-tools libaec-dev pkg-config
|
||||
python3 -m venv /opt/conformance
|
||||
/opt/conformance/bin/pip install --no-cache-dir -r conformance/requirements.txt
|
||||
/opt/conformance/bin/python -c "import h5py, hdf5plugin; print('h5py', h5py.__version__, 'HDF5', h5py.version.hdf5_version, 'hdf5plugin', hdf5plugin.version)"
|
||||
h5dump --version
|
||||
- name: Probe unit tests
|
||||
run: cargo test --release --manifest-path conformance/probe/Cargo.toml
|
||||
env:
|
||||
CARGO_TARGET_DIR: conformance/.cache/target
|
||||
- name: Sweep
|
||||
# The corpora come from GitHub (pinned commits, conformance/corpus.txt),
|
||||
# so this job needs a runner that reaches github.com.
|
||||
env:
|
||||
CLAWHDF5_PYTHON: /opt/conformance/bin/python
|
||||
run: bash conformance/run.sh
|
||||
- name: Report
|
||||
if: always()
|
||||
run: |
|
||||
if [ -f CONFORMANCE.md ]; then cat CONFORMANCE.md; else echo "no report was generated"; fi
|
||||
if [ -f conformance/.cache/results/summary.md ]; then
|
||||
echo; echo "---- per-file detail (conformance/.cache/results/summary.md) ----"
|
||||
cat conformance/.cache/results/summary.md
|
||||
fi
|
||||
@@ -243,6 +243,30 @@ index asked for ~16 000 candidates, where scanning the few hundred or thousand
|
||||
allowed records is exact and cheap. Re-ranking a 3k candidate pool and
|
||||
confidence rejection add about 3%.
|
||||
|
||||
### Signed checkpoints
|
||||
|
||||
Measured 2026-09-25 on tank (AMD Ryzen 7 7800X3D). A default store (float16,
|
||||
int8 index), 384-dim; each checkpoint rewrites the whole file, as every
|
||||
checkpoint does. Medians of five checkpoints and three verifies; three runs
|
||||
agreed to within the ranges shown.
|
||||
|
||||
```bash
|
||||
cargo run --release -p clawhdf5-bench --bin search_harness -- --signing-study --full
|
||||
```
|
||||
|
||||
| N | checkpoint, unsigned | checkpoint, signed | signing adds | `verify` | file size added |
|
||||
|---:|---:|---:|---:|---:|---:|
|
||||
| 1 000 | 5.4 ms | 6.4 ms | 0.7–1.0 ms | 2.1 ms | 0.03 MiB |
|
||||
| 10 000 | 46 ms | 55 ms | 8.1–9.4 ms | 18.6 ms | 0.31 MiB |
|
||||
| 100 000 | 495 ms | 598 ms | 89–112 ms | 247 ms | 3.05 MiB |
|
||||
|
||||
Signing costs about 20% of a checkpoint: every record is rehashed (SHA-256)
|
||||
and the Merkle root recomputed each time; the Ed25519 signature itself is
|
||||
microseconds. Caching per-record hashes between checkpoints would cut this to
|
||||
the records that changed. The per-record hashes stored for locating edits are
|
||||
32 bytes each (4% of a 100K float16 store). `verify` reads and rehashes the
|
||||
whole checkpoint.
|
||||
|
||||
### float16 embedding storage (`MemoryConfig::float16`)
|
||||
|
||||
Measured 2026-09-23 on tank (AMD Ryzen 7 7800X3D). The same clustered
|
||||
|
||||
+327
@@ -3,6 +3,58 @@
|
||||
## Unreleased
|
||||
|
||||
### Upgrade Notes
|
||||
- **HDF5 correctness audit (2026-09-25).** A sweep of 686 public files (the
|
||||
libhdf5 test files, the HDF Group's CVE reproducers, pyfive, netcdf-c,
|
||||
netcdf4-python, h5wasm, h5py and xarray corpora), a 567-case read matrix and
|
||||
a 96-case write matrix against HDF5 1.10–2.0 found bugs that returned wrong
|
||||
values with no error, and files we wrote that libhdf5 rejects. The fixes are
|
||||
listed under Correctness and Interop. What changes for callers:
|
||||
- **Chunked datasets whose max shape is larger than their current shape**,
|
||||
or whose unlimited dimension is not the first, were indexed by the current
|
||||
shape instead of the max shape, both when read and when written. Files from
|
||||
libhdf5 now read correctly. Files clawhdf5 wrote with such a max shape were
|
||||
laid out wrongly and now read the way libhdf5 always read them — rewrite
|
||||
them. Agent stores and ClawBrainHub files have no max shape and are
|
||||
unaffected.
|
||||
- Integer reads (`read_i32`/`read_i64`/`read_u64`/...) of float data now
|
||||
convert (truncate toward zero, saturate at the type's range, NaN reads as
|
||||
0) instead of returning the IEEE bit pattern, and out-of-range integers
|
||||
saturate instead of keeping the low bits.
|
||||
- `FileWriter::finish()` now returns an error instead of writing a corrupt
|
||||
file for: a header message over 64 KiB (e.g. an attribute larger than
|
||||
~64 KiB), a group/dataset/link name that is empty, `.` or contains `/`
|
||||
(nested paths were written as one literal link), a max shape smaller than
|
||||
the shape, a page size outside 512 B–1 GiB, and more than 65 535 chunks in
|
||||
a dataset with several unlimited dimensions.
|
||||
- **Breaking (format crate):** `ObjectHeaderWriter::serialize`,
|
||||
`BatchObjectHeaderWriter::compute_sizes`/`serialize_all` and
|
||||
`build_chunked_data_from_precompressed` return `Result`;
|
||||
`read_fixed_array_chunks`/`read_extensible_array_chunks` take `max_dims`;
|
||||
`build_fixed_array_at`/`ea_writer::build_extensible_array_at` take one
|
||||
`Option<WrittenChunk>` per index slot; `fill_value::dataset_fill_value`
|
||||
returns `UnresolvedSharedMessage` for a shared message it cannot resolve
|
||||
instead of `None`. `FillTime::default()` is `IfSet` (libhdf5's default;
|
||||
default files are byte-identical).
|
||||
- **ZeroClaw does not use clawhdf5.** The project described itself as
|
||||
ZeroClaw's memory backend ("imported as a `clawhdf5` Cargo feature"). Checked
|
||||
against ZeroClaw v0.8.5 (the latest release), the `osobh/zeroclaw` fork and
|
||||
their full history: no such feature or backend has ever existed. And
|
||||
`clawhdf5-migrate`'s "ZeroClaw layout" (`memory_chunks`, `sessions`,
|
||||
`entities`, `relations`) is not ZeroClaw's schema — ZeroClaw uses a single
|
||||
`memories` table — so the migrator cannot read a ZeroClaw database. The
|
||||
claims are withdrawn; the migrator's layout is documented as its own.
|
||||
- **OpenClaw is not supported, and never was.** The docs described a
|
||||
"drop-in" OpenClaw memory backend enabled with `memory.backend = "clawhdf5"`.
|
||||
That config was never valid in any OpenClaw release (v2026.2–v2026.7
|
||||
accepted only `builtin`/`qmd` and rejected unknown keys, so a Gateway given
|
||||
it refuses to start; OpenClaw 2.0 removed the key), no plugin was ever built,
|
||||
and `@redclaw/clawhdf5` was never published. The integration docs
|
||||
(`openclaw-integration.md`, `openclaw-config.md`, `migration-guide.md`) are
|
||||
removed; `docs/openclaw.md` explains the status and what a real plugin would
|
||||
need against OpenClaw v2026.9.6. `ClawhdfBackend` stays as a library API.
|
||||
- **Breaking:** `MemoryError` is now `#[non_exhaustive]` and gained
|
||||
`SigningKeyRequired`; a `match` on it needs a wildcard arm. Future variants
|
||||
will no longer be breaking.
|
||||
- **Breaking:** `clawhdf5-agent`'s `agent` feature is removed. It enabled
|
||||
nothing — the agent layer is always built — but the README and guides told
|
||||
people to pass it; drop `agent` from `features = [...]`.
|
||||
@@ -60,6 +112,29 @@
|
||||
`quantized_index = false`, or pass `create --f32-index` to the CLI, to opt
|
||||
out. The CLI's `--quantized-index` is still accepted but is now a no-op.
|
||||
|
||||
### Signing
|
||||
- `clawhdf5-agent`: **Ed25519-signed checkpoints** — the README's
|
||||
"cryptographically verifiable memory", now true. With
|
||||
`HDF5Memory::set_signing_key(key)`, every checkpoint stores a signed
|
||||
manifest: a SHA-256 per record (text, embedding as stored, channel,
|
||||
timestamp, session, tags, deleted flag, activation) in a Merkle tree, plus
|
||||
hashes of the settings (and WAL mark), sessions and knowledge graph, with
|
||||
the per-record hashes in `/integrity/record_hashes`.
|
||||
`HDF5Memory::verify(path, &public_key)` recomputes everything from the file
|
||||
and reports which part changed and which records (`changed_records`); a
|
||||
forged manifest fails the signature. The key is never persisted; a signed
|
||||
store refuses to checkpoint without it (`MemoryError::SigningKeyRequired`),
|
||||
and `remove_signature()` is the deliberate way back to unsigned. Saves still
|
||||
in the WAL are not covered (`wal_entries_unsigned`). Tests include every
|
||||
kind of edit, and an edit made with h5py in place, which verify pinpoints.
|
||||
Cost: ~20% of a checkpoint, 32 bytes per record (`BENCHMARKS.md`, "Signed
|
||||
checkpoints"). New dependencies `ed25519-dalek`, `sha2`, `rand_core` — pure
|
||||
Rust; the no-C check still passes.
|
||||
- `clawhdf5-cli`: `keygen --out <file>` (owner-only key file),
|
||||
`--signing-key <file>` / `CLAWHDF5_SIGNING_KEY` on writing commands
|
||||
(`create` signs immediately), `verify --public-key <hex|file>` (JSON report;
|
||||
exit status 2 if not valid), and `signed` in `create`/`stats` output.
|
||||
|
||||
### Migration
|
||||
- `clawhdf5-migrate`: writes through the agent's own API (`HDF5Memory::create`
|
||||
/ `open`, `save_batch`, the session cache and knowledge graph), so there is
|
||||
@@ -100,6 +175,13 @@
|
||||
activation of the `k` results it returns, not of the whole `3k` candidate
|
||||
pool it re-ranks.
|
||||
|
||||
### Documentation
|
||||
- OpenClaw claims withdrawn across the README, QUICKSTART, USE_CASES, ROADMAP
|
||||
(Track 7 marked withdrawn) and the `openclaw` module docs; the dead
|
||||
`github.com/redclawsystems/openclaw` link is gone. The Node package is
|
||||
marked unpublished and broken (now `"private": true` so it cannot be
|
||||
published by accident), with its bugs recorded in `docs/known-issues.md`.
|
||||
|
||||
### Benchmarks
|
||||
- Every undated or pre-September section of `BENCHMARKS.md` re-run on one
|
||||
machine on one day (tank, 2026-09-24, commit 5c8323c), with the command for
|
||||
@@ -115,6 +197,23 @@
|
||||
takes `--f32`; it had kept printing "f32" after the default changed.
|
||||
|
||||
### Interop
|
||||
- **Conformance sweep in the repo** (`conformance/`, report in
|
||||
`CONFORMANCE.md`). `conformance/run.sh` fetches eight public HDF5 corpora
|
||||
pinned by commit (libhdf5's test files, the HDF Group's CVE reproducers,
|
||||
pyfive, netcdf-c, netcdf4-python, h5wasm, h5py, xarray-data) into a
|
||||
gitignored cache, reads every file with clawhdf5 and with h5py/libhdf5 (and
|
||||
the CVE files with h5dump) under a timeout and memory limit, compares them
|
||||
object by object and regenerates the report — about 30 s once the corpus is
|
||||
cached. A nightly Gitea job (`.gitea/workflows/conformance.yml`) runs it and
|
||||
fails on any panic, hang, crash or out-of-memory, or when a file in
|
||||
`conformance/baseline.json` stops reading identically. First report, on
|
||||
42b81d9: 467 of 697 files identical to h5py, 123 our-error, 15 mismatch
|
||||
(2 of them an h5py bug), 92 that libhdf5 cannot read, no panics, hangs or
|
||||
crashes. Compared with the ad-hoc audit sweep, the probe now compares
|
||||
N-Bit floats (and integers with a bit offset) as the values libhdf5
|
||||
converts them to rather than raw file bytes — 8 files that were reported as
|
||||
mismatches read identically — and the reference side no longer flips
|
||||
between runs when libhdf5 aborts while freeing h5py objects.
|
||||
- `clawhdf5-format`: **every `f32` dataset was unreadable by h5py and
|
||||
libhdf5.** The float datatype encoder hard-coded the sign bit's position to
|
||||
63, correct only for `f64`; libhdf5 validates it and refused the dataset. It
|
||||
@@ -129,6 +228,37 @@
|
||||
`float16` rounding matches numpy's bit for bit on 4 020 probe values,
|
||||
including ties, subnormals and the overflow boundary), and an agent store —
|
||||
`f32` and `float16` — opened by h5py with every dataset decoded.
|
||||
- `clawhdf5-format` filters, checked against libhdf5 + hdf5plugin:
|
||||
- **LZ4 (32004) now uses the registered HDF5 LZ4 format** (8-byte BE size,
|
||||
4-byte BE block size, BE-length-prefixed blocks). Our old framing (4-byte
|
||||
LE size + one block) was readable only by clawhdf5, and we could not read
|
||||
libhdf5's (`h5ex_d_lz4.h5`). Old clawhdf5 LZ4 chunks still read; they are
|
||||
told apart unambiguously (a registered chunk starts with four zero bytes).
|
||||
- **Zstd (32015) frames now record the content size**, which libhdf5's zstd
|
||||
plugin needs; h5py could not read our zstd datasets.
|
||||
- **Pcodec moved from filter ID 32023 to 480.** 32023 is registered to
|
||||
Granular BitRound, whose decode is a pass-through — libhdf5 with that
|
||||
plugin would have returned compressed bytes as data. Pcodec has no
|
||||
registered ID; 480 is in the registry's private range (256–511) and only
|
||||
clawhdf5 can read it. Chunks written under 32023 with the filter name
|
||||
`pcodec` (clawhdf5 ≤ 2.7.0) still read.
|
||||
- **SZIP decode matches libhdf5.** It returned garbage or zeros with no
|
||||
error for libhdf5-written files (the 4-byte size prefix, 32/64-bit
|
||||
byte-plane interleaving, reference interval, scanline padding and byte
|
||||
order were all handled wrongly) and rejected 64-bit data.
|
||||
- N-Bit honours libhdf5's "need not compress" flag (multi-filter pipelines
|
||||
such as `tfilters.h5` failed) and reads enum/no-op members.
|
||||
- Scale-offset `float` decode uses libhdf5's single-precision arithmetic
|
||||
(was 1 ULP off for some values).
|
||||
- A pipeline with Fletcher32 ahead of the compressor (h5py
|
||||
`set_fletcher32()` then `set_deflate()`) no longer fails with "deflate:
|
||||
output exceeds size limit".
|
||||
- `clawhdf5-format`: **HDF5 1.4/1.6-era files are readable.** Data Layout
|
||||
message versions 1 and 2 (compact, contiguous, and chunked through the
|
||||
version-1 B-tree) failed with `InvalidLayoutVersion` — 84 of the 686 files in
|
||||
the 2026-09-25 audit sweep, 205 datasets. They now read as libhdf5 does;
|
||||
checked byte for byte against h5py on HDF5's own test files
|
||||
(`tests/legacy_format_interop.rs`).
|
||||
|
||||
### Storage
|
||||
- `clawhdf5-format`: **half-precision datasets.**
|
||||
@@ -166,6 +296,203 @@
|
||||
- CI keeps zlib-ng building and tested; the arm64 job no longer needs cmake.
|
||||
|
||||
### Correctness
|
||||
- `clawhdf5-format` VDS: variable-length and reference data from a source in
|
||||
another file is refused. Those elements are global-heap IDs and object
|
||||
addresses in the source file; copied into the virtual dataset they would
|
||||
be decoded against the wrong file and name another object.
|
||||
- `clawhdf5-agent`: a store whose `/meta` has an attribute that cannot be
|
||||
decoded fails to open (`MemoryError::Schema`). With `attrs()` now leaving
|
||||
unreadable attributes out, it would otherwise have opened with defaults in
|
||||
place of its settings (`float16`, `compression`, the WAL mark, ...).
|
||||
- `clawhdf5-format` reader: an old-style group whose local heap has a free
|
||||
list pointing outside the heap was listed with names read from the broken
|
||||
heap (garbage names on `cve-2021-36977.h5` once its user block was
|
||||
applied). libhdf5 refuses such a heap ("bad heap free list"); so do we now,
|
||||
with `FormatError::InvalidLocalHeapFreeList`. As in libhdf5 the free list
|
||||
is checked when the first name is read (`LocalHeap::validate_free_list`,
|
||||
new), so an empty group with a damaged heap still lists as empty.
|
||||
- **Files with a user block** (`h5py.File(..., userblock_size=N)`, `h5jam`;
|
||||
the superblock at 512, 1024, …) could not be read: every address in the
|
||||
file is relative to the superblock, but it was applied from byte 0
|
||||
(`InvalidObjectHeaderVersion` on the root group). `File` (mmap, buffered,
|
||||
`from_bytes`), `MmapFile`, `LazyFile`, `AsyncHDF5File`, the VOL readers,
|
||||
the HNSW loader and external VDS sources now view the file from the
|
||||
superblock on, using the signature's position as the base address as
|
||||
libhdf5 does; `user_block_size()` reports the user block (h5py's
|
||||
`userblock_size`), and `as_bytes()` returns the bytes from the superblock
|
||||
on. **Breaking (format crate):** `Superblock::parse` refuses a non-zero
|
||||
signature offset with `FormatError::UserBlockNotStripped`, since the
|
||||
addresses it returns would be applied to the wrong bytes; pass the slice
|
||||
from `signature::split_user_block` (new) and parse at offset 0.
|
||||
- `clawhdf5-format` reader: version-1 shared messages (HDF5 1.6-era files,
|
||||
e.g. a dataset using a committed datatype in libhdf5's `tcompound.h5`)
|
||||
read the heap-offset field of the embedded symbol-table entry as the
|
||||
target address and failed with `InvalidObjectHeaderVersion`. The address
|
||||
is now read after it, as libhdf5 does. **Breaking (format crate):**
|
||||
`shared_message::parse_shared_ref` takes `length_size`. A reference whose
|
||||
target header has no message of the referenced type is now
|
||||
`FormatError::SharedMessageTargetMissing` instead of returning the first
|
||||
other message found there (which decoded as garbage).
|
||||
- `clawhdf5-format` reader: array members of version-1 compound datatypes
|
||||
(HDF5 1.6-era files, e.g. libhdf5's `tcompound.h5`) were read as a single
|
||||
element: a `[4] i32` member came back as one `i32`, with the wrong size.
|
||||
The legacy per-member dimension fields are now decoded into an array type,
|
||||
as libhdf5 does; more than four dimensions, or a zero-sized one, is an
|
||||
error.
|
||||
- `clawhdf5-format` virtual datasets (VDS), checked against HDF5 2.0 through
|
||||
h5py (`crates/clawhdf5/tests/vds_interop.rs`):
|
||||
- **Wrong data:** elements no mapping supplies — unmapped regions, and
|
||||
mappings whose source file or dataset is missing — read as 0 instead of
|
||||
the virtual dataset's fill value (e.g. h5py `fillvalue=-1`). Assembly moved
|
||||
to the new `vds` module: `vds::read_virtual_dataset` takes the fill value
|
||||
and a resolver that can refuse a name (`VdsFileResolver`), and `File`
|
||||
passes the dataset's fill value. A missing source *dataset* read as an
|
||||
error; it is fill now, as in libhdf5. Source datasets are read with their
|
||||
own fill value for unallocated chunks, and a source whose datatype differs
|
||||
from the virtual dataset's is an error (libhdf5 converts; we do not).
|
||||
`File` now refuses a source name that leaves the virtual file's directory
|
||||
(`../x.h5`, absolute paths), or any external source of a `File::from_bytes`
|
||||
file, with an error — these used to read as fill.
|
||||
**Behaviour change:** the raw-read API (`read_raw_data_full*`), which has
|
||||
no fill value, now returns an error for a virtual dataset with unmapped
|
||||
elements instead of zeros.
|
||||
- Unlimited and printf-style mappings are supported (all 7 VDS files in the
|
||||
libhdf5 test set are such mappings, e.g. Eiger/Percival detector layouts).
|
||||
`%b` in a source file or dataset name is the block number and `%%` a
|
||||
literal `%` (other `%` sequences are an error, as in libhdf5); block `j`
|
||||
is read from the source named with `j`, probing from 0 up to the first
|
||||
missing source. Unlimited source/virtual selections cover as much as the
|
||||
source's current extent fills, including a partial last block. As
|
||||
libhdf5 does on `H5Dget_space`, the extent is recomputed from the sources
|
||||
present (default "last available" view, printf gap 0) —
|
||||
`vds::virtual_dataset_extent`, used by `Dataset::shape()` — so e.g.
|
||||
`vds-eiger.h5` is `[5, 10, 10]`, not its stored `[20, 10, 10]`. A source
|
||||
stored in the other byte order is byte-swapped (libhdf5 converts);
|
||||
other type conversions remain an error.
|
||||
- Hyperslab selection versions 1 and 2 were refused ("only version-3
|
||||
hyperslab selections are supported"). Version 1 is what libhdf5 writes for
|
||||
every VDS created with the default format bounds (h5py's default), so
|
||||
those could not be read at all; version 2 is its encoding of an unlimited
|
||||
selection. Both are decoded now, as are irregular hyperslabs (a union of
|
||||
blocks, read in row-major order as libhdf5 iterates them).
|
||||
`SerializedSelection` exposes the raw form, including unlimited counts.
|
||||
- The version-1 mapping list HDF5 2.0 writes (low version bound 2.0) was
|
||||
misparsed: each entry's flags byte was read as the start of the source
|
||||
file name, and names shared with an earlier entry (stored as that entry's
|
||||
index) were not followed. Now decoded as `H5D__virtual_load_layout` does.
|
||||
- `clawhdf5-format` reader — **values returned wrong with no error:**
|
||||
- Fixed Array and Extensible Array chunk indexes were laid out by the
|
||||
dataset's current shape instead of its max shape (23 libhdf5 test files,
|
||||
and any h5py file with e.g. `maxshape=(10, None)` or `(20, 10)` under
|
||||
`libver='latest'`).
|
||||
- Files with 4-byte offsets: unfiltered chunked datasets read as zeros.
|
||||
Chunk B-tree keys store offsets in 8 bytes whatever the file's offset
|
||||
size.
|
||||
- A chunk's filter mask skipped the whole pipeline when any bit was set;
|
||||
only the flagged filters are skipped now.
|
||||
- Float data read as an integer returned the bit pattern; narrowing integer
|
||||
reads kept the low bits; bfloat16 was decoded as IEEE half. Floats are now
|
||||
decoded from their datatype fields (bf16, FP8 E4M3/E5M2, IEEE half, single
|
||||
and double).
|
||||
- `vl_data::read_vl_bytes` truncated sequences of non-byte base types.
|
||||
- A shared fill-value message read as zero fill; it is resolved now,
|
||||
including from the file's shared-message (SOHM) table, which could never
|
||||
resolve because its index version byte was skipped.
|
||||
- Two threads reading two chunked datasets through one `File` could get each
|
||||
other's chunks (the shared chunk cache was switched between datasets
|
||||
across separate lock acquisitions). The cache is now keyed by dataset.
|
||||
- Compound datatype version 1 members with legacy array dimensions (HDF5
|
||||
before 1.4, which had no array class) were read as a single scalar at
|
||||
the member's offset; they are now array members, as in libhdf5
|
||||
(`tarrold.h5`, `tcompound.h5`). Only reachable once layout versions 1/2
|
||||
were readable, since the files that use it are that old.
|
||||
- `clawhdf5-format` reader — errors on valid files: a version-1 shared
|
||||
message (a committed datatype in HDF5 1.4/1.6-era files) was read as if the
|
||||
object header address followed the reserved bytes; it follows a link-name
|
||||
offset (the reference is an old-style symbol table entry), so the reader
|
||||
followed the name offset and failed with `InvalidObjectHeaderVersion`
|
||||
(`tcompound.h5`). New `shared_message::parse_shared_ref_sized` takes the
|
||||
superblock's length size; `parse_shared_ref` assumes it equals the offset
|
||||
size.
|
||||
- `clawhdf5-format` reader — errors on valid files: enum and bool datasets
|
||||
through the numeric readers; the "don't filter partial edge chunks" layout
|
||||
flag; Fletcher32 ahead of deflate (NetCDF-4's order). Unknown-message flags
|
||||
follow libhdf5 (`tbogus.h5`): "fail if unknown" is refused, "fail if unknown
|
||||
and writing" is ignored by a reader.
|
||||
- `clawhdf5-format` reader — dense groups and attributes (links or
|
||||
attributes kept in a fractal heap indexed by a v2 B-tree):
|
||||
- A link heap larger than the root indirect block's direct rows (512 KiB
|
||||
with libhdf5's defaults: a few thousand long link names, or ~20 000 short
|
||||
ones) could not be listed: child indirect blocks were given the wrong
|
||||
number of rows, so every link stored in one was unreachable.
|
||||
- v2 B-trees of depth 3 or more (a dense group of ~22 000+ links) were
|
||||
misparsed: internal-node child pointers were read with widths from an
|
||||
estimate instead of libhdf5's per-depth record capacities, and the
|
||||
listing failed. The same B-tree code indexes dense attributes, shared
|
||||
messages and chunks.
|
||||
- Fractal-heap "huge" objects (larger than the heap's managed-object
|
||||
limit, 4 KiB by default — e.g. an 8 KiB dense attribute or a link with a
|
||||
very long name) and "tiny" objects are now read; the ID type was taken
|
||||
from the wrong bits (6-7, the version, instead of 4-5), so a huge object
|
||||
failed and took every attribute on its object down with it (NetCDF-4
|
||||
files such as netcdf4-python's `issue671.nc`). Huge objects are found
|
||||
directly from the ID or through the huge-object v2 B-tree, filtered or
|
||||
not.
|
||||
- Heaps with an I/O filter pipeline (a group created with a filter on its
|
||||
creation property list compresses its link heap) are now read: the
|
||||
header's pipeline was skipped with the wrong size, so its checksum was
|
||||
looked for in the wrong place, and filtered direct blocks were read raw.
|
||||
- A user-defined link (link class 65-255, e.g. 187 in libhdf5's
|
||||
`tall.h5`/`tudlink.h5`) made its whole group unlistable. Such links
|
||||
cannot be followed without the application that registered the class, so
|
||||
they are now left out of `datasets()`/`groups()` and path lookup, as h5py
|
||||
leaves out links it cannot open; reserved link types are still an error.
|
||||
- `clawhdf5` — soft links are listed, as h5py lists them: `datasets()` and
|
||||
`groups()` on `Group`/`MmapGroup`/`LazyGroup` include each soft link under
|
||||
its own name as the kind of object it resolves to, and `dataset(name)` /
|
||||
`group(name)` open through it. Relative targets resolve from the group
|
||||
holding the link. Dangling or cyclic soft links, external links and
|
||||
user-defined links are left out (h5py lists their names but cannot open
|
||||
them). Previously soft links were missing from the listings, and in
|
||||
old-style (symbol table) groups a soft link made the listing fail. New
|
||||
`group_v2::resolve_group_children` / `resolve_path_from` and
|
||||
`group_v1::v1_soft_links` in `clawhdf5-format`.
|
||||
- `clawhdf5` — one unreadable attribute no longer fails `attrs()` for every
|
||||
attribute on its object: it is left out of the map, and the new
|
||||
`attrs_with_errors()` (on every group and dataset handle) returns the map
|
||||
plus one error per attribute left out. Returned values are always complete.
|
||||
An error in the attribute index itself (attribute info message, dense heap
|
||||
header or B-tree) still fails the call. `clawhdf5-format` gains
|
||||
`attribute::extract_attributes_tolerant`; `extract_attributes_full` stays
|
||||
strict.
|
||||
- `clawhdf5-format` reader — files with shared object header messages
|
||||
(SOHM, `H5Pset_shared_mesg_index`): a datatype, dataspace, filter pipeline
|
||||
or attribute stored in the file's SOHM heap failed with "invalid shared
|
||||
message version: 2" — only shared fill values loaded the SOHM table — so
|
||||
such files' datasets and attributes could not be read.
|
||||
`shared_message::resolve_shared_message` now loads the table when a
|
||||
reference needs it (36 cases of the audit's read matrix).
|
||||
- `clawhdf5-format` writer — **files libhdf5 rejects or reads wrong:**
|
||||
- Extensible Array (one unlimited dimension): chunks from index 244 on were
|
||||
written but never indexed and read as 0, by libhdf5 and by us.
|
||||
- Fixed Array: more than 1 024 chunks gave checksum errors (data blocks
|
||||
were never paged).
|
||||
- A finite max shape larger than the shape gave libhdf5 "addr overflow"; an
|
||||
unlimited dimension that is not the first scrambled the data; several
|
||||
unlimited dimensions (`(None, None)`) broke the whole file. These now
|
||||
write the index libhdf5 writes (swizzled Extensible Array, or a B-tree v2
|
||||
index for several unlimited dimensions).
|
||||
- Header messages over 64 KiB (the size field is 16 bits) and compact
|
||||
datasets at 65 534–65 535 bytes produced corrupt files.
|
||||
- Reference, Opaque, BitField and Time datatypes were written as empty
|
||||
messages; they now encode as HDF5 2.0 does.
|
||||
- `with_page_size` wrote a nonexistent superblock version 4; it now writes
|
||||
the v3 superblock and File Space Info message libhdf5 writes.
|
||||
- `FillTime` values were rotated on disk (NEVER was written as ALLOC, and so
|
||||
on). New `DatasetBuilder::with_fill_value`.
|
||||
- An empty-string attribute got a zero-size datatype, which made every
|
||||
attribute on the object unreadable in libhdf5.
|
||||
- `maxshape` equal to the shape no longer forces chunked layout.
|
||||
- `clawhdf5-format`: **a truncated deflate chunk read back short, with no
|
||||
error.** The deflate filter used flate2's streaming reader, which returns the
|
||||
bytes it has when the input runs out before the end-of-stream marker. It now
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
# clawhdf5
|
||||
|
||||
## Purpose
|
||||
Pure-Rust HDF5 format implementation with HNSW vector search, WAL-backed persistence, agent memory storage, and GPU-accelerated I/O. Used by ZeroClaw as its persistent memory and knowledge graph backend.
|
||||
Pure-Rust HDF5 format implementation with HNSW vector search, WAL-backed persistence, agent memory storage, and GPU-accelerated vector search. A standalone library. Its one verified consumer is ClawBrainHub (`.brain` files); no agent framework integrates it (OpenClaw and ZeroClaw claims were withdrawn on 2026-09-25 — neither was ever true).
|
||||
|
||||
## Architecture
|
||||
|
||||
@@ -11,13 +11,13 @@ Cargo workspace with 16 crates under `crates/` (plus `libaec-sys`, an internal F
|
||||
|-------|------|
|
||||
| `clawhdf5-format` | HDF5 binary spec parser (superblock, B-tree, heap) — also holds shared type definitions and physical constants |
|
||||
| `clawhdf5-io` | Read/write implementation |
|
||||
| `clawhdf5-filters` | Compression filters (gzip, LZ4, Zstd, Blosc) |
|
||||
| `clawhdf5-filters` | Deflate backends (zlib-rs, zlib-ng, Apple Compression); the HDF5 filter pipeline and the other codecs (LZ4, Zstd, SZIP, N-Bit, scale-offset, pcodec) live in `clawhdf5-format`. No Blosc. |
|
||||
| `clawhdf5-derive` | Proc-macro derive for HDF5-serializable structs |
|
||||
| `clawhdf5` | Main facade crate |
|
||||
| `clawhdf5-netcdf4` | NetCDF-4 compatibility layer |
|
||||
| `clawhdf5-ann` | HNSW approximate nearest-neighbor vector index |
|
||||
| `clawhdf5-agent` | Agent memory, session history, knowledge graph storage |
|
||||
| `clawhdf5-gpu` | GPU-accelerated I/O via wgpu (hand-written WGSL compute shaders) |
|
||||
| `clawhdf5-gpu` | GPU vector distance computation via wgpu (hand-written WGSL compute shaders) — not dataset I/O |
|
||||
| `clawhdf5-accel` | CPU SIMD acceleration path |
|
||||
| `clawhdf5-migrate` | SQLite → HDF5 agent-memory migration |
|
||||
| `clawhdf5-android` | Android JNI bindings |
|
||||
@@ -104,11 +104,34 @@ Cargo workspace with 16 crates under `crates/` (plus `libaec-sys`, an internal F
|
||||
the allowed records whenever cheaper than `pool × M` index distance
|
||||
evaluations, and as the fallback when the pool comes back short), fusion,
|
||||
activation scaling, optional re-ranking and confidence rejection.
|
||||
`hybrid_search`/`hybrid_search_with` are thin wrappers; the OpenClaw
|
||||
backend is `search` with re-rank + confidence on. Measure changes with
|
||||
`hybrid_search`/`hybrid_search_with` are thin wrappers; `ClawhdfBackend`
|
||||
(the `openclaw` module) is `search` with re-rank + confidence on.
|
||||
- **OpenClaw is not supported** (decided 2026-09-25): clawhdf5 is not an
|
||||
OpenClaw memory plugin and never was — the old `memory.backend = "clawhdf5"`
|
||||
config was never valid. Don't reintroduce OpenClaw claims; `docs/openclaw.md`
|
||||
records what a real plugin would need.
|
||||
- **ZeroClaw does not use clawhdf5** (checked 2026-09-25 against upstream
|
||||
v0.8.5 and the `osobh/zeroclaw` fork, and their full history): no
|
||||
`clawhdf5` feature or backend exists; ZeroClaw's memory backends are
|
||||
sqlite/lucid/postgres/qdrant/markdown/none behind its own `Memory` trait.
|
||||
`clawhdf5-migrate`'s default SQLite layout (`memory_chunks`, `sessions`,
|
||||
`entities`, `relations`) is not ZeroClaw's schema either (ZeroClaw's is a
|
||||
`memories` table). Don't reintroduce integration claims without an
|
||||
integration and a test against the real consumer. Measure changes with
|
||||
`search_harness --options-study`.
|
||||
- `MemoryConfig::compression` is off by default; when on, embeddings are
|
||||
deflate-compressed, or Zstd with the agent's `zstd` feature (links libzstd).
|
||||
- Signed checkpoints (`clawhdf5-agent` `signing` module): with
|
||||
`HDF5Memory::set_signing_key` every checkpoint stores an Ed25519-signed
|
||||
manifest (SHA-256 per record in a Merkle tree + settings/sessions/graph
|
||||
hashes; per-record hashes in `/integrity/record_hashes`);
|
||||
`HDF5Memory::verify(path, &pk)` locates edits. The hashes must cover exactly
|
||||
what the file persists in the form the loader returns it (strings lose
|
||||
trailing NULs; an empty WAL mark is not written) or untouched stores stop
|
||||
verifying — `tests/signed_store.rs` round-trips awkward strings. The key is
|
||||
never persisted; a signed store refuses to checkpoint without it
|
||||
(`MemoryError::SigningKeyRequired`, and `MemoryError` is `#[non_exhaustive]`).
|
||||
WAL entries after the checkpoint are not covered.
|
||||
- `Dataset::verify_provenance()` (clawhdf5 facade, `provenance` feature, on by
|
||||
default) recomputes a dataset's SHA-256 and compares it against the
|
||||
`_provenance_sha256` attribute written automatically on save when
|
||||
@@ -125,7 +148,7 @@ Cargo workspace with 16 crates under `crates/` (plus `libaec-sys`, an internal F
|
||||
Alerts never block a save — drain them with `HDF5Memory::take_anomaly_alerts`.
|
||||
`MemorySource` for this bookkeeping is inferred from the caller-supplied
|
||||
`source_channel` string (a heuristic, not an authenticated trust boundary).
|
||||
- GPU-accelerated batch I/O for large dataset processing
|
||||
- GPU-accelerated vector distance computation (`clawhdf5-gpu`, wgpu); HDF5 I/O itself is CPU-only
|
||||
- Python and Node.js bindings for cross-language use
|
||||
- NetCDF-4 compatibility for scientific data interop
|
||||
|
||||
@@ -173,4 +196,12 @@ python -c "import clawhdf5; print(clawhdf5.__version__)"
|
||||
```
|
||||
|
||||
## Integration
|
||||
ZeroClaw imports this as a Cargo feature (`clawhdf5` feature flag) to persist agent memory with HNSW vector search for context retrieval.
|
||||
- **ClawBrainHub** (`clawverse/clawbrainhub` on git.redclaw.dev) is the one
|
||||
verified consumer: `cbh-core` reads and writes `.brain` files through the
|
||||
facade (`File`, `FileBuilder`, `AttrValue`, `Selection`), `cbh-scanner`
|
||||
uses the facade, and `cbh-cli` uses `clawhdf5_agent::bm25::BM25Index`. It
|
||||
depends on this repo by path (`../clawhdf5`), so it builds against whatever
|
||||
is checked out — changes to those APIs reach it directly. Verified
|
||||
2026-09-25 against main: builds, and its 204 tests pass.
|
||||
- OpenClaw and ZeroClaw were both described as consumers; neither integrates
|
||||
clawhdf5 (see Key Features and `docs/openclaw.md`).
|
||||
|
||||
+300
@@ -0,0 +1,300 @@
|
||||
# clawhdf5 conformance report
|
||||
|
||||
Every HDF5 file of eight public corpora (pinned by commit) is read twice — by
|
||||
clawhdf5 (`conformance/probe`, the same `clawhdf5-format` calls the facade
|
||||
makes) and by h5py/libhdf5 (`conformance/ref.py`) — and the two readings are
|
||||
compared object by object: the set of hard-linked objects, each dataset's and
|
||||
attribute's shape, and a SHA-256 of its values in a canonical encoding. The
|
||||
CVE corpus is also run through `h5dump`. Each side runs under a timeout and an
|
||||
address-space limit, so a hang, crash or runaway allocation is recorded, not
|
||||
fatal. This file is generated by `conformance/run.sh`; do not edit it by hand.
|
||||
|
||||
## Run
|
||||
|
||||
| | |
|
||||
|---|---|
|
||||
| date | 2026-09-26 03:46 UTC |
|
||||
| clawhdf5 commit | `10d1029ead524e2fe64c2cd7f61b28067d9e449c` |
|
||||
| machine | `tank`: AMD Ryzen 7 7800X3D 8-Core Processor, 16 CPUs, 61 GiB, Linux 7.0.0-34-generic x86_64 |
|
||||
| command | `conformance/run.sh --no-fetch --update-baseline` |
|
||||
| rustc | rustc 1.98.1 (48a229cea 2026-09-01) |
|
||||
| reference | h5py 3.16.0, HDF5 2.0.0, numpy 2.5.3, hdf5plugin 7.1.0, Python 3.14.4 |
|
||||
| h5dump | Version 1.14.6 (CVE corpus only) |
|
||||
| limits | 20 s timeout (SIGKILL), 4096 MiB address space, per process; 16 files in parallel |
|
||||
| runtime | 22 s probing + comparing (0 s fetch/build before it) |
|
||||
|
||||
## Results
|
||||
|
||||
A file's class is the first that applies:
|
||||
|
||||
- **panic / hang / crash / oom** — clawhdf5 panicked (caught per object or not), hit the timeout, died on a signal, or failed an allocation. The CI gate fails on any of these.
|
||||
- **h5py-cannot-read** — libhdf5 could not open the file (or itself crashed or hung). Nothing to compare against; most are the deliberately malformed CVE reproducers.
|
||||
- **our-error** — clawhdf5 returned an error for something h5py reads.
|
||||
- **mismatch** — both read it, but the shapes, values, object set or attribute set differ.
|
||||
- **ok** — every object h5py reads, clawhdf5 reads identically.
|
||||
|
||||
| corpus | files | ok | our-error | mismatch | h5py-cannot-read | panic | hang | crash | oom |
|
||||
|---|---|---|---|---|---|---|---|---|---|
|
||||
| NCAS-CMS_pyfive | 33 | 32 | 0 | 1 | 0 | 0 | 0 | 0 | 0 |
|
||||
| cve_hdf5 | 147 | 100 | 6 | 9 | 32 | 0 | 0 | 0 | 0 |
|
||||
| h5py_data | 4 | 4 | 0 | 0 | 0 | 0 | 0 | 0 | 0 |
|
||||
| hdf5 | 466 | 386 | 8 | 12 | 60 | 0 | 0 | 0 | 0 |
|
||||
| netcdf-c | 20 | 20 | 0 | 0 | 0 | 0 | 0 | 0 | 0 |
|
||||
| netcdf4-python | 18 | 18 | 0 | 0 | 0 | 0 | 0 | 0 | 0 |
|
||||
| usnistgov_h5wasm | 5 | 5 | 0 | 0 | 0 | 0 | 0 | 0 | 0 |
|
||||
| xarray-data | 4 | 4 | 0 | 0 | 0 | 0 | 0 | 0 | 0 |
|
||||
| **all** | **697** | **569** | **14** | **22** | **92** | **0** | **0** | **0** | **0** |
|
||||
|
||||
2 of the 22 mismatches are a known h5py bug, not ours (see *Known not-our-bug*).
|
||||
|
||||
Corpora (fetched by `conformance/fetch-corpus.sh` into the gitignored `conformance/.cache/`):
|
||||
|
||||
| corpus | source | commit |
|
||||
|---|---|---|
|
||||
| hdf5 | https://github.com/HDFGroup/hdf5 | `a3cf1ea82cc7` |
|
||||
| cve_hdf5 | https://github.com/HDFGroup/cve_hdf5 | `3fd1f5ae3869` |
|
||||
| netcdf-c | https://github.com/Unidata/netcdf-c | `beb7b9585273` |
|
||||
| NCAS-CMS_pyfive | https://github.com/NCAS-CMS/pyfive | `8cf07b874913` |
|
||||
| usnistgov_h5wasm | https://github.com/usnistgov/h5wasm | `02f6336527d2` |
|
||||
| netcdf4-python | https://github.com/Unidata/netcdf4-python | `6e67576d39ae` |
|
||||
| xarray-data | https://github.com/pydata/xarray-data | `a35297e9da2c` |
|
||||
| h5py_data | https://github.com/h5py/h5py (`h5py/tests/data_files`) | `b2f0347c4200` |
|
||||
|
||||
## Panics, hangs, crashes, out-of-memory
|
||||
|
||||
None.
|
||||
|
||||
## Our-error root causes
|
||||
|
||||
Grouped by normalised error message. *files* counts files whose class this cause affects.
|
||||
|
||||
| files | objects | error | examples |
|
||||
|---:|---:|---|---|
|
||||
| 6 | 6 | `UnsupportedFilter(N)` | `hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_blosc.h5`, `hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_blosc2.h5`, `hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_bshuf.h5` (+3 more) |
|
||||
| 3 | 3 | `DataSizeMismatch { expected: N, actual: N }` | `cve_hdf5/cvefiles/cve-2020-18494.h5`, `cve_hdf5/cvefiles/cve-2024-32623.h5`, `cve_hdf5/cvefiles/cve-2025-2309.h5` |
|
||||
| 2 | 2 | `ChunkedReadError("…")` | `cve_hdf5/cvefiles/cve-2025-2308.h5`, `hdf5/test/testfiles/bad_nbit_parms_walk.h5` |
|
||||
| 1 | 1 | `UnexpectedEof { expected: N, available: N }` | `cve_hdf5/cvefiles/cve-2019-9151.h5` |
|
||||
| 1 | 1 | `MissingMessage(Dataspace)` | `cve_hdf5/cvefiles/cve-2024-33874.h5` |
|
||||
| 1 | 1 | `InvalidObjectHeaderVersion(N)` | `hdf5/tools/test/testfiles/h5clear_mdc_image.h5` |
|
||||
|
||||
## Mismatch root causes
|
||||
|
||||
| files | objects | cause | examples |
|
||||
|---:|---:|---|---|
|
||||
| 13 | 14 | `missing-object` | `cve_hdf5/cvefiles/cve-2019-8397.h5`, `cve_hdf5/cvefiles/cve-2019-8398.h5`, `cve_hdf5/cvefiles/cve-2021-46243.h5` (+10 more) |
|
||||
| 3 | 7 | `extra-attr` | `cve_hdf5/cvefiles/cve-2018-17438`, `cve_hdf5/cvefiles/cve-2018-17439`, `cve_hdf5/cvefiles/cve-2024-33874.h5` |
|
||||
| 3 | 6 | `extra-object` | `cve_hdf5/cvefiles/cve-2021-46244.h5`, `hdf5/tools/test/testfiles/h5clear_fsm_persist_less.h5`, `hdf5/tools/test/testfiles/h5stat_err_refcount.h5` |
|
||||
| 1 | 1 | `attr-values: ours=vlen(>u8) h5py=object layout=- filters=-` | `NCAS-CMS_pyfive/tests/data/attr_datatypes.hdf5` |
|
||||
| 1 | 1 | `values: ours=<f4 h5py=float32 layout=chunked filters=-` | `cve_hdf5/cvefiles/cve-2025-44904.h5` |
|
||||
| 1 | 1 | `values: ours=>i2 h5py=>i2 layout=chunked filters=[6]` | `cve_hdf5/cvefiles/cve-2025-44905.h5` |
|
||||
| 1 | 1 | `values: ours=>f4 h5py=>f4 layout=chunked filters=[2]` | `cve_hdf5/cvefiles/cve-2025-44905.h5` |
|
||||
| 1 | 1 | `values: ours=<f4 h5py=float32 layout=chunked filters=[2]` | `cve_hdf5/cvefiles/cve-2025-44905.h5` |
|
||||
| 1 | 1 | `values: ours=((<i4)[6, 3])[4] h5py=(('<i4', (6, 3)), (4,)) layout=contiguous filters=-` | `hdf5/tools/test/testfiles/tarray3.h5` |
|
||||
| 1 | 1 | `values: ours=vlen({r:>f4,i:>f4}8) h5py=object layout=contiguous filters=-` | `hdf5/tools/test/testfiles/tcomplex_be.h5` |
|
||||
|
||||
## CVE corpus: clawhdf5 vs h5dump vs h5py
|
||||
|
||||
The 147 files of [HDFGroup/cve_hdf5](https://github.com/HDFGroup/cve_hdf5) — reproducers for
|
||||
published libhdf5 CVEs and fuzzer finds. *read* = produced output (possibly with per-object
|
||||
errors), *error* = refused cleanly. h5dump exits non-zero on any error anywhere in a file, so
|
||||
its read/error split is not comparable with the other two rows; the panic, crash, hang and oom
|
||||
columns are.
|
||||
|
||||
| tool | read | error | panic | crash | hang | oom |
|
||||
|---|---:|---:|---:|---:|---:|---:|
|
||||
| clawhdf5 | 142 | 5 | 0 | 0 | 0 | 0 |
|
||||
| h5dump 1.14.6 | 16 | 129 | 0 | 2 | 0 | 0 |
|
||||
| h5py 3.16.0 / HDF5 2.0.0 | 115 | 31 | 0 | 1 | 0 | 0 |
|
||||
|
||||
<details><summary>Per-file outcomes</summary>
|
||||
|
||||
| file | h5dump | h5py | clawhdf5 | class |
|
||||
|---|---|---|---|---|
|
||||
| cvefiles/cve-2016-4330.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2016-4331.h5 | error exit | read 25 obj, 1 errors | read 25 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2016-4332-mtime-new.h5 | error exit | read 25 obj, 1 errors | read 25 obj | ok |
|
||||
| cvefiles/cve-2016-4332-mtime.h5 | error exit | read 4 obj, 3 errors | read 4 obj | ok |
|
||||
| cvefiles/cve-2016-4332-stab.h5 | error exit | open error | read 65 obj | h5py-cannot-read |
|
||||
| cvefiles/cve-2016-4333.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2017-17505.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2017-17506.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2017-17507.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2017-17508.h5 | error exit | read 2 obj, 1 errors | read 2 obj | ok |
|
||||
| cvefiles/cve-2017-17509.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-11202.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-11203.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-11204.h5 | error exit | read 2 obj, 1 errors | read 2 obj | ok |
|
||||
| cvefiles/cve-2018-11205.h5 | error exit | read 2 obj, 1 errors | read 2 obj | ok |
|
||||
| cvefiles/cve-2018-11206-new.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-11206-old.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-11207.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-13866.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||
| cvefiles/cve-2018-13867.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-13868.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-13869.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-13870.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-13871.h5 | error exit | read 2 obj | read 2 obj | ok |
|
||||
| cvefiles/cve-2018-13872.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-13873.h5 | error exit | read 1 obj, 1 errors | read 1 obj | ok |
|
||||
| cvefiles/cve-2018-13874.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2018-13875.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-13876.h5 | error exit | open error | read 2 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2018-14031.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-14033.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-14034.h5 | error exit | read 1 obj, 2 errors | read 1 obj | ok |
|
||||
| cvefiles/cve-2018-14035.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-14460.h5 | error exit | read 3 obj, 2 errors | read 3 obj, 2 errors | ok |
|
||||
| cvefiles/cve-2018-15671.h5 | ok | read 1 obj | read 1 obj | ok |
|
||||
| cvefiles/cve-2018-15672.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-16438.h5 | error exit | read 1 obj, 1 errors | read 1 obj | ok |
|
||||
| cvefiles/cve-2018-17233.h5 | error exit | read 6 obj, 1 errors | read 6 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-17234.h5 | error exit | read 6 obj, 1 errors | read 6 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-17237.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-17432.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-17433 | error exit | open error | open error | h5py-cannot-read |
|
||||
| cvefiles/cve-2018-17434.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-17435.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-17436 | error exit | open error | open error | h5py-cannot-read |
|
||||
| cvefiles/cve-2018-17437.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-17438 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | mismatch |
|
||||
| cvefiles/cve-2018-17439 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | mismatch |
|
||||
| cvefiles/cve-2019-8396.h5 | error exit | read 3 obj, 2 errors | read 3 obj, 2 errors | ok |
|
||||
| cvefiles/cve-2019-8397.h5 | error exit | read 3 obj, 2 errors | read 2 obj, 1 errors | mismatch |
|
||||
| cvefiles/cve-2019-8398.h5 | error exit | read 3 obj, 2 errors | read 2 obj, 1 errors | mismatch |
|
||||
| cvefiles/cve-2019-9151.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 2 errors | our-error |
|
||||
| cvefiles/cve-2019-9152.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2020-10809 | error exit | open error | open error | h5py-cannot-read |
|
||||
| cvefiles/cve-2020-10810.h5 | error exit | open error | read 2 obj | h5py-cannot-read |
|
||||
| cvefiles/cve-2020-10811.h5 | error exit | read 25 obj, 1 errors | read 25 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2020-10812.h5 | error exit | open error | read 2 obj | h5py-cannot-read |
|
||||
| cvefiles/cve-2020-18232.h5 | error exit | read 3 obj, 2 errors | read 3 obj, 2 errors | ok |
|
||||
| cvefiles/cve-2020-18494.h5 | ok | read 2 obj | read 2 obj, 1 errors | our-error |
|
||||
| cvefiles/cve-2021-36977.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2021-37501.h5 | error exit | read 18 obj, 1 errors | read 18 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2021-45829.h5 | error exit | read 1 obj, 2 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2021-45830.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2021-45833.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2021-46242.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2021-46243.h5 | error exit | read 3 obj, 2 errors | read 2 obj, 1 errors | mismatch |
|
||||
| cvefiles/cve-2021-46244.h5 | error exit | read 2 obj, 1 errors | read 6 obj, 3 errors | mismatch |
|
||||
| cvefiles/cve-2024-29157.h5 | error exit | read 4 obj, 7 errors | read 4 obj, 7 errors | ok |
|
||||
| cvefiles/cve-2024-29158.h5 | ok | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-29159.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-29160.h5 | error exit | read 4 obj, 1 errors | read 4 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-29161.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 2 errors | ok |
|
||||
| cvefiles/cve-2024-29162.h5 | error exit | read 17 obj, 4 errors | read 17 obj, 3 errors | ok |
|
||||
| cvefiles/cve-2024-29163.h5 | error exit | read 7 obj, 1 errors | read 7 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-29164.h5 | ok | read 3 obj | read 3 obj | ok |
|
||||
| cvefiles/cve-2024-29165.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-29166.h5 | error exit | read 17 obj, 2 errors | read 17 obj | ok |
|
||||
| cvefiles/cve-2024-32605.h5 | ok | read 6 obj, 1 errors | read 6 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-32606.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-32607-1.h5 | ok | read 10 obj | read 10 obj | ok |
|
||||
| cvefiles/cve-2024-32607-2.h5 | error exit | read 9 obj, 1 errors | read 9 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-32608.h5 | error exit | read 6 obj, 1 errors | read 6 obj | ok |
|
||||
| cvefiles/cve-2024-32609.h5 | error exit | SIGSEGV | read 3 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2024-32610.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-32611.h5 | ok | read 6 obj | read 6 obj | ok |
|
||||
| cvefiles/cve-2024-32612.h5 | ok | read 3 obj | read 3 obj | ok |
|
||||
| cvefiles/cve-2024-32613.h5 | error exit | read 7 obj, 1 errors | read 7 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-32614.h5 | error exit | read 25 obj, 2 errors | read 25 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-32615.h5 | error exit | read 4 obj, 1 errors | read 4 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-32616.h5 | error exit | read 10 obj, 7 errors | read 10 obj, 5 errors | ok |
|
||||
| cvefiles/cve-2024-32617.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-32618.h5 | error exit | read 4 obj, 2 errors | read 3 obj | mismatch |
|
||||
| cvefiles/cve-2024-32619.h5 | error exit | read 3 obj, 2 errors | read 3 obj | ok |
|
||||
| cvefiles/cve-2024-32620.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-32621.h5 | ok | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-32622.h5 | ok | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-32623.h5 | ok | read 6 obj | read 6 obj, 1 errors | our-error |
|
||||
| cvefiles/cve-2024-32624.h5 | error exit | read 6 obj, 1 errors | read 6 obj | ok |
|
||||
| cvefiles/cve-2024-33873.h5 | error exit | read 4 obj, 1 errors | read 4 obj | ok |
|
||||
| cvefiles/cve-2024-33874.h5 | ok | read 6 obj, 1 errors | read 6 obj, 1 errors | our-error |
|
||||
| cvefiles/cve-2024-33875.h5 | ok | read 2 obj | read 2 obj | ok |
|
||||
| cvefiles/cve-2024-33876.h5 | ok | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-33877.h5 | error exit | read 8 obj, 1 errors | read 8 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2025-2153.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2025-2308.h5 | error exit | read 25 obj, 1 errors | read 25 obj, 2 errors | our-error |
|
||||
| cvefiles/cve-2025-2309.h5 | ok | read 6 obj, 1 errors | read 6 obj, 1 errors | our-error |
|
||||
| cvefiles/cve-2025-2310.h5 | error exit | read 24 obj, 8 errors | read 24 obj, 8 errors | ok |
|
||||
| cvefiles/cve-2025-2912.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2025-2913.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2025-2914.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2025-2915.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2025-2923.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2025-2924.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2025-2925.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2025-2926.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2025-44904.h5 | error exit | read 25 obj, 1 errors | read 25 obj, 1 errors | mismatch |
|
||||
| cvefiles/cve-2025-44905.h5 | error exit | read 25 obj, 3 errors | read 25 obj, 3 errors | mismatch |
|
||||
| cvefiles/cve-2025-6269-1.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2025-6269-2.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2025-6269-3.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2025-6269-4.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2025-6270-1.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2025-6270-2.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2025-6270-3.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2025-6516.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2025-6750.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2025-6816.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2025-6817.h5 | error exit | open error | read 1 obj | h5py-cannot-read |
|
||||
| cvefiles/cve-2025-6818.h5 | error exit | open error | read 1 obj | h5py-cannot-read |
|
||||
| cvefiles/cve-2025-6856.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2025-6857.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2025-6858.h5 | SIGSEGV | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2025-7067.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2025-7068.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2025-7069.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2026-26200.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2026-34734.h5 | error exit | read 2 obj, 1 errors | read 2 obj | ok |
|
||||
| cvefiles/cve-2026-92627.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||
| cvefiles/unknown-1.h5 | error exit | read 11 obj, 1 errors | read 11 obj, 1 errors | ok |
|
||||
| fuzzerfiles/gh-4431-poc-03.h5 | error exit | read 1 obj | read 1 obj | ok |
|
||||
| fuzzerfiles/gh-4432-poc-05.h5 | SIGSEGV | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| fuzzerfiles/gh-4433-poc-08.h5 | error exit | read 1 obj, 1 errors | read 1 obj | ok |
|
||||
| fuzzerfiles/gh-4434-poc-09.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| fuzzerfiles/gh-4435-poc-10.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| fuzzerfiles/gh-4585.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||
| fuzzerfiles/gh_2649_flawed.h5 | error exit | read 9 obj, 1 errors | read 9 obj, 1 errors | ok |
|
||||
| fuzzerfiles/gh_2649_plain_model.h5 | ok | read 10 obj | read 10 obj | ok |
|
||||
|
||||
</details>
|
||||
|
||||
## Known not-our-bug
|
||||
|
||||
- **h5py big-endian variable-length sequences.** h5py returns the elements of a VL sequence
|
||||
whose base type is big-endian with the file's big-endian bytes but a native (little-endian)
|
||||
numpy dtype, so the values it reports are byte-swapped garbage; `h5dump` prints the values
|
||||
clawhdf5 reads. Reproducer: `h5py.vlen_dtype(np.dtype('>f4'))` dataset holding `[1.0, 2.0]`
|
||||
reads back in h5py as `[4.6e-41, 9.0e-44]`. Affected here: `NCAS-CMS_pyfive/tests/data/attr_datatypes.hdf5`, `hdf5/tools/test/testfiles/tcomplex_be.h5`.
|
||||
- **Non-IEEE floats and partial-precision integers (N-Bit).** libhdf5 converts a float whose
|
||||
bit layout is not IEEE (e.g. `H5Tset_precision` for the N-Bit filter) or an integer with a
|
||||
bit offset / reduced precision into the plain numpy type of the same size. The probe
|
||||
compares such values as converted numbers, not raw file bytes (before 2026-09-25 it compared
|
||||
raw bytes, which reported every N-Bit float dataset as a mismatch).
|
||||
- **Types h5py widens.** Where h5py reads a type into a numpy type of a different size
|
||||
(FP8 -> float16, bfloat16 -> float32, x87 long double -> float128) the values are not
|
||||
compared (shape and presence still are): dataset file type size 1 -> numpy float16 (2) (15x), attr file type size 1 -> numpy float16 (2) (15x), dataset file type size 2 -> numpy float32 (4) (2x), dataset file type size 8 -> numpy float128 (16) (1x), dataset file type size 12 -> numpy float128 (16) (1x), attr file type size 2 -> numpy float32 (4) (1x), dataset file type size 2 -> numpy >f4 (4) (1x), attr file type size 2 -> numpy >f4 (4) (1x).
|
||||
- **References** are compared by presence only (`R`), not by target.
|
||||
|
||||
## Objects h5py fails on but clawhdf5 reads
|
||||
|
||||
- 19 x `KeyError: '…'`
|
||||
- 19 x `OSError: Can't synchronously read data (no appropriate function for conversion path)`
|
||||
- 1 x `TypeError: unhandled dtype kind M (dtype('…'))`
|
||||
- 1 x `OSError: Can't synchronously read data (bad coordinate offset)`
|
||||
- 1 x `TypeError: No NumPy equivalent for TypeTimeID exists`
|
||||
- 1 x `KeyError: "…"`
|
||||
- 1 x `ValueError: Insufficient precision in available types to represent (N, N, N, N, N)`
|
||||
|
||||
## Reproduce
|
||||
|
||||
```sh
|
||||
# needs: Rust, python3 with h5py numpy hdf5plugin (conformance/requirements.txt), h5dump (hdf5-tools), git
|
||||
CLAWHDF5_PYTHON=/path/to/venv/bin/python conformance/run.sh
|
||||
```
|
||||
|
||||
The corpus (about 450 MB of sparse checkouts) is cached in `conformance/.cache/`; results for
|
||||
every file, both sides' raw JSON and stderr, are in `conformance/.cache/results/`.
|
||||
`conformance/baseline.json` holds the ok files the nightly CI job (`.gitea/workflows/conformance.yml`)
|
||||
must keep; `conformance/run.sh --update-baseline` rewrites it.
|
||||
@@ -8,7 +8,7 @@
|
||||
[](BENCHMARKS.md#longmemeval-results)
|
||||
[](BENCHMARKS.md#memory-footprint-1)
|
||||
|
||||
ClawHDF5 is a pure-Rust HDF5 implementation combined with a research-grade agent memory engine. It gives AI agents persistent, searchable, integrity-checked memory — all stored in a single portable file.
|
||||
ClawHDF5 is a pure-Rust HDF5 implementation combined with a research-grade agent memory engine. It gives AI agents persistent, searchable, cryptographically verifiable memory (Ed25519-signed checkpoints) — all stored in a single portable file.
|
||||
|
||||
> **Two things live here:**
|
||||
> - **A general-purpose, pure-Rust HDF5 library** — zero C dependencies, NetCDF-4 support, SIMD/GPU acceleration. See the **[Crate Map](#crate-map)** and **[BENCHMARKS.md](BENCHMARKS.md)** for the libhdf5 head-to-head numbers.
|
||||
@@ -71,7 +71,7 @@ breaking change, are in [CHANGELOG.md](CHANGELOG.md).
|
||||
100K). It no longer rebuilds BM25 or rewrites the store per query, and the
|
||||
HNSW graph is persisted (v2.4.0).
|
||||
- Default fusion weights are now the measured 0.4 / 0.6 (v2.5.0). Re-ranking had
|
||||
been discarding the retrieval score, costing the OpenClaw backend 40.6pp of
|
||||
been discarding the retrieval score, costing the Markdown backend 40.6pp of
|
||||
Hit@1; fixed in v2.6.0.
|
||||
- Selection reads decode only the chunks they touch (a 64×64 window: 105 ms to
|
||||
0.39 ms), and full reads are 1.2–1.9× faster (v2.5.0).
|
||||
@@ -93,7 +93,7 @@ breaking change, are in [CHANGELOG.md](CHANGELOG.md).
|
||||
identical LongMemEval retrieval on real embeddings.
|
||||
- `HDF5Memory::search` with `SearchOptions`: filter by source channel (exact
|
||||
filtered top-k, never slower than unfiltered), and opt-in re-ranking and
|
||||
confidence rejection, which used to be OpenClaw-only.
|
||||
confidence rejection, which used to be reachable only through `ClawhdfBackend`.
|
||||
|
||||
**Tooling**
|
||||
- CI now runs the h5py/netCDF4 interop suites for real (they had been skipping
|
||||
@@ -113,7 +113,7 @@ Every AI agent needs memory. Today that means scattered Markdown files, SQLite d
|
||||
| Memory consolidation | Manual pruning | Hippocampal-inspired automatic tiers |
|
||||
| Temporal queries | Custom code | Native temporal index (622 ns range query over 10K) |
|
||||
| Multi-modal | Multiple stores | Unified cross-modal search (exact scan: 842 µs over 1K records) |
|
||||
| Integrity | Hope for the best | Chained-CRC WAL, checksummed chunk indexes, write-anomaly alerts, opt-in SHA-256 dataset provenance |
|
||||
| Integrity | Hope for the best | Ed25519-signed checkpoints that pinpoint any edited record, chained-CRC WAL, checksummed chunk indexes, write-anomaly alerts |
|
||||
| Portability | Config + DB + files | **One `.h5` file. Copy it anywhere.** |
|
||||
|
||||
---
|
||||
@@ -228,7 +228,7 @@ is for. The weights matter more than the stages: a sweep of `vector_weight` from
|
||||
0.0 to 1.0 found the old `0.7/0.3` default is **strictly dominated** by
|
||||
`0.4/0.6` — better on Hit@1, Hit@5, Hit@10 and MRR at both granularities. Since
|
||||
v2.5.0 `0.4/0.6` is the default (`hybrid::DEFAULT_FUSION`, used by
|
||||
`unified_search`, `hybrid_search_with` and the OpenClaw backend); callers that
|
||||
`unified_search`, `hybrid_search_with` and `ClawhdfBackend`); callers that
|
||||
pass weights to `hybrid_search` explicitly choose their own. Use `0.3/0.7` if
|
||||
rank-1 precision matters most. Reciprocal rank fusion is selectable
|
||||
(`hybrid::Fusion::Rrf`) but measured worse than the weighted sum. See
|
||||
@@ -329,7 +329,7 @@ ClawhDF5's agent memory engine draws on 15+ recent papers on agentic memory syst
|
||||
│ × √(Hebbian activation) │
|
||||
└─────────────────┬──────────────────┘
|
||||
│ opt-in (SearchOptions);
|
||||
│ the OpenClaw backend turns both on
|
||||
│ ClawhdfBackend turns both on
|
||||
┌─────────────────▼──────────────────┐
|
||||
│ Multi-factor re-ranking │
|
||||
│ relevance · recency · authority · │
|
||||
@@ -365,13 +365,14 @@ directly; the store persists the records, sessions and graph they work over.
|
||||
| **`knowledge`** | Entity/relation graph with BFS traversal, spreading activation, fuzzy (Levenshtein) entity resolution |
|
||||
| **`consolidation`** | Three-tier memory (Working → Episodic → Semantic) with importance scoring, novelty, and time-decay |
|
||||
| **`hybrid`** | Vector + BM25 fusion. Default is a min-max-normalised weighted sum, vector 0.4 / keyword 0.6 (`hybrid::DEFAULT_FUSION`, tuned on LongMemEval); RRF is available via `Fusion::Rrf` / `hybrid_search_with`. The vector stage uses the HNSW index by default (`hnsw` feature); disable with `--no-default-features --features float16` for an exact linear scan |
|
||||
| **`reranker`** | Multi-factor re-ranking: retrieval relevance (leads, weight 1.0), temporal recency, source authority, activation weight. Opt-in via `SearchOptions::with_rerank`; on in the OpenClaw backend |
|
||||
| **`confidence`** | Low-confidence rejection — suppresses spurious recalls when nothing matches. Opt-in via `SearchOptions::with_confidence`; on in the OpenClaw backend |
|
||||
| **`reranker`** | Multi-factor re-ranking: retrieval relevance (leads, weight 1.0), temporal recency, source authority, activation weight. Opt-in via `SearchOptions::with_rerank`; on in `ClawhdfBackend` |
|
||||
| **`confidence`** | Low-confidence rejection — suppresses spurious recalls when nothing matches. Opt-in via `SearchOptions::with_confidence`; on in `ClawhdfBackend` |
|
||||
| **`temporal`** | Sorted timestamp index, session DAG, entity timeline, temporal query hints |
|
||||
| **`multimodal`** | Cross-modal search across text/image/audio/video embeddings |
|
||||
| **`signing`** | Ed25519-signed checkpoints: SHA-256 per record in a Merkle tree, plus hashes of settings, sessions and the knowledge graph; `HDF5Memory::verify` names any edited record |
|
||||
| **`provenance`** | Source attribution and an unkeyed FNV-1a content hash per record, held in memory for the session, for detecting accidental corruption (not tamper-proof) |
|
||||
| **`anomaly`** | Write rate limiting, 15 injection-pattern detectors, source-distribution analysis. Alerts never block a save; drain them with `take_anomaly_alerts` |
|
||||
| **`openclaw`** | OpenClaw integration: MemoryBackend trait, Markdown ↔ HDF5 conversion |
|
||||
| **`openclaw`** | `ClawhdfBackend`: a Markdown-oriented backend (ingest by section, search, read back by path, export). Named for OpenClaw, but **not an OpenClaw plugin** — see [docs/openclaw.md](docs/openclaw.md) |
|
||||
| **`vector_search`** | Flat cosine, pre-normed, SIMD, BLAS, GPU, parallel search paths |
|
||||
| **`ivf` / `pq`** | Standalone IVF and IVF-PQ indexes (benchmarked to 100K vectors); not used by `HDF5Memory`, whose ANN index is HNSW |
|
||||
| **`bm25`** | Incremental Okapi BM25 inverted index, kept for the life of the store; optional stemming |
|
||||
@@ -447,7 +448,7 @@ let work = memory.search(
|
||||
);
|
||||
|
||||
// Re-rank by relevance, recency, source authority and activation, then drop
|
||||
// low-confidence results — the pipeline the OpenClaw backend runs.
|
||||
// low-confidence results — the pipeline ClawhdfBackend runs.
|
||||
let careful = memory.search(
|
||||
&query_embedding,
|
||||
"user preferences",
|
||||
@@ -457,6 +458,35 @@ let careful = memory.search(
|
||||
);
|
||||
```
|
||||
|
||||
### Signed Checkpoints
|
||||
|
||||
```rust
|
||||
use clawhdf5_agent::signing;
|
||||
|
||||
// Once, somewhere safe: keep the secret key, publish the public key.
|
||||
let key = signing::generate_key();
|
||||
let public = key.verifying_key();
|
||||
|
||||
// Every checkpoint is signed from now on. The key is never written to disk;
|
||||
// a signed store refuses to checkpoint without it.
|
||||
memory.set_signing_key(key);
|
||||
memory.flush_wal()?;
|
||||
|
||||
// Anyone holding the public key can check the file, e.g. after copying it.
|
||||
let report = HDF5Memory::verify(std::path::Path::new("agent.h5"), &public)?;
|
||||
assert!(report.is_valid());
|
||||
// On a tampered file: report.changed_records lists the records that differ.
|
||||
```
|
||||
|
||||
The signature covers every record (text, embedding as stored, channel,
|
||||
timestamp, session, tags, deleted flag, activation), the store's settings,
|
||||
its sessions and its knowledge graph — a change made with any tool is caught.
|
||||
It covers checkpoints, not saves still in the WAL
|
||||
(`report.wal_entries_unsigned` counts those). CLI: `clawhdf5-cli keygen`,
|
||||
`--signing-key <file>` on writing commands, and `verify --public-key`.
|
||||
Signing adds about 20% to a checkpoint and 32 bytes per record to the file
|
||||
([BENCHMARKS.md § Signed checkpoints](BENCHMARKS.md#signed-checkpoints)).
|
||||
|
||||
### Knowledge Graph
|
||||
|
||||
```rust
|
||||
@@ -527,7 +557,13 @@ let ids = index.range_query(1700000000.0, 1700010800.0);
|
||||
let recent = index.latest(10);
|
||||
```
|
||||
|
||||
### OpenClaw Integration
|
||||
### Markdown Backend
|
||||
|
||||
`ClawhdfBackend` ingests Markdown by section and searches it with the full
|
||||
pipeline. It is a library API — clawhdf5 is **not** an OpenClaw memory plugin
|
||||
([docs/openclaw.md](docs/openclaw.md)). Sections stored this way carry no
|
||||
embedding, so their search is keyword-only unless you save records with
|
||||
vectors through `save_entry`.
|
||||
|
||||
```rust
|
||||
use clawhdf5_agent::openclaw::*;
|
||||
@@ -660,7 +696,7 @@ stores keep their setting. Opt out with `float16 = false` or
|
||||
| `fast-checksum` | no | crc32fast-accelerated checksums |
|
||||
| `lz4` | no | LZ4 block compression filter (id 32004) |
|
||||
| `zstd` | no | Zstandard compression filter (id 32015) |
|
||||
| `pcodec` | no | Pcodec lossless numerical codec (id 32023, via `pco` crate) |
|
||||
| `pcodec` | no | Pcodec lossless numerical codec (via `pco` crate). Private, unregistered filter id 480: **only clawhdf5 can read these datasets** (h5py/libhdf5 cannot). Files from clawhdf5 <= 2.7.0 used id 32023, which is registered to Granular BitRound; they still read. |
|
||||
| `system-zlib` | no | System zlib backend for deflate (C) |
|
||||
| `blake3_hash` | no | BLAKE3 content hashing for provenance |
|
||||
| `szip` | no | SZIP filter (id 4) via libaec (C, through the internal `libaec-sys` crate) |
|
||||
@@ -782,7 +818,9 @@ clawhdf5-migrate --sqlite old.db --hdf5 memory.h5 --agent-id my-agent --embedder
|
||||
|
||||
The output is an ordinary `clawhdf5-agent` store, written through the agent's
|
||||
own API: open it with `HDF5Memory::open` (or `clawhdf5-cli --path memory.h5 …`)
|
||||
and search it straight away. What carries over from the ZeroClaw tables:
|
||||
and search it straight away. The source must use the `memory_chunks` / `sessions` / `entities` / `relations` layout (names are
|
||||
configurable with `--*-table`); note that this is not ZeroClaw's schema, and
|
||||
ZeroClaw does not use clawhdf5. What carries over:
|
||||
|
||||
| SQLite | Agent store |
|
||||
|--------|-------------|
|
||||
@@ -821,10 +859,10 @@ See [ROADMAP.md](ROADMAP.md) for the full implementation tracker.
|
||||
- ✅ Temporal reasoning with sub-µs queries
|
||||
- ✅ Memory security + anomaly detection
|
||||
- ✅ Multi-modal memory (text/image/audio/video)
|
||||
- ✅ OpenClaw integration layer
|
||||
- ✅ Markdown ingest/export backend (`ClawhdfBackend`); an OpenClaw plugin was never built — see [docs/openclaw.md](docs/openclaw.md)
|
||||
- ✅ Comprehensive Criterion benchmarks
|
||||
|
||||
**Phase 2** — MemoryArena and LongMemEval academic benchmarks are done (see [BENCHMARKS.md](BENCHMARKS.md), reproduced on a second machine); remaining: publish the OpenClaw TypeScript bridge to npm, crates.io/PyPI publishing.
|
||||
**Phase 2** — MemoryArena and LongMemEval academic benchmarks are done (see [BENCHMARKS.md](BENCHMARKS.md), reproduced on a second machine); remaining: crates.io/PyPI publishing. The Node bindings are unpublished and known to be broken ([known issues](docs/known-issues.md)).
|
||||
|
||||
---
|
||||
|
||||
|
||||
+14
-8
@@ -105,24 +105,30 @@
|
||||
|
||||
---
|
||||
|
||||
## Track 7: OpenClaw Integration
|
||||
**Status:** 🟢 Complete
|
||||
## Track 7: OpenClaw Integration — withdrawn (2026-09-25)
|
||||
**Status:** ⚪ Withdrawn (the items below were library work; no OpenClaw integration shipped)
|
||||
**Priority:** Critical (for adoption)
|
||||
**Crates:** `clawhdf5-agent`, `clawhdf5-napi`
|
||||
|
||||
- [x] **7.1** Memory backend trait — MemoryBackend with search/get/write/ingest/export/stats
|
||||
- [x] **7.2** Hybrid retrieval pipeline — ClawhdfBackend wires RRF → reranker → confidence rejection
|
||||
- [x] **7.3** Markdown import/export — MarkdownParser + MarkdownExporter with line tracking + metadata
|
||||
- [x] **7.4** memory_search tool — backed by full hybrid retrieval pipeline
|
||||
- [x] **7.5** memory_get tool — get() with path + line range support
|
||||
- [x] **7.4** `search()` — backed by the full hybrid retrieval pipeline (a Rust method; no OpenClaw tool was ever registered)
|
||||
- [x] **7.5** `get()` — read back by path, with a line slice (not an OpenClaw tool either)
|
||||
- [x] **7.6** Compaction integration — run_compaction() (decay + compact + WAL flush), run_consolidation() (hippocampal engine), tick_session(), flush_wal()
|
||||
- [x] **7.7** Config surface — `memory.backend = "clawhdf5"` schema documented in docs/openclaw-config.md
|
||||
- [x] **7.8** Documentation + migration guide — docs/migration-guide.md, docs/openclaw-integration.md (architecture, full API reference, code patterns)
|
||||
- [ ] **7.7** ~~Config surface — `memory.backend = "clawhdf5"`~~ — never valid OpenClaw config; docs removed
|
||||
- [ ] **7.8** ~~Documentation + migration guide~~ — removed: they described an integration that never worked
|
||||
|
||||
**Node.js bridge:** `clawhdf5-napi` (napi-rs) → `@redclaw/clawhdf5` npm package with full TypeScript types.
|
||||
**Node.js bridge:** `clawhdf5-napi` (napi-rs) and a TypeScript wrapper in `packages/clawhdf5-node` exist but are unpublished, untested in CI and known to be broken (docs/known-issues.md).
|
||||
|
||||
---
|
||||
|
||||
> **Withdrawn.** None of this track produced a working OpenClaw integration: no
|
||||
> plugin was built, the documented `memory.backend = "clawhdf5"` config was never
|
||||
> valid in any OpenClaw release, and the Node package was never published. The
|
||||
> Rust `ClawhdfBackend` remains as a library API. Not pursued for now; see
|
||||
> [docs/openclaw.md](docs/openclaw.md) for what a plugin would need today.
|
||||
|
||||
## Track 8: Benchmarking & Validation
|
||||
**Status:** 🟢 Complete
|
||||
**Priority:** High
|
||||
@@ -142,7 +148,7 @@
|
||||
|
||||
**Phase 1:** ~~Tracks 1, 2, 3 — core memory intelligence~~ 🟢 Complete
|
||||
**Phase 2:** ~~Track 4 (temporal) + Track 5 (security)~~ 🟢 Complete
|
||||
**Phase 3:** ~~Track 6 (multi-modal) + Track 7 (OpenClaw integration)~~ 🟢 Complete
|
||||
**Phase 3:** ~~Track 6 (multi-modal)~~ 🟢 Complete; Track 7 (OpenClaw integration) withdrawn
|
||||
**Phase 4:** ~~Track 8 (benchmarking + validation)~~ 🟢 Complete
|
||||
|
||||
All 8 tracks delivered. 1,650+ tests passing, zero clippy warnings.
|
||||
|
||||
@@ -0,0 +1,3 @@
|
||||
/.cache/
|
||||
# pin the probe's dependencies (the workspace lock is not committed)
|
||||
!/probe/Cargo.lock
|
||||
@@ -0,0 +1,39 @@
|
||||
# Conformance sweep
|
||||
|
||||
Reads every HDF5 file of eight public corpora with clawhdf5 and with
|
||||
h5py/libhdf5, compares the two readings object by object, and writes
|
||||
[`CONFORMANCE.md`](../CONFORMANCE.md).
|
||||
|
||||
```sh
|
||||
CLAWHDF5_PYTHON=/path/to/venv/bin/python conformance/run.sh # ~30 s once the corpus is cached
|
||||
conformance/run.sh --update-baseline # after an intended change in results
|
||||
```
|
||||
|
||||
Needs Rust, `git`, `h5dump` (Debian/Ubuntu `hdf5-tools`), `libaec` (for the
|
||||
probe's `szip` feature; `libaec-dev`), and a Python with the packages in
|
||||
`requirements.txt`. The first run downloads about 450 MB of sparse checkouts.
|
||||
|
||||
| file | role |
|
||||
|---|---|
|
||||
| `corpus.txt` | the corpora: git URL, pinned commit, swept root, sparse-checkout patterns |
|
||||
| `fetch-corpus.sh` | shallow, sparse, blob-filtered checkout of each pinned commit into `.cache/src/` (gitignored); no-op when already there |
|
||||
| `list_files.py` | which files are probed (HDF5/netCDF-4 extensions minus netCDF classic, plus the CVE reproducers) |
|
||||
| `probe/` | the clawhdf5 side: a standalone crate (outside the workspace, so `cargo test --workspace` never builds it) that walks a file with `clawhdf5-format` and prints canonical JSON |
|
||||
| `ref.py` | the h5py side: the same JSON from h5py |
|
||||
| `run_one.sh` | runs both sides on one file (and `h5dump` on the CVE corpus) under a timeout and an address-space limit |
|
||||
| `compare.py` | classifies each file (ok / our-error / mismatch / h5py-cannot-read / panic / hang / crash / oom) and groups root causes |
|
||||
| `report.py` | writes `CONFORMANCE.md` |
|
||||
| `check.py` | the gate: fails on any panic/hang/crash/oom, on an ok count below `baseline.json`, or on a baseline-ok file that is no longer ok |
|
||||
| `baseline.json` | the ok files the gate holds the line on |
|
||||
| `requirements.txt` | pinned h5py / numpy / hdf5plugin / netCDF4 |
|
||||
|
||||
Results for every file (both sides' JSON and stderr, `results.csv`,
|
||||
`results.json`, `summary.md`) are left in `.cache/results/`.
|
||||
|
||||
The nightly job is `.gitea/workflows/conformance.yml`; it prints the report
|
||||
into the job log.
|
||||
|
||||
The canonical value encoding both sides hash is documented at the top of
|
||||
`probe/src/main.rs`. Values are compared as libhdf5 presents them: a float
|
||||
with a non-IEEE bit layout (N-Bit) or an integer with a bit offset is compared
|
||||
as the converted number, not as raw file bytes.
|
||||
@@ -0,0 +1,618 @@
|
||||
{
|
||||
"comment": "conformance/run.sh fails if the ok count drops below `ok` or a file in `ok_files` stops being ok. Regenerate with `conformance/run.sh --update-baseline` after an intended change.",
|
||||
"commit": "10d1029ead524e2fe64c2cd7f61b28067d9e449c",
|
||||
"date": "2026-09-26 03:46 UTC",
|
||||
"reference": "h5py 3.16.0 / HDF5 2.0.0",
|
||||
"files": 697,
|
||||
"ok": 569,
|
||||
"counts": {
|
||||
"h5py-cannot-read": 92,
|
||||
"mismatch": 22,
|
||||
"ok": 569,
|
||||
"our-error": 14
|
||||
},
|
||||
"per_corpus": {
|
||||
"NCAS-CMS_pyfive": {
|
||||
"mismatch": 1,
|
||||
"ok": 32
|
||||
},
|
||||
"cve_hdf5": {
|
||||
"h5py-cannot-read": 32,
|
||||
"mismatch": 9,
|
||||
"ok": 100,
|
||||
"our-error": 6
|
||||
},
|
||||
"h5py_data": {
|
||||
"ok": 4
|
||||
},
|
||||
"hdf5": {
|
||||
"h5py-cannot-read": 60,
|
||||
"mismatch": 12,
|
||||
"ok": 386,
|
||||
"our-error": 8
|
||||
},
|
||||
"netcdf-c": {
|
||||
"ok": 20
|
||||
},
|
||||
"netcdf4-python": {
|
||||
"ok": 18
|
||||
},
|
||||
"usnistgov_h5wasm": {
|
||||
"ok": 5
|
||||
},
|
||||
"xarray-data": {
|
||||
"ok": 4
|
||||
}
|
||||
},
|
||||
"ok_files": [
|
||||
"NCAS-CMS_pyfive/tests/compact.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/btreev2.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/chunked.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/cmip_bad_eg.nc",
|
||||
"NCAS-CMS_pyfive/tests/data/compressed.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/compressed_v1.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/dataset_datatypes.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/dataset_multidim.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/dim_scales.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/earliest.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/enum_h5variable.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/enum_variable.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/enum_variable.nc",
|
||||
"NCAS-CMS_pyfive/tests/data/enums_from_netcdf.nc",
|
||||
"NCAS-CMS_pyfive/tests/data/fillvalue_earliest.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/fillvalue_latest.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/filter_pipeline_v2.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/fletcher32.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/fractal_heap_no_mci_rlat.nc",
|
||||
"NCAS-CMS_pyfive/tests/data/groups.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/h5netcdf_test.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/issue23_A.nc",
|
||||
"NCAS-CMS_pyfive/tests/data/issue23_A_contiguous.nc",
|
||||
"NCAS-CMS_pyfive/tests/data/issue23_B.nc",
|
||||
"NCAS-CMS_pyfive/tests/data/latest.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/netcdf4_classic.nc",
|
||||
"NCAS-CMS_pyfive/tests/data/new_style_groups.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/noy_AERmonZ_UKESM1-0-LL_piControl_r1i1p1f2_gnz_200001-200012.nc",
|
||||
"NCAS-CMS_pyfive/tests/data/references.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/resizable.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/opaque_datetime.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/opaque_fixed.hdf5",
|
||||
"cve_hdf5/cvefiles/cve-2016-4330.h5",
|
||||
"cve_hdf5/cvefiles/cve-2016-4331.h5",
|
||||
"cve_hdf5/cvefiles/cve-2016-4332-mtime-new.h5",
|
||||
"cve_hdf5/cvefiles/cve-2016-4332-mtime.h5",
|
||||
"cve_hdf5/cvefiles/cve-2016-4333.h5",
|
||||
"cve_hdf5/cvefiles/cve-2017-17505.h5",
|
||||
"cve_hdf5/cvefiles/cve-2017-17506.h5",
|
||||
"cve_hdf5/cvefiles/cve-2017-17507.h5",
|
||||
"cve_hdf5/cvefiles/cve-2017-17508.h5",
|
||||
"cve_hdf5/cvefiles/cve-2017-17509.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-11202.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-11203.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-11204.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-11205.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-11206-new.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-11206-old.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-11207.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-13867.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-13868.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-13869.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-13870.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-13871.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-13872.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-13873.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-13875.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-14031.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-14033.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-14034.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-14035.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-14460.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-15671.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-15672.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-16438.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-17233.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-17234.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-17237.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-17432.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-17434.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-17435.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-17437.h5",
|
||||
"cve_hdf5/cvefiles/cve-2019-8396.h5",
|
||||
"cve_hdf5/cvefiles/cve-2019-9152.h5",
|
||||
"cve_hdf5/cvefiles/cve-2020-10811.h5",
|
||||
"cve_hdf5/cvefiles/cve-2020-18232.h5",
|
||||
"cve_hdf5/cvefiles/cve-2021-36977.h5",
|
||||
"cve_hdf5/cvefiles/cve-2021-37501.h5",
|
||||
"cve_hdf5/cvefiles/cve-2021-45829.h5",
|
||||
"cve_hdf5/cvefiles/cve-2021-45833.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-29157.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-29158.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-29159.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-29160.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-29161.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-29162.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-29163.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-29164.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-29165.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-29166.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32605.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32606.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32607-1.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32607-2.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32608.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32610.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32611.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32612.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32613.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32614.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32615.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32616.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32617.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32619.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32620.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32621.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32622.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32624.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-33873.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-33875.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-33876.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-33877.h5",
|
||||
"cve_hdf5/cvefiles/cve-2025-2310.h5",
|
||||
"cve_hdf5/cvefiles/cve-2025-2924.h5",
|
||||
"cve_hdf5/cvefiles/cve-2025-2925.h5",
|
||||
"cve_hdf5/cvefiles/cve-2025-6269-1.h5",
|
||||
"cve_hdf5/cvefiles/cve-2025-6269-2.h5",
|
||||
"cve_hdf5/cvefiles/cve-2025-6269-3.h5",
|
||||
"cve_hdf5/cvefiles/cve-2025-6269-4.h5",
|
||||
"cve_hdf5/cvefiles/cve-2025-6516.h5",
|
||||
"cve_hdf5/cvefiles/cve-2025-6857.h5",
|
||||
"cve_hdf5/cvefiles/cve-2025-7067.h5",
|
||||
"cve_hdf5/cvefiles/cve-2026-26200.h5",
|
||||
"cve_hdf5/cvefiles/cve-2026-34734.h5",
|
||||
"cve_hdf5/cvefiles/cve-2026-92627.h5",
|
||||
"cve_hdf5/cvefiles/unknown-1.h5",
|
||||
"cve_hdf5/fuzzerfiles/gh-4431-poc-03.h5",
|
||||
"cve_hdf5/fuzzerfiles/gh-4432-poc-05.h5",
|
||||
"cve_hdf5/fuzzerfiles/gh-4433-poc-08.h5",
|
||||
"cve_hdf5/fuzzerfiles/gh-4435-poc-10.h5",
|
||||
"cve_hdf5/fuzzerfiles/gh_2649_flawed.h5",
|
||||
"cve_hdf5/fuzzerfiles/gh_2649_plain_model.h5",
|
||||
"h5py_data/compound-dtype-complex.h5",
|
||||
"h5py_data/vlen_string_dset.h5",
|
||||
"h5py_data/vlen_string_dset_utc.h5",
|
||||
"h5py_data/vlen_string_s390x.h5",
|
||||
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_bitgroom.h5",
|
||||
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_granularbr.h5",
|
||||
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_jpeg.h5",
|
||||
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_lz4.h5",
|
||||
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_zstd.h5",
|
||||
"hdf5/HDF5Examples/C/H5G/16/h5ex_g_iterate.h5",
|
||||
"hdf5/HDF5Examples/C/H5G/16/h5ex_g_traverse.h5",
|
||||
"hdf5/HDF5Examples/C/H5G/h5ex_g_iterate.h5",
|
||||
"hdf5/HDF5Examples/C/H5G/h5ex_g_traverse.h5",
|
||||
"hdf5/HDF5Examples/C/H5G/h5ex_g_visit.h5",
|
||||
"hdf5/HDF5Examples/FORTRAN/H5G/h5ex_g_iterate.h5",
|
||||
"hdf5/HDF5Examples/FORTRAN/H5G/h5ex_g_traverse.h5",
|
||||
"hdf5/HDF5Examples/FORTRAN/H5G/h5ex_g_visit.h5",
|
||||
"hdf5/HDF5Examples/JAVA/H5G/h5ex_g_iterate.h5",
|
||||
"hdf5/HDF5Examples/JAVA/H5G/h5ex_g_visit.h5",
|
||||
"hdf5/HDF5Examples/JAVA/compat/H5G/110/h5ex_g_iterate.h5",
|
||||
"hdf5/HDF5Examples/JAVA/compat/H5G/110/h5ex_g_visit.h5",
|
||||
"hdf5/HDF5Examples/JAVA/compat/H5G/h5ex_g_iterate.h5",
|
||||
"hdf5/HDF5Examples/JAVA/compat/H5G/h5ex_g_visit.h5",
|
||||
"hdf5/c++/test/th5s.h5",
|
||||
"hdf5/hl/test/testfiles/test_ds_be.h5",
|
||||
"hdf5/hl/test/testfiles/test_ds_be_new_ref-32bit.h5",
|
||||
"hdf5/hl/test/testfiles/test_ds_be_new_ref.h5",
|
||||
"hdf5/hl/test/testfiles/test_ds_le.h5",
|
||||
"hdf5/hl/test/testfiles/test_ds_le_new_ref.h5",
|
||||
"hdf5/hl/test/testfiles/test_ld.h5",
|
||||
"hdf5/hl/test/testfiles/test_table_be.h5",
|
||||
"hdf5/hl/test/testfiles/test_table_cray.h5",
|
||||
"hdf5/hl/test/testfiles/test_table_le.h5",
|
||||
"hdf5/test/testfiles/aggr.h5",
|
||||
"hdf5/test/testfiles/bad_chunk_ndims.h5",
|
||||
"hdf5/test/testfiles/bad_compound.h5",
|
||||
"hdf5/test/testfiles/bad_offset.h5",
|
||||
"hdf5/test/testfiles/be_data.h5",
|
||||
"hdf5/test/testfiles/be_extlink1.h5",
|
||||
"hdf5/test/testfiles/be_extlink2.h5",
|
||||
"hdf5/test/testfiles/btree_idx_1_6.h5",
|
||||
"hdf5/test/testfiles/btree_idx_1_8.h5",
|
||||
"hdf5/test/testfiles/charsets.h5",
|
||||
"hdf5/test/testfiles/corrupt_stab_msg.h5",
|
||||
"hdf5/test/testfiles/deflate.h5",
|
||||
"hdf5/test/testfiles/file_image_core_test.h5",
|
||||
"hdf5/test/testfiles/filespace_1_6.h5",
|
||||
"hdf5/test/testfiles/filespace_1_8.h5",
|
||||
"hdf5/test/testfiles/fill18.h5",
|
||||
"hdf5/test/testfiles/fill_old.h5",
|
||||
"hdf5/test/testfiles/filter_error.h5",
|
||||
"hdf5/test/testfiles/fsm_aggr_nopersist.h5",
|
||||
"hdf5/test/testfiles/fsm_aggr_persist.h5",
|
||||
"hdf5/test/testfiles/group_old.h5",
|
||||
"hdf5/test/testfiles/h5fc_ext1_f.h5",
|
||||
"hdf5/test/testfiles/h5fc_ext1_i.h5",
|
||||
"hdf5/test/testfiles/h5fc_ext2_if.h5",
|
||||
"hdf5/test/testfiles/h5fc_ext2_sf.h5",
|
||||
"hdf5/test/testfiles/h5fc_ext3_isf.h5",
|
||||
"hdf5/test/testfiles/h5fc_ext_none.h5",
|
||||
"hdf5/test/testfiles/le_data.h5",
|
||||
"hdf5/test/testfiles/le_extlink1.h5",
|
||||
"hdf5/test/testfiles/le_extlink2.h5",
|
||||
"hdf5/test/testfiles/memleak_H5O_dtype_decode_helper_H5Odtype.h5",
|
||||
"hdf5/test/testfiles/mergemsg.h5",
|
||||
"hdf5/test/testfiles/noencoder.h5",
|
||||
"hdf5/test/testfiles/none.h5",
|
||||
"hdf5/test/testfiles/paged_nopersist.h5",
|
||||
"hdf5/test/testfiles/paged_persist.h5",
|
||||
"hdf5/test/testfiles/specmetaread.h5",
|
||||
"hdf5/test/testfiles/tarrold.h5",
|
||||
"hdf5/test/testfiles/tbad_msg_count.h5",
|
||||
"hdf5/test/testfiles/tbogus.h5",
|
||||
"hdf5/test/testfiles/test_filters_be.h5",
|
||||
"hdf5/test/testfiles/test_filters_le.h5",
|
||||
"hdf5/test/testfiles/th5s.h5",
|
||||
"hdf5/test/testfiles/tlayouto.h5",
|
||||
"hdf5/test/testfiles/tmisc38a.h5",
|
||||
"hdf5/test/testfiles/tmisc38b.h5",
|
||||
"hdf5/test/testfiles/tmtimen.h5",
|
||||
"hdf5/test/testfiles/tmtimeo.h5",
|
||||
"hdf5/test/testfiles/tnullspace.h5",
|
||||
"hdf5/test/testfiles/tsizeslheap.h5",
|
||||
"hdf5/tools/test/testfiles/bigendian/tdset2.h5",
|
||||
"hdf5/tools/test/testfiles/binfp64.h5",
|
||||
"hdf5/tools/test/testfiles/binin16.h5",
|
||||
"hdf5/tools/test/testfiles/binin32.h5",
|
||||
"hdf5/tools/test/testfiles/binin8.h5",
|
||||
"hdf5/tools/test/testfiles/binin8w.h5",
|
||||
"hdf5/tools/test/testfiles/binuin16.h5",
|
||||
"hdf5/tools/test/testfiles/binuin32.h5",
|
||||
"hdf5/tools/test/testfiles/bounds_latest_latest.h5",
|
||||
"hdf5/tools/test/testfiles/charsets.h5",
|
||||
"hdf5/tools/test/testfiles/compounds_array_vlen1.h5",
|
||||
"hdf5/tools/test/testfiles/compounds_array_vlen2.h5",
|
||||
"hdf5/tools/test/testfiles/err_attr_dspace.h5",
|
||||
"hdf5/tools/test/testfiles/file_space.h5",
|
||||
"hdf5/tools/test/testfiles/filter_fail.h5",
|
||||
"hdf5/tools/test/testfiles/h5clear_fsm_persist_equal.h5",
|
||||
"hdf5/tools/test/testfiles/h5clear_fsm_persist_noclose.h5",
|
||||
"hdf5/tools/test/testfiles/h5clear_fsm_persist_user_equal.h5",
|
||||
"hdf5/tools/test/testfiles/h5clear_fsm_persist_user_less.h5",
|
||||
"hdf5/tools/test/testfiles/h5clear_sec2_v0.h5",
|
||||
"hdf5/tools/test/testfiles/h5clear_sec2_v2.h5",
|
||||
"hdf5/tools/test/testfiles/h5copy_extlinks_src.h5",
|
||||
"hdf5/tools/test/testfiles/h5copy_extlinks_trg.h5",
|
||||
"hdf5/tools/test/testfiles/h5copy_ref.h5",
|
||||
"hdf5/tools/test/testfiles/h5copytst.h5",
|
||||
"hdf5/tools/test/testfiles/h5copytst_new.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_attr1.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_attr2.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_attr3.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_attr_v_level1.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_attr_v_level2.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_basic1.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_basic2.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_comp_vl_strs.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_danglelinks1.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_danglelinks2.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_dset1.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_dset2.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_dset3.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_dset_zero_dim_size1.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_dset_zero_dim_size2.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_dtypes.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_empty.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_enum_invalid_values.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_eps1.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_eps2.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_exclude1-1.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_exclude1-2.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_exclude2-1.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_exclude2-2.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_exclude3-1.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_exclude3-2.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_ext2softlink_src.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_ext2softlink_trg.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_extlink_src.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_extlink_trg.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_grp_recurse1.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_grp_recurse2.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_grp_recurse_ext1.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_grp_recurse_ext2-1.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_grp_recurse_ext2-2.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_grp_recurse_ext2-3.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_hyper1.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_hyper2.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_linked_softlink.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_links.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_onion_dset_1d.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_onion_dset_ext.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_onion_objs.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_softlinks.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_strings1.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_strings2.h5",
|
||||
"hdf5/tools/test/testfiles/h5fc_edge_v3.h5",
|
||||
"hdf5/tools/test/testfiles/h5fc_err_level.h5",
|
||||
"hdf5/tools/test/testfiles/h5fc_ext1_f.h5",
|
||||
"hdf5/tools/test/testfiles/h5fc_ext1_i.h5",
|
||||
"hdf5/tools/test/testfiles/h5fc_ext1_s.h5",
|
||||
"hdf5/tools/test/testfiles/h5fc_ext2_if.h5",
|
||||
"hdf5/tools/test/testfiles/h5fc_ext2_is.h5",
|
||||
"hdf5/tools/test/testfiles/h5fc_ext2_sf.h5",
|
||||
"hdf5/tools/test/testfiles/h5fc_ext3_isf.h5",
|
||||
"hdf5/tools/test/testfiles/h5fc_ext_none.h5",
|
||||
"hdf5/tools/test/testfiles/h5fc_non_v3.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_CVE-2018-14460.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_CVE-2018-17432.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_aggr.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_attr.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_attr_refs.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_deflate.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_early.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_ext.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_f32le.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_f32le_ex.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_fill.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_filters.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_fletcher.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_fsm_aggr_nopersist.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_fsm_aggr_persist.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_hlink.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_int32le_1d.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_int32le_1d_ex.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_int32le_2d.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_int32le_2d_ex.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_int32le_3d.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_int32le_3d_ex.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_layout.UD.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_layout.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_layout2.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_layout3.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_layouto.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_named_dtypes.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_nbit.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_nested_8bit_enum.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_nested_8bit_enum_deflated.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_none.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_objs.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_paged_nopersist.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_paged_persist.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_refs.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_shuffle.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_soffset.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_szip.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_uint8be.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_uint8be_ex.h5",
|
||||
"hdf5/tools/test/testfiles/h5stat_err_old_fill.h5",
|
||||
"hdf5/tools/test/testfiles/h5stat_err_old_layout.h5",
|
||||
"hdf5/tools/test/testfiles/h5stat_filters.h5",
|
||||
"hdf5/tools/test/testfiles/h5stat_idx.h5",
|
||||
"hdf5/tools/test/testfiles/h5stat_newgrat.h5",
|
||||
"hdf5/tools/test/testfiles/h5stat_threshold.h5",
|
||||
"hdf5/tools/test/testfiles/h5stat_tsohm.h5",
|
||||
"hdf5/tools/test/testfiles/mod_h5clear_mdc_image.h5",
|
||||
"hdf5/tools/test/testfiles/non_comparables1.h5",
|
||||
"hdf5/tools/test/testfiles/non_comparables2.h5",
|
||||
"hdf5/tools/test/testfiles/old_h5fc_ext1_f.h5",
|
||||
"hdf5/tools/test/testfiles/old_h5fc_ext1_i.h5",
|
||||
"hdf5/tools/test/testfiles/old_h5fc_ext1_s.h5",
|
||||
"hdf5/tools/test/testfiles/old_h5fc_ext2_if.h5",
|
||||
"hdf5/tools/test/testfiles/old_h5fc_ext2_is.h5",
|
||||
"hdf5/tools/test/testfiles/old_h5fc_ext2_sf.h5",
|
||||
"hdf5/tools/test/testfiles/old_h5fc_ext3_isf.h5",
|
||||
"hdf5/tools/test/testfiles/old_h5fc_ext_none.h5",
|
||||
"hdf5/tools/test/testfiles/packedbits.h5",
|
||||
"hdf5/tools/test/testfiles/t128bit_float.h5",
|
||||
"hdf5/tools/test/testfiles/tCVE-2021-37501_attr_decode.h5",
|
||||
"hdf5/tools/test/testfiles/tCVE_2018_11206_fill_new.h5",
|
||||
"hdf5/tools/test/testfiles/tCVE_2018_11206_fill_old.h5",
|
||||
"hdf5/tools/test/testfiles/taindices.h5",
|
||||
"hdf5/tools/test/testfiles/tarray1.h5",
|
||||
"hdf5/tools/test/testfiles/tarray1_big.h5",
|
||||
"hdf5/tools/test/testfiles/tarray2.h5",
|
||||
"hdf5/tools/test/testfiles/tarray4.h5",
|
||||
"hdf5/tools/test/testfiles/tarray5.h5",
|
||||
"hdf5/tools/test/testfiles/tarray8.h5",
|
||||
"hdf5/tools/test/testfiles/tattr.h5",
|
||||
"hdf5/tools/test/testfiles/tattr2.h5",
|
||||
"hdf5/tools/test/testfiles/tattr4_be.h5",
|
||||
"hdf5/tools/test/testfiles/tattrintsize.h5",
|
||||
"hdf5/tools/test/testfiles/tattrreg.h5",
|
||||
"hdf5/tools/test/testfiles/tbfloat16.h5",
|
||||
"hdf5/tools/test/testfiles/tbfloat16_be.h5",
|
||||
"hdf5/tools/test/testfiles/tbigdims.h5",
|
||||
"hdf5/tools/test/testfiles/tbinary.h5",
|
||||
"hdf5/tools/test/testfiles/tbitnopaque.h5",
|
||||
"hdf5/tools/test/testfiles/tchar.h5",
|
||||
"hdf5/tools/test/testfiles/tcmpdattrintsize.h5",
|
||||
"hdf5/tools/test/testfiles/tcmpdintarray.h5",
|
||||
"hdf5/tools/test/testfiles/tcmpdints.h5",
|
||||
"hdf5/tools/test/testfiles/tcmpdintsize.h5",
|
||||
"hdf5/tools/test/testfiles/tcomplex.h5",
|
||||
"hdf5/tools/test/testfiles/tcompound.h5",
|
||||
"hdf5/tools/test/testfiles/tcompound_complex.h5",
|
||||
"hdf5/tools/test/testfiles/tcompound_complex2.h5",
|
||||
"hdf5/tools/test/testfiles/tdatareg.h5",
|
||||
"hdf5/tools/test/testfiles/tdset.h5",
|
||||
"hdf5/tools/test/testfiles/tdset2.h5",
|
||||
"hdf5/tools/test/testfiles/tdset_idx.h5",
|
||||
"hdf5/tools/test/testfiles/tempty.h5",
|
||||
"hdf5/tools/test/testfiles/textlink.h5",
|
||||
"hdf5/tools/test/testfiles/textlinkfar.h5",
|
||||
"hdf5/tools/test/testfiles/textlinksrc.h5",
|
||||
"hdf5/tools/test/testfiles/textlinktar.h5",
|
||||
"hdf5/tools/test/testfiles/textpfe.h5",
|
||||
"hdf5/tools/test/testfiles/tfcontents2.h5",
|
||||
"hdf5/tools/test/testfiles/tfilters.h5",
|
||||
"hdf5/tools/test/testfiles/tfloat16.h5",
|
||||
"hdf5/tools/test/testfiles/tfloat16_be.h5",
|
||||
"hdf5/tools/test/testfiles/tfloat4.h5",
|
||||
"hdf5/tools/test/testfiles/tfloat6.h5",
|
||||
"hdf5/tools/test/testfiles/tfloat8.h5",
|
||||
"hdf5/tools/test/testfiles/tfloatsattrs.h5",
|
||||
"hdf5/tools/test/testfiles/tfpformat.h5",
|
||||
"hdf5/tools/test/testfiles/tfvalues.h5",
|
||||
"hdf5/tools/test/testfiles/tgroup.h5",
|
||||
"hdf5/tools/test/testfiles/tgrp_comments.h5",
|
||||
"hdf5/tools/test/testfiles/tgrpnullspace.h5",
|
||||
"hdf5/tools/test/testfiles/thlink.h5",
|
||||
"hdf5/tools/test/testfiles/thyperslab.h5",
|
||||
"hdf5/tools/test/testfiles/tintascii.h5",
|
||||
"hdf5/tools/test/testfiles/tints4dims.h5",
|
||||
"hdf5/tools/test/testfiles/tintsattrs.h5",
|
||||
"hdf5/tools/test/testfiles/tintsnodata.h5",
|
||||
"hdf5/tools/test/testfiles/tlarge_objname.h5",
|
||||
"hdf5/tools/test/testfiles/tldouble.h5",
|
||||
"hdf5/tools/test/testfiles/tldouble_scalar.h5",
|
||||
"hdf5/tools/test/testfiles/tlonglinks.h5",
|
||||
"hdf5/tools/test/testfiles/tloop.h5",
|
||||
"hdf5/tools/test/testfiles/tnamed_dtype_attr.h5",
|
||||
"hdf5/tools/test/testfiles/tnestedcmpddt.h5",
|
||||
"hdf5/tools/test/testfiles/tnestedcomp.h5",
|
||||
"hdf5/tools/test/testfiles/tno-subset.h5",
|
||||
"hdf5/tools/test/testfiles/tnullspace.h5",
|
||||
"hdf5/tools/test/testfiles/torderattr.h5",
|
||||
"hdf5/tools/test/testfiles/tordergr.h5",
|
||||
"hdf5/tools/test/testfiles/trefer_attr.h5",
|
||||
"hdf5/tools/test/testfiles/trefer_compat.h5",
|
||||
"hdf5/tools/test/testfiles/trefer_ext1.h5",
|
||||
"hdf5/tools/test/testfiles/trefer_ext2.h5",
|
||||
"hdf5/tools/test/testfiles/trefer_grp.h5",
|
||||
"hdf5/tools/test/testfiles/trefer_obj.h5",
|
||||
"hdf5/tools/test/testfiles/trefer_obj_del.h5",
|
||||
"hdf5/tools/test/testfiles/trefer_param.h5",
|
||||
"hdf5/tools/test/testfiles/trefer_reg.h5",
|
||||
"hdf5/tools/test/testfiles/trefer_reg_1d.h5",
|
||||
"hdf5/tools/test/testfiles/tsaf.h5",
|
||||
"hdf5/tools/test/testfiles/tscalarattrintsize.h5",
|
||||
"hdf5/tools/test/testfiles/tscalarintattrsize.h5",
|
||||
"hdf5/tools/test/testfiles/tscalarintsize.h5",
|
||||
"hdf5/tools/test/testfiles/tscalarstring.h5",
|
||||
"hdf5/tools/test/testfiles/tslink.h5",
|
||||
"hdf5/tools/test/testfiles/tsoftlinks.h5",
|
||||
"hdf5/tools/test/testfiles/tst_onion_dset_1d.h5",
|
||||
"hdf5/tools/test/testfiles/tst_onion_dset_ext.h5",
|
||||
"hdf5/tools/test/testfiles/tst_onion_objs.h5",
|
||||
"hdf5/tools/test/testfiles/tstr.h5",
|
||||
"hdf5/tools/test/testfiles/tstr2.h5",
|
||||
"hdf5/tools/test/testfiles/tstr3.h5",
|
||||
"hdf5/tools/test/testfiles/tudfilter.h5",
|
||||
"hdf5/tools/test/testfiles/tudfilter2.h5",
|
||||
"hdf5/tools/test/testfiles/tvldtypes1.h5",
|
||||
"hdf5/tools/test/testfiles/tvldtypes2.h5",
|
||||
"hdf5/tools/test/testfiles/tvldtypes3.h5",
|
||||
"hdf5/tools/test/testfiles/tvldtypes4.h5",
|
||||
"hdf5/tools/test/testfiles/tvldtypes5.h5",
|
||||
"hdf5/tools/test/testfiles/tvlenstr_array.h5",
|
||||
"hdf5/tools/test/testfiles/tvlstr.h5",
|
||||
"hdf5/tools/test/testfiles/tvms.h5",
|
||||
"hdf5/tools/test/testfiles/txtfp32.h5",
|
||||
"hdf5/tools/test/testfiles/txtfp64.h5",
|
||||
"hdf5/tools/test/testfiles/txtin16.h5",
|
||||
"hdf5/tools/test/testfiles/txtin32.h5",
|
||||
"hdf5/tools/test/testfiles/txtin8.h5",
|
||||
"hdf5/tools/test/testfiles/txtstr.h5",
|
||||
"hdf5/tools/test/testfiles/txtuin16.h5",
|
||||
"hdf5/tools/test/testfiles/txtuin32.h5",
|
||||
"hdf5/tools/test/testfiles/vds/1_a.h5",
|
||||
"hdf5/tools/test/testfiles/vds/1_b.h5",
|
||||
"hdf5/tools/test/testfiles/vds/1_c.h5",
|
||||
"hdf5/tools/test/testfiles/vds/1_d.h5",
|
||||
"hdf5/tools/test/testfiles/vds/1_e.h5",
|
||||
"hdf5/tools/test/testfiles/vds/1_f.h5",
|
||||
"hdf5/tools/test/testfiles/vds/1_vds.h5",
|
||||
"hdf5/tools/test/testfiles/vds/2_a.h5",
|
||||
"hdf5/tools/test/testfiles/vds/2_b.h5",
|
||||
"hdf5/tools/test/testfiles/vds/2_c.h5",
|
||||
"hdf5/tools/test/testfiles/vds/2_d.h5",
|
||||
"hdf5/tools/test/testfiles/vds/2_e.h5",
|
||||
"hdf5/tools/test/testfiles/vds/2_vds.h5",
|
||||
"hdf5/tools/test/testfiles/vds/3_1_vds.h5",
|
||||
"hdf5/tools/test/testfiles/vds/3_2_vds.h5",
|
||||
"hdf5/tools/test/testfiles/vds/4_0.h5",
|
||||
"hdf5/tools/test/testfiles/vds/4_1.h5",
|
||||
"hdf5/tools/test/testfiles/vds/4_2.h5",
|
||||
"hdf5/tools/test/testfiles/vds/4_vds.h5",
|
||||
"hdf5/tools/test/testfiles/vds/5_a.h5",
|
||||
"hdf5/tools/test/testfiles/vds/5_b.h5",
|
||||
"hdf5/tools/test/testfiles/vds/5_c.h5",
|
||||
"hdf5/tools/test/testfiles/vds/5_vds.h5",
|
||||
"hdf5/tools/test/testfiles/vds/a.h5",
|
||||
"hdf5/tools/test/testfiles/vds/b.h5",
|
||||
"hdf5/tools/test/testfiles/vds/c.h5",
|
||||
"hdf5/tools/test/testfiles/vds/d.h5",
|
||||
"hdf5/tools/test/testfiles/vds/f-0.h5",
|
||||
"hdf5/tools/test/testfiles/vds/f-3.h5",
|
||||
"hdf5/tools/test/testfiles/vds/vds-eiger.h5",
|
||||
"hdf5/tools/test/testfiles/vds/vds-percival-unlim-maxmin.h5",
|
||||
"hdf5/tools/test/testfiles/xml/tbitfields.h5",
|
||||
"hdf5/tools/test/testfiles/xml/tcompound2.h5",
|
||||
"hdf5/tools/test/testfiles/xml/tdset2.h5",
|
||||
"hdf5/tools/test/testfiles/xml/tenum.h5",
|
||||
"hdf5/tools/test/testfiles/xml/test35.nc",
|
||||
"hdf5/tools/test/testfiles/xml/tloop2.h5",
|
||||
"hdf5/tools/test/testfiles/xml/tname-amp.h5",
|
||||
"hdf5/tools/test/testfiles/xml/tname-apos.h5",
|
||||
"hdf5/tools/test/testfiles/xml/tname-gt.h5",
|
||||
"hdf5/tools/test/testfiles/xml/tname-lt.h5",
|
||||
"hdf5/tools/test/testfiles/xml/tname-quot.h5",
|
||||
"hdf5/tools/test/testfiles/xml/tname-sp.h5",
|
||||
"hdf5/tools/test/testfiles/xml/tnodata.h5",
|
||||
"hdf5/tools/test/testfiles/xml/tobjref.h5",
|
||||
"hdf5/tools/test/testfiles/xml/topaque.h5",
|
||||
"hdf5/tools/test/testfiles/xml/tref-escapes-at.h5",
|
||||
"hdf5/tools/test/testfiles/xml/tref-escapes.h5",
|
||||
"hdf5/tools/test/testfiles/xml/tref.h5",
|
||||
"hdf5/tools/test/testfiles/xml/tstring-at.h5",
|
||||
"hdf5/tools/test/testfiles/xml/tstring.h5",
|
||||
"hdf5/tools/test/testfiles/zerodim.h5",
|
||||
"netcdf-c/h5_test/ref_tst_h_compounds.h5",
|
||||
"netcdf-c/h5_test/ref_tst_h_compounds2.h5",
|
||||
"netcdf-c/nc_test4/ref_hdf5_compat1.nc",
|
||||
"netcdf-c/nc_test4/ref_hdf5_compat2.nc",
|
||||
"netcdf-c/nc_test4/ref_hdf5_compat3.nc",
|
||||
"netcdf-c/nc_test4/ref_szip.h5",
|
||||
"netcdf-c/nc_test4/ref_tst_compounds.nc",
|
||||
"netcdf-c/nc_test4/ref_tst_dims.nc",
|
||||
"netcdf-c/nc_test4/ref_tst_interops4.nc",
|
||||
"netcdf-c/nc_test4/ref_tst_xplatform2_1.nc",
|
||||
"netcdf-c/nc_test4/ref_tst_xplatform2_2.nc",
|
||||
"netcdf-c/nc_test4/tdset.h5",
|
||||
"netcdf-c/ncdump/ref_nc_test_netcdf4_4_0.nc",
|
||||
"netcdf-c/ncdump/ref_no_ncproperty.nc",
|
||||
"netcdf-c/ncdump/ref_provenance_v1.nc",
|
||||
"netcdf-c/ncdump/ref_test_corrupt_magic.nc",
|
||||
"netcdf-c/ncdump/ref_tst_compounds2.nc",
|
||||
"netcdf-c/ncdump/ref_tst_compounds3.nc",
|
||||
"netcdf-c/ncdump/ref_tst_compounds4.nc",
|
||||
"netcdf-c/ncdump/ref_tst_irish_rover.nc",
|
||||
"netcdf4-python/examples/data/prmsl.2000.nc",
|
||||
"netcdf4-python/examples/data/prmsl.2001.nc",
|
||||
"netcdf4-python/examples/data/prmsl.2002.nc",
|
||||
"netcdf4-python/examples/data/prmsl.2003.nc",
|
||||
"netcdf4-python/examples/data/prmsl.2004.nc",
|
||||
"netcdf4-python/examples/data/prmsl.2005.nc",
|
||||
"netcdf4-python/examples/data/prmsl.2006.nc",
|
||||
"netcdf4-python/examples/data/prmsl.2007.nc",
|
||||
"netcdf4-python/examples/data/prmsl.2008.nc",
|
||||
"netcdf4-python/examples/data/prmsl.2009.nc",
|
||||
"netcdf4-python/examples/data/prmsl.2010.nc",
|
||||
"netcdf4-python/examples/data/prmsl.2011.nc",
|
||||
"netcdf4-python/examples/data/rtofs_glo_3dz_f006_6hrly_reg3.nc",
|
||||
"netcdf4-python/test/20171025_2056.Cloud_Top_Height.nc",
|
||||
"netcdf4-python/test/issue1152.nc",
|
||||
"netcdf4-python/test/issue671.nc",
|
||||
"netcdf4-python/test/issue672.nc",
|
||||
"netcdf4-python/test/test_gold.nc",
|
||||
"usnistgov_h5wasm/test/array.h5",
|
||||
"usnistgov_h5wasm/test/compressed.h5",
|
||||
"usnistgov_h5wasm/test/empty.h5",
|
||||
"usnistgov_h5wasm/test/float16.h5",
|
||||
"usnistgov_h5wasm/test/vlen.h5",
|
||||
"xarray-data/ROMS_example.nc",
|
||||
"xarray-data/basin_mask.nc",
|
||||
"xarray-data/imerghh_730.hdf5",
|
||||
"xarray-data/precipitation.nc4"
|
||||
]
|
||||
}
|
||||
Executable
+88
@@ -0,0 +1,88 @@
|
||||
#!/usr/bin/env python3
|
||||
"""check.py <results_dir> <baseline.json> [--update]
|
||||
|
||||
The conformance gate. Fails (exit 1) when
|
||||
* clawhdf5 panicked, hung, crashed or ran out of memory on any file, or
|
||||
* the ok count fell below the baseline's, or
|
||||
* a file the baseline lists as ok is no longer ok (even if another file
|
||||
became ok and the total held).
|
||||
New ok files are reported so the baseline can be raised (--update rewrites it
|
||||
from the results).
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
|
||||
FATAL = ("panic", "hang", "crash", "oom")
|
||||
|
||||
|
||||
def main():
|
||||
args = [a for a in sys.argv[1:] if not a.startswith("--")]
|
||||
update = "--update" in sys.argv
|
||||
res_dir, base_path = args
|
||||
res = json.load(open(os.path.join(res_dir, "results.json")))
|
||||
rows = res["rows"]
|
||||
counts = {}
|
||||
per_corpus = {}
|
||||
for r in rows:
|
||||
counts[r["class"]] = counts.get(r["class"], 0) + 1
|
||||
pc = per_corpus.setdefault(r["corpus"], {})
|
||||
pc[r["class"]] = pc.get(r["class"], 0) + 1
|
||||
ok_files = sorted(r["file"] for r in rows if r["class"] == "ok")
|
||||
|
||||
if update:
|
||||
meta = {}
|
||||
mp = os.path.join(res_dir, "report-meta.json")
|
||||
if os.path.exists(mp):
|
||||
meta = json.load(open(mp))
|
||||
base = {
|
||||
"comment": "conformance/run.sh fails if the ok count drops below `ok` or a file in `ok_files` stops being ok. "
|
||||
"Regenerate with `conformance/run.sh --update-baseline` after an intended change.",
|
||||
"commit": meta.get("commit", ""),
|
||||
"date": meta.get("date", ""),
|
||||
"reference": meta.get("reference", ""),
|
||||
"files": len(rows),
|
||||
"ok": len(ok_files),
|
||||
"counts": dict(sorted(counts.items())),
|
||||
"per_corpus": {k: dict(sorted(v.items())) for k, v in sorted(per_corpus.items())},
|
||||
"ok_files": ok_files,
|
||||
}
|
||||
with open(base_path, "w") as fh:
|
||||
json.dump(base, fh, indent=1)
|
||||
fh.write("\n")
|
||||
print(f"baseline updated: {len(ok_files)} ok of {len(rows)} files -> {base_path}")
|
||||
return 0
|
||||
|
||||
base = json.load(open(base_path))
|
||||
failures = []
|
||||
fatal = [r for r in rows if r["class"] in FATAL]
|
||||
for r in fatal:
|
||||
failures.append(f"{r['class']}: {r['file']}: {r['ours_detail'][:200]}")
|
||||
if len(ok_files) < base["ok"]:
|
||||
failures.append(f"ok count dropped: {len(ok_files)} < baseline {base['ok']}")
|
||||
now_ok = set(ok_files)
|
||||
by_file = {r["file"]: r for r in rows}
|
||||
for f in base["ok_files"]:
|
||||
if f not in now_ok:
|
||||
r = by_file.get(f)
|
||||
why = f"now {r['class']}: {(r['ours_detail'] or r['first_issue'])[:200]}" if r else "no longer in the corpus"
|
||||
failures.append(f"regressed: {f}: {why}")
|
||||
gained = sorted(now_ok - set(base["ok_files"]))
|
||||
|
||||
print(f"conformance: {len(ok_files)} ok of {len(rows)} files (baseline {base['ok']} of {base['files']}); "
|
||||
+ ", ".join(f"{k} {v}" for k, v in sorted(counts.items())))
|
||||
if gained:
|
||||
print(f"{len(gained)} file(s) newly ok — raise the baseline with `conformance/run.sh --update-baseline`:")
|
||||
for f in gained:
|
||||
print(f" + {f}")
|
||||
if failures:
|
||||
print(f"CONFORMANCE GATE FAILED ({len(failures)}):")
|
||||
for f in failures:
|
||||
print(f" - {f}")
|
||||
return 1
|
||||
print("conformance gate passed")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
Executable
+289
@@ -0,0 +1,289 @@
|
||||
#!/usr/bin/env python3
|
||||
"""compare.py <results_dir>: classify each file and group failures by root cause.
|
||||
|
||||
Writes <results_dir>/results.csv, results.json and summary.md.
|
||||
File classes (first match wins):
|
||||
hang, oom, crash, panic ours: timeout / allocation failure / signal / any panic (caught or not)
|
||||
h5py-cannot-read libhdf5/h5py failed to open the file (or crashed/hung)
|
||||
our-error we fail to open, list, or read something h5py reads
|
||||
mismatch we read something with different shape/values, or a different object set
|
||||
ok
|
||||
"""
|
||||
import collections
|
||||
import csv
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
|
||||
R = sys.argv[1]
|
||||
RUNS = os.path.join(R, "runs")
|
||||
|
||||
|
||||
def load(d, name):
|
||||
rc_p = os.path.join(d, name + ".rc")
|
||||
if not os.path.exists(rc_p):
|
||||
return None
|
||||
rc = int(open(rc_p).read().strip() or -1)
|
||||
err = open(os.path.join(d, name + ".err"), errors="replace").read()
|
||||
js = None
|
||||
try:
|
||||
js = json.load(open(os.path.join(d, name + ".json")))
|
||||
except Exception: # noqa: BLE001
|
||||
pass
|
||||
return {"rc": rc, "err": err, "json": js}
|
||||
|
||||
|
||||
def proc_status(p):
|
||||
"""-> (status, detail)"""
|
||||
if p is None:
|
||||
return "missing", ""
|
||||
rc, err = p["rc"], p["err"]
|
||||
first_panic = next((ln for ln in err.splitlines() if ln.startswith("PANIC:") or "panicked at" in ln), "")
|
||||
if rc == 0 and p["json"] is not None:
|
||||
return "ok", ""
|
||||
if rc == 137 or rc == 124:
|
||||
return "hang", f"timeout ({os.environ.get('TMO', '20')} s)"
|
||||
if "memory allocation of" in err or "MemoryError" in err or "std::bad_alloc" in err:
|
||||
m = re.search(r"memory allocation of \d+ bytes failed", err)
|
||||
return "oom", m.group(0) if m else "allocation failure"
|
||||
if "overflowed its stack" in err:
|
||||
return "crash", "stack overflow"
|
||||
if rc == 101:
|
||||
return "panic", first_panic or (err.strip().splitlines() or [""])[-1]
|
||||
if rc in (134, 139, 136, 135, 132) or rc > 128:
|
||||
sig = {134: "SIGABRT", 139: "SIGSEGV", 136: "SIGFPE", 135: "SIGBUS", 132: "SIGILL"}.get(rc, f"signal {rc - 128}")
|
||||
tail = [ln for ln in err.strip().splitlines() if ln.strip()][-1:]
|
||||
return "crash", f"{sig}: {tail[0][:200] if tail else ''}"
|
||||
tail = [ln for ln in err.strip().splitlines() if ln.strip()][-1:]
|
||||
return "crash", f"rc={rc}: {tail[0][:200] if tail else ''}"
|
||||
|
||||
|
||||
def norm(msg):
|
||||
m = msg.split("\n")[0]
|
||||
m = re.sub(r"0x[0-9a-fA-F]+", "X", m)
|
||||
m = re.sub(r'"[^"]*"', '"…"', m)
|
||||
m = re.sub(r"'[^']*'", "'…'", m)
|
||||
m = re.sub(r"\d+", "N", m)
|
||||
return m[:160]
|
||||
|
||||
|
||||
def panic_head(msg):
|
||||
"""First line + first clawhdf5 frame of a PANIC record."""
|
||||
lines = msg.split("\n")
|
||||
frame = next((ln.strip() for ln in lines[1:] if "clawhdf5_format" in ln), "")
|
||||
return lines[0][:300], frame[:300]
|
||||
|
||||
|
||||
def eq_shape(a, b):
|
||||
return a == b
|
||||
|
||||
|
||||
rows = []
|
||||
issues_by_file = {}
|
||||
root_causes = collections.defaultdict(lambda: {"files": set(), "count": 0, "examples": []})
|
||||
mismatch_causes = collections.defaultdict(lambda: {"files": set(), "count": 0, "examples": []})
|
||||
panics = []
|
||||
ref_only_errors = collections.Counter()
|
||||
incomparable = collections.Counter()
|
||||
|
||||
|
||||
def add(bucket, key, file, example):
|
||||
b = bucket[key]
|
||||
b["count"] += 1
|
||||
if file not in b["files"] and len(b["examples"]) < 6:
|
||||
b["examples"].append(example)
|
||||
b["files"].add(file)
|
||||
|
||||
|
||||
files = [ln.strip() for ln in open(os.path.join(R, "files.txt")) if ln.strip()]
|
||||
for rel in files:
|
||||
d = os.path.join(RUNS, rel.replace("/", "__"))
|
||||
corpus = rel.split("/")[0]
|
||||
ours, ref = load(d, "ours"), load(d, "ref")
|
||||
h5dump = load(d, "h5dump")
|
||||
os_, od = proc_status(ours)
|
||||
rs, rd = proc_status(ref)
|
||||
oj = ours["json"] if ours else None
|
||||
rj = ref["json"] if ref else None
|
||||
issues = [] # (kind, detail)
|
||||
caught_panics = []
|
||||
|
||||
def scan_err(path, what, msg):
|
||||
if msg.startswith("PANIC:"):
|
||||
caught_panics.append((path, what, msg))
|
||||
|
||||
if oj:
|
||||
for o in oj.get("objects", []):
|
||||
for k in ("error", "attrs_error", "list_error"):
|
||||
if k in o:
|
||||
scan_err(o["path"], k, o[k])
|
||||
for an, av in (o.get("attrs") or {}).items():
|
||||
if "error" in av:
|
||||
scan_err(o["path"], f"attr {an}", av["error"])
|
||||
if oj.get("open_error", "").startswith("PANIC:"):
|
||||
caught_panics.append(("<open>", "open", oj["open_error"]))
|
||||
|
||||
ref_open_fail = rs != "ok" or (rj is not None and "open_error" in rj)
|
||||
ours_open_err = oj.get("open_error") if oj else None
|
||||
n_obj = n_ok = 0
|
||||
if os_ == "ok" and rj and not ref_open_fail and not ours_open_err:
|
||||
ro = {x["path"]: x for x in rj.get("objects", [])}
|
||||
oo = {x["path"]: x for x in oj.get("objects", [])}
|
||||
our_list_errors = [x for x in oo.values() if "list_error" in x]
|
||||
for p in sorted(set(ro) | set(oo)):
|
||||
a, b = ro.get(p), oo.get(p)
|
||||
n_obj += 1
|
||||
if a is None:
|
||||
issues.append(("mismatch", f"extra object {p} (kind={b.get('kind')})", "extra-object", b))
|
||||
continue
|
||||
if b is None:
|
||||
if our_list_errors:
|
||||
continue # accounted for by the list_error
|
||||
issues.append(("mismatch", f"missing object {p} (kind={a.get('kind')})", "missing-object", a))
|
||||
continue
|
||||
ok = True
|
||||
if a.get("kind") != b.get("kind") and "error" not in b and "error" not in a:
|
||||
issues.append(("mismatch", f"{p}: kind {a.get('kind')} vs ours {b.get('kind')}", "kind", b))
|
||||
ok = False
|
||||
for k in ("error", "list_error", "attrs_error"):
|
||||
if k in b and k not in a:
|
||||
issues.append(("our-error", f"{p}: {k}: {b[k]}", b[k], b))
|
||||
ok = False
|
||||
elif k in a and k not in b and k == "error":
|
||||
ref_only_errors[norm(a[k])] += 1
|
||||
if a.get("kind") == "dataset" and "error" not in a and "error" not in b:
|
||||
if "skipped" in a or "skipped" in b:
|
||||
pass
|
||||
elif a.get("converted"):
|
||||
incomparable[f"dataset {a['converted']}"] += 1
|
||||
elif a.get("shape") != b.get("shape"):
|
||||
issues.append(("mismatch", f"{p}: shape {a.get('shape')} vs ours {b.get('shape')}", "shape", b))
|
||||
ok = False
|
||||
elif a.get("hash") != b.get("hash"):
|
||||
issues.append(("mismatch", f"{p}: values differ (h5py {a.get('dtype')} vs ours {b.get('dtype')})", "values", b | {"ref_head": a.get("head"), "ref_dtype": a.get("dtype")}))
|
||||
ok = False
|
||||
ra, oa = a.get("attrs") or {}, b.get("attrs") or {}
|
||||
if "attrs_error" not in b and "attrs_error" not in a:
|
||||
for an in sorted(set(ra) | set(oa)):
|
||||
x, y = ra.get(an), oa.get(an)
|
||||
if x is None:
|
||||
issues.append(("mismatch", f"{p}@{an}: extra attribute", "extra-attr", y or {}))
|
||||
elif y is None:
|
||||
issues.append(("mismatch", f"{p}@{an}: missing attribute", "missing-attr", x))
|
||||
elif "error" in y and "error" not in x:
|
||||
issues.append(("our-error", f"{p}@{an}: {y['error']}", y["error"], y))
|
||||
elif "error" in x:
|
||||
continue
|
||||
elif x.get("converted"):
|
||||
incomparable[f"attr {x['converted']}"] += 1
|
||||
elif x.get("shape") != y.get("shape"):
|
||||
issues.append(("mismatch", f"{p}@{an}: attr shape {x.get('shape')} vs ours {y.get('shape')}", "attr-shape", y | {"ref_dtype": x.get("dtype")}))
|
||||
elif x.get("hash") != y.get("hash"):
|
||||
issues.append(("mismatch", f"{p}@{an}: attr values differ (h5py {x.get('dtype')} vs ours {y.get('dtype')})", "attr-values", y | {"ref_head": x.get("head"), "ref_dtype": x.get("dtype")}))
|
||||
if ok:
|
||||
n_ok += 1
|
||||
|
||||
# classify
|
||||
if os_ in ("hang", "oom", "crash", "panic"):
|
||||
cls = os_
|
||||
elif caught_panics:
|
||||
cls = "panic"
|
||||
elif ref_open_fail:
|
||||
cls = "h5py-cannot-read"
|
||||
elif ours_open_err:
|
||||
cls = "our-error"
|
||||
issues.append(("our-error", f"open: {ours_open_err}", ours_open_err, {}))
|
||||
elif any(i[0] == "our-error" for i in issues):
|
||||
cls = "our-error"
|
||||
elif issues:
|
||||
cls = "mismatch"
|
||||
else:
|
||||
cls = "ok"
|
||||
|
||||
if os_ in ("hang", "oom", "crash", "panic") or caught_panics:
|
||||
panics.append({
|
||||
"file": rel, "class": cls, "detail": od,
|
||||
"stderr": (ours["err"] if ours else "")[:3000],
|
||||
"caught": [(p, w, m[:2500]) for p, w, m in caught_panics[:3]],
|
||||
"n_caught": len(caught_panics),
|
||||
})
|
||||
for kind, detail, key, rec in issues:
|
||||
if kind == "our-error":
|
||||
add(root_causes, norm(key), rel, detail[:300])
|
||||
else:
|
||||
if key in ("values", "attr-values", "shape", "attr-shape"):
|
||||
mk = f"{key}: ours={rec.get('dtype')} h5py={rec.get('ref_dtype')} layout={rec.get('layout','-')} filters={rec.get('filters','-')}"
|
||||
else:
|
||||
mk = key
|
||||
add(mismatch_causes, mk, rel, detail[:300] + (f" | ref_head={rec.get('ref_head')} our_head={rec.get('head')}" if rec.get("ref_head") else ""))
|
||||
ref_detail = rd if rs != "ok" else ((rj or {}).get("open_error") or "")
|
||||
h5d = ""
|
||||
if h5dump:
|
||||
rc = h5dump["rc"]
|
||||
h5d = {0: "ok", 1: "error", 137: "hang", 124: "hang", 134: "SIGABRT", 139: "SIGSEGV", 136: "SIGFPE", 135: "SIGBUS"}.get(rc, f"rc={rc}")
|
||||
if "memory allocation" in h5dump["err"] or "Cannot allocate" in h5dump["err"]:
|
||||
h5d += "(oom)"
|
||||
rows.append({
|
||||
"file": rel, "corpus": corpus, "class": cls,
|
||||
"ours": os_ if os_ != "ok" else ("open-error" if ours_open_err else ("panic" if caught_panics else "ok")),
|
||||
"ours_detail": (od or ours_open_err or (caught_panics[0][2].split("\n")[0] if caught_panics else ""))[:300],
|
||||
"ref": rs if rs != "ok" else ("open-error" if (rj or {}).get("open_error") else "ok"),
|
||||
"ref_detail": ref_detail[:300],
|
||||
"h5dump_1_14_6": h5d,
|
||||
"h5dump_detail": ([ln for ln in h5dump["err"].splitlines() if ln.strip()][-1:] or [""])[0][:200] if h5dump else "",
|
||||
"objects": n_obj, "objects_ok": n_ok,
|
||||
"issues": len(issues), "first_issue": issues[0][1][:300] if issues else "",
|
||||
"superblock": (oj or {}).get("superblock_version", ""),
|
||||
})
|
||||
# the first issues of each file, for report.py's known-cause matching
|
||||
issues_by_file[rel] = [
|
||||
{"kind": k, "key": key, "detail": det[:300], "ours_dtype": rec.get("dtype"), "ref_dtype": rec.get("ref_dtype")}
|
||||
for k, det, key, rec in issues[:50]
|
||||
]
|
||||
|
||||
with open(os.path.join(R, "results.csv"), "w", newline="") as fh:
|
||||
w = csv.DictWriter(fh, fieldnames=list(rows[0].keys()))
|
||||
w.writeheader()
|
||||
w.writerows(rows)
|
||||
|
||||
|
||||
def ser(b):
|
||||
return {k: {"files": len(v["files"]), "count": v["count"], "examples": v["examples"], "file_list": sorted(v["files"])} for k, v in sorted(b.items(), key=lambda kv: -len(kv[1]["files"]))}
|
||||
|
||||
|
||||
json.dump({"rows": rows, "issues": issues_by_file, "root_causes": ser(root_causes), "mismatch_causes": ser(mismatch_causes),
|
||||
"panics": panics, "incomparable": incomparable.most_common(), "ref_only_errors": ref_only_errors.most_common()},
|
||||
open(os.path.join(R, "results.json"), "w"), indent=1)
|
||||
|
||||
classes = ["ok", "our-error", "mismatch", "h5py-cannot-read", "hang", "panic", "crash", "oom"]
|
||||
by_corpus = collections.defaultdict(collections.Counter)
|
||||
for r in rows:
|
||||
by_corpus[r["corpus"]][r["class"]] += 1
|
||||
by_corpus["ALL"][r["class"]] += 1
|
||||
lines = ["# Conformance sweep summary", "", "| corpus | files | " + " | ".join(classes) + " |", "|---" * (len(classes) + 2) + "|"]
|
||||
for c in sorted(by_corpus, key=lambda k: (k == "ALL", k)):
|
||||
cnt = by_corpus[c]
|
||||
lines.append(f"| {c} | {sum(cnt.values())} | " + " | ".join(str(cnt.get(k, 0)) for k in classes) + " |")
|
||||
lines += ["", "## Panics / hangs / crashes / OOM", ""]
|
||||
for p in panics:
|
||||
lines.append(f"- **{p['file']}** [{p['class']}] {p['detail']}")
|
||||
for path, what, m in p["caught"][:1]:
|
||||
lines.append(" ```\n " + f"{path} ({what}): " + m.replace("\n", "\n ")[:1500] + "\n ```")
|
||||
if not p["caught"] and p["stderr"]:
|
||||
lines.append(" ```\n " + p["stderr"].strip()[:1500].replace("\n", "\n ") + "\n ```")
|
||||
lines += ["", "## Our-error root causes (files affected)", ""]
|
||||
for k, v in ser(root_causes).items():
|
||||
lines.append(f"- [{v['files']} files, {v['count']} objs] `{k}`")
|
||||
for ex in v["examples"][:3]:
|
||||
lines.append(f" - {ex}")
|
||||
lines += ["", "## Mismatch root causes", ""]
|
||||
for k, v in ser(mismatch_causes).items():
|
||||
lines.append(f"- [{v['files']} files, {v['count']} objs] `{k}`")
|
||||
for ex in v["examples"][:3]:
|
||||
lines.append(f" - {ex}")
|
||||
lines += ["", "## Objects h5py fails on but we read (top)", ""]
|
||||
for k, n in ref_only_errors.most_common(15):
|
||||
lines.append(f"- {n} x `{k}`")
|
||||
open(os.path.join(R, "summary.md"), "w").write("\n".join(lines) + "\n")
|
||||
print("\n".join(lines[:4 + len(by_corpus)]))
|
||||
@@ -0,0 +1,19 @@
|
||||
# Conformance corpora, pinned by commit. fetch-corpus.sh reads this file.
|
||||
#
|
||||
# name git-url commit root [sparse-checkout patterns...]
|
||||
#
|
||||
# `root` is the directory inside the checkout that is swept ("." = all of it).
|
||||
# Patterns are git non-cone sparse-checkout patterns; none = whole repository.
|
||||
# Every file under <root> with an HDF5/netCDF-4 extension is probed; for
|
||||
# cve_hdf5 the extension-less files in cvefiles/ and fuzzerfiles/ are too.
|
||||
# Licences: each corpus keeps its upstream licence; nothing here is committed
|
||||
# to this repository — the files are downloaded into the gitignored cache.
|
||||
hdf5 https://github.com/HDFGroup/hdf5.git a3cf1ea82cc7a66e50029a688121e1b105a7ce88 . *.h5 *.he5 *.nc *.hdf5 *.h5f
|
||||
cve_hdf5 https://github.com/HDFGroup/cve_hdf5.git 3fd1f5ae3869e01b8ae02b41d7108de7ffb1a374 .
|
||||
netcdf-c https://github.com/Unidata/netcdf-c.git beb7b9585273c1548386231a59b809d906359033 . /nc_test4/*.nc /ncdump/*.nc /nc_test4/*.h5 /ncdump/*.h5 /h5_test/*.h5 /hdf5_test/*.h5
|
||||
NCAS-CMS_pyfive https://github.com/NCAS-CMS/pyfive.git 8cf07b8749133f41c5e30b8a4c604486f687fe74 . *.h5 *.hdf5 *.hdf *.nc *.he5
|
||||
usnistgov_h5wasm https://github.com/usnistgov/h5wasm.git 02f6336527d2812783fcedabfbf42127ec8d06d2 . *.h5 *.hdf5 *.hdf *.nc *.he5
|
||||
netcdf4-python https://github.com/Unidata/netcdf4-python.git 6e67576d39aef8091fb20bd767b4f1a52ddc1bec . *.nc *.h5
|
||||
xarray-data https://github.com/pydata/xarray-data.git a35297e9da2cc99c811014f0c8a4297345a5c28d . /basin_mask.nc /precipitation.nc4 /imerghh_730.hdf5 /eraint_uvz.nc /ROMS_example.nc /tiny.nc
|
||||
# h5py 3.16.0 (tag 3.16.0), its test data files.
|
||||
h5py_data https://github.com/h5py/h5py.git b2f0347c4200333acd89b43733f1caa0c115162f h5py/tests/data_files /h5py/tests/data_files/*
|
||||
Executable
+39
@@ -0,0 +1,39 @@
|
||||
#!/usr/bin/env bash
|
||||
# fetch-corpus.sh [cache_dir]
|
||||
#
|
||||
# Download the corpora pinned in conformance/corpus.txt into the (gitignored)
|
||||
# cache: <cache>/src/<name> is a shallow, sparse, blob-filtered checkout of the
|
||||
# pinned commit and <cache>/corpus/<name> links to the swept root inside it.
|
||||
# A corpus already checked out at its pinned commit is left alone, so a second
|
||||
# run costs nothing and needs no network.
|
||||
set -euo pipefail
|
||||
HERE="$(cd "$(dirname "$0")" && pwd)"
|
||||
CACHE="${1:-${CONFORMANCE_CACHE:-$HERE/.cache}}"
|
||||
mkdir -p "$CACHE/src" "$CACHE/corpus"
|
||||
CACHE="$(cd "$CACHE" && pwd)"
|
||||
|
||||
retry() { local i; for i in 1 2 3 4; do "$@" && return 0; sleep $((i * 5)); done; return 1; }
|
||||
|
||||
grep -v '^[[:space:]]*\(#\|$\)' "$HERE/corpus.txt" | while read -r name url commit root patterns; do
|
||||
src="$CACHE/src/$name"
|
||||
if [ -d "$src/.git" ] && [ "$(git -C "$src" rev-parse HEAD 2>/dev/null)" = "$commit" ]; then
|
||||
echo "cached $name @ ${commit:0:12}"
|
||||
else
|
||||
echo "fetching $name @ ${commit:0:12} from $url"
|
||||
rm -rf "$src"
|
||||
git init -q "$src"
|
||||
git -C "$src" remote add origin "$url"
|
||||
git -C "$src" config advice.detachedHead false
|
||||
if [ -n "$patterns" ]; then
|
||||
git -C "$src" config core.sparseCheckout true
|
||||
# no-cone patterns (globs); `set -f` keeps the shell from expanding them
|
||||
(set -f; printf '%s\n' $patterns) > "$src/.git/info/sparse-checkout"
|
||||
fi
|
||||
retry git -C "$src" fetch -q --depth 1 --filter=blob:none origin "$commit"
|
||||
retry git -C "$src" checkout -q FETCH_HEAD
|
||||
got="$(git -C "$src" rev-parse HEAD)"
|
||||
[ "$got" = "$commit" ] || { echo "error: $name checked out $got, expected $commit" >&2; exit 1; }
|
||||
fi
|
||||
ln -sfn "$src/$root" "$CACHE/corpus/$name"
|
||||
done
|
||||
echo "corpus ready in $CACHE/corpus"
|
||||
@@ -0,0 +1,48 @@
|
||||
#!/usr/bin/env python3
|
||||
"""list_files.py <corpus_dir>: print the files the sweep probes, one per line,
|
||||
as <corpus>/<path> in byte order.
|
||||
|
||||
* every file named *.h5 *.hdf5 *.he5 *.nc *.nc4 *.hdf *.h5f in each corpus,
|
||||
except netCDF classic / 64-bit-offset / CDF5 files (magic "CDF"): they are
|
||||
not HDF5, so neither side can read them and they say nothing;
|
||||
* plus, for cve_hdf5, every file in cvefiles/ and fuzzerfiles/ except
|
||||
.md/.c sources — the reproducers are mostly extension-less, and they are
|
||||
kept whatever their bytes look like (that is their point).
|
||||
"""
|
||||
import os
|
||||
import sys
|
||||
|
||||
EXTS = (".h5", ".hdf5", ".he5", ".nc", ".nc4", ".hdf", ".h5f")
|
||||
|
||||
|
||||
def walk(top):
|
||||
for dirpath, dirnames, filenames in os.walk(top):
|
||||
dirnames[:] = [d for d in dirnames if d != ".git"]
|
||||
for fn in filenames:
|
||||
p = os.path.join(dirpath, fn)
|
||||
if os.path.isfile(p) and not os.path.islink(p):
|
||||
yield os.path.relpath(p, top)
|
||||
|
||||
|
||||
def main(root):
|
||||
out = set()
|
||||
for corpus in sorted(os.listdir(root)):
|
||||
top = os.path.join(root, corpus)
|
||||
if not os.path.isdir(top):
|
||||
continue
|
||||
for rel in walk(top):
|
||||
path = os.path.join(top, rel)
|
||||
if rel.lower().endswith(EXTS):
|
||||
with open(path, "rb") as fh:
|
||||
if fh.read(3) == b"CDF":
|
||||
continue
|
||||
out.add(f"{corpus}/{rel}")
|
||||
elif corpus == "cve_hdf5" and rel.split(os.sep)[0] in ("cvefiles", "fuzzerfiles") \
|
||||
and not rel.endswith((".md", ".c")):
|
||||
out.add(f"{corpus}/{rel}")
|
||||
for f in sorted(out, key=lambda s: s.encode()):
|
||||
print(f)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main(sys.argv[1])
|
||||
Generated
+458
@@ -0,0 +1,458 @@
|
||||
# This file is automatically @generated by Cargo.
|
||||
# It is not intended for manual editing.
|
||||
version = 4
|
||||
|
||||
[[package]]
|
||||
name = "adler2"
|
||||
version = "2.0.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "320119579fcad9c21884f5c4861d16174d0e06250625266f50fe6898340abefa"
|
||||
|
||||
[[package]]
|
||||
name = "better_io"
|
||||
version = "0.2.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ef0a3155e943e341e557863e69a708999c94ede624e37865c8e2a91b94efa78f"
|
||||
|
||||
[[package]]
|
||||
name = "block-buffer"
|
||||
version = "0.10.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "3078c7629b62d3f0439517fa394996acacc5cbc91c5a20d8c658e77abd503a71"
|
||||
dependencies = [
|
||||
"generic-array",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "byteorder"
|
||||
version = "1.5.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "1fd0f2584146f6f2ef48085050886acf353beff7305ebd1ae69500e27c67f64b"
|
||||
|
||||
[[package]]
|
||||
name = "cc"
|
||||
version = "1.5.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f360145194ee8e21db5ee7f3fcd4fe52210864c75c985dae33218202c8bbe040"
|
||||
dependencies = [
|
||||
"find-msvc-tools",
|
||||
"jobserver",
|
||||
"libc",
|
||||
"shlex",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "cfg-if"
|
||||
version = "1.0.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "4e7648175b45a9a48536d676f68d918270699102aa8dab5496df06904c914600"
|
||||
|
||||
[[package]]
|
||||
name = "clawhdf5-format"
|
||||
version = "2.7.0"
|
||||
dependencies = [
|
||||
"byteorder",
|
||||
"flate2",
|
||||
"libaec-sys",
|
||||
"lz4_flex",
|
||||
"pco",
|
||||
"portable-atomic",
|
||||
"sha2",
|
||||
"zstd",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "conformance-probe"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"clawhdf5-format",
|
||||
"serde_json",
|
||||
"sha2",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "cpufeatures"
|
||||
version = "0.2.17"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "59ed5838eebb26a2bb2e58f6d5b5316989ae9d08bab10e0e6d103e656d1b0280"
|
||||
dependencies = [
|
||||
"libc",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "crc32fast"
|
||||
version = "1.5.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "01a7799fd6b852db0e61728dde9a204c423b44d689dbd432522543614b490e78"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "crunchy"
|
||||
version = "0.2.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "460fbee9c2c2f33933d720630a6a0bac33ba7053db5344fac858d4b8952d77d5"
|
||||
|
||||
[[package]]
|
||||
name = "crypto-common"
|
||||
version = "0.1.7"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "78c8292055d1c1df0cce5d180393dc8cce0abec0a7102adb6c7b1eef6016d60a"
|
||||
dependencies = [
|
||||
"generic-array",
|
||||
"typenum",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "digest"
|
||||
version = "0.10.7"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292"
|
||||
dependencies = [
|
||||
"block-buffer",
|
||||
"crypto-common",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "dtype_dispatch"
|
||||
version = "0.2.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ab23e69df104e2fd85ee63a533a22d2132ef5975dc6b36f9f3e5a7305e4a8ed7"
|
||||
|
||||
[[package]]
|
||||
name = "find-msvc-tools"
|
||||
version = "0.1.14"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "aedcfb3409746eddb02b9e19ebda1c3394f759a152e48ee875a0844d1b955484"
|
||||
|
||||
[[package]]
|
||||
name = "flate2"
|
||||
version = "1.1.10"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "6e634e2e0ebac1ee034020da1ca582e17ffe4e0f5e985823721e168928136dcb"
|
||||
dependencies = [
|
||||
"crc32fast",
|
||||
"miniz_oxide",
|
||||
"zlib-rs",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "generic-array"
|
||||
version = "0.14.7"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "85649ca51fd72272d7821adaf274ad91c288277713d9c18820d8499a7ff69e9a"
|
||||
dependencies = [
|
||||
"typenum",
|
||||
"version_check",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "getrandom"
|
||||
version = "0.4.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "300e883d756b2e4ec94e02791f39b04b522276138852cfc41d9fb7e904106099"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"libc",
|
||||
"r-efi",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "half"
|
||||
version = "2.7.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "6ea2d84b969582b4b1864a92dc5d27cd2b77b622a8d79306834f1be5ba20d84b"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"crunchy",
|
||||
"zerocopy",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "itoa"
|
||||
version = "1.0.18"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682"
|
||||
|
||||
[[package]]
|
||||
name = "jobserver"
|
||||
version = "0.1.35"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "1c00acbd29eabad4a2392fa0e921c874934dbbf4194312ad20f04a0ed67a3cb3"
|
||||
dependencies = [
|
||||
"getrandom",
|
||||
"libc",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "libaec-sys"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"pkg-config",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "libc"
|
||||
version = "0.2.189"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2"
|
||||
|
||||
[[package]]
|
||||
name = "lz4_flex"
|
||||
version = "0.11.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "373f5eceeeab7925e0c1098212f2fbc4d416adec9d35051a6ab251e824c1854a"
|
||||
dependencies = [
|
||||
"twox-hash",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "memchr"
|
||||
version = "2.8.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98"
|
||||
|
||||
[[package]]
|
||||
name = "miniz_oxide"
|
||||
version = "0.9.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b63fbc4a50860e98e7b2aa7804ded1db5cbc3aff9193adaff57a6931bf7c4b4c"
|
||||
dependencies = [
|
||||
"adler2",
|
||||
"simd-adler32",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "pco"
|
||||
version = "1.0.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "386342cad4c6e97f081568e5d910ea7d871314c843aa8fc564f2a6b64cab9456"
|
||||
dependencies = [
|
||||
"better_io",
|
||||
"dtype_dispatch",
|
||||
"half",
|
||||
"rand_xoshiro",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "pkg-config"
|
||||
version = "0.3.34"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f6b464fbc74e149a392436b17d523f769e057cb6877f6a5c4618bc6f11800548"
|
||||
|
||||
[[package]]
|
||||
name = "portable-atomic"
|
||||
version = "1.15.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "05c8b63e8d9609db387f0324918f81d68fe27748f084ef092fb35954d0539a85"
|
||||
|
||||
[[package]]
|
||||
name = "proc-macro2"
|
||||
version = "1.0.107"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9"
|
||||
dependencies = [
|
||||
"unicode-ident",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "quote"
|
||||
version = "1.0.47"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "r-efi"
|
||||
version = "6.0.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f8dcc9c7d52a811697d2151c701e0d08956f92b0e24136cf4cf27b57a6a0d9bf"
|
||||
|
||||
[[package]]
|
||||
name = "rand_core"
|
||||
version = "0.6.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ec0be4795e2f6a28069bec0b5ff3e2ac9bafc99e6a9a7dc3547996c5c816922c"
|
||||
|
||||
[[package]]
|
||||
name = "rand_xoshiro"
|
||||
version = "0.6.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "6f97cdb2a36ed4183de61b2f824cc45c9f1037f28afe0a322e9fff4c108b5aaa"
|
||||
dependencies = [
|
||||
"rand_core",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "serde"
|
||||
version = "1.0.229"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba"
|
||||
dependencies = [
|
||||
"serde_core",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "serde_core"
|
||||
version = "1.0.229"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48"
|
||||
dependencies = [
|
||||
"serde_derive",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "serde_derive"
|
||||
version = "1.0.229"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"syn 3.0.6",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "serde_json"
|
||||
version = "1.0.151"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14"
|
||||
dependencies = [
|
||||
"itoa",
|
||||
"memchr",
|
||||
"serde",
|
||||
"serde_core",
|
||||
"zmij",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "sha2"
|
||||
version = "0.10.9"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "a7507d819769d01a365ab707794a4084392c824f54a7a6a7862f8c3d0892b283"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"cpufeatures",
|
||||
"digest",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "shlex"
|
||||
version = "2.0.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba"
|
||||
|
||||
[[package]]
|
||||
name = "simd-adler32"
|
||||
version = "0.3.10"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "3a219298ac11a56ea9a6d2120044824d6f01aeb034955e7af7bc16858527deea"
|
||||
|
||||
[[package]]
|
||||
name = "syn"
|
||||
version = "2.0.119"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "872831b642d1a07999a962a351ed35b955ea2cfc8f3862091e2a240a84f17297"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"unicode-ident",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "syn"
|
||||
version = "3.0.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "8593e8e72159ed2257d083c7a454a85cbf854f37a0966d8d483aff8c8a3ebcee"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"unicode-ident",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "twox-hash"
|
||||
version = "2.1.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "5283634e518fe9e82c7b20520bb4bc209009fd16c82077c802f8111ecbb0117a"
|
||||
|
||||
[[package]]
|
||||
name = "typenum"
|
||||
version = "1.20.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b6f5e870be6c3b371b77fe0ee0bafb859fa4964b4404c27de1d380043c4dda20"
|
||||
|
||||
[[package]]
|
||||
name = "unicode-ident"
|
||||
version = "1.0.26"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d245f478577f809a851594d02313b640fb437e0bb33866753cff937863096954"
|
||||
|
||||
[[package]]
|
||||
name = "version_check"
|
||||
version = "0.9.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a"
|
||||
|
||||
[[package]]
|
||||
name = "zerocopy"
|
||||
version = "0.8.59"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "6df92bf3d9227be3d53173901ddbffac2babc27ae50f397776ffd6dc33f800cb"
|
||||
dependencies = [
|
||||
"zerocopy-derive",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "zerocopy-derive"
|
||||
version = "0.8.59"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ac4f328cf2f05d084e496c3e9c3f33ed0a183656a16e1fcec4d464d8373aec82"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"syn 2.0.119",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "zlib-rs"
|
||||
version = "0.6.8"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b268e58e7c693d7c271f93ffc4ba3b380412554231c85bf61ca7af91042a4112"
|
||||
|
||||
[[package]]
|
||||
name = "zmij"
|
||||
version = "1.0.23"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b"
|
||||
|
||||
[[package]]
|
||||
name = "zstd"
|
||||
version = "0.13.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e91ee311a569c327171651566e07972200e76fcfe2242a4fa446149a3881c08a"
|
||||
dependencies = [
|
||||
"zstd-safe",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "zstd-safe"
|
||||
version = "7.3.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "64d80649ab6db9d9f6f9c80a40becd948eda4714a0a5ac8c4d157a32231c7882"
|
||||
dependencies = [
|
||||
"zstd-sys",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "zstd-sys"
|
||||
version = "2.1.0+zstd.1.5.7"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0ef0a8027ec3ee71300ab3bcbcd0393f434aa72b91ca6d635a39941deae8eea0"
|
||||
dependencies = [
|
||||
"cc",
|
||||
"pkg-config",
|
||||
]
|
||||
@@ -0,0 +1,25 @@
|
||||
[package]
|
||||
name = "conformance-probe"
|
||||
version = "0.1.0"
|
||||
edition = "2024"
|
||||
rust-version = "1.92"
|
||||
publish = false
|
||||
description = "Walks an HDF5 file with clawhdf5-format and prints a canonical JSON description (see conformance/README.md)"
|
||||
|
||||
# Deliberately outside the main workspace: `cargo test --workspace` never
|
||||
# builds it, and it links the optional C codecs (zstd, libaec) that the core
|
||||
# crates' default build must not.
|
||||
[workspace]
|
||||
|
||||
[dependencies]
|
||||
clawhdf5-format = { path = "../../crates/clawhdf5-format", features = ["lz4", "zstd", "szip", "pcodec"] }
|
||||
serde_json = "1"
|
||||
sha2 = "0.10"
|
||||
|
||||
[profile.release]
|
||||
# Keep panics catchable (the probe records them per object) and turn integer
|
||||
# overflow into a reported panic instead of silent wraparound.
|
||||
debug = 1
|
||||
overflow-checks = true
|
||||
debug-assertions = true
|
||||
panic = "unwind"
|
||||
@@ -0,0 +1,898 @@
|
||||
//! Conformance probe: walks an HDF5 file with clawhdf5-format (the same calls
|
||||
//! the `clawhdf5` facade makes) and prints a canonical JSON description:
|
||||
//! every hard-linked object (sorted-name DFS, deduplicated by header address),
|
||||
//! and for each dataset / attribute its shape plus the SHA-256 of its values
|
||||
//! in a canonical encoding shared with `ref.py`.
|
||||
//!
|
||||
//! Canonical value encoding (per element, concatenated, row-major):
|
||||
//! int / float / bitfield / enum / time : element bytes, little-endian
|
||||
//! non-IEEE-layout float (e.g. N-Bit) : the IEEE float of the same size it converts to
|
||||
//! int with bit offset / short precision: the full-width integer it converts to
|
||||
//! opaque : raw bytes
|
||||
//! compound : members in declaration order (padding dropped)
|
||||
//! array : base elements row-major
|
||||
//! string (fixed or VL) : b'S' + u32le len + bytes (cut at first NUL, trailing spaces stripped)
|
||||
//! VL sequence : b'V' + u32le count + base elements
|
||||
//! reference : b'R' (payload not compared)
|
||||
//!
|
||||
//! Every object is processed inside catch_unwind; a caught panic is recorded
|
||||
//! with its message, location and the clawhdf5 frames of its backtrace.
|
||||
|
||||
use std::cell::RefCell;
|
||||
use std::collections::{HashMap, HashSet};
|
||||
use std::panic::{self, AssertUnwindSafe};
|
||||
use std::rc::Rc;
|
||||
|
||||
use clawhdf5_format::attribute::extract_attributes_full;
|
||||
use clawhdf5_format::data_layout::DataLayout;
|
||||
use clawhdf5_format::data_read;
|
||||
use clawhdf5_format::dataspace::{Dataspace, DataspaceType};
|
||||
use clawhdf5_format::datatype::{Datatype, DatatypeByteOrder};
|
||||
use clawhdf5_format::filter_pipeline::FilterPipeline;
|
||||
use clawhdf5_format::global_heap::GlobalHeapCollection;
|
||||
use clawhdf5_format::group_v1::{self, GroupEntry};
|
||||
use clawhdf5_format::group_v2;
|
||||
use clawhdf5_format::message_type::MessageType;
|
||||
use clawhdf5_format::object_header::ObjectHeader;
|
||||
use clawhdf5_format::signature;
|
||||
use clawhdf5_format::superblock::Superblock;
|
||||
use clawhdf5_format::symbol_table::SymbolTableMessage;
|
||||
use serde_json::{Map, Value, json};
|
||||
use sha2::{Digest, Sha256};
|
||||
|
||||
const MAX_BYTES: u64 = 200 * 1024 * 1024;
|
||||
const MAX_OBJECTS: usize = 200_000;
|
||||
|
||||
thread_local! {
|
||||
static LAST_PANIC: RefCell<Option<String>> = const { RefCell::new(None) };
|
||||
}
|
||||
|
||||
fn install_hook() {
|
||||
panic::set_hook(Box::new(|info| {
|
||||
let msg = if let Some(s) = info.payload().downcast_ref::<&str>() {
|
||||
s.to_string()
|
||||
} else if let Some(s) = info.payload().downcast_ref::<String>() {
|
||||
s.clone()
|
||||
} else {
|
||||
"<non-string panic>".into()
|
||||
};
|
||||
let loc = info
|
||||
.location()
|
||||
.map(|l| format!("{}:{}", l.file(), l.line()))
|
||||
.unwrap_or_default();
|
||||
let bt = std::backtrace::Backtrace::force_capture().to_string();
|
||||
// keep only frames from clawhdf5 code
|
||||
let mut frames = Vec::new();
|
||||
let lines: Vec<&str> = bt.lines().collect();
|
||||
for (i, l) in lines.iter().enumerate() {
|
||||
let t = l.trim();
|
||||
if t.contains("clawhdf5_format::") || t.contains("conformance_probe::") {
|
||||
let at = lines
|
||||
.get(i + 1)
|
||||
.map(|n| n.trim())
|
||||
.filter(|n| n.starts_with("at "))
|
||||
.map(|n| {
|
||||
let n = n.trim_start_matches("at ");
|
||||
match n.find("/crates/") {
|
||||
Some(p) => n[p + 1..].to_string(),
|
||||
None => n.to_string(),
|
||||
}
|
||||
})
|
||||
.unwrap_or_default();
|
||||
let name = t.split_once(": ").map(|x| x.1).unwrap_or(t);
|
||||
frames.push(format!("{name} ({at})"));
|
||||
if frames.len() >= 12 {
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
let full = format!("PANIC: {msg} @ {loc}\n {}", frames.join("\n "));
|
||||
eprintln!("{full}");
|
||||
LAST_PANIC.with(|p| *p.borrow_mut() = Some(full));
|
||||
}));
|
||||
}
|
||||
|
||||
/// Run `f`, turning a panic into Err("PANIC: ...").
|
||||
fn guarded<T>(f: impl FnOnce() -> Result<T, String>) -> Result<T, String> {
|
||||
match panic::catch_unwind(AssertUnwindSafe(f)) {
|
||||
Ok(r) => r,
|
||||
Err(_) => Err(LAST_PANIC
|
||||
.with(|p| p.borrow_mut().take())
|
||||
.unwrap_or_else(|| "PANIC: <unknown>".into())),
|
||||
}
|
||||
}
|
||||
|
||||
fn e<E: std::fmt::Debug>(x: E) -> String {
|
||||
format!("{x:?}")
|
||||
}
|
||||
|
||||
struct Ctx<'a> {
|
||||
data: &'a [u8],
|
||||
os: u8,
|
||||
ls: u8,
|
||||
base_dir: std::path::PathBuf,
|
||||
heaps: RefCell<HashMap<u64, Result<Rc<GlobalHeapCollection>, String>>>,
|
||||
}
|
||||
|
||||
impl<'a> Ctx<'a> {
|
||||
fn header(&self, addr: u64) -> Result<ObjectHeader, String> {
|
||||
ObjectHeader::parse(self.data, addr as usize, self.os, self.ls).map_err(e)
|
||||
}
|
||||
|
||||
fn payload(&self, h: &ObjectHeader, t: MessageType) -> Result<Option<Vec<u8>>, String> {
|
||||
match h.messages.iter().find(|m| m.msg_type == t) {
|
||||
None => Ok(None),
|
||||
Some(m) => {
|
||||
clawhdf5_format::shared_message::message_data(self.data, m, self.os, self.ls)
|
||||
.map(|c| Some(c.into_owned()))
|
||||
.map_err(e)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn heap_obj(&self, addr: u64, idx: u32) -> Result<Vec<u8>, String> {
|
||||
let coll = {
|
||||
let mut cache = self.heaps.borrow_mut();
|
||||
cache
|
||||
.entry(addr)
|
||||
.or_insert_with(|| {
|
||||
GlobalHeapCollection::parse(self.data, addr as usize, self.ls)
|
||||
.map(Rc::new)
|
||||
.map_err(e)
|
||||
})
|
||||
.clone()?
|
||||
};
|
||||
coll.get_object(idx as u16)
|
||||
.map(|o| o.data.clone())
|
||||
.ok_or_else(|| {
|
||||
format!("GlobalHeapObjectNotFound {{ collection_address: {addr}, index: {idx} }}")
|
||||
})
|
||||
}
|
||||
|
||||
fn read_offset(&self, b: &[u8]) -> u64 {
|
||||
let mut v = 0u64;
|
||||
for (i, x) in b.iter().take(self.os as usize).enumerate() {
|
||||
v |= (*x as u64) << (8 * i);
|
||||
}
|
||||
v
|
||||
}
|
||||
|
||||
fn canon(&self, dt: &Datatype, b: &[u8], out: &mut Vec<u8>) -> Result<(), String> {
|
||||
let size = dt.type_size() as usize;
|
||||
if b.len() < size {
|
||||
return Err(format!(
|
||||
"canon: element slice {} < type size {size}",
|
||||
b.len()
|
||||
));
|
||||
}
|
||||
match dt {
|
||||
Datatype::FloatingPoint { .. } if !ieee_layout(dt) => {
|
||||
canon_custom_float(dt, &b[..size], out)?
|
||||
}
|
||||
Datatype::FixedPoint { .. } if partial_int(dt) => {
|
||||
canon_partial_int(dt, &b[..size], out)?
|
||||
}
|
||||
Datatype::FixedPoint { byte_order, .. }
|
||||
| Datatype::BitField { byte_order, .. }
|
||||
| Datatype::FloatingPoint { byte_order, .. } => match byte_order {
|
||||
DatatypeByteOrder::LittleEndian => out.extend_from_slice(&b[..size]),
|
||||
DatatypeByteOrder::BigEndian => out.extend(b[..size].iter().rev()),
|
||||
DatatypeByteOrder::Vax => return Err("canon: VAX byte order".into()),
|
||||
},
|
||||
Datatype::Time { .. } | Datatype::Opaque { .. } => out.extend_from_slice(&b[..size]),
|
||||
Datatype::String { .. } => canon_str(&b[..size], out),
|
||||
Datatype::Compound { members, .. } => {
|
||||
for m in members {
|
||||
let off = m.byte_offset as usize;
|
||||
let ms = m.datatype.type_size() as usize;
|
||||
if off.checked_add(ms).is_none_or(|end| end > size) {
|
||||
return Err(format!("canon: member {} out of bounds", m.name));
|
||||
}
|
||||
self.canon(&m.datatype, &b[off..off + ms], out)?;
|
||||
}
|
||||
}
|
||||
Datatype::Reference { .. } => out.push(b'R'),
|
||||
Datatype::Enumeration { base_type, .. } => self.canon(base_type, b, out)?,
|
||||
Datatype::Array {
|
||||
base_type,
|
||||
dimensions,
|
||||
} => {
|
||||
let n: usize = dimensions.iter().map(|d| *d as usize).product();
|
||||
let bs = base_type.type_size() as usize;
|
||||
for i in 0..n {
|
||||
self.canon(base_type, &b[i * bs..], out)?;
|
||||
}
|
||||
}
|
||||
Datatype::VariableLength {
|
||||
is_string,
|
||||
base_type,
|
||||
..
|
||||
} => {
|
||||
let len = u32::from_le_bytes([b[0], b[1], b[2], b[3]]) as usize;
|
||||
let addr = self.read_offset(&b[4..]);
|
||||
let idx_off = 4 + self.os as usize;
|
||||
let idx = u32::from_le_bytes([
|
||||
b[idx_off],
|
||||
b[idx_off + 1],
|
||||
b[idx_off + 2],
|
||||
b[idx_off + 3],
|
||||
]);
|
||||
let obj = if len == 0 || addr == 0 || addr == u64::MAX >> (64 - 8 * self.os as u32)
|
||||
{
|
||||
Vec::new()
|
||||
} else {
|
||||
self.heap_obj(addr, idx)?
|
||||
};
|
||||
if *is_string {
|
||||
let l = len.min(obj.len());
|
||||
canon_str(&obj[..l], out);
|
||||
} else {
|
||||
let bs = base_type.type_size() as usize;
|
||||
if bs == 0 {
|
||||
return Err("canon: VL base size 0".into());
|
||||
}
|
||||
let need = len.checked_mul(bs).ok_or("canon: VL overflow")?;
|
||||
if len > 0 && obj.len() < need {
|
||||
return Err(format!("canon: VL object {} < {need}", obj.len()));
|
||||
}
|
||||
out.push(b'V');
|
||||
out.extend_from_slice(&(len as u32).to_le_bytes());
|
||||
for i in 0..len {
|
||||
self.canon(base_type, &obj[i * bs..], out)?;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Returns (shape json, n_elements)
|
||||
fn shape(ds: &Dataspace) -> (Value, u64) {
|
||||
match ds.space_type {
|
||||
DataspaceType::Null => (Value::String("null".into()), 0),
|
||||
DataspaceType::Scalar => (json!([]), 1),
|
||||
DataspaceType::Simple => {
|
||||
let n = ds.dimensions.iter().fold(1u64, |a, d| a.saturating_mul(*d));
|
||||
(json!(ds.dimensions), n)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn hash_values(
|
||||
&self,
|
||||
dt: &Datatype,
|
||||
raw: &[u8],
|
||||
n: u64,
|
||||
rec: &mut Map<String, Value>,
|
||||
) -> Result<(), String> {
|
||||
let size = dt.type_size() as usize;
|
||||
let need = (n as usize).checked_mul(size).ok_or("n*size overflow")?;
|
||||
if raw.len() != need {
|
||||
return Err(format!(
|
||||
"raw length {} != n_elements {n} * type_size {size}",
|
||||
raw.len()
|
||||
));
|
||||
}
|
||||
let mut canon = Vec::with_capacity(need);
|
||||
for i in 0..n as usize {
|
||||
self.canon(dt, &raw[i * size..(i + 1) * size], &mut canon)?;
|
||||
}
|
||||
let h = Sha256::digest(&canon);
|
||||
rec.insert("hash".into(), Value::String(hex(&h)));
|
||||
rec.insert(
|
||||
"head".into(),
|
||||
Value::String(hex(&canon[..canon.len().min(48)])),
|
||||
);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// VDS source files resolve next to the virtual file; like the library,
|
||||
/// refuse absolute paths and `..`.
|
||||
fn vds_resolver(
|
||||
&self,
|
||||
) -> impl Fn(&str) -> Result<Option<Vec<u8>>, clawhdf5_format::error::FormatError> + use<> {
|
||||
let base = self.base_dir.clone();
|
||||
move |name: &str| {
|
||||
use clawhdf5_format::error::FormatError;
|
||||
let p = std::path::Path::new(name);
|
||||
if p.is_absolute()
|
||||
|| p.components()
|
||||
.any(|c| matches!(c, std::path::Component::ParentDir))
|
||||
{
|
||||
return Err(FormatError::ChunkedReadError(format!("refused {name}")));
|
||||
}
|
||||
match std::fs::read(base.join(p)) {
|
||||
Ok(b) => Ok(Some(b)),
|
||||
Err(err) if err.kind() == std::io::ErrorKind::NotFound => Ok(None),
|
||||
Err(err) => Err(FormatError::ChunkedReadError(err.to_string())),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn read_dataset(&self, h: &ObjectHeader, rec: &mut Map<String, Value>) -> Result<(), String> {
|
||||
let dtb = self
|
||||
.payload(h, MessageType::Datatype)?
|
||||
.ok_or("MissingMessage(Datatype)")?;
|
||||
let (dt, _) = Datatype::parse(&dtb).map_err(e)?;
|
||||
rec.insert("dtype".into(), Value::String(dtype_str(&dt)));
|
||||
let dsb = self
|
||||
.payload(h, MessageType::Dataspace)?
|
||||
.ok_or("MissingMessage(Dataspace)")?;
|
||||
let mut ds = Dataspace::parse(&dsb, self.ls).map_err(e)?;
|
||||
// A virtual dataset's extent can come from its sources (unlimited /
|
||||
// printf mappings), as h5py reports it, rather than the stored one.
|
||||
if let Some(lm) = h
|
||||
.messages
|
||||
.iter()
|
||||
.find(|m| m.msg_type == MessageType::DataLayout)
|
||||
&& let Ok(dl @ DataLayout::Virtual { .. }) =
|
||||
DataLayout::parse(&lm.data, self.os, self.ls)
|
||||
{
|
||||
let resolver = self.vds_resolver();
|
||||
ds.dimensions = clawhdf5_format::vds::virtual_dataset_extent(
|
||||
self.data,
|
||||
&dl,
|
||||
&ds,
|
||||
self.os,
|
||||
self.ls,
|
||||
Some(&resolver),
|
||||
)
|
||||
.map_err(e)?;
|
||||
}
|
||||
let (shape, n) = Self::shape(&ds);
|
||||
rec.insert("shape".into(), shape);
|
||||
if n.saturating_mul(dt.type_size() as u64) > MAX_BYTES {
|
||||
rec.insert("skipped".into(), Value::String("too large".into()));
|
||||
return Ok(());
|
||||
}
|
||||
let lm = h
|
||||
.messages
|
||||
.iter()
|
||||
.find(|m| m.msg_type == MessageType::DataLayout)
|
||||
.ok_or("MissingMessage(DataLayout)")?;
|
||||
let dl = DataLayout::parse(&lm.data, self.os, self.ls).map_err(e)?;
|
||||
rec.insert(
|
||||
"layout".into(),
|
||||
Value::String(
|
||||
match &dl {
|
||||
DataLayout::Compact { .. } => "compact",
|
||||
DataLayout::Contiguous { .. } => "contiguous",
|
||||
DataLayout::Chunked { .. } => "chunked",
|
||||
DataLayout::Virtual { .. } => "virtual",
|
||||
}
|
||||
.into(),
|
||||
),
|
||||
);
|
||||
let pipeline = match self.payload(h, MessageType::FilterPipeline)? {
|
||||
Some(p) => Some(FilterPipeline::parse(&p).map_err(e)?),
|
||||
None => None,
|
||||
};
|
||||
if let Some(p) = &pipeline {
|
||||
rec.insert(
|
||||
"filters".into(),
|
||||
json!(p.filters.iter().map(|f| f.filter_id).collect::<Vec<_>>()),
|
||||
);
|
||||
}
|
||||
let raw = if matches!(dl, DataLayout::Virtual { .. }) {
|
||||
let resolver = self.vds_resolver();
|
||||
let fill = clawhdf5_format::fill_value::dataset_fill_value_in(
|
||||
self.data,
|
||||
&h.messages,
|
||||
self.os,
|
||||
self.ls,
|
||||
)
|
||||
.map_err(e)?;
|
||||
clawhdf5_format::vds::read_virtual_dataset(
|
||||
self.data,
|
||||
&dl,
|
||||
&ds,
|
||||
&dt,
|
||||
fill.as_deref(),
|
||||
self.os,
|
||||
self.ls,
|
||||
Some(&resolver),
|
||||
)
|
||||
.map_err(e)?
|
||||
.data
|
||||
} else {
|
||||
let cache = clawhdf5_format::chunk_cache::ChunkCache::new();
|
||||
clawhdf5_format::fill_value::read_full_with_fill::<clawhdf5_format::error::FormatError>(
|
||||
&h.messages,
|
||||
self.data,
|
||||
&dl,
|
||||
&ds,
|
||||
dt.type_size() as usize,
|
||||
self.os,
|
||||
self.ls,
|
||||
|| {
|
||||
data_read::read_raw_data_cached(
|
||||
self.data,
|
||||
&dl,
|
||||
&ds,
|
||||
&dt,
|
||||
pipeline.as_ref(),
|
||||
self.os,
|
||||
self.ls,
|
||||
&cache,
|
||||
)
|
||||
},
|
||||
)
|
||||
.map_err(e)?
|
||||
};
|
||||
self.hash_values(&dt, &raw, n, rec)
|
||||
}
|
||||
|
||||
fn attrs(&self, h: &ObjectHeader) -> Result<Map<String, Value>, String> {
|
||||
let msgs = extract_attributes_full(self.data, h, self.os, self.ls).map_err(e)?;
|
||||
let mut out = Map::new();
|
||||
for a in &msgs {
|
||||
let r = guarded(|| {
|
||||
let mut rec = Map::new();
|
||||
rec.insert("dtype".into(), Value::String(dtype_str(&a.datatype)));
|
||||
let (shape, n) = Self::shape(&a.dataspace);
|
||||
rec.insert("shape".into(), shape);
|
||||
self.hash_values(&a.datatype, &a.raw_data, n, &mut rec)?;
|
||||
Ok(rec)
|
||||
});
|
||||
let v = match r {
|
||||
Ok(rec) => Value::Object(rec),
|
||||
Err(msg) => json!({ "error": msg }),
|
||||
};
|
||||
out.insert(a.name.clone(), v);
|
||||
}
|
||||
Ok(out)
|
||||
}
|
||||
|
||||
fn entries(&self, h: &ObjectHeader) -> Result<Vec<GroupEntry>, String> {
|
||||
let v1 = h
|
||||
.messages
|
||||
.iter()
|
||||
.find(|m| m.msg_type == MessageType::SymbolTable);
|
||||
if let Some(m) = v1 {
|
||||
let stm = SymbolTableMessage::parse(&m.data, self.os).map_err(e)?;
|
||||
group_v1::resolve_v1_group_entries(self.data, &stm, self.os, self.ls).map_err(e)
|
||||
} else if h
|
||||
.messages
|
||||
.iter()
|
||||
.any(|m| m.msg_type == MessageType::LinkInfo || m.msg_type == MessageType::Link)
|
||||
{
|
||||
group_v2::resolve_v2_group_entries(self.data, h, self.os, self.ls).map_err(e)
|
||||
} else {
|
||||
Ok(Vec::new())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Element bytes as an unsigned integer (at most 16 bytes), honouring byte order.
|
||||
fn element_bits(b: &[u8], byte_order: &DatatypeByteOrder) -> Result<u128, String> {
|
||||
if b.len() > 16 {
|
||||
return Err(format!("canon: {}-byte numeric element", b.len()));
|
||||
}
|
||||
let mut v = 0u128;
|
||||
match byte_order {
|
||||
DatatypeByteOrder::LittleEndian => {
|
||||
for (i, x) in b.iter().enumerate() {
|
||||
v |= u128::from(*x) << (8 * i);
|
||||
}
|
||||
}
|
||||
DatatypeByteOrder::BigEndian => {
|
||||
for x in b {
|
||||
v = (v << 8) | u128::from(*x);
|
||||
}
|
||||
}
|
||||
DatatypeByteOrder::Vax => return Err("canon: VAX byte order".into()),
|
||||
}
|
||||
Ok(v)
|
||||
}
|
||||
|
||||
fn field(v: u128, pos: u32, len: u32) -> u128 {
|
||||
if len == 0 || pos >= 128 {
|
||||
return 0;
|
||||
}
|
||||
let v = v >> pos;
|
||||
if len >= 128 {
|
||||
v
|
||||
} else {
|
||||
v & ((1u128 << len) - 1)
|
||||
}
|
||||
}
|
||||
|
||||
/// True when a float's bit fields are exactly IEEE 754 binary16/32/64 for its
|
||||
/// size. h5py hands back such a type's bytes untouched; any other layout (an
|
||||
/// N-Bit `H5Tset_precision` float, say) is *converted* by libhdf5 into the
|
||||
/// numpy float of the same size, so comparing raw bytes would be meaningless.
|
||||
fn ieee_layout(dt: &Datatype) -> bool {
|
||||
let Datatype::FloatingPoint {
|
||||
size,
|
||||
bit_offset,
|
||||
bit_precision,
|
||||
exponent_location,
|
||||
exponent_size,
|
||||
mantissa_location,
|
||||
mantissa_size,
|
||||
exponent_bias,
|
||||
..
|
||||
} = dt
|
||||
else {
|
||||
return true;
|
||||
};
|
||||
let std = match size {
|
||||
2 => (16, 10, 5, 10, 15),
|
||||
4 => (32, 23, 8, 23, 127),
|
||||
8 => (64, 52, 11, 52, 1023),
|
||||
_ => return true, // no same-size numpy float to convert to: compare raw
|
||||
};
|
||||
*bit_offset == 0
|
||||
&& (
|
||||
*bit_precision,
|
||||
*exponent_location,
|
||||
*exponent_size,
|
||||
*mantissa_size,
|
||||
*exponent_bias,
|
||||
) == (std.0, std.1, std.2, std.3, std.4)
|
||||
&& *mantissa_location == 0
|
||||
}
|
||||
|
||||
/// Canonicalise a non-IEEE-layout float the way libhdf5's float->float
|
||||
/// conversion presents it to h5py: as the IEEE float of the same size.
|
||||
/// Assumes the implied-leading-one normalisation and the sign bit at the top
|
||||
/// of the precision (what `H5Tset_precision` produces; the parser does not
|
||||
/// keep either field).
|
||||
fn canon_custom_float(dt: &Datatype, b: &[u8], out: &mut Vec<u8>) -> Result<(), String> {
|
||||
let Datatype::FloatingPoint {
|
||||
size,
|
||||
byte_order,
|
||||
bit_offset,
|
||||
bit_precision,
|
||||
exponent_location,
|
||||
exponent_size,
|
||||
mantissa_location,
|
||||
mantissa_size,
|
||||
exponent_bias,
|
||||
} = dt
|
||||
else {
|
||||
unreachable!()
|
||||
};
|
||||
let (esize, msize) = (u32::from(*exponent_size), u32::from(*mantissa_size));
|
||||
if esize == 0 || esize > 30 || msize > 64 {
|
||||
return Err(format!("canon: unsupported float layout e{esize} m{msize}"));
|
||||
}
|
||||
let v = element_bits(b, byte_order)?;
|
||||
let sign_pos = (u32::from(*bit_offset) + u32::from(*bit_precision)).saturating_sub(1);
|
||||
let neg = field(v, sign_pos, 1) == 1;
|
||||
let e = field(v, u32::from(*exponent_location), esize) as i64;
|
||||
let m = field(v, u32::from(*mantissa_location), msize);
|
||||
let emax = (1i64 << esize) - 1;
|
||||
let bias = i64::from(*exponent_bias);
|
||||
let mag = if e == emax {
|
||||
if m == 0 { f64::INFINITY } else { f64::NAN }
|
||||
} else if e == 0 {
|
||||
(m as f64) * 2f64.powi((1 - bias - msize as i64) as i32)
|
||||
} else {
|
||||
((1u128 << msize) as f64 + m as f64) * 2f64.powi((e - bias - msize as i64) as i32)
|
||||
};
|
||||
let x = if neg { -mag } else { mag };
|
||||
match size {
|
||||
2 => out
|
||||
.extend_from_slice(&clawhdf5_format::float16::f32_to_f16_bits(x as f32).to_le_bytes()),
|
||||
4 => out.extend_from_slice(&(x as f32).to_le_bytes()),
|
||||
8 => out.extend_from_slice(&x.to_le_bytes()),
|
||||
_ => unreachable!("ieee_layout keeps other sizes raw"),
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Integers stored with a bit offset or reduced precision (N-Bit): libhdf5
|
||||
/// converts them to the full-width integer of the same size, shifting the
|
||||
/// value down and sign-extending from the top precision bit.
|
||||
fn canon_partial_int(dt: &Datatype, b: &[u8], out: &mut Vec<u8>) -> Result<(), String> {
|
||||
let Datatype::FixedPoint {
|
||||
size,
|
||||
byte_order,
|
||||
signed,
|
||||
bit_offset,
|
||||
bit_precision,
|
||||
} = dt
|
||||
else {
|
||||
unreachable!()
|
||||
};
|
||||
let prec = u32::from(*bit_precision);
|
||||
let v = element_bits(b, byte_order)?;
|
||||
let mut x = field(v, u32::from(*bit_offset), prec);
|
||||
if *signed && prec > 0 && prec < 128 && field(x, prec - 1, 1) == 1 {
|
||||
x |= !0u128 << prec;
|
||||
}
|
||||
out.extend_from_slice(&x.to_le_bytes()[..*size as usize]);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn partial_int(dt: &Datatype) -> bool {
|
||||
matches!(dt, Datatype::FixedPoint { size, bit_offset, bit_precision, .. }
|
||||
if *bit_offset != 0 || u32::from(*bit_precision) != size * 8)
|
||||
}
|
||||
|
||||
fn canon_str(b: &[u8], out: &mut Vec<u8>) {
|
||||
let cut = b.iter().position(|&c| c == 0).unwrap_or(b.len());
|
||||
let mut s = &b[..cut];
|
||||
while let [rest @ .., b' '] = s {
|
||||
s = rest;
|
||||
}
|
||||
out.push(b'S');
|
||||
out.extend_from_slice(&(s.len() as u32).to_le_bytes());
|
||||
out.extend_from_slice(s);
|
||||
}
|
||||
|
||||
fn hex(b: &[u8]) -> String {
|
||||
b.iter().map(|x| format!("{x:02x}")).collect()
|
||||
}
|
||||
|
||||
fn dtype_str(dt: &Datatype) -> String {
|
||||
match dt {
|
||||
Datatype::FixedPoint {
|
||||
size,
|
||||
signed,
|
||||
byte_order,
|
||||
..
|
||||
} => {
|
||||
format!(
|
||||
"{}{}{}",
|
||||
bo(byte_order),
|
||||
if *signed { "i" } else { "u" },
|
||||
size
|
||||
)
|
||||
}
|
||||
Datatype::FloatingPoint {
|
||||
size, byte_order, ..
|
||||
} => format!("{}f{}", bo(byte_order), size),
|
||||
Datatype::BitField {
|
||||
size, byte_order, ..
|
||||
} => format!("{}b{}", bo(byte_order), size),
|
||||
Datatype::Time { size, .. } => format!("time{size}"),
|
||||
Datatype::String { size, .. } => format!("S{size}"),
|
||||
Datatype::Opaque { size, .. } => format!("V{size}"),
|
||||
Datatype::Compound { size, members } => format!(
|
||||
"{{{}}}{size}",
|
||||
members
|
||||
.iter()
|
||||
.map(|m| format!("{}:{}", m.name, dtype_str(&m.datatype)))
|
||||
.collect::<Vec<_>>()
|
||||
.join(",")
|
||||
),
|
||||
Datatype::Reference { ref_type, .. } => format!("ref({ref_type:?})"),
|
||||
Datatype::Enumeration { base_type, .. } => format!("enum({})", dtype_str(base_type)),
|
||||
Datatype::VariableLength {
|
||||
is_string: true, ..
|
||||
} => "vlstr".into(),
|
||||
Datatype::VariableLength { base_type, .. } => format!("vlen({})", dtype_str(base_type)),
|
||||
Datatype::Array {
|
||||
base_type,
|
||||
dimensions,
|
||||
} => format!("({}){dimensions:?}", dtype_str(base_type)),
|
||||
}
|
||||
}
|
||||
|
||||
fn bo(b: &DatatypeByteOrder) -> &'static str {
|
||||
match b {
|
||||
DatatypeByteOrder::LittleEndian => "<",
|
||||
DatatypeByteOrder::BigEndian => ">",
|
||||
DatatypeByteOrder::Vax => "vax",
|
||||
}
|
||||
}
|
||||
|
||||
fn is_group(h: &ObjectHeader) -> bool {
|
||||
h.messages.iter().any(|m| {
|
||||
matches!(
|
||||
m.msg_type,
|
||||
MessageType::LinkInfo | MessageType::Link | MessageType::SymbolTable
|
||||
)
|
||||
})
|
||||
}
|
||||
|
||||
fn main() {
|
||||
install_hook();
|
||||
let path = std::env::args().nth(1).expect("usage: probe <file>");
|
||||
let mut top = Map::new();
|
||||
top.insert("file".into(), Value::String(path.clone()));
|
||||
let data = match std::fs::read(&path) {
|
||||
Ok(d) => d,
|
||||
Err(err) => {
|
||||
top.insert("open_error".into(), Value::String(format!("Io({err})")));
|
||||
println!("{}", Value::Object(top));
|
||||
return;
|
||||
}
|
||||
};
|
||||
// Every address is relative to the superblock: look at the file from
|
||||
// there on (past any user block), as libhdf5 does.
|
||||
let hdf5: &[u8] = match signature::find_signature(&data) {
|
||||
Ok(off) => &data[off..],
|
||||
Err(_) => &data,
|
||||
};
|
||||
let sb = guarded(|| Superblock::parse(hdf5, 0).map_err(e));
|
||||
let sb = match sb {
|
||||
Ok(sb) => sb,
|
||||
Err(msg) => {
|
||||
top.insert("open_error".into(), Value::String(msg));
|
||||
println!("{}", Value::Object(top));
|
||||
return;
|
||||
}
|
||||
};
|
||||
top.insert("superblock_version".into(), json!(sb.version));
|
||||
let ctx = Ctx {
|
||||
data: hdf5,
|
||||
os: sb.offset_size,
|
||||
ls: sb.length_size,
|
||||
base_dir: std::path::Path::new(&path)
|
||||
.parent()
|
||||
.map(|p| p.to_path_buf())
|
||||
.unwrap_or_default(),
|
||||
heaps: RefCell::new(HashMap::new()),
|
||||
};
|
||||
let mut objects: Vec<Value> = Vec::new();
|
||||
let mut visited = HashSet::new();
|
||||
let mut soft_v1 = 0u64;
|
||||
// explicit DFS stack: (address, path)
|
||||
let mut stack: Vec<(u64, String)> = vec![(sb.root_group_address, "/".to_string())];
|
||||
while let Some((addr, p)) = stack.pop() {
|
||||
if objects.len() >= MAX_OBJECTS {
|
||||
top.insert("truncated".into(), json!(true));
|
||||
break;
|
||||
}
|
||||
if !visited.insert(addr) {
|
||||
continue;
|
||||
}
|
||||
let mut rec = Map::new();
|
||||
rec.insert("path".into(), Value::String(p.clone()));
|
||||
let r = guarded(|| {
|
||||
let h = ctx.header(addr)?;
|
||||
Ok(h)
|
||||
});
|
||||
let h = match r {
|
||||
Ok(h) => h,
|
||||
Err(msg) => {
|
||||
rec.insert("kind".into(), Value::String("unknown".into()));
|
||||
rec.insert("error".into(), Value::String(msg));
|
||||
objects.push(Value::Object(rec));
|
||||
continue;
|
||||
}
|
||||
};
|
||||
let is_ds = h
|
||||
.messages
|
||||
.iter()
|
||||
.any(|m| m.msg_type == MessageType::DataLayout);
|
||||
let kind = if is_ds {
|
||||
"dataset"
|
||||
} else if is_group(&h) || addr == sb.root_group_address {
|
||||
"group"
|
||||
} else if h
|
||||
.messages
|
||||
.iter()
|
||||
.any(|m| m.msg_type == MessageType::Datatype)
|
||||
{
|
||||
"datatype"
|
||||
} else {
|
||||
"unknown"
|
||||
};
|
||||
rec.insert("kind".into(), Value::String(kind.into()));
|
||||
if kind == "dataset"
|
||||
&& let Err(msg) = guarded(|| ctx.read_dataset(&h, &mut rec))
|
||||
{
|
||||
rec.insert("error".into(), Value::String(msg));
|
||||
}
|
||||
if kind != "datatype" {
|
||||
match guarded(|| ctx.attrs(&h)) {
|
||||
Ok(m) => {
|
||||
rec.insert("attrs".into(), Value::Object(m));
|
||||
}
|
||||
Err(msg) => {
|
||||
rec.insert("attrs_error".into(), Value::String(msg));
|
||||
}
|
||||
}
|
||||
}
|
||||
if kind == "group" {
|
||||
match guarded(|| ctx.entries(&h)) {
|
||||
Ok(mut ents) => {
|
||||
ents.retain(|en| {
|
||||
if en.cache_type == 2 {
|
||||
soft_v1 += 1;
|
||||
false
|
||||
} else {
|
||||
true
|
||||
}
|
||||
});
|
||||
ents.sort_by(|a, b| a.name.cmp(&b.name));
|
||||
let base = if p == "/" { String::new() } else { p.clone() };
|
||||
for en in ents.into_iter().rev() {
|
||||
stack.push((en.object_header_address, format!("{base}/{}", en.name)));
|
||||
}
|
||||
}
|
||||
Err(msg) => {
|
||||
rec.insert("list_error".into(), Value::String(msg));
|
||||
}
|
||||
}
|
||||
}
|
||||
objects.push(Value::Object(rec));
|
||||
}
|
||||
if soft_v1 > 0 {
|
||||
top.insert("v1_soft_link_entries".into(), json!(soft_v1));
|
||||
}
|
||||
top.insert("objects".into(), Value::Array(objects));
|
||||
println!("{}", Value::Object(top));
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// The N-Bit float of libhdf5's `test/testfiles/le_data.h5`
|
||||
/// (`Nbit_float_data_le`): offset 7, precision 20, sign bit 26, exponent
|
||||
/// 20+6 (bias 31), mantissa 7+13.
|
||||
fn nbit_f32(byte_order: DatatypeByteOrder) -> Datatype {
|
||||
Datatype::FloatingPoint {
|
||||
size: 4,
|
||||
byte_order,
|
||||
bit_offset: 7,
|
||||
bit_precision: 20,
|
||||
exponent_location: 20,
|
||||
exponent_size: 6,
|
||||
mantissa_location: 7,
|
||||
mantissa_size: 13,
|
||||
exponent_bias: 31,
|
||||
}
|
||||
}
|
||||
|
||||
fn canon_one(dt: &Datatype, bytes: &[u8]) -> Vec<u8> {
|
||||
let mut out = Vec::new();
|
||||
canon_custom_float(dt, bytes, &mut out).unwrap();
|
||||
out
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn nbit_float_canonicalises_to_the_value_libhdf5_returns() {
|
||||
let le = nbit_f32(DatatypeByteOrder::LittleEndian);
|
||||
let be = nbit_f32(DatatypeByteOrder::BigEndian);
|
||||
assert!(!ieee_layout(&le));
|
||||
// 1.0: exponent = bias, mantissa 0
|
||||
let one: u32 = 31 << 20;
|
||||
assert_eq!(canon_one(&le, &one.to_le_bytes()), 1.0f32.to_le_bytes());
|
||||
assert_eq!(canon_one(&be, &one.to_be_bytes()), 1.0f32.to_le_bytes());
|
||||
// -2.1999512 (h5py's reading of the file's -2.2): sign, e = 32, m = 819
|
||||
let v: u32 = (1 << 26) | (32 << 20) | (819 << 7);
|
||||
assert_eq!(
|
||||
canon_one(&le, &v.to_le_bytes()),
|
||||
(-2.199_951_2f32).to_le_bytes()
|
||||
);
|
||||
assert_eq!(canon_one(&le, &[0; 4]), 0.0f32.to_le_bytes());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ieee_floats_keep_their_raw_bytes() {
|
||||
let f32le = Datatype::FloatingPoint {
|
||||
size: 4,
|
||||
byte_order: DatatypeByteOrder::LittleEndian,
|
||||
bit_offset: 0,
|
||||
bit_precision: 32,
|
||||
exponent_location: 23,
|
||||
exponent_size: 8,
|
||||
mantissa_location: 0,
|
||||
mantissa_size: 23,
|
||||
exponent_bias: 127,
|
||||
};
|
||||
assert!(ieee_layout(&f32le));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn partial_precision_int_is_shifted_and_sign_extended() {
|
||||
let dt = Datatype::FixedPoint {
|
||||
size: 4,
|
||||
byte_order: DatatypeByteOrder::BigEndian,
|
||||
signed: true,
|
||||
bit_offset: 4,
|
||||
bit_precision: 17,
|
||||
};
|
||||
assert!(partial_int(&dt));
|
||||
let stored = (((-5i32) as u32) & 0x1_FFFF) << 4;
|
||||
let mut out = Vec::new();
|
||||
canon_partial_int(&dt, &stored.to_be_bytes(), &mut out).unwrap();
|
||||
assert_eq!(out, (-5i32).to_le_bytes());
|
||||
}
|
||||
}
|
||||
Executable
+259
@@ -0,0 +1,259 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Reference probe: same JSON as the Rust `conformance-probe`, produced with h5py.
|
||||
|
||||
Walk: iterative DFS from '/', children in sorted (UTF-8 byte) name order, hard
|
||||
links only, each object once (first path wins, deduplicated by object identity).
|
||||
Canonical value encoding: see harness/src/main.rs.
|
||||
"""
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
import struct
|
||||
import sys
|
||||
|
||||
import numpy as np
|
||||
import h5py
|
||||
|
||||
try:
|
||||
import hdf5plugin # noqa: F401 registers blosc/lz4/zstd/bzip2/... filters
|
||||
except Exception: # pragma: no cover
|
||||
pass
|
||||
|
||||
MAX_BYTES = 200 * 1024 * 1024
|
||||
MAX_OBJECTS = 200_000
|
||||
|
||||
|
||||
def canon_str(b, out):
|
||||
if isinstance(b, str):
|
||||
b = b.encode("utf-8", "surrogateescape")
|
||||
b = bytes(b)
|
||||
cut = b.find(b"\x00")
|
||||
if cut >= 0:
|
||||
b = b[:cut]
|
||||
b = b.rstrip(b" ")
|
||||
out += b"S" + struct.pack("<I", len(b)) + b
|
||||
|
||||
|
||||
def simple(dt):
|
||||
if dt.fields:
|
||||
return all(simple(dt.fields[n][0]) for n in dt.names)
|
||||
if dt.subdtype:
|
||||
return simple(dt.subdtype[0])
|
||||
return dt.kind in "iufcbV"
|
||||
|
||||
|
||||
def packed(dt):
|
||||
if dt.fields:
|
||||
return np.dtype([(n, packed(dt.fields[n][0])) for n in dt.names])
|
||||
if dt.subdtype:
|
||||
base, shape = dt.subdtype
|
||||
return np.dtype((packed(base), shape))
|
||||
if dt.kind in "iufcb":
|
||||
return dt.newbyteorder("<")
|
||||
return dt
|
||||
|
||||
|
||||
def canon_el(dt, val, out):
|
||||
if dt.fields:
|
||||
for n in dt.names:
|
||||
canon_el(dt.fields[n][0], val[n], out)
|
||||
return
|
||||
if dt.subdtype:
|
||||
base, _ = dt.subdtype
|
||||
for x in np.asarray(val).reshape(-1):
|
||||
canon_el(base, x, out)
|
||||
return
|
||||
k = dt.kind
|
||||
if k in "iufcb":
|
||||
out += np.asarray(val, dtype=dt).astype(dt.newbyteorder("<")).tobytes()
|
||||
elif k == "V":
|
||||
out += np.asarray(val, dtype=dt).tobytes()
|
||||
elif k == "S":
|
||||
canon_str(val, out)
|
||||
elif k == "O":
|
||||
if h5py.check_string_dtype(dt) is not None:
|
||||
canon_str(val if val is not None else b"", out)
|
||||
elif h5py.check_ref_dtype(dt) is not None:
|
||||
out += b"R"
|
||||
else:
|
||||
base = h5py.check_vlen_dtype(dt)
|
||||
if base is None:
|
||||
raise TypeError(f"unhandled object dtype {dt!r}")
|
||||
arr = np.asarray(val if val is not None else [], dtype=base).reshape(-1)
|
||||
out += b"V" + struct.pack("<I", arr.shape[0])
|
||||
if simple(base):
|
||||
out += arr.astype(packed(base)).tobytes()
|
||||
else:
|
||||
for x in arr:
|
||||
canon_el(base, x, out)
|
||||
elif k == "U":
|
||||
canon_str(str(val), out)
|
||||
else:
|
||||
raise TypeError(f"unhandled dtype kind {k} ({dt!r})")
|
||||
|
||||
|
||||
def has_obj(dt):
|
||||
if dt.fields:
|
||||
return any(has_obj(dt.fields[n][0]) for n in dt.names)
|
||||
if dt.subdtype:
|
||||
return has_obj(dt.subdtype[0])
|
||||
return dt.kind == "O"
|
||||
|
||||
|
||||
def note_conversion(tid, dt, rec):
|
||||
"""h5py converts some file types (FP8, bfloat16, x87 long double, ...) to a
|
||||
different-sized numpy type; then value bytes are not comparable."""
|
||||
try:
|
||||
if not has_obj(dt) and tid.get_size() != dt.itemsize:
|
||||
rec["converted"] = f"file type size {tid.get_size()} -> numpy {dt} ({dt.itemsize})"
|
||||
except Exception: # noqa: BLE001
|
||||
pass
|
||||
|
||||
|
||||
def hash_values(arr, dt, rec):
|
||||
if dt.subdtype is not None:
|
||||
# h5py expands an HDF5 array element type into trailing array dims
|
||||
dt = dt.subdtype[0]
|
||||
arr = np.asarray(arr, dtype=dt)
|
||||
if simple(dt):
|
||||
c = np.ascontiguousarray(arr).astype(packed(dt)).tobytes()
|
||||
else:
|
||||
out = bytearray()
|
||||
for x in arr.reshape(-1):
|
||||
canon_el(dt, x, out)
|
||||
c = bytes(out)
|
||||
rec["hash"] = hashlib.sha256(c).hexdigest()
|
||||
rec["head"] = c[:48].hex()
|
||||
|
||||
|
||||
def err(e):
|
||||
s = f"{type(e).__name__}: {e}"
|
||||
return s.splitlines()[0][:400] if s else type(e).__name__
|
||||
|
||||
|
||||
def shape_of(s):
|
||||
return "null" if s is None else list(s)
|
||||
|
||||
|
||||
def n_bytes(shape, tid):
|
||||
n = 1
|
||||
for d in shape or ():
|
||||
n *= d
|
||||
return n * tid.get_size()
|
||||
|
||||
|
||||
def read_attrs(obj):
|
||||
out = {}
|
||||
names = sorted(obj.attrs.keys(), key=lambda s: s.encode("utf-8", "surrogateescape"))
|
||||
for name in names:
|
||||
rec = {}
|
||||
try:
|
||||
aid = obj.attrs.get_id(name)
|
||||
rec["dtype"] = str(aid.dtype)
|
||||
rec["shape"] = shape_of(aid.shape)
|
||||
note_conversion(aid.get_type(), aid.dtype, rec)
|
||||
if aid.shape is None:
|
||||
hash_values(np.empty((0,), dtype=aid.dtype), aid.dtype, rec)
|
||||
else:
|
||||
val = obj.attrs[name]
|
||||
hash_values(val, aid.dtype, rec)
|
||||
except Exception as e: # noqa: BLE001
|
||||
rec = {"error": err(e)}
|
||||
out[name] = rec
|
||||
return out
|
||||
|
||||
|
||||
def main(path):
|
||||
top = {"file": path}
|
||||
try:
|
||||
f = h5py.File(path, "r")
|
||||
except Exception as e: # noqa: BLE001
|
||||
top["open_error"] = err(e)
|
||||
print(json.dumps(top))
|
||||
return
|
||||
objects = []
|
||||
seen = set()
|
||||
stack = [("/", None)]
|
||||
while stack:
|
||||
p, obj = stack.pop()
|
||||
if len(objects) >= MAX_OBJECTS:
|
||||
top["truncated"] = True
|
||||
break
|
||||
rec = {"path": p}
|
||||
try:
|
||||
if obj is None:
|
||||
obj = f[p]
|
||||
key = hash(obj.id) # h5py ObjectID hash = (fileno, object address/token)
|
||||
except Exception as e: # noqa: BLE001
|
||||
rec["kind"] = "unknown"
|
||||
rec["error"] = err(e)
|
||||
objects.append(rec)
|
||||
continue
|
||||
if key in seen:
|
||||
continue
|
||||
seen.add(key)
|
||||
if isinstance(obj, h5py.Dataset):
|
||||
kind = "dataset"
|
||||
elif isinstance(obj, h5py.Group):
|
||||
kind = "group"
|
||||
elif isinstance(obj, h5py.Datatype):
|
||||
kind = "datatype"
|
||||
else:
|
||||
kind = "unknown"
|
||||
rec["kind"] = kind
|
||||
if kind == "dataset":
|
||||
try:
|
||||
dt = obj.dtype
|
||||
rec["dtype"] = str(dt)
|
||||
rec["shape"] = shape_of(obj.shape)
|
||||
note_conversion(obj.id.get_type(), dt, rec)
|
||||
if obj.shape is None:
|
||||
hash_values(np.empty((0,), dtype=dt), dt, rec)
|
||||
elif n_bytes(obj.shape, obj.id.get_type()) > MAX_BYTES:
|
||||
rec["skipped"] = "too large"
|
||||
else:
|
||||
arr = np.empty(obj.shape, dtype=dt)
|
||||
if arr.size:
|
||||
try:
|
||||
obj.read_direct(arr)
|
||||
except Exception: # noqa: BLE001
|
||||
arr = obj[()]
|
||||
hash_values(arr, dt, rec)
|
||||
except Exception as e: # noqa: BLE001
|
||||
rec["error"] = err(e)
|
||||
if kind != "datatype":
|
||||
try:
|
||||
rec["attrs"] = read_attrs(obj)
|
||||
except Exception as e: # noqa: BLE001
|
||||
rec["attrs_error"] = err(e)
|
||||
if kind == "group":
|
||||
try:
|
||||
names = sorted(obj.keys(), key=lambda s: s.encode("utf-8", "surrogateescape"))
|
||||
base = "" if p == "/" else p
|
||||
kids = []
|
||||
for n in names:
|
||||
try:
|
||||
link = obj.get(n, getlink=True)
|
||||
except Exception: # noqa: BLE001
|
||||
link = None
|
||||
if link is not None and not isinstance(link, h5py.HardLink):
|
||||
continue
|
||||
kids.append(f"{base}/{n}")
|
||||
for k in reversed(kids):
|
||||
stack.append((k, None))
|
||||
except Exception as e: # noqa: BLE001
|
||||
rec["list_error"] = err(e)
|
||||
objects.append(rec)
|
||||
top["objects"] = objects
|
||||
print(json.dumps(top), flush=True)
|
||||
# Exit without tearing down the h5py objects: freeing them for some files
|
||||
# that hold references (hdf5's h5repack_attr_refs.h5, cve-2024-32623.h5)
|
||||
# makes libhdf5 2.0 abort with "free(): chunks in smallbin corrupted"
|
||||
# about half the time. That happens after the reading is done, so it says
|
||||
# nothing about what h5py read, but it flipped those files between ok and
|
||||
# h5py-cannot-read from one run to the next.
|
||||
os._exit(0)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main(sys.argv[1])
|
||||
@@ -0,0 +1,335 @@
|
||||
#!/usr/bin/env python3
|
||||
"""report.py <results_dir> <CONFORMANCE.md> <corpus_dir>
|
||||
|
||||
Render the sweep's results (compare.py's results.json plus the raw per-side
|
||||
runs) as CONFORMANCE.md, and write <results_dir>/report-meta.json (commit,
|
||||
date, versions) for check.py --update.
|
||||
"""
|
||||
import collections
|
||||
import datetime
|
||||
import json
|
||||
import os
|
||||
import platform
|
||||
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
import h5py
|
||||
import numpy
|
||||
|
||||
try:
|
||||
import hdf5plugin
|
||||
HDF5PLUGIN = hdf5plugin.version
|
||||
except Exception: # noqa: BLE001
|
||||
HDF5PLUGIN = "not installed"
|
||||
|
||||
R, OUT_MD, CORPUS = sys.argv[1], sys.argv[2], sys.argv[3]
|
||||
HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
ROOT = os.path.dirname(HERE)
|
||||
CLASSES = ["ok", "our-error", "mismatch", "h5py-cannot-read", "panic", "hang", "crash", "oom"]
|
||||
|
||||
|
||||
def sh(*cmd, cwd=ROOT):
|
||||
try:
|
||||
return subprocess.run(cmd, cwd=cwd, capture_output=True, text=True, timeout=30).stdout.strip()
|
||||
except Exception: # noqa: BLE001
|
||||
return ""
|
||||
|
||||
|
||||
def cpu_model():
|
||||
try:
|
||||
for ln in open("/proc/cpuinfo"):
|
||||
if ln.startswith(("model name", "Model")):
|
||||
return ln.split(":", 1)[1].strip()
|
||||
except OSError:
|
||||
pass
|
||||
return platform.processor() or "unknown"
|
||||
|
||||
|
||||
def mem_gib():
|
||||
try:
|
||||
for ln in open("/proc/meminfo"):
|
||||
if ln.startswith("MemTotal:"):
|
||||
return f"{int(ln.split()[1]) / 1048576:.0f} GiB"
|
||||
except OSError:
|
||||
pass
|
||||
return "?"
|
||||
|
||||
|
||||
res = json.load(open(os.path.join(R, "results.json")))
|
||||
meta_run = json.load(open(os.path.join(R, "meta.json"))) if os.path.exists(os.path.join(R, "meta.json")) else {}
|
||||
rows = res["rows"]
|
||||
issues = res.get("issues", {})
|
||||
|
||||
# safe.directory: a checkout owned by another user (a container) is still ours to read
|
||||
commit = sh("git", "-c", "safe.directory=*", "rev-parse", "HEAD") or os.environ.get("GITHUB_SHA", "unknown")
|
||||
lib_dirty = sh("git", "-c", "safe.directory=*", "status", "--porcelain", "--", "crates", "Cargo.toml")
|
||||
h5dump_v = sh("h5dump", "--version").replace("h5dump: ", "")
|
||||
meta = {
|
||||
"date": datetime.datetime.now(datetime.timezone.utc).strftime("%Y-%m-%d %H:%M UTC"),
|
||||
"commit": commit + (" (library sources modified)" if lib_dirty else ""),
|
||||
"reference": f"h5py {h5py.__version__} / HDF5 {h5py.version.hdf5_version}",
|
||||
}
|
||||
json.dump(meta, open(os.path.join(R, "report-meta.json"), "w"), indent=1)
|
||||
|
||||
pins = []
|
||||
for ln in open(os.path.join(HERE, "corpus.txt")):
|
||||
if ln.strip() and not ln.lstrip().startswith("#"):
|
||||
name, url, rev, root, *_ = ln.split()
|
||||
pins.append((name, url, rev, root))
|
||||
|
||||
by_corpus = collections.defaultdict(collections.Counter)
|
||||
for r in rows:
|
||||
by_corpus[r["corpus"]][r["class"]] += 1
|
||||
total = collections.Counter(r["class"] for r in rows)
|
||||
|
||||
|
||||
def ex_list(files, n=3):
|
||||
s = ", ".join(f"`{f}`" for f in files[:n])
|
||||
return s + (f" (+{len(files) - n} more)" if len(files) > n else "")
|
||||
|
||||
|
||||
# --- known causes that are not clawhdf5 bugs --------------------------------
|
||||
def is_h5py_be_vlen(i):
|
||||
"""h5py returns the elements of a VL sequence of a big-endian base type
|
||||
with their file (big-endian) bytes but a native-endian dtype."""
|
||||
return (i["kind"] == "mismatch" and i["key"] in ("values", "attr-values")
|
||||
and (i.get("ref_dtype") == "object") and (i.get("ours_dtype") or "").startswith("vlen(")
|
||||
and ">" in (i.get("ours_dtype") or ""))
|
||||
|
||||
|
||||
known = collections.defaultdict(list)
|
||||
for r in rows:
|
||||
if r["class"] != "mismatch":
|
||||
continue
|
||||
iss = issues.get(r["file"], [])
|
||||
if iss and all(is_h5py_be_vlen(i) for i in iss):
|
||||
known["h5py-be-vlen"].append(r["file"])
|
||||
|
||||
|
||||
# --- the CVE corpus: clawhdf5 vs h5dump vs h5py ------------------------------
|
||||
def side(run, name):
|
||||
p = os.path.join(R, "runs", run, name)
|
||||
if not os.path.exists(p + ".rc"):
|
||||
return None
|
||||
rc = int(open(p + ".rc").read().strip() or -1)
|
||||
err = open(p + ".err", errors="replace").read()
|
||||
try:
|
||||
j = json.load(open(p + ".json"))
|
||||
except Exception: # noqa: BLE001
|
||||
j = None
|
||||
return rc, err, j
|
||||
|
||||
|
||||
def outcome(s, rust=False):
|
||||
"""-> (bucket, text). bucket in read / error / panic / crash / hang / oom."""
|
||||
if s is None:
|
||||
return "missing", "not run"
|
||||
rc, err, j = s
|
||||
if rc in (137, 124):
|
||||
return "hang", "hang (killed at timeout)"
|
||||
if "memory allocation of" in err or "MemoryError" in err or "bad_alloc" in err or "Cannot allocate" in err:
|
||||
return "oom", "out of memory"
|
||||
if rust and (rc == 101 or "PANIC:" in err):
|
||||
return "panic", "panic"
|
||||
if "overflowed its stack" in err:
|
||||
return "crash", "stack overflow"
|
||||
if rc == 139:
|
||||
return "crash", "SIGSEGV"
|
||||
if rc == 134:
|
||||
return "crash", "SIGABRT" + (" (heap corruption)" if ("corrupted" in err or "free()" in err) else "")
|
||||
if rc > 128:
|
||||
return "crash", f"signal {rc - 128}"
|
||||
if j is None:
|
||||
return ("error", "error exit") if rc in (0, 1) else ("crash", f"exit {rc}")
|
||||
if "open_error" in j:
|
||||
return "error", "open error"
|
||||
objs = j.get("objects", [])
|
||||
ne = sum(1 for o in objs for k in ("error", "attrs_error", "list_error") if k in o)
|
||||
ne += sum(1 for o in objs for a in (o.get("attrs") or {}).values() if "error" in a)
|
||||
return "read", f"read {len(objs)} obj" + (f", {ne} errors" if ne else "")
|
||||
|
||||
|
||||
def h5dump_outcome(s):
|
||||
if s is None:
|
||||
return "missing", "not run"
|
||||
rc, err, _ = s
|
||||
if rc in (137, 124):
|
||||
return "hang", "hang (killed at timeout)"
|
||||
if "memory allocation" in err or "Cannot allocate" in err:
|
||||
return "oom", "out of memory"
|
||||
if rc == 139:
|
||||
return "crash", "SIGSEGV"
|
||||
if rc == 134:
|
||||
return "crash", "SIGABRT" + (" (heap corruption)" if ("corrupted" in err or "free()" in err) else "")
|
||||
if rc > 128:
|
||||
return "crash", f"signal {rc - 128}"
|
||||
return ("read", "ok") if rc == 0 else ("error", "error exit")
|
||||
|
||||
|
||||
cve_rows = []
|
||||
buckets = {"clawhdf5": collections.Counter(), "h5dump": collections.Counter(), "h5py": collections.Counter()}
|
||||
ours_panic = {r["file"] for r in rows if r["class"] == "panic"}
|
||||
for r in rows:
|
||||
if r["corpus"] != "cve_hdf5":
|
||||
continue
|
||||
run = r["file"].replace("/", "__")
|
||||
o = outcome(side(run, "ours"), rust=True)
|
||||
if o[0] == "read" and r["file"] in ours_panic:
|
||||
o = ("panic", "caught panic")
|
||||
p = outcome(side(run, "ref"))
|
||||
d = h5dump_outcome(side(run, "h5dump"))
|
||||
buckets["clawhdf5"][o[0]] += 1
|
||||
buckets["h5py"][p[0]] += 1
|
||||
buckets["h5dump"][d[0]] += 1
|
||||
cve_rows.append((r["file"].split("/", 1)[1], d[1], p[1], o[1], r["class"]))
|
||||
|
||||
# --- render -----------------------------------------------------------------
|
||||
L = []
|
||||
w = L.append
|
||||
w("# clawhdf5 conformance report")
|
||||
w("")
|
||||
w("Every HDF5 file of eight public corpora (pinned by commit) is read twice — by")
|
||||
w("clawhdf5 (`conformance/probe`, the same `clawhdf5-format` calls the facade")
|
||||
w("makes) and by h5py/libhdf5 (`conformance/ref.py`) — and the two readings are")
|
||||
w("compared object by object: the set of hard-linked objects, each dataset's and")
|
||||
w("attribute's shape, and a SHA-256 of its values in a canonical encoding. The")
|
||||
w("CVE corpus is also run through `h5dump`. Each side runs under a timeout and an")
|
||||
w("address-space limit, so a hang, crash or runaway allocation is recorded, not")
|
||||
w("fatal. This file is generated by `conformance/run.sh`; do not edit it by hand.")
|
||||
w("")
|
||||
w("## Run")
|
||||
w("")
|
||||
w("| | |")
|
||||
w("|---|---|")
|
||||
w(f"| date | {meta['date']} |")
|
||||
w(f"| clawhdf5 commit | `{meta['commit']}` |")
|
||||
w(f"| machine | `{platform.node()}`: {cpu_model()}, {os.cpu_count()} CPUs, {mem_gib()}, {platform.system()} {platform.release()} {platform.machine()} |")
|
||||
w(f"| command | `{os.environ.get('CONFORMANCE_CMD', 'conformance/run.sh')}` |")
|
||||
w(f"| rustc | {sh('rustc', '-V')} |")
|
||||
w(f"| reference | h5py {h5py.__version__}, HDF5 {h5py.version.hdf5_version}, numpy {numpy.__version__}, hdf5plugin {HDF5PLUGIN}, Python {platform.python_version()} |")
|
||||
w(f"| h5dump | {h5dump_v} (CVE corpus only) |")
|
||||
if meta_run:
|
||||
w(f"| limits | {meta_run.get('timeout_s')} s timeout (SIGKILL), {int(meta_run.get('mem_kb', 0)) // 1024} MiB address space, per process; {meta_run.get('jobs')} files in parallel |")
|
||||
w(f"| runtime | {meta_run.get('probe_seconds')} s probing + comparing ({meta_run.get('build_seconds')} s fetch/build before it) |")
|
||||
w("")
|
||||
w("## Results")
|
||||
w("")
|
||||
w("A file's class is the first that applies:")
|
||||
w("")
|
||||
w("- **panic / hang / crash / oom** — clawhdf5 panicked (caught per object or not), hit the timeout, died on a signal, or failed an allocation. The CI gate fails on any of these.")
|
||||
w("- **h5py-cannot-read** — libhdf5 could not open the file (or itself crashed or hung). Nothing to compare against; most are the deliberately malformed CVE reproducers.")
|
||||
w("- **our-error** — clawhdf5 returned an error for something h5py reads.")
|
||||
w("- **mismatch** — both read it, but the shapes, values, object set or attribute set differ.")
|
||||
w("- **ok** — every object h5py reads, clawhdf5 reads identically.")
|
||||
w("")
|
||||
w("| corpus | files | " + " | ".join(CLASSES) + " |")
|
||||
w("|---" * (len(CLASSES) + 2) + "|")
|
||||
for c in sorted(by_corpus):
|
||||
cnt = by_corpus[c]
|
||||
w(f"| {c} | {sum(cnt.values())} | " + " | ".join(str(cnt.get(k, 0)) for k in CLASSES) + " |")
|
||||
w(f"| **all** | **{len(rows)}** | " + " | ".join(f"**{total.get(k, 0)}**" for k in CLASSES) + " |")
|
||||
w("")
|
||||
n_known = sum(len(v) for v in known.values())
|
||||
if n_known:
|
||||
w(f"{n_known} of the {total.get('mismatch', 0)} mismatches are a known h5py bug, not ours (see *Known not-our-bug*).")
|
||||
w("")
|
||||
w("Corpora (fetched by `conformance/fetch-corpus.sh` into the gitignored `conformance/.cache/`):")
|
||||
w("")
|
||||
w("| corpus | source | commit |")
|
||||
w("|---|---|---|")
|
||||
for name, url, rev, root in pins:
|
||||
w(f"| {name} | {url.removesuffix('.git')}" + ("" if root == "." else f" (`{root}`)") + f" | `{rev[:12]}` |")
|
||||
w("")
|
||||
|
||||
w("## Panics, hangs, crashes, out-of-memory")
|
||||
w("")
|
||||
if not res["panics"]:
|
||||
w("None.")
|
||||
else:
|
||||
for p in res["panics"]:
|
||||
w(f"- `{p['file']}` [{p['class']}] {p['detail']}")
|
||||
w("")
|
||||
|
||||
w("## Our-error root causes")
|
||||
w("")
|
||||
w("Grouped by normalised error message. *files* counts files whose class this cause affects.")
|
||||
w("")
|
||||
w("| files | objects | error | examples |")
|
||||
w("|---:|---:|---|---|")
|
||||
for k, v in res["root_causes"].items():
|
||||
w(f"| {v['files']} | {v['count']} | `{k.replace('|', '/')}` | {ex_list(v['file_list'])} |")
|
||||
w("")
|
||||
w("## Mismatch root causes")
|
||||
w("")
|
||||
w("| files | objects | cause | examples |")
|
||||
w("|---:|---:|---|---|")
|
||||
for k, v in res["mismatch_causes"].items():
|
||||
w(f"| {v['files']} | {v['count']} | `{k.replace('|', '/')}` | {ex_list(v['file_list'])} |")
|
||||
w("")
|
||||
|
||||
w("## CVE corpus: clawhdf5 vs h5dump vs h5py")
|
||||
w("")
|
||||
w(f"The {len(cve_rows)} files of [HDFGroup/cve_hdf5](https://github.com/HDFGroup/cve_hdf5) — reproducers for")
|
||||
w("published libhdf5 CVEs and fuzzer finds. *read* = produced output (possibly with per-object")
|
||||
w("errors), *error* = refused cleanly. h5dump exits non-zero on any error anywhere in a file, so")
|
||||
w("its read/error split is not comparable with the other two rows; the panic, crash, hang and oom")
|
||||
w("columns are.")
|
||||
w("")
|
||||
w("| tool | read | error | panic | crash | hang | oom |")
|
||||
w("|---|---:|---:|---:|---:|---:|---:|")
|
||||
for tool, label in (("clawhdf5", "clawhdf5"), ("h5dump", f"h5dump {h5dump_v.split()[-1] if h5dump_v else ''}"),
|
||||
("h5py", f"h5py {h5py.__version__} / HDF5 {h5py.version.hdf5_version}")):
|
||||
b = buckets[tool]
|
||||
w(f"| {label} | " + " | ".join(str(b.get(k, 0)) for k in ("read", "error", "panic", "crash", "hang", "oom")) + " |")
|
||||
w("")
|
||||
w("<details><summary>Per-file outcomes</summary>")
|
||||
w("")
|
||||
w("| file | h5dump | h5py | clawhdf5 | class |")
|
||||
w("|---|---|---|---|---|")
|
||||
for f, d, p, o, cls in cve_rows:
|
||||
w(f"| {f} | {d} | {p} | {o} | {cls} |")
|
||||
w("")
|
||||
w("</details>")
|
||||
w("")
|
||||
|
||||
w("## Known not-our-bug")
|
||||
w("")
|
||||
w("- **h5py big-endian variable-length sequences.** h5py returns the elements of a VL sequence")
|
||||
w(" whose base type is big-endian with the file's big-endian bytes but a native (little-endian)")
|
||||
w(" numpy dtype, so the values it reports are byte-swapped garbage; `h5dump` prints the values")
|
||||
w(" clawhdf5 reads. Reproducer: `h5py.vlen_dtype(np.dtype('>f4'))` dataset holding `[1.0, 2.0]`")
|
||||
w(" reads back in h5py as `[4.6e-41, 9.0e-44]`. Affected here: "
|
||||
+ (ex_list(sorted(known["h5py-be-vlen"]), 10) if known["h5py-be-vlen"] else "none") + ".")
|
||||
w("- **Non-IEEE floats and partial-precision integers (N-Bit).** libhdf5 converts a float whose")
|
||||
w(" bit layout is not IEEE (e.g. `H5Tset_precision` for the N-Bit filter) or an integer with a")
|
||||
w(" bit offset / reduced precision into the plain numpy type of the same size. The probe")
|
||||
w(" compares such values as converted numbers, not raw file bytes (before 2026-09-25 it compared")
|
||||
w(" raw bytes, which reported every N-Bit float dataset as a mismatch).")
|
||||
if res["incomparable"]:
|
||||
w("- **Types h5py widens.** Where h5py reads a type into a numpy type of a different size")
|
||||
w(" (FP8 -> float16, bfloat16 -> float32, x87 long double -> float128) the values are not")
|
||||
w(" compared (shape and presence still are): "
|
||||
+ ", ".join(f"{k} ({n}x)" for k, n in res["incomparable"]) + ".")
|
||||
w("- **References** are compared by presence only (`R`), not by target.")
|
||||
w("")
|
||||
if res.get("ref_only_errors"):
|
||||
w("## Objects h5py fails on but clawhdf5 reads")
|
||||
w("")
|
||||
for k, n in res["ref_only_errors"][:15]:
|
||||
w(f"- {n} x `{k}`")
|
||||
w("")
|
||||
w("## Reproduce")
|
||||
w("")
|
||||
w("```sh")
|
||||
w("# needs: Rust, python3 with h5py numpy hdf5plugin (conformance/requirements.txt), h5dump (hdf5-tools), git")
|
||||
w("CLAWHDF5_PYTHON=/path/to/venv/bin/python conformance/run.sh")
|
||||
w("```")
|
||||
w("")
|
||||
w("The corpus (about 450 MB of sparse checkouts) is cached in `conformance/.cache/`; results for")
|
||||
w("every file, both sides' raw JSON and stderr, are in `conformance/.cache/results/`.")
|
||||
w("`conformance/baseline.json` holds the ok files the nightly CI job (`.gitea/workflows/conformance.yml`)")
|
||||
w("must keep; `conformance/run.sh --update-baseline` rewrites it.")
|
||||
|
||||
with open(OUT_MD, "w") as fh:
|
||||
fh.write("\n".join(L) + "\n")
|
||||
@@ -0,0 +1,6 @@
|
||||
# The reference side of the conformance sweep. Pinned so the nightly job and a
|
||||
# local run compare against the same libhdf5 (h5py wheels bundle it).
|
||||
h5py==3.16.0
|
||||
numpy==2.5.3
|
||||
hdf5plugin==7.1.0
|
||||
netCDF4==1.7.4
|
||||
Executable
+88
@@ -0,0 +1,88 @@
|
||||
#!/usr/bin/env bash
|
||||
# conformance/run.sh — the clawhdf5 conformance sweep, end to end.
|
||||
#
|
||||
# fetch the pinned corpora (cached) -> build the probe -> probe every file
|
||||
# with clawhdf5 and with h5py (and h5dump for the CVE corpus), each under a
|
||||
# timeout and a memory limit -> compare -> write CONFORMANCE.md -> check the
|
||||
# result against conformance/baseline.json.
|
||||
#
|
||||
# Usage: conformance/run.sh [--no-fetch] [--no-report] [--update-baseline]
|
||||
#
|
||||
# Environment:
|
||||
# CLAWHDF5_PYTHON python with h5py, numpy, hdf5plugin (default: repo .venv, then python3)
|
||||
# CONFORMANCE_CACHE corpus / build / results cache (default: conformance/.cache)
|
||||
# CONFORMANCE_OUT results directory (default: $CONFORMANCE_CACHE/results)
|
||||
# CONFORMANCE_REPORT report path (default: CONFORMANCE.md at the repo root)
|
||||
# JOBS parallel files (default: nproc)
|
||||
# CONFORMANCE_PROBE use this prebuilt probe binary instead of building one
|
||||
# TMO / MEM_KB per-process timeout in seconds (20) / address-space limit in KiB (4 GiB)
|
||||
#
|
||||
# Exit status: 0 = gate passed; 1 = a panic/hang/crash/oom in clawhdf5, or the
|
||||
# ok count fell below the baseline, or a baseline-ok file regressed; 2 = setup error.
|
||||
set -euo pipefail
|
||||
HERE="$(cd "$(dirname "$0")" && pwd)"
|
||||
ROOT="$(cd "$HERE/.." && pwd)"
|
||||
FETCH=1 REPORT=1 UPDATE=0
|
||||
for a in "$@"; do
|
||||
case "$a" in
|
||||
--no-fetch) FETCH=0 ;;
|
||||
--no-report) REPORT=0 ;;
|
||||
--update-baseline) UPDATE=1 ;;
|
||||
-h|--help) sed -n '2,23p' "$0"; exit 0 ;;
|
||||
*) echo "unknown argument: $a" >&2; exit 2 ;;
|
||||
esac
|
||||
done
|
||||
|
||||
export PATH="$HOME/.cargo/bin:$PATH"
|
||||
CACHE="${CONFORMANCE_CACHE:-$HERE/.cache}"
|
||||
mkdir -p "$CACHE"; CACHE="$(cd "$CACHE" && pwd)"
|
||||
OUT="${CONFORMANCE_OUT:-$CACHE/results}"
|
||||
REPORT_PATH="${CONFORMANCE_REPORT:-$ROOT/CONFORMANCE.md}"
|
||||
JOBS="${JOBS:-$(nproc 2>/dev/null || echo 4)}"
|
||||
if [ -n "${CLAWHDF5_PYTHON:-}" ]; then PY="$CLAWHDF5_PYTHON"
|
||||
elif [ -x "$ROOT/.venv/bin/python" ]; then PY="$ROOT/.venv/bin/python"
|
||||
else PY="$(command -v python3)"; fi
|
||||
export PY TMO="${TMO:-20}" MEM_KB="${MEM_KB:-4194304}"
|
||||
command -v h5dump >/dev/null || { echo "error: h5dump not found (install hdf5-tools)" >&2; exit 2; }
|
||||
"$PY" -c 'import h5py, numpy, hdf5plugin' || { echo "error: $PY lacks h5py/numpy/hdf5plugin" >&2; exit 2; }
|
||||
|
||||
t0=$(date +%s)
|
||||
[ "$FETCH" = 1 ] && bash "$HERE/fetch-corpus.sh" "$CACHE"
|
||||
C="$CACHE/corpus"
|
||||
[ -d "$C" ] || { echo "error: no corpus in $C (run without --no-fetch)" >&2; exit 2; }
|
||||
|
||||
if [ -n "${CONFORMANCE_PROBE:-}" ]; then
|
||||
export PROBE="$CONFORMANCE_PROBE" # a prebuilt probe, e.g. an older one for a before/after
|
||||
else
|
||||
echo "== building the probe"
|
||||
CARGO_TARGET_DIR="${CARGO_TARGET_DIR:-$CACHE/target}" \
|
||||
cargo build -q --release --manifest-path "$HERE/probe/Cargo.toml"
|
||||
export PROBE="${CARGO_TARGET_DIR:-$CACHE/target}/release/conformance-probe"
|
||||
fi
|
||||
t1=$(date +%s)
|
||||
|
||||
rm -rf "$OUT"; mkdir -p "$OUT"
|
||||
"$PY" "$HERE/list_files.py" "$C" > "$OUT/files.txt"
|
||||
echo "== probing $(wc -l <"$OUT/files.txt") files, $JOBS at a time (timeout ${TMO}s, limit $((MEM_KB / 1024)) MiB)"
|
||||
export C OUT HERE
|
||||
# The shell's "Segmentation fault (core dumped)" notices go to probe.log; the
|
||||
# signals themselves are recorded in each side's .rc.
|
||||
xargs -a "$OUT/files.txt" -d '\n' -P "$JOBS" -I{} bash -c '
|
||||
f="$1"; d="$OUT/runs/${f//\//__}"
|
||||
case "$f" in cve_hdf5/*) export WITH_H5DUMP=1 ;; esac
|
||||
"$HERE/run_one.sh" "$C/$f" "$d"' _ {} 2>"$OUT/probe.log"
|
||||
echo "== comparing"
|
||||
"$PY" "$HERE/compare.py" "$OUT" >/dev/null
|
||||
t2=$(date +%s)
|
||||
cat > "$OUT/meta.json" <<EOF
|
||||
{"build_seconds": $((t1 - t0)), "probe_seconds": $((t2 - t1)), "jobs": $JOBS, "timeout_s": $TMO, "mem_kb": $MEM_KB}
|
||||
EOF
|
||||
export CONFORMANCE_CMD="${CONFORMANCE_CMD:-conformance/run.sh${*:+ $*}}"
|
||||
if [ "$REPORT" = 1 ]; then
|
||||
"$PY" "$HERE/report.py" "$OUT" "$REPORT_PATH" "$C"
|
||||
echo "== wrote $REPORT_PATH"
|
||||
fi
|
||||
if [ "$UPDATE" = 1 ]; then
|
||||
"$PY" "$HERE/check.py" "$OUT" "$HERE/baseline.json" --update
|
||||
fi
|
||||
"$PY" "$HERE/check.py" "$OUT" "$HERE/baseline.json"
|
||||
Executable
+27
@@ -0,0 +1,27 @@
|
||||
#!/usr/bin/env bash
|
||||
# run_one.sh <file> <outdir>
|
||||
#
|
||||
# Probe one file with clawhdf5 (PROBE) and with h5py (PY ref.py), and with
|
||||
# h5dump too when WITH_H5DUMP is set. Each side runs under a timeout (TMO
|
||||
# seconds, SIGKILL) and an address-space limit (MEM_KB), with core dumps off.
|
||||
# Writes <outdir>/<side>.{json,err,rc}; rc 137 = killed by the timeout.
|
||||
set -u
|
||||
f="$1"; out="$2"; mkdir -p "$out"
|
||||
HERE="$(cd "$(dirname "$0")" && pwd)"
|
||||
: "${PROBE:?PROBE must name the conformance-probe binary}"
|
||||
: "${PY:?PY must name a python with h5py}"
|
||||
TMO="${TMO:-20}"
|
||||
MEM_KB="${MEM_KB:-4194304}"
|
||||
run() { # name cmd...
|
||||
local name=$1; shift
|
||||
( ulimit -v "$MEM_KB"; ulimit -c 0; RUST_BACKTRACE=1 exec timeout -s KILL "$TMO" "$@" ) \
|
||||
>"$out/$name.json" 2>"$out/$name.err"
|
||||
echo $? >"$out/$name.rc"
|
||||
}
|
||||
run ours "$PROBE" "$f"
|
||||
run ref "$PY" "$HERE/ref.py" "$f"
|
||||
if [ -n "${WITH_H5DUMP:-}" ]; then
|
||||
run h5dump h5dump "$f"
|
||||
: >"$out/h5dump.json" # h5dump's text dump is not compared, only its exit status
|
||||
fi
|
||||
exit 0
|
||||
@@ -19,6 +19,10 @@ clawhdf5-ann = { path = "../clawhdf5-ann", version = "2.7.0", optional = true }
|
||||
clawhdf5-gpu = { path = "../clawhdf5-gpu", version = "2.7.0", optional = true, default-features = false }
|
||||
serde = { workspace = true }
|
||||
byteorder = "1"
|
||||
# Signed checkpoints (MemoryConfig-independent; see `signing`). Pure Rust.
|
||||
ed25519-dalek = { version = "2", features = ["rand_core"] }
|
||||
sha2 = "0.10"
|
||||
rand_core = { version = "0.6", features = ["getrandom"] }
|
||||
half = { workspace = true, optional = true }
|
||||
rayon = { version = "1", optional = true }
|
||||
matrixmultiply = { version = "0.3", optional = true }
|
||||
|
||||
@@ -203,7 +203,7 @@ impl ImportanceScorer {
|
||||
/// Novelty score: 1.0 − max cosine similarity against all existing records.
|
||||
/// Returns 1.0 when there are no existing memories.
|
||||
///
|
||||
/// Same result as [`Self::cosine_similarity`] against each record, but the
|
||||
/// Same result as the reference cosine similarity against each record, but the
|
||||
/// new embedding's norm is computed once rather than per record, each
|
||||
/// record costs one fused pass (dot product and its norm together) rather
|
||||
/// than three, and a large working set is scored in parallel. Every insert
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
//! ZeroClaw agent memory HDF5 backend.
|
||||
//! Agent memory stored in a single HDF5 file.
|
||||
//!
|
||||
//! Provides persistent memory storage for AI agents using HDF5 files.
|
||||
//! All data is cached in-memory for fast access and flushed to disk
|
||||
@@ -36,6 +36,7 @@ pub mod reranker;
|
||||
pub mod schema;
|
||||
pub mod search;
|
||||
pub mod session;
|
||||
pub mod signing;
|
||||
pub mod storage;
|
||||
mod store_lock;
|
||||
pub mod temporal;
|
||||
@@ -78,6 +79,7 @@ pub use session::{SessionCache, SessionEntry};
|
||||
// --- Error type ---
|
||||
|
||||
#[derive(Debug)]
|
||||
#[non_exhaustive]
|
||||
pub enum MemoryError {
|
||||
Io(std::io::Error),
|
||||
Hdf5(String),
|
||||
@@ -88,6 +90,11 @@ pub enum MemoryError {
|
||||
/// A record the store cannot hold as given, e.g. an embedding value
|
||||
/// outside the half-precision range of a `float16` store.
|
||||
InvalidEntry(String),
|
||||
/// The store's checkpoints are signed and no signing key is set, so a
|
||||
/// checkpoint would leave it unsigned. Set the key with
|
||||
/// [`HDF5Memory::set_signing_key`], or drop the signature on purpose with
|
||||
/// [`HDF5Memory::remove_signature`].
|
||||
SigningKeyRequired(String),
|
||||
}
|
||||
|
||||
impl std::fmt::Display for MemoryError {
|
||||
@@ -99,6 +106,7 @@ impl std::fmt::Display for MemoryError {
|
||||
MemoryError::NotFound(e) => write!(f, "not found: {e}"),
|
||||
MemoryError::Locked(e) => write!(f, "store is locked: {e}"),
|
||||
MemoryError::InvalidEntry(e) => write!(f, "invalid entry: {e}"),
|
||||
MemoryError::SigningKeyRequired(e) => write!(f, "signing key required: {e}"),
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -317,6 +325,12 @@ pub struct HDF5Memory {
|
||||
activations_dirty: bool,
|
||||
/// Opened with [`HDF5Memory::open_read_only`]: nothing may reach the disk.
|
||||
read_only: bool,
|
||||
/// Key that signs every checkpoint; never persisted. See
|
||||
/// [`HDF5Memory::set_signing_key`].
|
||||
signing_key: Option<signing::SigningKey>,
|
||||
/// Checkpoints of this store are signed: the file on disk is, or a key
|
||||
/// has been set. A checkpoint without a key is then refused.
|
||||
signed: bool,
|
||||
/// A WAL that `open()` could not read and moved aside; see
|
||||
/// [`HDF5Memory::quarantined_wal`].
|
||||
quarantined_wal: Option<PathBuf>,
|
||||
@@ -372,6 +386,8 @@ impl HDF5Memory {
|
||||
bm25_filter: bm25::TokenFilter::default(),
|
||||
activations_dirty: false,
|
||||
read_only: false,
|
||||
signing_key: None,
|
||||
signed: false,
|
||||
quarantined_wal: None,
|
||||
_lock: Some(lock),
|
||||
})
|
||||
@@ -550,6 +566,8 @@ impl HDF5Memory {
|
||||
bm25_filter: bm25::TokenFilter::default(),
|
||||
activations_dirty: false,
|
||||
read_only,
|
||||
signing_key: None,
|
||||
signed: checkpoint.signed,
|
||||
quarantined_wal,
|
||||
_lock: lock,
|
||||
})
|
||||
@@ -710,6 +728,39 @@ impl HDF5Memory {
|
||||
}
|
||||
}
|
||||
|
||||
/// Sign every checkpoint from now on with `key` (Ed25519). The key is
|
||||
/// never written anywhere; set it again after every `open`. Once a store
|
||||
/// is signed, a checkpoint without the key is refused
|
||||
/// ([`MemoryError::SigningKeyRequired`]) rather than silently leaving it
|
||||
/// unsigned. Setting a different key re-signs the store under that key
|
||||
/// from the next checkpoint; a verifier trusting the old key will then
|
||||
/// reject it, which is the point. Call [`AgentMemory::flush_wal`] to sign
|
||||
/// right away.
|
||||
pub fn set_signing_key(&mut self, key: signing::SigningKey) {
|
||||
self.signing_key = Some(key);
|
||||
self.signed = true;
|
||||
}
|
||||
|
||||
/// Stop signing: the next checkpoint writes the store unsigned. The
|
||||
/// deliberate way out of [`MemoryError::SigningKeyRequired`].
|
||||
pub fn remove_signature(&mut self) {
|
||||
self.signing_key = None;
|
||||
self.signed = false;
|
||||
}
|
||||
|
||||
/// Checkpoints of this store are signed (on disk, or from the next
|
||||
/// checkpoint because a key has been set).
|
||||
pub fn is_signed(&self) -> bool {
|
||||
self.signed
|
||||
}
|
||||
|
||||
/// Check the checkpoint at `path` against the public key the caller
|
||||
/// trusts; see [`signing::verify_store`]. Reads the file only: it works
|
||||
/// on a store another process has open.
|
||||
pub fn verify(path: &Path, trusted: &signing::VerifyingKey) -> Result<signing::VerifyReport> {
|
||||
signing::verify_store(path, trusted)
|
||||
}
|
||||
|
||||
/// Flush current state to disk and truncate the WAL.
|
||||
///
|
||||
/// Every code path that persists the full cache to the .h5 file must
|
||||
@@ -725,10 +776,28 @@ impl HDF5Memory {
|
||||
// Record which WAL prefix this checkpoint contains, so a crash before
|
||||
// the truncate below can't replay those entries a second time.
|
||||
let wal_applied = self.wal.as_ref().map(|w| w.mark());
|
||||
let signature = match &self.signing_key {
|
||||
Some(key) => Some(signing::sign(
|
||||
key,
|
||||
&self.config,
|
||||
&self.cache,
|
||||
&self.sessions,
|
||||
&self.knowledge,
|
||||
wal_applied,
|
||||
)),
|
||||
None if self.signed => {
|
||||
return Err(MemoryError::SigningKeyRequired(format!(
|
||||
"{} is signed; set its signing key before a checkpoint \
|
||||
(saves so far are held in the WAL or in memory)",
|
||||
self.config.path.display()
|
||||
)));
|
||||
}
|
||||
None => None,
|
||||
};
|
||||
// Written before the .h5 so a crash in between leaves a sidecar whose
|
||||
// generation matches no checkpoint (ignored), never the reverse.
|
||||
let ann_generation = self.persist_vector_index();
|
||||
storage::write_to_disk_with_meta(
|
||||
storage::write_to_disk_signed(
|
||||
&self.config.path,
|
||||
&self.config,
|
||||
&self.cache,
|
||||
@@ -737,7 +806,9 @@ impl HDF5Memory {
|
||||
&schema::CheckpointMeta {
|
||||
wal_applied,
|
||||
ann_generation,
|
||||
signed: signature.is_some(),
|
||||
},
|
||||
signature.as_ref(),
|
||||
)?;
|
||||
if let Some(ref mut w) = self.wal {
|
||||
w.truncate()?;
|
||||
|
||||
@@ -1,10 +1,12 @@
|
||||
//! OpenClaw Integration Layer.
|
||||
//! A Markdown-oriented memory backend over [`crate::HDF5Memory`].
|
||||
//!
|
||||
//! Bridge between OpenClaw agent gateway (Markdown + sqlite-vec) and the
|
||||
//! clawhdf5 HDF5-backed memory backend. Provides:
|
||||
//! Named for OpenClaw, whose workspace memory is Markdown, but **not an
|
||||
//! OpenClaw plugin**: nothing here registers with OpenClaw, and the
|
||||
//! integration it was written for never worked (see `docs/openclaw.md`).
|
||||
//! Provides:
|
||||
//!
|
||||
//! - [`MemoryBackend`] — the trait OpenClaw implements against.
|
||||
//! - [`ClawhdfBackend`] — concrete HDF5-backed implementation.
|
||||
//! - [`MemoryBackend`] — search / read back / write / ingest / export.
|
||||
//! - [`ClawhdfBackend`] — the HDF5-backed implementation.
|
||||
//! - [`MarkdownParser`] — splits Markdown into [`MarkdownSection`] records.
|
||||
//! - [`MarkdownExporter`] — renders sections back to Markdown text.
|
||||
|
||||
@@ -61,7 +63,8 @@ pub struct BackendStats {
|
||||
// MemoryBackend trait
|
||||
// ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
/// Interface that OpenClaw uses to interact with a memory backend.
|
||||
/// A Markdown-oriented memory backend: search, read back by path, write,
|
||||
/// ingest and export.
|
||||
///
|
||||
/// Implementors provide persistent storage, full-text + vector search,
|
||||
/// Markdown ingestion / export, and statistics.
|
||||
@@ -318,7 +321,7 @@ impl MarkdownExporter {
|
||||
///
|
||||
/// # Path mapping
|
||||
///
|
||||
/// OpenClaw addresses memories by file path (e.g. `"memory/user.md"`).
|
||||
/// Memories are addressed by file path (e.g. `"memory/user.md"`).
|
||||
/// Internally every [`MemoryEntry`] stores the originating path as its
|
||||
/// `source_channel`. Section sub-paths are stored as
|
||||
/// `"<path>::<heading>"`.
|
||||
@@ -421,7 +424,7 @@ impl ClawhdfBackend {
|
||||
|
||||
// ── Compaction & Consolidation hooks (7.6) ────────────────────────────
|
||||
|
||||
/// Run a compaction cycle — called by OpenClaw during session compaction.
|
||||
/// Run a compaction cycle (decay, compaction, WAL flush).
|
||||
///
|
||||
/// Sequence:
|
||||
/// 1. `tick_session()` — apply Hebbian decay to all activation weights.
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
//!
|
||||
//! Records the origin, authorship, and a content hash of every memory chunk
|
||||
//! so the system can detect *accidental* corruption and trace data lineage.
|
||||
//! The hash is unkeyed (see [`fnv1a_64`]) — this is not a tamper-evidence or
|
||||
//! The hash is unkeyed (FNV-1a) — this is not a tamper-evidence or
|
||||
//! authenticity guarantee.
|
||||
|
||||
use std::collections::HashMap;
|
||||
|
||||
@@ -15,6 +15,9 @@ use crate::session::SessionCache;
|
||||
use crate::wal::WalMark;
|
||||
|
||||
pub const SCHEMA_VERSION: &str = "1.0";
|
||||
/// Writer-version tag stored in `/meta` as `edgehdf5_version`. Kept for file
|
||||
/// compatibility; despite the name it has nothing to do with ZeroClaw, which
|
||||
/// does not use clawhdf5.
|
||||
pub const ZEROCLAW_VERSION: &str = "0.8.0";
|
||||
|
||||
/// `/meta` attributes holding the [`WalMark`] of the WAL prefix already folded
|
||||
@@ -23,6 +26,7 @@ pub const ZEROCLAW_VERSION: &str = "0.8.0";
|
||||
const WAL_APPLIED_LEN_ATTR: &str = "wal_applied_len";
|
||||
const WAL_APPLIED_CRC_ATTR: &str = "wal_applied_crc";
|
||||
const ANN_GENERATION_ATTR: &str = "ann_generation";
|
||||
const SIG_VERSION_ATTR: &str = "sig_version";
|
||||
|
||||
/// Build a complete HDF5 file from the in-memory state.
|
||||
pub fn build_hdf5_file(
|
||||
@@ -46,7 +50,7 @@ pub fn build_hdf5_file_with_mark(
|
||||
) -> Result<Vec<u8>, MemoryError> {
|
||||
let meta = CheckpointMeta {
|
||||
wal_applied,
|
||||
ann_generation: None,
|
||||
..CheckpointMeta::default()
|
||||
};
|
||||
build_hdf5_file_with_meta(config, cache, sessions, knowledge, &meta)
|
||||
}
|
||||
@@ -61,6 +65,10 @@ pub struct CheckpointMeta {
|
||||
/// one left over from another checkpoint can never be attached to records
|
||||
/// it wasn't built from.
|
||||
pub ann_generation: Option<u64>,
|
||||
/// The checkpoint carries an Ed25519 signature (see [`crate::signing`]).
|
||||
/// Read-only: whether a checkpoint is *written* signed is decided by the
|
||||
/// signature passed to [`build_hdf5_file_signed`].
|
||||
pub signed: bool,
|
||||
}
|
||||
|
||||
/// [`build_hdf5_file`] with checkpoint bookkeeping.
|
||||
@@ -70,6 +78,19 @@ pub fn build_hdf5_file_with_meta(
|
||||
sessions: &SessionCache,
|
||||
knowledge: &KnowledgeCache,
|
||||
checkpoint: &CheckpointMeta,
|
||||
) -> Result<Vec<u8>, MemoryError> {
|
||||
build_hdf5_file_signed(config, cache, sessions, knowledge, checkpoint, None)
|
||||
}
|
||||
|
||||
/// [`build_hdf5_file_with_meta`], plus a signed manifest of the contents
|
||||
/// (see [`crate::signing`]).
|
||||
pub fn build_hdf5_file_signed(
|
||||
config: &MemoryConfig,
|
||||
cache: &MemoryCache,
|
||||
sessions: &SessionCache,
|
||||
knowledge: &KnowledgeCache,
|
||||
checkpoint: &CheckpointMeta,
|
||||
signature: Option<&crate::signing::StoredSignature>,
|
||||
) -> Result<Vec<u8>, MemoryError> {
|
||||
let wal_applied = checkpoint.wal_applied;
|
||||
let mut builder = clawhdf5::FileBuilder::new();
|
||||
@@ -130,11 +151,42 @@ pub fn build_hdf5_file_with_meta(
|
||||
// round trip through every reader.
|
||||
meta.set_attr(ANN_GENERATION_ATTR, AttrValue::I64(generation as i64));
|
||||
}
|
||||
if let Some(sig) = signature {
|
||||
use crate::signing::to_hex;
|
||||
let m = &sig.manifest;
|
||||
meta.set_attr(
|
||||
SIG_VERSION_ATTR,
|
||||
AttrValue::I64(crate::signing::MANIFEST_VERSION),
|
||||
);
|
||||
meta.set_attr("sig_algorithm", AttrValue::String("ed25519".into()));
|
||||
meta.set_attr("sig_public_key", AttrValue::String(to_hex(&sig.public_key)));
|
||||
meta.set_attr("sig_signature", AttrValue::String(to_hex(&sig.signature)));
|
||||
meta.set_attr("sig_record_count", AttrValue::I64(m.record_count as i64));
|
||||
meta.set_attr(
|
||||
"sig_records_root",
|
||||
AttrValue::String(to_hex(&m.records_root)),
|
||||
);
|
||||
meta.set_attr("sig_settings", AttrValue::String(to_hex(&m.settings)));
|
||||
meta.set_attr("sig_sessions", AttrValue::String(to_hex(&m.sessions)));
|
||||
meta.set_attr("sig_graph", AttrValue::String(to_hex(&m.graph)));
|
||||
}
|
||||
// Need at least one dataset in the group for it to be a proper group
|
||||
meta.create_dataset("_marker").with_u8_data(&[1]).compact();
|
||||
let finished_meta = meta.finish();
|
||||
builder.add_group(finished_meta);
|
||||
|
||||
// /integrity: the signed per-record hashes, so verification can say
|
||||
// which records changed.
|
||||
if let Some(sig) = signature {
|
||||
let mut group = builder.create_group("integrity");
|
||||
let flat: Vec<u8> = sig.record_hashes.iter().flatten().copied().collect();
|
||||
group
|
||||
.create_dataset("record_hashes")
|
||||
.with_u8_data(&flat)
|
||||
.with_shape(&[sig.record_hashes.len() as u64, 32]);
|
||||
builder.add_group(group.finish());
|
||||
}
|
||||
|
||||
// /memory group
|
||||
build_memory_group(&mut builder, config, cache)?;
|
||||
|
||||
@@ -425,10 +477,34 @@ fn write_string_dataset(
|
||||
}
|
||||
}
|
||||
|
||||
/// `/meta`'s attributes, failing if any of them cannot be read.
|
||||
///
|
||||
/// `Group::attrs` leaves out an attribute it cannot decode. For the store's
|
||||
/// settings that would silently fall back to defaults (e.g. `float16`, the
|
||||
/// WAL mark), so an unreadable attribute is an error here, as it was before
|
||||
/// `attrs` became tolerant.
|
||||
fn meta_attrs(
|
||||
file: &clawhdf5::File,
|
||||
) -> Result<std::collections::HashMap<String, AttrValue>, MemoryError> {
|
||||
let meta = file
|
||||
.group("meta")
|
||||
.map_err(|e| MemoryError::Schema(format!("missing /meta group: {e}")))?;
|
||||
let (attrs, errors) = meta
|
||||
.attrs_with_errors()
|
||||
.map_err(|e| MemoryError::Schema(format!("cannot read /meta attrs: {e}")))?;
|
||||
if let Some(e) = errors.first() {
|
||||
return Err(MemoryError::Schema(format!(
|
||||
"cannot read /meta attrs: {} unreadable, first: {e}",
|
||||
errors.len()
|
||||
)));
|
||||
}
|
||||
Ok(attrs)
|
||||
}
|
||||
|
||||
/// Validate an HDF5 file has the correct schema and load all data.
|
||||
/// Read the checkpoint's [`WalMark`] from `/meta`, if it has one.
|
||||
pub fn read_wal_mark(file: &clawhdf5::File) -> Option<WalMark> {
|
||||
let attrs = file.group("meta").ok()?.attrs().ok()?;
|
||||
let attrs = meta_attrs(file).ok()?;
|
||||
let len = match attrs.get(WAL_APPLIED_LEN_ATTR)? {
|
||||
AttrValue::I64(v) => u64::try_from(*v).ok()?,
|
||||
_ => return None,
|
||||
@@ -440,19 +516,75 @@ pub fn read_wal_mark(file: &clawhdf5::File) -> Option<WalMark> {
|
||||
Some(WalMark { len, crc })
|
||||
}
|
||||
|
||||
/// Read a checkpoint's signature, if it has one. A signature whose
|
||||
/// attributes are present but malformed is an error, not "unsigned".
|
||||
pub fn read_signature(
|
||||
file: &clawhdf5::File,
|
||||
) -> Result<Option<crate::signing::StoredSignature>, MemoryError> {
|
||||
use crate::signing::{Manifest, StoredSignature, from_hex};
|
||||
let attrs = meta_attrs(file)?;
|
||||
let version = match attrs.get(SIG_VERSION_ATTR) {
|
||||
None => return Ok(None),
|
||||
Some(AttrValue::I64(v)) => *v,
|
||||
Some(_) => return Err(MemoryError::Schema("malformed sig_version".into())),
|
||||
};
|
||||
if version != crate::signing::MANIFEST_VERSION {
|
||||
return Err(MemoryError::Schema(format!(
|
||||
"unsupported signature version {version}"
|
||||
)));
|
||||
}
|
||||
fn hex<const N: usize>(
|
||||
attrs: &std::collections::HashMap<String, AttrValue>,
|
||||
name: &str,
|
||||
) -> Result<[u8; N], MemoryError> {
|
||||
match attrs.get(name) {
|
||||
Some(AttrValue::String(s)) => from_hex::<N>(s),
|
||||
_ => None,
|
||||
}
|
||||
.ok_or_else(|| MemoryError::Schema(format!("malformed or missing {name}")))
|
||||
}
|
||||
let record_count = match attrs.get("sig_record_count") {
|
||||
Some(AttrValue::I64(v)) if *v >= 0 => *v as u64,
|
||||
_ => return Err(MemoryError::Schema("malformed sig_record_count".into())),
|
||||
};
|
||||
let group = file
|
||||
.group("integrity")
|
||||
.map_err(|e| MemoryError::Schema(format!("signed checkpoint without /integrity: {e}")))?;
|
||||
let flat = read_u8_dataset(&group, "record_hashes")?;
|
||||
if flat.len() % 32 != 0 {
|
||||
return Err(MemoryError::Schema(
|
||||
"/integrity/record_hashes is not a whole number of hashes".into(),
|
||||
));
|
||||
}
|
||||
let record_hashes = flat.as_chunks::<32>().0.to_vec();
|
||||
Ok(Some(StoredSignature {
|
||||
manifest: Manifest {
|
||||
record_count,
|
||||
records_root: hex::<32>(&attrs, "sig_records_root")?,
|
||||
settings: hex::<32>(&attrs, "sig_settings")?,
|
||||
sessions: hex::<32>(&attrs, "sig_sessions")?,
|
||||
graph: hex::<32>(&attrs, "sig_graph")?,
|
||||
},
|
||||
record_hashes,
|
||||
public_key: hex::<32>(&attrs, "sig_public_key")?,
|
||||
signature: hex::<64>(&attrs, "sig_signature")?,
|
||||
}))
|
||||
}
|
||||
|
||||
/// Read the checkpoint bookkeeping from `/meta`.
|
||||
pub fn read_checkpoint_meta(file: &clawhdf5::File) -> CheckpointMeta {
|
||||
let ann_generation = file
|
||||
.group("meta")
|
||||
let ann_generation =
|
||||
meta_attrs(file)
|
||||
.ok()
|
||||
.and_then(|g| g.attrs().ok())
|
||||
.and_then(|attrs| match attrs.get(ANN_GENERATION_ATTR) {
|
||||
Some(AttrValue::I64(v)) => Some(*v as u64),
|
||||
_ => None,
|
||||
});
|
||||
let signed = meta_attrs(file).is_ok_and(|attrs| attrs.contains_key(SIG_VERSION_ATTR));
|
||||
CheckpointMeta {
|
||||
wal_applied: read_wal_mark(file),
|
||||
ann_generation,
|
||||
signed,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -460,12 +592,7 @@ pub fn validate_and_load(
|
||||
file: &clawhdf5::File,
|
||||
) -> Result<(MemoryConfig, MemoryCache, SessionCache, KnowledgeCache), MemoryError> {
|
||||
// Read /meta group attributes
|
||||
let meta = file
|
||||
.group("meta")
|
||||
.map_err(|e| MemoryError::Schema(format!("missing /meta group: {e}")))?;
|
||||
let attrs = meta
|
||||
.attrs()
|
||||
.map_err(|e| MemoryError::Schema(format!("cannot read /meta attrs: {e}")))?;
|
||||
let attrs = meta_attrs(file)?;
|
||||
|
||||
let schema_version = match attrs.get("schema_version") {
|
||||
Some(AttrValue::String(s)) => s.clone(),
|
||||
|
||||
@@ -0,0 +1,419 @@
|
||||
//! Ed25519-signed checkpoints.
|
||||
//!
|
||||
//! When a signing key is set ([`crate::HDF5Memory::set_signing_key`]), every
|
||||
//! checkpoint writes a signed manifest of the store: a SHA-256 per memory
|
||||
//! record rolled into a Merkle root, plus hashes of the store's settings, its
|
||||
//! sessions and its knowledge graph. [`verify_store`] recomputes all of it from
|
||||
//! the file and checks the signature against a public key the caller trusts,
|
||||
//! so any change to the checkpointed file — a record's text or embedding, a
|
||||
//! setting, a session, a graph edge, made through this crate or any other HDF5
|
||||
//! tool — is detected, and the per-record hashes say which records changed.
|
||||
//!
|
||||
//! What it does not cover: saves still only in the WAL (made since the last
|
||||
//! checkpoint). [`VerifyReport::wal_entries_unsigned`] counts them.
|
||||
//!
|
||||
//! The hashes cover exactly what the file persists, in the form the loader
|
||||
//! returns it, so a store verifies after any number of reopen/checkpoint
|
||||
//! cycles. Derived data (L2 norms, the vector index) is not covered; it is
|
||||
//! recomputed from covered data.
|
||||
|
||||
use ed25519_dalek::{Signature, Signer, Verifier};
|
||||
pub use ed25519_dalek::{SigningKey, VerifyingKey};
|
||||
use sha2::{Digest, Sha256};
|
||||
|
||||
use crate::MemoryConfig;
|
||||
use crate::cache::MemoryCache;
|
||||
use crate::knowledge::KnowledgeCache;
|
||||
use crate::session::SessionCache;
|
||||
use crate::wal::WalMark;
|
||||
|
||||
/// Version of the manifest encoding; part of what is signed.
|
||||
pub const MANIFEST_VERSION: i64 = 1;
|
||||
|
||||
type Hash = [u8; 32];
|
||||
|
||||
/// The hashes a signature covers.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct Manifest {
|
||||
pub record_count: u64,
|
||||
/// Merkle root over the per-record hashes.
|
||||
pub records_root: Hash,
|
||||
/// Settings persisted in `/meta`, plus the checkpoint's WAL mark.
|
||||
pub settings: Hash,
|
||||
pub sessions: Hash,
|
||||
pub graph: Hash,
|
||||
}
|
||||
|
||||
impl Manifest {
|
||||
/// The exact bytes that are signed.
|
||||
pub fn signed_bytes(&self) -> Vec<u8> {
|
||||
let mut m = Vec::with_capacity(160);
|
||||
m.extend_from_slice(b"clawhdf5-agent signed checkpoint\0");
|
||||
m.extend_from_slice(&MANIFEST_VERSION.to_le_bytes());
|
||||
m.extend_from_slice(&self.record_count.to_le_bytes());
|
||||
m.extend_from_slice(&self.records_root);
|
||||
m.extend_from_slice(&self.settings);
|
||||
m.extend_from_slice(&self.sessions);
|
||||
m.extend_from_slice(&self.graph);
|
||||
m
|
||||
}
|
||||
}
|
||||
|
||||
/// A signature as stored in a checkpoint.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct StoredSignature {
|
||||
pub manifest: Manifest,
|
||||
pub record_hashes: Vec<Hash>,
|
||||
pub public_key: [u8; 32],
|
||||
pub signature: [u8; 64],
|
||||
}
|
||||
|
||||
/// Build the manifest (and per-record hashes) for the state about to be
|
||||
/// checkpointed, and sign it.
|
||||
pub fn sign(
|
||||
key: &SigningKey,
|
||||
config: &MemoryConfig,
|
||||
cache: &MemoryCache,
|
||||
sessions: &SessionCache,
|
||||
knowledge: &KnowledgeCache,
|
||||
wal_applied: Option<WalMark>,
|
||||
) -> StoredSignature {
|
||||
let (manifest, record_hashes) = manifest(config, cache, sessions, knowledge, wal_applied);
|
||||
let signature = key.sign(&manifest.signed_bytes()).to_bytes();
|
||||
StoredSignature {
|
||||
manifest,
|
||||
record_hashes,
|
||||
public_key: key.verifying_key().to_bytes(),
|
||||
signature,
|
||||
}
|
||||
}
|
||||
|
||||
/// Compute the manifest of a store's state.
|
||||
pub fn manifest(
|
||||
config: &MemoryConfig,
|
||||
cache: &MemoryCache,
|
||||
sessions: &SessionCache,
|
||||
knowledge: &KnowledgeCache,
|
||||
wal_applied: Option<WalMark>,
|
||||
) -> (Manifest, Vec<Hash>) {
|
||||
let record_hashes: Vec<Hash> = (0..cache.len()).map(|i| record_hash(cache, i)).collect();
|
||||
let manifest = Manifest {
|
||||
record_count: cache.len() as u64,
|
||||
records_root: merkle_root(&record_hashes),
|
||||
settings: settings_hash(config, wal_applied),
|
||||
sessions: sessions_hash(sessions),
|
||||
graph: graph_hash(knowledge),
|
||||
};
|
||||
(manifest, record_hashes)
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Canonical encoding
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// A SHA-256 over length-prefixed fields, so no two different field lists
|
||||
/// hash the same bytes.
|
||||
struct Fields(Sha256);
|
||||
|
||||
impl Fields {
|
||||
fn new(domain: &str) -> Self {
|
||||
let mut h = Sha256::new();
|
||||
h.update((domain.len() as u64).to_le_bytes());
|
||||
h.update(domain.as_bytes());
|
||||
Self(h)
|
||||
}
|
||||
fn bytes(&mut self, b: &[u8]) -> &mut Self {
|
||||
self.0.update((b.len() as u64).to_le_bytes());
|
||||
self.0.update(b);
|
||||
self
|
||||
}
|
||||
/// Strings as the loader returns them: stored null-padded, so a trailing
|
||||
/// NUL cannot survive a round trip and must not be part of the hash.
|
||||
fn str(&mut self, s: &str) -> &mut Self {
|
||||
self.bytes(s.trim_end_matches('\0').as_bytes())
|
||||
}
|
||||
fn u64(&mut self, v: u64) -> &mut Self {
|
||||
self.0.update(v.to_le_bytes());
|
||||
self
|
||||
}
|
||||
fn f64(&mut self, v: f64) -> &mut Self {
|
||||
self.0.update(v.to_bits().to_le_bytes());
|
||||
self
|
||||
}
|
||||
fn f32(&mut self, v: f32) -> &mut Self {
|
||||
self.0.update(v.to_bits().to_le_bytes());
|
||||
self
|
||||
}
|
||||
fn finish(self) -> Hash {
|
||||
self.0.finalize().into()
|
||||
}
|
||||
}
|
||||
|
||||
/// Everything persisted about record `i`, including its position. The
|
||||
/// embedding is hashed as the cache holds it — for a `float16` store that is
|
||||
/// the half-rounded value the file holds.
|
||||
fn record_hash(cache: &MemoryCache, i: usize) -> Hash {
|
||||
let mut f = Fields::new("clawhdf5-agent/record");
|
||||
f.u64(i as u64).str(&cache.chunks[i]);
|
||||
let emb: Vec<u8> = cache.embeddings[i]
|
||||
.iter()
|
||||
.flat_map(|v| v.to_bits().to_le_bytes())
|
||||
.collect();
|
||||
f.bytes(&emb)
|
||||
.str(&cache.source_channels[i])
|
||||
.f64(cache.timestamps[i])
|
||||
.str(&cache.session_ids[i])
|
||||
.str(&cache.tags[i])
|
||||
.u64(u64::from(cache.tombstones[i]))
|
||||
.f32(cache.activation_weights[i]);
|
||||
f.finish()
|
||||
}
|
||||
|
||||
/// Binary Merkle tree: leaves are the record hashes; a parent hashes its two
|
||||
/// children with a node prefix; an odd node is carried up unchanged.
|
||||
fn merkle_root(leaves: &[Hash]) -> Hash {
|
||||
if leaves.is_empty() {
|
||||
return Fields::new("clawhdf5-agent/merkle-empty").finish();
|
||||
}
|
||||
let mut level: Vec<Hash> = leaves.to_vec();
|
||||
while level.len() > 1 {
|
||||
level = level
|
||||
.chunks(2)
|
||||
.map(|pair| match pair {
|
||||
[l, r] => {
|
||||
let mut h = Sha256::new();
|
||||
h.update([1u8]);
|
||||
h.update(l);
|
||||
h.update(r);
|
||||
h.finalize().into()
|
||||
}
|
||||
[only] => *only,
|
||||
_ => unreachable!(),
|
||||
})
|
||||
.collect();
|
||||
}
|
||||
level[0]
|
||||
}
|
||||
|
||||
fn settings_hash(c: &MemoryConfig, wal_applied: Option<WalMark>) -> Hash {
|
||||
let mut f = Fields::new("clawhdf5-agent/settings");
|
||||
f.str(crate::schema::SCHEMA_VERSION)
|
||||
.str(&c.created_at)
|
||||
.str(&c.agent_id)
|
||||
.str(&c.embedder)
|
||||
.u64(c.embedding_dim as u64)
|
||||
.u64(c.chunk_size as u64)
|
||||
.u64(c.overlap as u64)
|
||||
.u64(u64::from(c.float16))
|
||||
.u64(u64::from(c.compression))
|
||||
.u64(u64::from(c.compression_level))
|
||||
.f32(c.compact_threshold)
|
||||
.f32(c.hebbian_boost)
|
||||
.f32(c.decay_factor)
|
||||
.u64(u64::from(c.wal_enabled))
|
||||
.u64(c.wal_max_entries as u64)
|
||||
.u64(u64::from(c.quantized_index))
|
||||
.u64(c.hnsw_m as u64)
|
||||
.u64(c.hnsw_ef_construction as u64)
|
||||
.u64(c.hnsw_ef_search as u64);
|
||||
// An empty mark is not written to the file, so it must hash as none.
|
||||
match wal_applied.filter(|m| m.len > 0) {
|
||||
Some(m) => f.u64(1).u64(m.len).u64(u64::from(m.crc)),
|
||||
None => f.u64(0),
|
||||
};
|
||||
f.finish()
|
||||
}
|
||||
|
||||
fn sessions_hash(s: &SessionCache) -> Hash {
|
||||
let mut f = Fields::new("clawhdf5-agent/sessions");
|
||||
f.u64(s.entries.len() as u64);
|
||||
for (i, e) in s.entries.iter().enumerate() {
|
||||
f.str(&e.id)
|
||||
.u64(e.start_idx)
|
||||
.u64(e.end_idx)
|
||||
.str(&e.channel)
|
||||
.f64(e.ts)
|
||||
.str(s.summaries.get(i).map(String::as_str).unwrap_or(""));
|
||||
}
|
||||
f.finish()
|
||||
}
|
||||
|
||||
fn graph_hash(k: &KnowledgeCache) -> Hash {
|
||||
let mut f = Fields::new("clawhdf5-agent/graph");
|
||||
f.u64(k.entities.len() as u64);
|
||||
for e in &k.entities {
|
||||
f.u64(e.id)
|
||||
.str(&e.name)
|
||||
.str(&e.entity_type)
|
||||
.u64(e.embedding_idx as u64);
|
||||
}
|
||||
f.u64(k.relations.len() as u64);
|
||||
for r in &k.relations {
|
||||
f.u64(r.src)
|
||||
.u64(r.tgt)
|
||||
.str(&r.relation)
|
||||
.f32(r.weight)
|
||||
.f64(r.ts);
|
||||
}
|
||||
f.u64(k.alias_strings.len() as u64);
|
||||
for (s, id) in k.alias_strings.iter().zip(&k.alias_entity_ids) {
|
||||
f.str(s).u64(*id as u64);
|
||||
}
|
||||
f.finish()
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Verification
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// The outcome of [`verify_store`].
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct VerifyReport {
|
||||
/// The checkpoint carries a signature.
|
||||
pub signed: bool,
|
||||
/// The signature was made by the key the caller trusts.
|
||||
pub key_matches: bool,
|
||||
/// The signature over the stored manifest is valid.
|
||||
pub signature_valid: bool,
|
||||
/// The file's current contents match the signed manifest.
|
||||
pub records_match: bool,
|
||||
pub settings_match: bool,
|
||||
pub sessions_match: bool,
|
||||
pub graph_match: bool,
|
||||
/// Records whose contents differ from what was signed (by position),
|
||||
/// when the stored per-record hashes are themselves authentic.
|
||||
pub changed_records: Vec<usize>,
|
||||
/// Records in the file versus in the signed manifest.
|
||||
pub record_count: u64,
|
||||
pub signed_record_count: u64,
|
||||
/// The public key the checkpoint claims to be signed by.
|
||||
pub public_key: Option<[u8; 32]>,
|
||||
/// Saves in the WAL after the checkpoint: not covered by the signature.
|
||||
pub wal_entries_unsigned: usize,
|
||||
}
|
||||
|
||||
impl VerifyReport {
|
||||
/// Signed by the trusted key, signature valid, and every part of the
|
||||
/// file unchanged since it was signed.
|
||||
pub fn is_valid(&self) -> bool {
|
||||
self.signed
|
||||
&& self.key_matches
|
||||
&& self.signature_valid
|
||||
&& self.records_match
|
||||
&& self.settings_match
|
||||
&& self.sessions_match
|
||||
&& self.graph_match
|
||||
}
|
||||
}
|
||||
|
||||
/// Check a store file against the public key the caller trusts.
|
||||
///
|
||||
/// Reads the checkpoint (not the WAL), recomputes every hash from its
|
||||
/// contents and checks the signature. Never writes.
|
||||
pub fn verify_store(
|
||||
path: &std::path::Path,
|
||||
trusted: &VerifyingKey,
|
||||
) -> Result<VerifyReport, crate::MemoryError> {
|
||||
let file = clawhdf5::File::open(path)
|
||||
.map_err(|e| crate::MemoryError::Hdf5(format!("cannot open {}: {e}", path.display())))?;
|
||||
let (config, cache, sessions, knowledge) = crate::schema::validate_and_load(&file)?;
|
||||
let checkpoint = crate::schema::read_checkpoint_meta(&file);
|
||||
let stored = crate::schema::read_signature(&file)?;
|
||||
let wal_entries_unsigned = count_wal_entries_after(path, checkpoint.wal_applied);
|
||||
|
||||
let (current, current_hashes) = manifest(
|
||||
&config,
|
||||
&cache,
|
||||
&sessions,
|
||||
&knowledge,
|
||||
checkpoint.wal_applied,
|
||||
);
|
||||
|
||||
let Some(stored) = stored else {
|
||||
return Ok(VerifyReport {
|
||||
signed: false,
|
||||
key_matches: false,
|
||||
signature_valid: false,
|
||||
records_match: false,
|
||||
settings_match: false,
|
||||
sessions_match: false,
|
||||
graph_match: false,
|
||||
changed_records: Vec::new(),
|
||||
record_count: current.record_count,
|
||||
signed_record_count: 0,
|
||||
public_key: None,
|
||||
wal_entries_unsigned,
|
||||
});
|
||||
};
|
||||
|
||||
let key_matches = stored.public_key == trusted.to_bytes();
|
||||
let signature_valid = trusted
|
||||
.verify(
|
||||
&stored.manifest.signed_bytes(),
|
||||
&Signature::from_bytes(&stored.signature),
|
||||
)
|
||||
.is_ok();
|
||||
// The stored per-record hashes can localise a change only if they are
|
||||
// the ones that were signed.
|
||||
let hashes_authentic = signature_valid
|
||||
&& stored.record_hashes.len() as u64 == stored.manifest.record_count
|
||||
&& merkle_root(&stored.record_hashes) == stored.manifest.records_root;
|
||||
let changed_records = if hashes_authentic {
|
||||
let n = current_hashes.len().max(stored.record_hashes.len());
|
||||
(0..n)
|
||||
.filter(|&i| current_hashes.get(i) != stored.record_hashes.get(i))
|
||||
.collect()
|
||||
} else {
|
||||
Vec::new()
|
||||
};
|
||||
|
||||
Ok(VerifyReport {
|
||||
signed: true,
|
||||
key_matches,
|
||||
signature_valid,
|
||||
records_match: signature_valid
|
||||
&& current.record_count == stored.manifest.record_count
|
||||
&& current.records_root == stored.manifest.records_root,
|
||||
settings_match: signature_valid && current.settings == stored.manifest.settings,
|
||||
sessions_match: signature_valid && current.sessions == stored.manifest.sessions,
|
||||
graph_match: signature_valid && current.graph == stored.manifest.graph,
|
||||
changed_records,
|
||||
record_count: current.record_count,
|
||||
signed_record_count: stored.manifest.record_count,
|
||||
public_key: Some(stored.public_key),
|
||||
wal_entries_unsigned,
|
||||
})
|
||||
}
|
||||
|
||||
fn count_wal_entries_after(store: &std::path::Path, mark: Option<WalMark>) -> usize {
|
||||
let wal = store.with_extension("h5.wal");
|
||||
if !wal.exists() {
|
||||
return 0;
|
||||
}
|
||||
crate::wal::WalFile::read_entries_for_migration(&wal, mark)
|
||||
.map(|e| e.len())
|
||||
.unwrap_or(0)
|
||||
}
|
||||
|
||||
/// A new random signing key from the operating system's RNG.
|
||||
pub fn generate_key() -> SigningKey {
|
||||
SigningKey::generate(&mut rand_core::OsRng)
|
||||
}
|
||||
|
||||
/// Hex encoding for keys and signatures in attributes and the CLI.
|
||||
pub fn to_hex(bytes: &[u8]) -> String {
|
||||
bytes.iter().map(|b| format!("{b:02x}")).collect()
|
||||
}
|
||||
|
||||
/// Parse hex into exactly `N` bytes.
|
||||
pub fn from_hex<const N: usize>(s: &str) -> Option<[u8; N]> {
|
||||
let s = s.trim();
|
||||
if s.len() != 2 * N {
|
||||
return None;
|
||||
}
|
||||
let mut out = [0u8; N];
|
||||
for (i, byte) in out.iter_mut().enumerate() {
|
||||
*byte = u8::from_str_radix(&s[2 * i..2 * i + 2], 16).ok()?;
|
||||
}
|
||||
Some(out)
|
||||
}
|
||||
@@ -36,7 +36,7 @@ pub fn write_to_disk_with_mark(
|
||||
) -> Result<(), MemoryError> {
|
||||
let meta = schema::CheckpointMeta {
|
||||
wal_applied,
|
||||
ann_generation: None,
|
||||
..schema::CheckpointMeta::default()
|
||||
};
|
||||
write_to_disk_with_meta(path, config, cache, sessions, knowledge, &meta)
|
||||
}
|
||||
@@ -50,7 +50,21 @@ pub fn write_to_disk_with_meta(
|
||||
knowledge: &KnowledgeCache,
|
||||
checkpoint: &schema::CheckpointMeta,
|
||||
) -> Result<(), MemoryError> {
|
||||
let bytes = schema::build_hdf5_file_with_meta(config, cache, sessions, knowledge, checkpoint)?;
|
||||
write_to_disk_signed(path, config, cache, sessions, knowledge, checkpoint, None)
|
||||
}
|
||||
|
||||
/// [`write_to_disk_with_meta`] with a signed manifest of the contents.
|
||||
pub fn write_to_disk_signed(
|
||||
path: &Path,
|
||||
config: &MemoryConfig,
|
||||
cache: &MemoryCache,
|
||||
sessions: &SessionCache,
|
||||
knowledge: &KnowledgeCache,
|
||||
checkpoint: &schema::CheckpointMeta,
|
||||
signature: Option<&crate::signing::StoredSignature>,
|
||||
) -> Result<(), MemoryError> {
|
||||
let bytes =
|
||||
schema::build_hdf5_file_signed(config, cache, sessions, knowledge, checkpoint, signature)?;
|
||||
|
||||
if bytes.is_empty() {
|
||||
return Err(MemoryError::Hdf5("build_hdf5_file produced 0 bytes".into()));
|
||||
|
||||
@@ -258,3 +258,43 @@ fn an_existing_f32_store_stays_f32() {
|
||||
assert_eq!(&values[..before.1.len()], before.1.as_slice());
|
||||
assert_eq!(&values[before.1.len()..], odd.as_slice());
|
||||
}
|
||||
|
||||
/// `Group::attrs` leaves out an attribute it cannot decode. A store whose
|
||||
/// `float16` setting is unreadable must not open as `float16 = false` (or with
|
||||
/// any other default in place of a setting it has): it is an error.
|
||||
#[test]
|
||||
fn unreadable_meta_attribute_fails_open_instead_of_defaulting() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let path = dir.path().join("store.h5");
|
||||
{
|
||||
let mut m = HDF5Memory::create(config(&dir, "store.h5", true)).unwrap();
|
||||
m.save(entry(1)).unwrap();
|
||||
m.flush_wal().unwrap();
|
||||
}
|
||||
assert!(HDF5Memory::open_read_only(&path).is_ok());
|
||||
|
||||
// Give the `float16` attribute message an unknown version (the name is
|
||||
// at +8 in a version-1 message and +9 in a version-3 one).
|
||||
let mut bytes = std::fs::read(&path).unwrap();
|
||||
let name = b"float16\0";
|
||||
let mut hit = false;
|
||||
let positions: Vec<usize> = (9..bytes.len() - name.len())
|
||||
.filter(|&p| &bytes[p..p + name.len()] == name)
|
||||
.collect();
|
||||
for pos in positions {
|
||||
for (back, version) in [(8, 1u8), (9, 3u8)] {
|
||||
if bytes[pos - back] == version {
|
||||
bytes[pos - back] = 0x7f;
|
||||
hit = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
assert!(hit, "float16 attribute message not found");
|
||||
std::fs::write(&path, &bytes).unwrap();
|
||||
|
||||
match HDF5Memory::open_read_only(&path) {
|
||||
Err(MemoryError::Schema(msg)) => assert!(msg.contains("/meta"), "{msg}"),
|
||||
Err(e) => panic!("unexpected error: {e}"),
|
||||
Ok(_) => panic!("store opened with an unreadable float16 setting"),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -92,3 +92,64 @@ print(len(names))
|
||||
assert!(n >= 10, "only {n} datasets");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_edit_made_with_h5py_breaks_the_signature_and_names_the_record() {
|
||||
if !h5py_available() {
|
||||
assert!(
|
||||
std::env::var("CLAWHDF5_REQUIRE_INTEROP").as_deref() != Ok("1"),
|
||||
"CLAWHDF5_REQUIRE_INTEROP=1 but python3 with h5py is not available"
|
||||
);
|
||||
eprintln!("SKIP: python3 with h5py not available");
|
||||
return;
|
||||
}
|
||||
use clawhdf5_agent::signing::SigningKey;
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let path = dir.path().join("signed.h5");
|
||||
let key = SigningKey::from_bytes(&[42; 32]);
|
||||
let mut m = HDF5Memory::create(MemoryConfig::new(path.clone(), "agent", 8)).unwrap();
|
||||
m.set_signing_key(key.clone());
|
||||
m.save_batch(
|
||||
(0..10)
|
||||
.map(|i| MemoryEntry {
|
||||
chunk: format!("memory {i}"),
|
||||
embedding: (0..8).map(|j| ((i * 8 + j) as f32).cos()).collect(),
|
||||
source_channel: "test".into(),
|
||||
timestamp: i as f64,
|
||||
session_id: "s".into(),
|
||||
tags: String::new(),
|
||||
})
|
||||
.collect(),
|
||||
)
|
||||
.unwrap();
|
||||
drop(m);
|
||||
assert!(
|
||||
HDF5Memory::verify(&path, &key.verifying_key())
|
||||
.unwrap()
|
||||
.is_valid()
|
||||
);
|
||||
|
||||
// Someone edits one timestamp in place with h5py.
|
||||
let script = format!(
|
||||
r#"
|
||||
import h5py
|
||||
with h5py.File("{}", "r+") as f:
|
||||
ts = f["memory/timestamps"]
|
||||
ts[3] = 12345.0
|
||||
"#,
|
||||
path.display()
|
||||
);
|
||||
let out = Command::new(python())
|
||||
.args(["-c", &script])
|
||||
.output()
|
||||
.unwrap();
|
||||
assert!(
|
||||
out.status.success(),
|
||||
"{}",
|
||||
String::from_utf8_lossy(&out.stderr)
|
||||
);
|
||||
|
||||
let r = HDF5Memory::verify(&path, &key.verifying_key()).unwrap();
|
||||
assert!(r.signature_valid && !r.is_valid(), "{r:?}");
|
||||
assert_eq!(r.changed_records, vec![3]);
|
||||
}
|
||||
|
||||
@@ -0,0 +1,330 @@
|
||||
//! Ed25519-signed checkpoints: `HDF5Memory::set_signing_key` and
|
||||
//! `HDF5Memory::verify`.
|
||||
|
||||
use std::path::Path;
|
||||
|
||||
use clawhdf5_agent::signing::{SigningKey, VerifyReport, VerifyingKey};
|
||||
use clawhdf5_agent::storage;
|
||||
use clawhdf5_agent::{AgentMemory, HDF5Memory, MemoryConfig, MemoryEntry, MemoryError, schema};
|
||||
use tempfile::TempDir;
|
||||
|
||||
const DIM: usize = 16;
|
||||
|
||||
fn key(seed: u8) -> SigningKey {
|
||||
SigningKey::from_bytes(&[seed; 32])
|
||||
}
|
||||
|
||||
fn entry(i: usize, chunk: &str) -> MemoryEntry {
|
||||
MemoryEntry {
|
||||
chunk: chunk.to_string(),
|
||||
embedding: (0..DIM)
|
||||
.map(|j| ((i * DIM + j) as f32 * 0.37).sin())
|
||||
.collect(),
|
||||
source_channel: "chat".into(),
|
||||
timestamp: 1_700_000_000.0 + i as f64,
|
||||
session_id: format!("s{}", i % 3),
|
||||
tags: format!("t{i}"),
|
||||
}
|
||||
}
|
||||
|
||||
/// Awkward strings on purpose: they must hash the same after a round trip.
|
||||
const TEXTS: [&str; 6] = [
|
||||
"plain text",
|
||||
"ünïcödé — 日本語 🙂",
|
||||
"",
|
||||
"trailing spaces ",
|
||||
"tab\tand\nnewline",
|
||||
"x",
|
||||
];
|
||||
|
||||
fn signed_store(dir: &TempDir, float16: bool, k: &SigningKey) -> std::path::PathBuf {
|
||||
let mut cfg = MemoryConfig::new(dir.path().join("s.h5"), "agent", DIM);
|
||||
cfg.float16 = float16;
|
||||
let path = cfg.path.clone();
|
||||
let mut m = HDF5Memory::create(cfg).unwrap();
|
||||
m.set_signing_key(k.clone());
|
||||
let entries = (0..30).map(|i| entry(i, TEXTS[i % TEXTS.len()])).collect();
|
||||
m.save_batch(entries).unwrap();
|
||||
// Some graph and a deleted record, so every part of the manifest is used.
|
||||
let a = m.knowledge_mut().add_entity("Alice", "person", 0);
|
||||
let b = m.knowledge_mut().add_entity("Acme", "org", -1);
|
||||
m.knowledge_mut().add_relation(a, b, "works_at", 0.75);
|
||||
m.sessions_mut()
|
||||
.add_at("s0", 0, 9, "chat", "first session", 1_700_000_000.0);
|
||||
m.delete(4).unwrap();
|
||||
m.flush_wal().unwrap();
|
||||
path
|
||||
}
|
||||
|
||||
fn verify(path: &Path, k: &SigningKey) -> VerifyReport {
|
||||
HDF5Memory::verify(path, &k.verifying_key()).unwrap()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_signed_store_verifies_through_reopen_and_checkpoint_cycles() {
|
||||
for float16 in [true, false] {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let k = key(7);
|
||||
let path = signed_store(&dir, float16, &k);
|
||||
let r = verify(&path, &k);
|
||||
assert!(r.is_valid(), "float16={float16}: {r:?}");
|
||||
assert_eq!(r.public_key, Some(k.verifying_key().to_bytes()));
|
||||
assert_eq!(r.record_count, 30);
|
||||
assert!(r.changed_records.is_empty());
|
||||
|
||||
// Reopen, change nothing, checkpoint again (with the key): still valid.
|
||||
for _ in 0..3 {
|
||||
let mut m = HDF5Memory::open(&path).unwrap();
|
||||
assert!(m.is_signed());
|
||||
m.set_signing_key(k.clone());
|
||||
m.flush_wal().unwrap();
|
||||
drop(m);
|
||||
assert!(verify(&path, &k).is_valid());
|
||||
}
|
||||
// And after real changes, re-signed.
|
||||
let mut m = HDF5Memory::open(&path).unwrap();
|
||||
m.set_signing_key(k.clone());
|
||||
m.save(entry(99, "added later")).unwrap();
|
||||
m.hybrid_search(&entry(1, "").embedding, "text", 0.4, 0.6, 5);
|
||||
m.flush_wal().unwrap();
|
||||
drop(m);
|
||||
let r = verify(&path, &k);
|
||||
assert!(r.is_valid(), "{r:?}");
|
||||
assert_eq!(r.record_count, 31);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_signed_store_refuses_to_checkpoint_without_its_key() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let k = key(1);
|
||||
let path = signed_store(&dir, true, &k);
|
||||
|
||||
let mut m = HDF5Memory::open(&path).unwrap();
|
||||
m.save(entry(50, "pending")).unwrap();
|
||||
match m.flush_wal() {
|
||||
Err(MemoryError::SigningKeyRequired(msg)) => assert!(msg.contains("signed"), "{msg}"),
|
||||
other => panic!("expected SigningKeyRequired, got {other:?}"),
|
||||
}
|
||||
// The file is untouched and still valid; the save is still in the WAL.
|
||||
let r = verify(&path, &k);
|
||||
assert!(r.is_valid());
|
||||
assert_eq!(r.wal_entries_unsigned, 1);
|
||||
|
||||
// Supplying the key lets the checkpoint through, signed.
|
||||
m.set_signing_key(k.clone());
|
||||
m.flush_wal().unwrap();
|
||||
drop(m);
|
||||
let r = verify(&path, &k);
|
||||
assert!(r.is_valid());
|
||||
assert_eq!((r.record_count, r.wal_entries_unsigned), (31, 0));
|
||||
|
||||
// Removing the signature on purpose writes it unsigned.
|
||||
let mut m = HDF5Memory::open(&path).unwrap();
|
||||
m.remove_signature();
|
||||
m.flush_wal().unwrap();
|
||||
drop(m);
|
||||
let r = verify(&path, &k);
|
||||
assert!(!r.signed && !r.is_valid());
|
||||
assert!(!HDF5Memory::open(&path).unwrap().is_signed());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_wrong_key_does_not_verify_and_a_new_key_re_signs() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let (a, b) = (key(1), key(2));
|
||||
let path = signed_store(&dir, true, &a);
|
||||
let r = verify(&path, &b);
|
||||
assert!(r.signed && !r.key_matches && !r.signature_valid && !r.is_valid());
|
||||
|
||||
let mut m = HDF5Memory::open(&path).unwrap();
|
||||
m.set_signing_key(b.clone());
|
||||
m.flush_wal().unwrap();
|
||||
drop(m);
|
||||
assert!(verify(&path, &b).is_valid());
|
||||
assert!(!verify(&path, &a).is_valid());
|
||||
}
|
||||
|
||||
/// Rewrite the store with changed contents but the *old* signature — what
|
||||
/// someone with write access to the file, but not the key, can do.
|
||||
fn tamper(path: &Path, change: impl FnOnce(&mut Tampered)) {
|
||||
let file = clawhdf5::File::open(path).unwrap();
|
||||
let (config, cache, sessions, knowledge) = schema::validate_and_load(&file).unwrap();
|
||||
let checkpoint = schema::read_checkpoint_meta(&file);
|
||||
let signature = schema::read_signature(&file).unwrap().unwrap();
|
||||
drop(file);
|
||||
let mut t = Tampered {
|
||||
config,
|
||||
cache,
|
||||
sessions,
|
||||
knowledge,
|
||||
};
|
||||
change(&mut t);
|
||||
storage::write_to_disk_signed(
|
||||
path,
|
||||
&t.config,
|
||||
&t.cache,
|
||||
&t.sessions,
|
||||
&t.knowledge,
|
||||
&checkpoint,
|
||||
Some(&signature),
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
struct Tampered {
|
||||
config: MemoryConfig,
|
||||
cache: clawhdf5_agent::cache::MemoryCache,
|
||||
sessions: clawhdf5_agent::SessionCache,
|
||||
knowledge: clawhdf5_agent::knowledge::KnowledgeCache,
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn every_kind_of_edit_is_detected_and_located() {
|
||||
let k = key(3);
|
||||
type Edit = Box<dyn FnOnce(&mut Tampered)>;
|
||||
type Case = (&'static str, Edit, fn(&VerifyReport) -> bool);
|
||||
let cases: Vec<Case> = vec![
|
||||
(
|
||||
"record text",
|
||||
Box::new(|t: &mut Tampered| t.cache.chunks[7] = "rewritten".into()),
|
||||
|r| !r.records_match && r.changed_records == vec![7],
|
||||
),
|
||||
(
|
||||
"one embedding value",
|
||||
Box::new(|t: &mut Tampered| {
|
||||
let mut e = t.cache.embeddings[12].to_vec();
|
||||
e[3] = 0.5;
|
||||
t.cache.embeddings.set(12, &e);
|
||||
}),
|
||||
|r| r.changed_records == vec![12],
|
||||
),
|
||||
(
|
||||
"undelete",
|
||||
Box::new(|t: &mut Tampered| t.cache.tombstones[4] = 0),
|
||||
|r| r.changed_records == vec![4],
|
||||
),
|
||||
(
|
||||
"timestamp",
|
||||
Box::new(|t: &mut Tampered| t.cache.timestamps[20] += 1.0),
|
||||
|r| r.changed_records == vec![20],
|
||||
),
|
||||
(
|
||||
"record appended",
|
||||
Box::new(|t: &mut Tampered| {
|
||||
t.cache.push(
|
||||
"new".into(),
|
||||
vec![0.1; DIM],
|
||||
"x".into(),
|
||||
1.0,
|
||||
"s".into(),
|
||||
"".into(),
|
||||
);
|
||||
}),
|
||||
|r| !r.records_match && r.changed_records == vec![30] && r.record_count == 31,
|
||||
),
|
||||
(
|
||||
"setting",
|
||||
Box::new(|t: &mut Tampered| t.config.agent_id = "someone-else".into()),
|
||||
|r| !r.settings_match && r.records_match,
|
||||
),
|
||||
(
|
||||
"session summary",
|
||||
Box::new(|t: &mut Tampered| t.sessions.summaries[0] = "edited".into()),
|
||||
|r| !r.sessions_match && r.records_match,
|
||||
),
|
||||
(
|
||||
"graph edge",
|
||||
Box::new(|t: &mut Tampered| t.knowledge.relations[0].weight = 1.0),
|
||||
|r| !r.graph_match && r.records_match,
|
||||
),
|
||||
];
|
||||
for (name, edit, check) in cases {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let path = signed_store(&dir, true, &k);
|
||||
tamper(&path, edit);
|
||||
let r = verify(&path, &k);
|
||||
assert!(
|
||||
r.signed && r.key_matches && r.signature_valid,
|
||||
"{name}: {r:?}"
|
||||
);
|
||||
assert!(!r.is_valid(), "{name}: edit not detected: {r:?}");
|
||||
assert!(check(&r), "{name}: {r:?}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_forged_manifest_fails_the_signature() {
|
||||
// Recomputing the hashes for tampered contents does not help without the
|
||||
// key: the signature no longer matches the manifest.
|
||||
let dir = TempDir::new().unwrap();
|
||||
let k = key(5);
|
||||
let path = signed_store(&dir, true, &k);
|
||||
let file = clawhdf5::File::open(&path).unwrap();
|
||||
let (config, mut cache, sessions, knowledge) = schema::validate_and_load(&file).unwrap();
|
||||
let checkpoint = schema::read_checkpoint_meta(&file);
|
||||
let mut sig = schema::read_signature(&file).unwrap().unwrap();
|
||||
drop(file);
|
||||
cache.chunks[0] = "forged".into();
|
||||
// Re-sign with an attacker key, then splice the victim's public key back.
|
||||
let forged = clawhdf5_agent::signing::sign(
|
||||
&key(66),
|
||||
&config,
|
||||
&cache,
|
||||
&sessions,
|
||||
&knowledge,
|
||||
checkpoint.wal_applied,
|
||||
);
|
||||
sig.manifest = forged.manifest;
|
||||
sig.record_hashes = forged.record_hashes;
|
||||
storage::write_to_disk_signed(
|
||||
&path,
|
||||
&config,
|
||||
&cache,
|
||||
&sessions,
|
||||
&knowledge,
|
||||
&checkpoint,
|
||||
Some(&sig),
|
||||
)
|
||||
.unwrap();
|
||||
let r = verify(&path, &k);
|
||||
assert!(
|
||||
r.key_matches && !r.signature_valid && !r.is_valid(),
|
||||
"{r:?}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_unsigned_store_reports_unsigned() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let mut m = HDF5Memory::create(MemoryConfig::new(dir.path().join("u.h5"), "a", DIM)).unwrap();
|
||||
m.save_batch(vec![entry(0, "hello")]).unwrap();
|
||||
drop(m);
|
||||
let r = HDF5Memory::verify(&dir.path().join("u.h5"), &VerifyingKey::from(&key(1))).unwrap();
|
||||
assert!(!r.signed && !r.is_valid());
|
||||
assert_eq!(r.record_count, 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn nul_bytes_in_text_still_verify() {
|
||||
// Strings are stored null-padded; the hash must follow what a reopened
|
||||
// store actually holds, or an untouched store would fail to verify.
|
||||
let dir = TempDir::new().unwrap();
|
||||
let k = key(9);
|
||||
let mut m = HDF5Memory::create(MemoryConfig::new(dir.path().join("n.h5"), "a", DIM)).unwrap();
|
||||
m.set_signing_key(k.clone());
|
||||
m.save_batch(vec![
|
||||
entry(0, "inner\0nul"),
|
||||
entry(1, "trailing nul\0"),
|
||||
entry(2, "\0leading"),
|
||||
])
|
||||
.unwrap();
|
||||
drop(m);
|
||||
let r = verify(&dir.path().join("n.h5"), &k);
|
||||
assert!(r.is_valid(), "{r:?}");
|
||||
let m = HDF5Memory::open(&dir.path().join("n.h5")).unwrap();
|
||||
eprintln!(
|
||||
"reloaded: {:?}",
|
||||
(0..3).map(|i| m.get_chunk(i)).collect::<Vec<_>>()
|
||||
);
|
||||
}
|
||||
@@ -13,7 +13,7 @@ use clawhdf5_format::filter_pipeline::FilterPipeline;
|
||||
use clawhdf5_format::group_v2::resolve_path_any;
|
||||
use clawhdf5_format::message_type::MessageType;
|
||||
use clawhdf5_format::object_header::ObjectHeader;
|
||||
use clawhdf5_format::signature::find_signature;
|
||||
use clawhdf5_format::signature::split_user_block;
|
||||
use clawhdf5_format::superblock::Superblock;
|
||||
use clawhdf5_io::FileWriter as IoFileWriter;
|
||||
|
||||
@@ -861,8 +861,9 @@ impl HnswIndex {
|
||||
/// The HDF5 data must contain the `/ann/vectors`, `/ann/graph_layer_*`,
|
||||
/// and `/ann/config` datasets as produced by [`to_hdf5_bytes`].
|
||||
pub fn load_from_hdf5(data: &[u8]) -> Result<Self, FormatError> {
|
||||
let sig_offset = find_signature(data)?;
|
||||
let sb = Superblock::parse(data, sig_offset)?;
|
||||
// Addresses are relative to the superblock: skip any user block.
|
||||
let (_, data) = split_user_block(data)?;
|
||||
let sb = Superblock::parse(data, 0)?;
|
||||
|
||||
// Read config dataset and its attributes
|
||||
let config_attrs = read_dataset_attrs(data, &sb, "ann/config")?;
|
||||
|
||||
@@ -21,6 +21,7 @@
|
||||
//! cargo run --release -p clawhdf5-bench --bin search_harness -- --ann-only --uniform
|
||||
//! cargo run --release -p clawhdf5-bench --bin search_harness -- --float16-study --full
|
||||
//! cargo run --release -p clawhdf5-bench --bin search_harness -- --options-study --full
|
||||
//! cargo run --release -p clawhdf5-bench --bin search_harness -- --signing-study --full
|
||||
//! ```
|
||||
|
||||
use std::time::{Duration, Instant};
|
||||
@@ -488,6 +489,81 @@ fn bench_end_to_end(n: usize, json: &mut Vec<serde_json::Value>) {
|
||||
}));
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Signing study: what does an Ed25519-signed checkpoint cost?
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// `--signing-study`: checkpoint time unsigned vs signed, `verify` time, and
|
||||
/// the file-size cost of the stored per-record hashes. Default store
|
||||
/// settings (float16, int8 index). Medians of five checkpoints / three
|
||||
/// verifies.
|
||||
fn signing_study(n: usize) {
|
||||
use clawhdf5_agent::signing::SigningKey;
|
||||
let data = make_dataset(n, 0x516 ^ n as u64);
|
||||
let mut rng = Rng(9);
|
||||
let entries: Vec<MemoryEntry> = data
|
||||
.vectors
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(i, v)| MemoryEntry {
|
||||
chunk: text_for(data.cluster_of[i], i, &mut rng),
|
||||
embedding: v.clone(),
|
||||
source_channel: "bench".into(),
|
||||
timestamp: i as f64,
|
||||
session_id: format!("s{}", i % 50),
|
||||
tags: format!("t{i}"),
|
||||
})
|
||||
.collect();
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let path = dir.path().join("sign.h5");
|
||||
let mut mem = HDF5Memory::create(MemoryConfig::new(path.clone(), "bench", DIM)).unwrap();
|
||||
mem.save_batch(entries).unwrap();
|
||||
std::hint::black_box(mem.hybrid_search(&data.queries[0], "", 1.0, 0.0, K));
|
||||
|
||||
let median = |mut v: Vec<Duration>| {
|
||||
v.sort();
|
||||
v[v.len() / 2]
|
||||
};
|
||||
let checkpoint = |mem: &mut HDF5Memory| {
|
||||
median(
|
||||
(0..5)
|
||||
.map(|_| {
|
||||
let t = Instant::now();
|
||||
mem.flush_wal().unwrap();
|
||||
t.elapsed()
|
||||
})
|
||||
.collect(),
|
||||
)
|
||||
};
|
||||
let unsigned = checkpoint(&mut mem);
|
||||
let unsigned_bytes = std::fs::metadata(&path).unwrap().len();
|
||||
let key = SigningKey::from_bytes(&[7; 32]);
|
||||
mem.set_signing_key(key.clone());
|
||||
let signed = checkpoint(&mut mem);
|
||||
let signed_bytes = std::fs::metadata(&path).unwrap().len();
|
||||
drop(mem);
|
||||
let vk = key.verifying_key();
|
||||
let verify = median(
|
||||
(0..3)
|
||||
.map(|_| {
|
||||
let t = Instant::now();
|
||||
let r = HDF5Memory::verify(&path, &vk).unwrap();
|
||||
let d = t.elapsed();
|
||||
assert!(r.is_valid());
|
||||
d
|
||||
})
|
||||
.collect(),
|
||||
);
|
||||
println!(
|
||||
"| {n} | {:.1} | {:.1} | {:+.1} | {:.1} | {:+.2} |",
|
||||
millis(unsigned),
|
||||
millis(signed),
|
||||
millis(signed) - millis(unsigned),
|
||||
millis(verify),
|
||||
(signed_bytes as f64 - unsigned_bytes as f64) / (1024.0 * 1024.0),
|
||||
);
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Search options study: source filters, re-ranking, confidence rejection
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -960,6 +1036,21 @@ fn main() {
|
||||
}
|
||||
return;
|
||||
}
|
||||
if args.iter().any(|a| a == "--signing-study") {
|
||||
println!("## Signed checkpoints ({DIM}-dim, float16, int8 index)\n");
|
||||
println!(
|
||||
"| N | checkpoint ms, unsigned | checkpoint ms, signed | signing adds ms | verify ms | file MiB added |"
|
||||
);
|
||||
println!("|---:|---:|---:|---:|---:|---:|");
|
||||
for &n in if full {
|
||||
&[1_000, 10_000, 100_000][..]
|
||||
} else {
|
||||
&[1_000, 10_000][..]
|
||||
} {
|
||||
signing_study(n);
|
||||
}
|
||||
return;
|
||||
}
|
||||
if args.iter().any(|a| a == "--options-study") {
|
||||
println!("## Search options ({DIM}-dim, k = {K}, Hebbian boost off)\n");
|
||||
println!("| N | options | filtered recall@10 | p50 ms | p99 ms |");
|
||||
|
||||
+126
-16
@@ -1,15 +1,22 @@
|
||||
use std::path::PathBuf;
|
||||
use std::path::{Path, PathBuf};
|
||||
|
||||
use clap::{Parser, Subcommand};
|
||||
use clawhdf5_agent::signing::{self, SigningKey, VerifyingKey};
|
||||
use clawhdf5_agent::{AgentMemory, HDF5Memory, MemoryConfig, MemoryEntry};
|
||||
|
||||
/// ClawhDF5 — HDF5-backed cognitive memory for AI agents
|
||||
#[derive(Parser)]
|
||||
#[command(name = "clawhdf5", version, about)]
|
||||
struct Cli {
|
||||
/// Path to the .h5 memory file
|
||||
/// Path to the .h5 memory file (not needed for `keygen`)
|
||||
#[arg(short, long, env = "CLAWHDF5_PATH")]
|
||||
path: PathBuf,
|
||||
path: Option<PathBuf>,
|
||||
|
||||
/// File holding an Ed25519 signing key (64 hex characters, from
|
||||
/// `keygen`). Every checkpoint this command makes is then signed; a
|
||||
/// signed store refuses to checkpoint without it.
|
||||
#[arg(long, env = "CLAWHDF5_SIGNING_KEY", global = true)]
|
||||
signing_key: Option<PathBuf>,
|
||||
|
||||
#[command(subcommand)]
|
||||
command: Commands,
|
||||
@@ -91,6 +98,38 @@ enum Commands {
|
||||
/// Destination path
|
||||
dest: PathBuf,
|
||||
},
|
||||
/// Generate an Ed25519 signing key for signed checkpoints
|
||||
Keygen {
|
||||
/// Where to write the secret key (created new, owner-only on Unix)
|
||||
#[arg(long)]
|
||||
out: PathBuf,
|
||||
},
|
||||
/// Verify a signed store against a public key; exit status 2 if not valid
|
||||
Verify {
|
||||
/// The trusted public key: 64 hex characters, or a file holding them
|
||||
#[arg(long)]
|
||||
public_key: String,
|
||||
},
|
||||
}
|
||||
|
||||
fn read_signing_key(path: &Path) -> Result<SigningKey, Box<dyn std::error::Error>> {
|
||||
let text = std::fs::read_to_string(path)
|
||||
.map_err(|e| format!("cannot read signing key {}: {e}", path.display()))?;
|
||||
let bytes = signing::from_hex::<32>(&text)
|
||||
.ok_or_else(|| format!("{} is not a 64-hex-character key", path.display()))?;
|
||||
Ok(SigningKey::from_bytes(&bytes))
|
||||
}
|
||||
|
||||
/// Open for writing, with the signing key applied if one was given.
|
||||
fn open_writable(
|
||||
path: &Path,
|
||||
key: &Option<SigningKey>,
|
||||
) -> Result<HDF5Memory, Box<dyn std::error::Error>> {
|
||||
let mut mem = HDF5Memory::open(path)?;
|
||||
if let Some(k) = key {
|
||||
mem.set_signing_key(k.clone());
|
||||
}
|
||||
Ok(mem)
|
||||
}
|
||||
|
||||
fn main() {
|
||||
@@ -103,6 +142,37 @@ fn main() {
|
||||
}
|
||||
|
||||
fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
||||
if let Commands::Keygen { out } = &cli.command {
|
||||
let key = signing::generate_key();
|
||||
let mut opts = std::fs::OpenOptions::new();
|
||||
opts.write(true).create_new(true);
|
||||
#[cfg(unix)]
|
||||
{
|
||||
use std::os::unix::fs::OpenOptionsExt;
|
||||
opts.mode(0o600);
|
||||
}
|
||||
use std::io::Write;
|
||||
let mut f = opts
|
||||
.open(out)
|
||||
.map_err(|e| format!("cannot create {}: {e}", out.display()))?;
|
||||
writeln!(f, "{}", signing::to_hex(&key.to_bytes()))?;
|
||||
let j = serde_json::json!({
|
||||
"status": "generated",
|
||||
"secret_key_file": out.display().to_string(),
|
||||
"public_key": signing::to_hex(&key.verifying_key().to_bytes()),
|
||||
});
|
||||
println!("{}", serde_json::to_string_pretty(&j)?);
|
||||
return Ok(());
|
||||
}
|
||||
let path = cli
|
||||
.path
|
||||
.clone()
|
||||
.ok_or("--path (or CLAWHDF5_PATH) is required")?;
|
||||
let key = cli
|
||||
.signing_key
|
||||
.as_deref()
|
||||
.map(read_signing_key)
|
||||
.transpose()?;
|
||||
match cli.command {
|
||||
Commands::Create {
|
||||
agent_id,
|
||||
@@ -113,7 +183,7 @@ fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
||||
f32,
|
||||
float16: _,
|
||||
} => {
|
||||
let mut config = MemoryConfig::new(cli.path.clone(), &agent_id, dim);
|
||||
let mut config = MemoryConfig::new(path.clone(), &agent_id, dim);
|
||||
config.wal_enabled = wal;
|
||||
// As with --f32-index: only ever switch the library default off.
|
||||
if f32 {
|
||||
@@ -127,15 +197,21 @@ fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
||||
config.quantized_index = false;
|
||||
}
|
||||
let config_quantized = config.quantized_index;
|
||||
let mem = HDF5Memory::create(config)?;
|
||||
let mut mem = HDF5Memory::create(config)?;
|
||||
// Sign straight away, so the store is never on disk unsigned.
|
||||
if let Some(k) = &key {
|
||||
mem.set_signing_key(k.clone());
|
||||
mem.flush_wal()?;
|
||||
}
|
||||
let j = serde_json::json!({
|
||||
"status": "created",
|
||||
"path": cli.path.display().to_string(),
|
||||
"path": path.display().to_string(),
|
||||
"agent_id": agent_id,
|
||||
"embedding_dim": dim,
|
||||
"wal_enabled": wal,
|
||||
"quantized_index": config_quantized,
|
||||
"float16": config_float16,
|
||||
"signed": mem.is_signed(),
|
||||
"count": mem.count(),
|
||||
});
|
||||
println!("{}", serde_json::to_string_pretty(&j)?);
|
||||
@@ -152,7 +228,7 @@ fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
||||
}
|
||||
};
|
||||
let entry: MemoryEntry = serde_json::from_str(&input)?;
|
||||
let mut mem = HDF5Memory::open(&cli.path)?;
|
||||
let mut mem = open_writable(&path, &key)?;
|
||||
let idx = mem.save(entry)?;
|
||||
let j = serde_json::json!({ "status": "saved", "index": idx, "count": mem.count() });
|
||||
println!("{}", serde_json::to_string(&j)?);
|
||||
@@ -166,7 +242,7 @@ fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
||||
keyword_weight,
|
||||
} => {
|
||||
let emb: Vec<f32> = serde_json::from_str(&embedding)?;
|
||||
let mut mem = HDF5Memory::open(&cli.path)?;
|
||||
let mut mem = open_writable(&path, &key)?;
|
||||
let results = mem.hybrid_search(&emb, &query, vector_weight, keyword_weight, top_k);
|
||||
let j: Vec<serde_json::Value> = results
|
||||
.iter()
|
||||
@@ -184,7 +260,7 @@ fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
||||
}
|
||||
|
||||
Commands::Recall { index } => {
|
||||
let mem = HDF5Memory::open_read_only(&cli.path)?;
|
||||
let mem = HDF5Memory::open_read_only(&path)?;
|
||||
match mem.get_chunk(index) {
|
||||
Some(content) => {
|
||||
let j = serde_json::json!({ "index": index, "chunk": content });
|
||||
@@ -198,22 +274,23 @@ fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
||||
}
|
||||
|
||||
Commands::Stats => {
|
||||
let mem = HDF5Memory::open_read_only(&cli.path)?;
|
||||
let mem = HDF5Memory::open_read_only(&path)?;
|
||||
let cfg = mem.config();
|
||||
let j = serde_json::json!({
|
||||
"path": cli.path.display().to_string(),
|
||||
"path": path.display().to_string(),
|
||||
"agent_id": cfg.agent_id,
|
||||
"embedding_dim": cfg.embedding_dim,
|
||||
"count": mem.count(),
|
||||
"active": mem.count_active(),
|
||||
"wal_enabled": cfg.wal_enabled,
|
||||
"wal_pending": mem.wal_pending_count(),
|
||||
"signed": mem.is_signed(),
|
||||
});
|
||||
println!("{}", serde_json::to_string_pretty(&j)?);
|
||||
}
|
||||
|
||||
Commands::FlushWal => {
|
||||
let mut mem = HDF5Memory::open(&cli.path)?;
|
||||
let mut mem = open_writable(&path, &key)?;
|
||||
let before = mem.wal_pending_count();
|
||||
mem.flush_wal()?;
|
||||
let j = serde_json::json!({
|
||||
@@ -225,7 +302,7 @@ fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
||||
}
|
||||
|
||||
Commands::AgentsMd { output } => {
|
||||
let mem = HDF5Memory::open_read_only(&cli.path)?;
|
||||
let mem = HDF5Memory::open_read_only(&path)?;
|
||||
let md = mem.generate_agents_md();
|
||||
match output {
|
||||
Some(p) => {
|
||||
@@ -237,7 +314,7 @@ fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
||||
}
|
||||
|
||||
Commands::Export => {
|
||||
let mem = HDF5Memory::open_read_only(&cli.path)?;
|
||||
let mem = HDF5Memory::open_read_only(&path)?;
|
||||
for i in 0..mem.count() {
|
||||
if let Some(chunk) = mem.get_chunk(i) {
|
||||
let j = serde_json::json!({ "index": i, "chunk": chunk });
|
||||
@@ -246,11 +323,44 @@ fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
||||
}
|
||||
}
|
||||
|
||||
Commands::Keygen { .. } => unreachable!("handled before opening a store"),
|
||||
|
||||
Commands::Verify { public_key } => {
|
||||
let text = if Path::new(&public_key).is_file() {
|
||||
std::fs::read_to_string(&public_key)?
|
||||
} else {
|
||||
public_key
|
||||
};
|
||||
let bytes = signing::from_hex::<32>(&text)
|
||||
.ok_or("--public-key must be 64 hex characters or a file holding them")?;
|
||||
let trusted = VerifyingKey::from_bytes(&bytes)?;
|
||||
let r = HDF5Memory::verify(&path, &trusted)?;
|
||||
let j = serde_json::json!({
|
||||
"valid": r.is_valid(),
|
||||
"signed": r.signed,
|
||||
"key_matches": r.key_matches,
|
||||
"signature_valid": r.signature_valid,
|
||||
"records_match": r.records_match,
|
||||
"settings_match": r.settings_match,
|
||||
"sessions_match": r.sessions_match,
|
||||
"graph_match": r.graph_match,
|
||||
"changed_records": r.changed_records,
|
||||
"record_count": r.record_count,
|
||||
"signed_record_count": r.signed_record_count,
|
||||
"signed_by": r.public_key.map(|k| signing::to_hex(&k)),
|
||||
"wal_entries_unsigned": r.wal_entries_unsigned,
|
||||
});
|
||||
println!("{}", serde_json::to_string_pretty(&j)?);
|
||||
if !r.is_valid() {
|
||||
std::process::exit(2);
|
||||
}
|
||||
}
|
||||
|
||||
Commands::Snapshot { dest } => {
|
||||
let _result = clawhdf5_agent::storage::snapshot_file(&cli.path, &dest)?;
|
||||
let _result = clawhdf5_agent::storage::snapshot_file(&path, &dest)?;
|
||||
let j = serde_json::json!({
|
||||
"status": "snapshot_created",
|
||||
"source": cli.path.display().to_string(),
|
||||
"source": path.display().to_string(),
|
||||
"dest": dest.display().to_string(),
|
||||
});
|
||||
println!("{}", serde_json::to_string(&j)?);
|
||||
|
||||
Binary file not shown.
@@ -97,7 +97,7 @@ impl AttributeMessage {
|
||||
return Ok(Cow::Borrowed(bytes));
|
||||
}
|
||||
let (file_data, offset_size) = file.ok_or(FormatError::UnresolvedSharedMessage)?;
|
||||
let shared_ref = shared_message::parse_shared_ref(bytes, offset_size)?;
|
||||
let shared_ref = shared_message::parse_shared_ref_sized(bytes, offset_size, length_size)?;
|
||||
shared_message::resolve_shared_message(
|
||||
file_data,
|
||||
&shared_ref,
|
||||
@@ -394,42 +394,80 @@ pub fn find_attribute<'a>(
|
||||
///
|
||||
/// Use this instead of `extract_attributes` when reading files that may use dense storage
|
||||
/// (e.g., objects with many attributes, typically >8).
|
||||
///
|
||||
/// Fails if any attribute cannot be read; see [`extract_attributes_tolerant`]
|
||||
/// to read the others.
|
||||
pub fn extract_attributes_full(
|
||||
file_data: &[u8],
|
||||
header: &ObjectHeader,
|
||||
offset_size: u8,
|
||||
length_size: u8,
|
||||
) -> Result<Vec<AttributeMessage>, FormatError> {
|
||||
extract_attributes_with(file_data, header, offset_size, length_size, &mut Err)
|
||||
}
|
||||
|
||||
/// Like [`extract_attributes_full`], but an attribute that cannot be read
|
||||
/// (a corrupt or unsupported attribute message, or a heap object that cannot
|
||||
/// be located) is left out and its error returned alongside the attributes
|
||||
/// that could be read, instead of failing them all.
|
||||
///
|
||||
/// Errors in the structures that index the attributes (the Attribute Info
|
||||
/// message, the dense-storage heap header or B-tree) still fail the call:
|
||||
/// then it is unknown which attributes exist at all.
|
||||
pub fn extract_attributes_tolerant(
|
||||
file_data: &[u8],
|
||||
header: &ObjectHeader,
|
||||
offset_size: u8,
|
||||
length_size: u8,
|
||||
) -> Result<(Vec<AttributeMessage>, Vec<FormatError>), FormatError> {
|
||||
let mut errors = Vec::new();
|
||||
let attrs = extract_attributes_with(file_data, header, offset_size, length_size, &mut |e| {
|
||||
errors.push(e);
|
||||
Ok(())
|
||||
})?;
|
||||
Ok((attrs, errors))
|
||||
}
|
||||
|
||||
/// Read every attribute; each one that fails goes to `on_error`, which
|
||||
/// either stops the read (returns the error) or skips that attribute.
|
||||
fn extract_attributes_with(
|
||||
file_data: &[u8],
|
||||
header: &ObjectHeader,
|
||||
offset_size: u8,
|
||||
length_size: u8,
|
||||
on_error: &mut dyn FnMut(FormatError) -> Result<(), FormatError>,
|
||||
) -> Result<Vec<AttributeMessage>, FormatError> {
|
||||
let mut attrs = Vec::new();
|
||||
|
||||
// Collect compact attributes (inline in OH)
|
||||
for msg in &header.messages {
|
||||
if msg.msg_type == MessageType::Attribute {
|
||||
if shared_message::is_shared(msg.flags) {
|
||||
let attr = if shared_message::is_shared(msg.flags) {
|
||||
// Shared attribute: resolve the reference to get actual attribute data
|
||||
let shared_ref = shared_message::parse_shared_ref(&msg.data, offset_size)?;
|
||||
let resolved_data = shared_message::resolve_shared_message(
|
||||
shared_message::parse_shared_ref_sized(&msg.data, offset_size, length_size)
|
||||
.and_then(|shared_ref| {
|
||||
shared_message::resolve_shared_message(
|
||||
file_data,
|
||||
&shared_ref,
|
||||
MessageType::Attribute,
|
||||
offset_size,
|
||||
length_size,
|
||||
)?;
|
||||
let attr = AttributeMessage::parse_in_file(
|
||||
&resolved_data,
|
||||
)
|
||||
})
|
||||
.and_then(|resolved| {
|
||||
AttributeMessage::parse_in_file(
|
||||
&resolved,
|
||||
file_data,
|
||||
offset_size,
|
||||
length_size,
|
||||
)?;
|
||||
attrs.push(attr);
|
||||
)
|
||||
})
|
||||
} else {
|
||||
let attr = AttributeMessage::parse_in_file(
|
||||
&msg.data,
|
||||
file_data,
|
||||
offset_size,
|
||||
length_size,
|
||||
)?;
|
||||
attrs.push(attr);
|
||||
AttributeMessage::parse_in_file(&msg.data, file_data, offset_size, length_size)
|
||||
};
|
||||
match attr {
|
||||
Ok(attr) => attrs.push(attr),
|
||||
Err(e) => on_error(e)?,
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -439,9 +477,15 @@ pub fn extract_attributes_full(
|
||||
if let Some(info) = attr_info
|
||||
&& let Some(fh_addr) = info.fractal_heap_address
|
||||
{
|
||||
let dense_attrs =
|
||||
extract_dense_attributes(file_data, &info, fh_addr, offset_size, length_size)?;
|
||||
attrs.extend(dense_attrs);
|
||||
extract_dense_attributes(
|
||||
file_data,
|
||||
&info,
|
||||
fh_addr,
|
||||
offset_size,
|
||||
length_size,
|
||||
&mut attrs,
|
||||
on_error,
|
||||
)?;
|
||||
}
|
||||
|
||||
Ok(attrs)
|
||||
@@ -468,7 +512,9 @@ fn extract_dense_attributes(
|
||||
fh_addr: u64,
|
||||
offset_size: u8,
|
||||
length_size: u8,
|
||||
) -> Result<Vec<AttributeMessage>, FormatError> {
|
||||
attrs: &mut Vec<AttributeMessage>,
|
||||
on_error: &mut dyn FnMut(FormatError) -> Result<(), FormatError>,
|
||||
) -> Result<(), FormatError> {
|
||||
// Parse fractal heap
|
||||
let fh = FractalHeapHeader::parse(file_data, fh_addr as usize, offset_size, length_size)?;
|
||||
|
||||
@@ -482,28 +528,32 @@ fn extract_dense_attributes(
|
||||
let btree_hdr = BTreeV2Header::parse(file_data, btree_addr as usize, offset_size, length_size)?;
|
||||
let records = collect_btree_v2_records(file_data, &btree_hdr, offset_size, length_size)?;
|
||||
|
||||
let mut attrs = Vec::new();
|
||||
for record in &records {
|
||||
// Per HDF5 spec, both type 8 and type 9 records start with heap_id:
|
||||
// Type 8: heap_id(8) + msg_flags(1) + creation_order(4) + hash(4)
|
||||
// Type 9: heap_id(8) + msg_flags(1) + creation_order(4)
|
||||
let id_offset = 0;
|
||||
|
||||
if record.data.len() < id_offset + fh.heap_id_length as usize {
|
||||
let id_len = fh.heap_id_length as usize;
|
||||
let Some(id_bytes) = record.data.get(..id_len) else {
|
||||
on_error(FormatError::UnexpectedEof {
|
||||
expected: id_len,
|
||||
available: record.data.len(),
|
||||
})?;
|
||||
continue;
|
||||
}
|
||||
let id_bytes = &record.data[id_offset..id_offset + fh.heap_id_length as usize];
|
||||
|
||||
// Read attribute message from fractal heap
|
||||
let attr_data = fh.read_managed_object(file_data, id_bytes, offset_size)?;
|
||||
};
|
||||
|
||||
// The data in the heap is a complete attribute message
|
||||
let attr =
|
||||
AttributeMessage::parse_in_file(&attr_data, file_data, offset_size, length_size)?;
|
||||
attrs.push(attr);
|
||||
let attr = fh
|
||||
.read_managed_object(file_data, id_bytes, offset_size)
|
||||
.and_then(|attr_data| {
|
||||
AttributeMessage::parse_in_file(&attr_data, file_data, offset_size, length_size)
|
||||
});
|
||||
match attr {
|
||||
Ok(attr) => attrs.push(attr),
|
||||
Err(e) => on_error(e)?,
|
||||
}
|
||||
}
|
||||
|
||||
Ok(attrs)
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
|
||||
@@ -323,39 +323,21 @@ fn collect_internal_records(
|
||||
let records_start = pos;
|
||||
pos += records_total;
|
||||
|
||||
// Compute sizes for child pointers
|
||||
// max_records at child depth - for variable-width nrec encoding
|
||||
// Child pointer layout, as libhdf5 computes it (H5B2__hdr_init): the
|
||||
// child's record count is always encoded in the width needed for a
|
||||
// *leaf's* maximum, and — below the first internal level — the child
|
||||
// subtree's total record count in the width needed for the most records
|
||||
// a subtree of that depth can hold.
|
||||
let child_depth = depth - 1;
|
||||
let max_nrec_child = if child_depth == 0 {
|
||||
max_leaf_nrec
|
||||
} else {
|
||||
// For internal nodes at child_depth, the true max_nrec depends on the
|
||||
// node size, record size, and the recursive width of child pointer
|
||||
// entries (which themselves depend on max_nrec at deeper levels).
|
||||
// Computing the exact value requires iterating from the leaf level
|
||||
// upward, as described in the HDF5 spec (III.A.2 "Computing the Size
|
||||
// of B-tree Nodes").
|
||||
//
|
||||
// We use `max_leaf_nrec * 2` as a conservative upper bound. This
|
||||
// over-estimates the nrec encoding width, which means we may read
|
||||
// slightly more bytes per child pointer than strictly necessary, but
|
||||
// never fewer. The over-read bytes are harmless because we only
|
||||
// decode `num_records` entries (the actual count from the node header).
|
||||
//
|
||||
// Known limitation: for very deep trees (depth > 3) with small record
|
||||
// sizes, the true max could exceed this estimate, causing us to
|
||||
// under-allocate the nrec encoding width and misparse child pointers.
|
||||
// In practice, HDF5 B-tree v2 depths rarely exceed 2-3.
|
||||
max_leaf_nrec * 2
|
||||
};
|
||||
let nrec_width = bytes_for_max_records(max_nrec_child);
|
||||
|
||||
// Total records in subtree width (only if depth > 1)
|
||||
let nrec_width = bytes_for_max_records(max_leaf_nrec);
|
||||
let total_nrec_width = if depth > 1 {
|
||||
// Width to hold total records in a subtree
|
||||
// We compute max possible total records at this subtree depth
|
||||
let max_total = header_max_total_records(max_leaf_nrec, depth - 1);
|
||||
bytes_for_max_records(max_total)
|
||||
bytes_for_max_records(cum_max_records(
|
||||
node_size,
|
||||
record_size,
|
||||
offset_size,
|
||||
max_leaf_nrec,
|
||||
child_depth,
|
||||
))
|
||||
} else {
|
||||
0
|
||||
};
|
||||
@@ -435,14 +417,36 @@ fn collect_internal_records(
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Estimate maximum total records at a given depth (for variable-width encoding).
|
||||
fn header_max_total_records(max_leaf_nrec: u64, depth: u16) -> u64 {
|
||||
// Conservative: branching factor * max_leaf at each level
|
||||
let mut total = max_leaf_nrec;
|
||||
for _ in 0..depth {
|
||||
total = total.saturating_mul(max_leaf_nrec.max(2));
|
||||
/// Most records a subtree whose root is at `depth` can hold (libhdf5's
|
||||
/// `cum_max_nrec`): a leaf holds `max_leaf_nrec`; an internal node at depth
|
||||
/// `d` holds `max_nrec(d)` records and `max_nrec(d) + 1` subtrees of depth
|
||||
/// `d - 1`, where `max_nrec(d)` is what fits in a node once each record is
|
||||
/// paired with a child pointer of the width depth `d` needs.
|
||||
fn cum_max_records(
|
||||
node_size: u32,
|
||||
record_size: u16,
|
||||
offset_size: u8,
|
||||
max_leaf_nrec: u64,
|
||||
depth: u16,
|
||||
) -> u64 {
|
||||
// Internal node overhead: signature(4) + version(1) + type(1) + checksum(4).
|
||||
const PREFIX: u64 = 10;
|
||||
let nrec_width = bytes_for_max_records(max_leaf_nrec) as u64;
|
||||
let mut cum = max_leaf_nrec;
|
||||
let mut cum_width = 0u64;
|
||||
for d in 1..=depth {
|
||||
let ptr = u64::from(offset_size) + nrec_width + if d > 1 { cum_width } else { 0 };
|
||||
let max_nrec = u64::from(node_size)
|
||||
.saturating_sub(PREFIX)
|
||||
.saturating_sub(ptr)
|
||||
/ (u64::from(record_size) + ptr).max(1);
|
||||
cum = max_nrec
|
||||
.saturating_add(1)
|
||||
.saturating_mul(cum)
|
||||
.saturating_add(max_nrec);
|
||||
cum_width = bytes_for_max_records(cum) as u64;
|
||||
}
|
||||
total
|
||||
cum
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
@@ -512,9 +516,15 @@ mod tests {
|
||||
child_nrec: u64,
|
||||
) -> Vec<u8> {
|
||||
let max_leaf = max_records_leaf(node_size, record_size);
|
||||
let nrec_width = bytes_for_max_records(if depth == 1 { max_leaf } else { max_leaf * 2 });
|
||||
let nrec_width = bytes_for_max_records(max_leaf);
|
||||
let total_width = if depth > 1 {
|
||||
bytes_for_max_records(header_max_total_records(max_leaf, depth - 1))
|
||||
bytes_for_max_records(cum_max_records(
|
||||
node_size,
|
||||
record_size,
|
||||
8,
|
||||
max_leaf,
|
||||
depth - 1,
|
||||
))
|
||||
} else {
|
||||
0
|
||||
};
|
||||
@@ -673,4 +683,18 @@ mod tests {
|
||||
let records = collect_btree_v2_records(&header, &hdr, 8, 8).unwrap();
|
||||
assert!(records.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn subtree_capacity_matches_libhdf5() {
|
||||
// A link-name index (11-byte records, 512-byte nodes, 8-byte
|
||||
// addresses): libhdf5's H5B2__hdr_init gives 45 records per leaf,
|
||||
// then cum_max_nrec 1 149 at depth 1 and 26 449 at depth 2 — two
|
||||
// bytes of subtree count in a depth-3 root's child pointers, where
|
||||
// leaf_max^3 = 91 125 would need three.
|
||||
let leaf = max_records_leaf(512, 11);
|
||||
assert_eq!(leaf, 45);
|
||||
assert_eq!(cum_max_records(512, 11, 8, leaf, 0), 45);
|
||||
assert_eq!(cum_max_records(512, 11, 8, leaf, 1), 1_149);
|
||||
assert_eq!(cum_max_records(512, 11, 8, leaf, 2), 26_449);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -223,13 +223,32 @@ pub const DEFAULT_CACHE_BYTES: usize = 16 * 1024 * 1024; // 16 MiB
|
||||
/// coordinate map and reduces collision chains compared to power-of-two sizes.
|
||||
pub const DEFAULT_MAX_SLOTS: usize = 521;
|
||||
|
||||
/// Most datasets whose chunk index a [`ChunkCache`] keeps at once.
|
||||
pub const MAX_INDEXED_DATASETS: usize = 64;
|
||||
|
||||
/// Most chunk-index entries, summed over all datasets, a [`ChunkCache`] keeps.
|
||||
/// Least-recently-used datasets' indexes are dropped past this (the dataset
|
||||
/// being read is always kept), so a file with many or huge chunked datasets
|
||||
/// cannot grow the cache without bound.
|
||||
pub const MAX_INDEXED_CHUNKS: usize = 1 << 20;
|
||||
|
||||
/// The dataset key the address-less (legacy) methods use when
|
||||
/// [`ChunkCache::ensure_dataset`] has not been called.
|
||||
#[cfg(feature = "std")]
|
||||
const UNBOUND_DATASET: u64 = u64::MAX;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// LRU entry
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Decompressed chunks are keyed by dataset *and* coordinate: every chunked
|
||||
/// dataset has a chunk at (0, 0, ...), so the coordinate alone is ambiguous.
|
||||
#[cfg(feature = "std")]
|
||||
type SlotKey = (u64, ChunkCoord);
|
||||
|
||||
#[cfg(feature = "std")]
|
||||
struct CachedChunk {
|
||||
coord: ChunkCoord,
|
||||
key: SlotKey,
|
||||
/// Shared so a cache hit is a refcount bump, not a copy of the whole
|
||||
/// (potentially large) decompressed chunk.
|
||||
data: Arc<CacheAlignedBuffer>,
|
||||
@@ -237,21 +256,48 @@ struct CachedChunk {
|
||||
last_access: u64,
|
||||
}
|
||||
|
||||
/// Per-dataset index state.
|
||||
#[cfg(feature = "std")]
|
||||
#[derive(Default)]
|
||||
struct DatasetEntry {
|
||||
/// Chunk coordinate -> ChunkInfo (offset + size in file).
|
||||
index: Option<Arc<HashMap<ChunkCoord, ChunkInfo>>>,
|
||||
/// Pre-built chunk index for O(1) coordinate lookups.
|
||||
chunk_index: Option<Arc<ChunkIndex>>,
|
||||
/// Pre-computed chunk layout for fast assembly.
|
||||
chunk_layout: Option<Arc<ChunkLayout>>,
|
||||
/// Tick of the last use, for dropping the least recently used dataset.
|
||||
last_used: u64,
|
||||
}
|
||||
|
||||
#[cfg(feature = "std")]
|
||||
impl DatasetEntry {
|
||||
fn weight(&self) -> usize {
|
||||
self.index.as_ref().map_or(0, |m| m.len())
|
||||
+ self.chunk_index.as_ref().map_or(0, |c| c.num_chunks())
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// ChunkCache
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// A per-dataset chunk cache with hash-based index and LRU eviction.
|
||||
/// A per-file chunk cache: chunk indexes per dataset, plus an LRU of
|
||||
/// decompressed chunks, all keyed by dataset.
|
||||
///
|
||||
/// # Usage
|
||||
/// A dataset is identified by the address of its chunk index (B-tree, fixed
|
||||
/// or extensible array, ...), which is unique within a file. Every method
|
||||
/// that takes an `addr` works on that dataset only, so threads reading
|
||||
/// different datasets through one shared cache never see each other's
|
||||
/// chunks. The address-less methods (`has_index`, `populate_index`,
|
||||
/// `get_decompressed`, ...) act on the dataset last bound with
|
||||
/// [`Self::ensure_dataset`]; that binding is shared state, so concurrent
|
||||
/// readers must use the `*_in` / `*_for` methods instead (the chunked
|
||||
/// readers in [`crate::chunked_read`] do).
|
||||
///
|
||||
/// ```ignore
|
||||
/// let cache = ChunkCache::new();
|
||||
/// // Pass &cache to read_chunked_data — it will populate the index lazily.
|
||||
/// ```
|
||||
///
|
||||
/// The cache is wrapped in `Mutex` internally so it can be mutated through
|
||||
/// shared references (thread-safe).
|
||||
/// Memory is bounded: decompressed data by `max_bytes`/`max_slots` across
|
||||
/// all datasets, indexes by [`MAX_INDEXED_DATASETS`] and
|
||||
/// [`MAX_INDEXED_CHUNKS`].
|
||||
///
|
||||
/// Only available with the `std` feature because it requires `std::sync::Mutex`.
|
||||
#[cfg(feature = "std")]
|
||||
@@ -261,26 +307,20 @@ pub struct ChunkCache {
|
||||
|
||||
#[cfg(feature = "std")]
|
||||
struct CacheInner {
|
||||
/// Hash index: chunk coordinate -> ChunkInfo (offset + size in file).
|
||||
/// Populated once per dataset on first access.
|
||||
index: Option<HashMap<ChunkCoord, ChunkInfo>>,
|
||||
/// Per-dataset chunk indexes, keyed by chunk-index address.
|
||||
datasets: HashMap<u64, DatasetEntry>,
|
||||
|
||||
/// Address of the dataset (its chunk-index base address) that the cached
|
||||
/// index, chunk index, layout, and decompressed slots currently belong to.
|
||||
/// The cache is shared per file across datasets, so every cached-read entry
|
||||
/// checks this and resets the per-dataset state when the dataset changes —
|
||||
/// otherwise one dataset's chunk index (with its own rank) would be reused
|
||||
/// for another, corrupting reads.
|
||||
index_addr: Option<u64>,
|
||||
/// Dataset the address-less methods act on (see `ensure_dataset`).
|
||||
current: Option<u64>,
|
||||
|
||||
/// LRU cache of decompressed chunk data.
|
||||
slots: Vec<CachedChunk>,
|
||||
|
||||
/// Coordinate -> index into `slots`, for O(1) lookup instead of a linear
|
||||
/// Key -> index into `slots`, for O(1) lookup instead of a linear
|
||||
/// scan. Kept in sync with `slots` on every insert/evict/clear — in
|
||||
/// particular, `slots.swap_remove(i)` moves the last element into slot
|
||||
/// `i`, so the moved element's index entry must be updated too.
|
||||
slot_index: HashMap<ChunkCoord, usize>,
|
||||
slot_index: HashMap<SlotKey, usize>,
|
||||
|
||||
/// Current total bytes of cached decompressed data.
|
||||
current_bytes: usize,
|
||||
@@ -294,17 +334,145 @@ struct CacheInner {
|
||||
/// Monotonic counter for LRU ordering.
|
||||
tick: u64,
|
||||
|
||||
/// Last accessed chunk coordinate (for sequential detection).
|
||||
last_coord: Option<ChunkCoord>,
|
||||
/// Last accessed chunk (for sequential detection).
|
||||
last_coord: Option<SlotKey>,
|
||||
|
||||
/// Access pattern statistics.
|
||||
stats: AccessStats,
|
||||
}
|
||||
|
||||
/// Pre-built chunk index for O(1) coordinate lookups.
|
||||
chunk_index: Option<ChunkIndex>,
|
||||
#[cfg(feature = "std")]
|
||||
impl CacheInner {
|
||||
fn current(&self) -> u64 {
|
||||
self.current.unwrap_or(UNBOUND_DATASET)
|
||||
}
|
||||
|
||||
/// Pre-computed chunk layout for fast assembly.
|
||||
chunk_layout: Option<ChunkLayout>,
|
||||
fn touch(&mut self, addr: u64) -> &mut DatasetEntry {
|
||||
self.tick += 1;
|
||||
let tick = self.tick;
|
||||
let entry = self.datasets.entry(addr).or_default();
|
||||
entry.last_used = tick;
|
||||
entry
|
||||
}
|
||||
|
||||
fn entry(&self, addr: u64) -> Option<&DatasetEntry> {
|
||||
self.datasets.get(&addr)
|
||||
}
|
||||
|
||||
/// Drop least-recently-used datasets' indexes (never `keep`'s) until the
|
||||
/// dataset and chunk-entry budgets hold.
|
||||
fn trim_datasets(&mut self, keep: u64) {
|
||||
loop {
|
||||
let total: usize = self.datasets.values().map(DatasetEntry::weight).sum();
|
||||
if self.datasets.len() <= MAX_INDEXED_DATASETS && total <= MAX_INDEXED_CHUNKS {
|
||||
return;
|
||||
}
|
||||
let victim = self
|
||||
.datasets
|
||||
.iter()
|
||||
.filter(|(a, _)| **a != keep)
|
||||
.min_by_key(|(_, e)| e.last_used)
|
||||
.map(|(a, _)| *a);
|
||||
match victim {
|
||||
Some(a) => {
|
||||
self.datasets.remove(&a);
|
||||
}
|
||||
None => return,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn get_decompressed(&mut self, addr: u64, coord: &[u64]) -> Option<Arc<CacheAlignedBuffer>> {
|
||||
self.tick += 1;
|
||||
let tick = self.tick;
|
||||
|
||||
// Track sequential vs random access
|
||||
let is_sequential = self.last_coord.as_ref().is_some_and(|(prev_addr, prev)| {
|
||||
// Sequential if exactly one dimension changed
|
||||
let changes: usize = prev
|
||||
.iter()
|
||||
.zip(coord.iter())
|
||||
.filter(|(a, b)| a != b)
|
||||
.count();
|
||||
*prev_addr == addr && changes <= 1
|
||||
});
|
||||
if is_sequential {
|
||||
self.stats.sequential_count += 1;
|
||||
} else if self.last_coord.is_some() {
|
||||
self.stats.random_count += 1;
|
||||
}
|
||||
let key: SlotKey = (addr, coord.to_vec());
|
||||
let found = if let Some(&idx) = self.slot_index.get(&key) {
|
||||
self.slots[idx].last_access = tick;
|
||||
Some(Arc::clone(&self.slots[idx].data))
|
||||
} else {
|
||||
None
|
||||
};
|
||||
self.last_coord = Some(key);
|
||||
if let Some(ref data) = found {
|
||||
self.stats.hits += 1;
|
||||
self.stats.bytes_read += data.len() as u64;
|
||||
} else {
|
||||
self.stats.misses += 1;
|
||||
}
|
||||
found
|
||||
}
|
||||
|
||||
fn put_decompressed(
|
||||
&mut self,
|
||||
key: SlotKey,
|
||||
data: Arc<CacheAlignedBuffer>,
|
||||
) -> Arc<CacheAlignedBuffer> {
|
||||
let data_len = data.len();
|
||||
|
||||
// Don't cache if single chunk exceeds budget — still return the data
|
||||
// to the caller, just don't retain it.
|
||||
if data_len > self.max_bytes {
|
||||
return data;
|
||||
}
|
||||
|
||||
// Check if already present
|
||||
self.tick += 1;
|
||||
let tick = self.tick;
|
||||
if let Some(&idx) = self.slot_index.get(&key) {
|
||||
self.slots[idx].last_access = tick;
|
||||
return Arc::clone(&self.slots[idx].data); // already cached
|
||||
}
|
||||
|
||||
// Evict until we have room
|
||||
while self.slots.len() >= self.max_slots
|
||||
|| (self.current_bytes + data_len > self.max_bytes && !self.slots.is_empty())
|
||||
{
|
||||
// Find LRU slot
|
||||
let lru_idx = self
|
||||
.slots
|
||||
.iter()
|
||||
.enumerate()
|
||||
.min_by_key(|(_, s)| s.last_access)
|
||||
.map(|(i, _)| i)
|
||||
.unwrap();
|
||||
let removed = self.slots.swap_remove(lru_idx);
|
||||
self.slot_index.remove(&removed.key);
|
||||
// swap_remove moved the former last element into `lru_idx` (unless
|
||||
// it *was* the last element) — fix up that element's index entry.
|
||||
if lru_idx < self.slots.len() {
|
||||
let moved_key = self.slots[lru_idx].key.clone();
|
||||
self.slot_index.insert(moved_key, lru_idx);
|
||||
}
|
||||
self.current_bytes -= removed.data.len();
|
||||
self.stats.evictions += 1;
|
||||
}
|
||||
|
||||
self.current_bytes += data_len;
|
||||
let new_idx = self.slots.len();
|
||||
self.slot_index.insert(key.clone(), new_idx);
|
||||
self.slots.push(CachedChunk {
|
||||
key,
|
||||
data: Arc::clone(&data),
|
||||
last_access: tick,
|
||||
});
|
||||
data
|
||||
}
|
||||
}
|
||||
|
||||
/// Access pattern statistics tracked by the chunk cache.
|
||||
@@ -356,8 +524,8 @@ impl ChunkCache {
|
||||
pub fn with_capacity(max_bytes: usize, max_slots: usize) -> Self {
|
||||
Self {
|
||||
inner: std::sync::Mutex::new(CacheInner {
|
||||
index: None,
|
||||
index_addr: None,
|
||||
datasets: HashMap::new(),
|
||||
current: None,
|
||||
slots: Vec::with_capacity(max_slots.min(64)),
|
||||
slot_index: HashMap::with_capacity(max_slots.min(64)),
|
||||
current_bytes: 0,
|
||||
@@ -366,340 +534,331 @@ impl ChunkCache {
|
||||
tick: 0,
|
||||
last_coord: None,
|
||||
stats: AccessStats::default(),
|
||||
chunk_index: None,
|
||||
chunk_layout: None,
|
||||
}),
|
||||
}
|
||||
}
|
||||
|
||||
// ----- Index operations -----
|
||||
fn lock(&self) -> std::sync::MutexGuard<'_, CacheInner> {
|
||||
self.inner.lock().unwrap_or_else(|e| e.into_inner())
|
||||
}
|
||||
|
||||
/// The most decompressed bytes this cache will hold.
|
||||
pub fn max_bytes(&self) -> usize {
|
||||
self.inner.lock().map(|g| g.max_bytes).unwrap_or(0)
|
||||
self.lock().max_bytes
|
||||
}
|
||||
|
||||
/// Bind the cache to the dataset at chunk-index address `addr`.
|
||||
// ----- Dataset-keyed operations (safe to use concurrently) -----
|
||||
|
||||
/// The chunk list of the dataset whose chunk index is at `addr`.
|
||||
///
|
||||
/// The cache is shared per file across all of its datasets. If the cache
|
||||
/// currently holds state for a different dataset, all per-dataset state
|
||||
/// (chunk index, chunk-index map, layout, and decompressed slots) is
|
||||
/// dropped so the next access rebuilds it for this dataset. Reading the
|
||||
/// same dataset again is a no-op, preserving the cache's benefit for
|
||||
/// repeated/sequential access. Returns `true` if a reset occurred.
|
||||
/// On the first call for a dataset, `build` scans its chunk index; the
|
||||
/// result is kept (offsets truncated to `rank` for the lookup key), so
|
||||
/// later calls skip the scan. `build` runs without the cache lock held;
|
||||
/// if two threads race to build the same dataset's index, the first
|
||||
/// stored one wins and both return equivalent lists.
|
||||
pub fn chunks_for<E>(
|
||||
&self,
|
||||
addr: u64,
|
||||
rank: usize,
|
||||
build: impl FnOnce() -> Result<Vec<ChunkInfo>, E>,
|
||||
) -> Result<Vec<ChunkInfo>, E> {
|
||||
Ok(self
|
||||
.index_for(addr, rank, build)?
|
||||
.values()
|
||||
.cloned()
|
||||
.collect())
|
||||
}
|
||||
|
||||
fn index_for<E>(
|
||||
&self,
|
||||
addr: u64,
|
||||
rank: usize,
|
||||
build: impl FnOnce() -> Result<Vec<ChunkInfo>, E>,
|
||||
) -> Result<Arc<HashMap<ChunkCoord, ChunkInfo>>, E> {
|
||||
if let Some(index) = self.lock().touch(addr).index.clone() {
|
||||
return Ok(index);
|
||||
}
|
||||
let chunks = build()?;
|
||||
let map: HashMap<ChunkCoord, ChunkInfo> = chunks
|
||||
.into_iter()
|
||||
.map(|ci| (ci.offsets.iter().take(rank).copied().collect(), ci))
|
||||
.collect();
|
||||
let mut inner = self.lock();
|
||||
let entry = inner.touch(addr);
|
||||
let index = Arc::clone(entry.index.get_or_insert_with(|| Arc::new(map)));
|
||||
inner.trim_datasets(addr);
|
||||
Ok(index)
|
||||
}
|
||||
|
||||
/// The pre-computed assembly layout of the dataset at `addr`, building
|
||||
/// its chunk index (via `build`, as in [`Self::chunks_for`]) and layout on
|
||||
/// first use.
|
||||
pub fn chunk_layout_for<E>(
|
||||
&self,
|
||||
addr: u64,
|
||||
rank: usize,
|
||||
build: impl FnOnce() -> Result<Vec<ChunkInfo>, E>,
|
||||
ds_dims: &[usize],
|
||||
chunk_dims: &[usize],
|
||||
elem_size: usize,
|
||||
) -> Result<Arc<ChunkLayout>, E> {
|
||||
let (layout, chunk_index) = {
|
||||
let mut inner = self.lock();
|
||||
let entry = inner.touch(addr);
|
||||
(entry.chunk_layout.clone(), entry.chunk_index.clone())
|
||||
};
|
||||
if let Some(layout) = layout {
|
||||
return Ok(layout);
|
||||
}
|
||||
let chunk_index = match chunk_index {
|
||||
Some(ci) => ci,
|
||||
None => {
|
||||
let index = self.index_for(addr, rank, build)?;
|
||||
let chunks: Vec<ChunkInfo> = index.values().cloned().collect();
|
||||
Arc::new(ChunkIndex::build(&chunks, rank))
|
||||
}
|
||||
};
|
||||
let layout = ChunkLayout::build(&chunk_index, ds_dims, chunk_dims, elem_size);
|
||||
let mut inner = self.lock();
|
||||
let entry = inner.touch(addr);
|
||||
entry.chunk_index.get_or_insert(chunk_index);
|
||||
let layout = Arc::clone(entry.chunk_layout.get_or_insert_with(|| Arc::new(layout)));
|
||||
inner.trim_datasets(addr);
|
||||
Ok(layout)
|
||||
}
|
||||
|
||||
/// Cached decompressed chunk at `coord` of the dataset at `addr`.
|
||||
///
|
||||
/// O(1) lookup; the clone is an `Arc` refcount bump, not a copy of the
|
||||
/// underlying decompressed data.
|
||||
pub fn get_decompressed_in(&self, addr: u64, coord: &[u64]) -> Option<Arc<CacheAlignedBuffer>> {
|
||||
self.lock().get_decompressed(addr, coord)
|
||||
}
|
||||
|
||||
/// Cache decompressed chunk data for `coord` of the dataset at `addr`.
|
||||
/// Returns the `Arc`-shared buffer now cached (or already cached).
|
||||
pub fn put_decompressed_in(
|
||||
&self,
|
||||
addr: u64,
|
||||
coord: ChunkCoord,
|
||||
data: Vec<u8>,
|
||||
) -> Arc<CacheAlignedBuffer> {
|
||||
self.put_decompressed_aligned_in(addr, coord, CacheAlignedBuffer::from_vec(data))
|
||||
}
|
||||
|
||||
/// [`Self::put_decompressed_in`] for an already-aligned buffer.
|
||||
pub fn put_decompressed_aligned_in(
|
||||
&self,
|
||||
addr: u64,
|
||||
coord: ChunkCoord,
|
||||
data: CacheAlignedBuffer,
|
||||
) -> Arc<CacheAlignedBuffer> {
|
||||
let data = Arc::new(data);
|
||||
self.lock().put_decompressed((addr, coord), data)
|
||||
}
|
||||
|
||||
/// Record that the given chunk coordinates of the dataset at `addr` are
|
||||
/// predicted to be accessed soon (bookkeeping only).
|
||||
///
|
||||
/// This does **not** prefetch or pre-decompress anything — it only
|
||||
/// checks whether each coordinate is already in the chunk index and
|
||||
/// updates access-pattern stats accordingly.
|
||||
pub fn prefetch_hint_in(&self, addr: u64, next_coords: &[ChunkCoord]) {
|
||||
let mut inner = self.lock();
|
||||
let Some(index) = inner.entry(addr).and_then(|e| e.index.clone()) else {
|
||||
return;
|
||||
};
|
||||
let known = next_coords
|
||||
.iter()
|
||||
.filter(|c| index.contains_key(*c))
|
||||
.count();
|
||||
inner.stats.sequential_count += known as u64;
|
||||
}
|
||||
|
||||
// ----- Address-less operations on the bound dataset -----
|
||||
|
||||
/// Bind the address-less methods to the dataset at chunk-index address
|
||||
/// `addr`. Returns `true` if this changed the bound dataset.
|
||||
///
|
||||
/// Each dataset's state is kept separately, so switching loses nothing
|
||||
/// and never exposes one dataset's index or chunks to another. The
|
||||
/// binding itself is shared, though: concurrent readers should use the
|
||||
/// `addr`-taking methods rather than bind and then call these.
|
||||
pub fn ensure_dataset(&self, addr: u64) -> bool {
|
||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
||||
if inner.index_addr == Some(addr) {
|
||||
return false;
|
||||
}
|
||||
inner.index = None;
|
||||
inner.chunk_index = None;
|
||||
inner.chunk_layout = None;
|
||||
inner.slots.clear();
|
||||
inner.slot_index.clear();
|
||||
inner.current_bytes = 0;
|
||||
inner.last_coord = None;
|
||||
inner.index_addr = Some(addr);
|
||||
true
|
||||
let mut inner = self.lock();
|
||||
let changed = inner.current != Some(addr);
|
||||
inner.current = Some(addr);
|
||||
changed
|
||||
}
|
||||
|
||||
/// Returns `true` if the chunk index has been built.
|
||||
/// Returns `true` if the bound dataset's chunk index has been built.
|
||||
pub fn has_index(&self) -> bool {
|
||||
self.inner
|
||||
.lock()
|
||||
.unwrap_or_else(|e| e.into_inner())
|
||||
.index
|
||||
.is_some()
|
||||
let inner = self.lock();
|
||||
inner
|
||||
.entry(inner.current())
|
||||
.is_some_and(|e| e.index.is_some())
|
||||
}
|
||||
|
||||
/// Build the chunk index from a pre-collected list of `ChunkInfo`.
|
||||
/// Build the bound dataset's chunk index from a pre-collected list of
|
||||
/// `ChunkInfo`.
|
||||
///
|
||||
/// The `rank` parameter is used to truncate offsets to spatial dims only
|
||||
/// (B-tree v1 stores rank+1 offsets).
|
||||
pub fn populate_index(&self, chunks: &[ChunkInfo], rank: usize) {
|
||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
||||
if inner.index.is_some() {
|
||||
return; // already populated
|
||||
}
|
||||
let mut map = HashMap::with_capacity(chunks.len());
|
||||
|
||||
for ci in chunks {
|
||||
let coord: ChunkCoord = ci.offsets.iter().take(rank).copied().collect();
|
||||
map.insert(coord, ci.clone());
|
||||
}
|
||||
inner.index = Some(map);
|
||||
let addr = self.lock().current();
|
||||
let _ = self.index_for::<core::convert::Infallible>(addr, rank, || Ok(chunks.to_vec()));
|
||||
}
|
||||
|
||||
/// Look up a chunk by its spatial coordinate in the index.
|
||||
/// Look up a chunk by its spatial coordinate in the bound dataset's index.
|
||||
pub fn lookup_index(&self, coord: &[u64]) -> Option<ChunkInfo> {
|
||||
let inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
||||
inner.index.as_ref()?.get(coord).cloned()
|
||||
let inner = self.lock();
|
||||
inner
|
||||
.entry(inner.current())?
|
||||
.index
|
||||
.as_ref()?
|
||||
.get(coord)
|
||||
.cloned()
|
||||
}
|
||||
|
||||
/// Return all indexed chunks as a `Vec<ChunkInfo>` (order unspecified).
|
||||
/// Return all of the bound dataset's indexed chunks (order unspecified).
|
||||
pub fn all_indexed_chunks(&self) -> Option<Vec<ChunkInfo>> {
|
||||
let inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
||||
inner.index.as_ref().map(|m| m.values().cloned().collect())
|
||||
let inner = self.lock();
|
||||
let index = inner.entry(inner.current())?.index.as_ref()?;
|
||||
Some(index.values().cloned().collect())
|
||||
}
|
||||
|
||||
// ----- Chunk index (pre-built coordinate → ChunkInfo map) -----
|
||||
|
||||
/// Returns `true` if the chunk B-tree index has been built.
|
||||
/// Returns `true` if the bound dataset's `ChunkIndex` has been built.
|
||||
pub fn has_chunk_index(&self) -> bool {
|
||||
self.inner
|
||||
.lock()
|
||||
.unwrap_or_else(|e| e.into_inner())
|
||||
.chunk_index
|
||||
.is_some()
|
||||
let inner = self.lock();
|
||||
inner
|
||||
.entry(inner.current())
|
||||
.is_some_and(|e| e.chunk_index.is_some())
|
||||
}
|
||||
|
||||
/// Build and store the chunk B-tree index from a pre-collected list of `ChunkInfo`.
|
||||
/// Build and store the bound dataset's `ChunkIndex`.
|
||||
pub fn populate_chunk_index(&self, chunks: &[ChunkInfo], rank: usize) {
|
||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
||||
if inner.chunk_index.is_some() {
|
||||
return;
|
||||
}
|
||||
inner.chunk_index = Some(ChunkIndex::build(chunks, rank));
|
||||
let built = Arc::new(ChunkIndex::build(chunks, rank));
|
||||
let mut inner = self.lock();
|
||||
let addr = inner.current();
|
||||
inner.touch(addr).chunk_index.get_or_insert(built);
|
||||
inner.trim_datasets(addr);
|
||||
}
|
||||
|
||||
// ----- Chunk layout (pre-computed assembly plan) -----
|
||||
|
||||
/// Returns `true` if the chunk layout has been computed.
|
||||
/// Returns `true` if the bound dataset's chunk layout has been computed.
|
||||
pub fn has_chunk_layout(&self) -> bool {
|
||||
self.inner
|
||||
.lock()
|
||||
.unwrap_or_else(|e| e.into_inner())
|
||||
.chunk_layout
|
||||
.is_some()
|
||||
let inner = self.lock();
|
||||
inner
|
||||
.entry(inner.current())
|
||||
.is_some_and(|e| e.chunk_layout.is_some())
|
||||
}
|
||||
|
||||
/// Build and store the pre-computed chunk layout for fast assembly.
|
||||
/// Build and store the bound dataset's chunk layout (needs its
|
||||
/// `ChunkIndex`; does nothing without one).
|
||||
pub fn populate_chunk_layout(&self, ds_dims: &[usize], chunk_dims: &[usize], elem_size: usize) {
|
||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
||||
if inner.chunk_layout.is_some() {
|
||||
let mut inner = self.lock();
|
||||
let addr = inner.current();
|
||||
let entry = inner.touch(addr);
|
||||
if entry.chunk_layout.is_some() {
|
||||
return;
|
||||
}
|
||||
if let Some(ref idx) = inner.chunk_index {
|
||||
inner.chunk_layout = Some(ChunkLayout::build(idx, ds_dims, chunk_dims, elem_size));
|
||||
if let Some(idx) = entry.chunk_index.clone() {
|
||||
entry.chunk_layout = Some(Arc::new(ChunkLayout::build(
|
||||
&idx, ds_dims, chunk_dims, elem_size,
|
||||
)));
|
||||
}
|
||||
}
|
||||
|
||||
/// Execute a function with a reference to the chunk layout.
|
||||
///
|
||||
/// Returns `None` if the layout hasn't been computed yet.
|
||||
/// Execute a function with a reference to the bound dataset's chunk
|
||||
/// layout. Returns `None` if the layout hasn't been computed yet.
|
||||
pub fn with_chunk_layout<F, R>(&self, f: F) -> Option<R>
|
||||
where
|
||||
F: FnOnce(&ChunkLayout) -> R,
|
||||
{
|
||||
let inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
||||
inner.chunk_layout.as_ref().map(f)
|
||||
let layout = {
|
||||
let inner = self.lock();
|
||||
inner.entry(inner.current())?.chunk_layout.clone()?
|
||||
};
|
||||
Some(f(&layout))
|
||||
}
|
||||
|
||||
// ----- Decompressed data cache (LRU) -----
|
||||
|
||||
/// Try to get cached decompressed data for a chunk coordinate.
|
||||
/// Try to get cached decompressed data for a chunk of the bound dataset.
|
||||
///
|
||||
/// O(1) lookup. Returns an owned copy for API compatibility with callers
|
||||
/// that need a `Vec<u8>`; prefer [`Self::get_decompressed_aligned`] when
|
||||
/// an `Arc`-shared buffer works for the caller, since that avoids the
|
||||
/// copy entirely.
|
||||
/// Returns an owned copy; prefer [`Self::get_decompressed_aligned`] when
|
||||
/// an `Arc`-shared buffer works for the caller.
|
||||
pub fn get_decompressed(&self, coord: &[u64]) -> Option<Vec<u8>> {
|
||||
self.get_decompressed_aligned(coord)
|
||||
.map(|arc| arc.as_slice().to_vec())
|
||||
}
|
||||
|
||||
/// Try to get a reference-counted clone of the aligned buffer for a chunk.
|
||||
///
|
||||
/// O(1) index lookup; the clone is an `Arc` refcount bump, not a copy of
|
||||
/// the underlying decompressed data.
|
||||
/// Reference-counted cached buffer for a chunk of the bound dataset.
|
||||
pub fn get_decompressed_aligned(&self, coord: &[u64]) -> Option<Arc<CacheAlignedBuffer>> {
|
||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
||||
inner.tick += 1;
|
||||
let tick = inner.tick;
|
||||
|
||||
// Track sequential vs random access
|
||||
let is_sequential = inner.last_coord.as_ref().is_some_and(|prev| {
|
||||
// Sequential if exactly one dimension changed
|
||||
let changes: usize = prev
|
||||
.iter()
|
||||
.zip(coord.iter())
|
||||
.filter(|(a, b)| a != b)
|
||||
.count();
|
||||
changes <= 1
|
||||
});
|
||||
if is_sequential {
|
||||
inner.stats.sequential_count += 1;
|
||||
} else if inner.last_coord.is_some() {
|
||||
inner.stats.random_count += 1;
|
||||
}
|
||||
inner.last_coord = Some(coord.to_vec());
|
||||
|
||||
let found = if let Some(&idx) = inner.slot_index.get(coord) {
|
||||
inner.slots[idx].last_access = tick;
|
||||
Some(Arc::clone(&inner.slots[idx].data))
|
||||
} else {
|
||||
None
|
||||
};
|
||||
if let Some(ref data) = found {
|
||||
inner.stats.hits += 1;
|
||||
inner.stats.bytes_read += data.len() as u64;
|
||||
} else {
|
||||
inner.stats.misses += 1;
|
||||
}
|
||||
found
|
||||
let mut inner = self.lock();
|
||||
let addr = inner.current();
|
||||
inner.get_decompressed(addr, coord)
|
||||
}
|
||||
|
||||
/// Insert decompressed chunk data into the LRU cache.
|
||||
///
|
||||
/// The data is stored in a [`CacheAlignedBuffer`] so subsequent reads
|
||||
/// return cache-line-aligned memory. Returns the `Arc`-shared buffer that
|
||||
/// is now cached (or already was), so the caller can reuse it directly
|
||||
/// instead of holding a separate copy of the same data.
|
||||
/// Insert decompressed chunk data for the bound dataset into the LRU
|
||||
/// cache, returning the `Arc`-shared buffer now cached.
|
||||
pub fn put_decompressed(&self, coord: ChunkCoord, data: Vec<u8>) -> Arc<CacheAlignedBuffer> {
|
||||
let aligned = CacheAlignedBuffer::from_vec(data);
|
||||
self.put_decompressed_aligned(coord, aligned)
|
||||
self.put_decompressed_aligned(coord, CacheAlignedBuffer::from_vec(data))
|
||||
}
|
||||
|
||||
/// Insert an already-aligned buffer into the LRU cache.
|
||||
///
|
||||
/// Returns the `Arc`-shared buffer now held by the cache (the one just
|
||||
/// inserted, or the existing cached copy if `coord` was already present).
|
||||
/// Insert an already-aligned buffer for the bound dataset.
|
||||
pub fn put_decompressed_aligned(
|
||||
&self,
|
||||
coord: ChunkCoord,
|
||||
data: CacheAlignedBuffer,
|
||||
) -> Arc<CacheAlignedBuffer> {
|
||||
let data = Arc::new(data);
|
||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
||||
let data_len = data.len();
|
||||
|
||||
// Don't cache if single chunk exceeds budget — still return the data
|
||||
// to the caller, just don't retain it.
|
||||
if data_len > inner.max_bytes {
|
||||
return data;
|
||||
let mut inner = self.lock();
|
||||
let addr = inner.current();
|
||||
inner.put_decompressed((addr, coord), data)
|
||||
}
|
||||
|
||||
// Check if already present
|
||||
inner.tick += 1;
|
||||
let tick = inner.tick;
|
||||
if let Some(&idx) = inner.slot_index.get(&coord) {
|
||||
inner.slots[idx].last_access = tick;
|
||||
return Arc::clone(&inner.slots[idx].data); // already cached
|
||||
/// [`Self::prefetch_hint_in`] for the bound dataset.
|
||||
pub fn prefetch_hint(&self, next_coords: &[ChunkCoord]) {
|
||||
let addr = self.lock().current();
|
||||
self.prefetch_hint_in(addr, next_coords);
|
||||
}
|
||||
|
||||
// Evict until we have room
|
||||
while inner.slots.len() >= inner.max_slots
|
||||
|| (inner.current_bytes + data_len > inner.max_bytes && !inner.slots.is_empty())
|
||||
{
|
||||
// Find LRU slot
|
||||
let lru_idx = inner
|
||||
.slots
|
||||
.iter()
|
||||
.enumerate()
|
||||
.min_by_key(|(_, s)| s.last_access)
|
||||
.map(|(i, _)| i)
|
||||
.unwrap();
|
||||
let removed = inner.slots.swap_remove(lru_idx);
|
||||
inner.slot_index.remove(&removed.coord);
|
||||
// swap_remove moved the former last element into `lru_idx` (unless
|
||||
// it *was* the last element) — fix up that element's index entry.
|
||||
if lru_idx < inner.slots.len() {
|
||||
let moved_coord = inner.slots[lru_idx].coord.clone();
|
||||
inner.slot_index.insert(moved_coord, lru_idx);
|
||||
}
|
||||
inner.current_bytes -= removed.data.len();
|
||||
inner.stats.evictions += 1;
|
||||
}
|
||||
// ----- Whole-cache operations -----
|
||||
|
||||
inner.current_bytes += data_len;
|
||||
let new_idx = inner.slots.len();
|
||||
inner.slot_index.insert(coord.clone(), new_idx);
|
||||
inner.slots.push(CachedChunk {
|
||||
coord,
|
||||
data: Arc::clone(&data),
|
||||
last_access: tick,
|
||||
});
|
||||
data
|
||||
}
|
||||
|
||||
/// Clear the entire cache (index + decompressed data).
|
||||
/// Clear the entire cache (indexes + decompressed data + stats).
|
||||
pub fn clear(&self) {
|
||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
||||
inner.index = None;
|
||||
inner.index_addr = None;
|
||||
let mut inner = self.lock();
|
||||
inner.datasets.clear();
|
||||
inner.current = None;
|
||||
inner.slots.clear();
|
||||
inner.slot_index.clear();
|
||||
inner.current_bytes = 0;
|
||||
inner.tick = 0;
|
||||
inner.last_coord = None;
|
||||
inner.stats = AccessStats::default();
|
||||
inner.chunk_index = None;
|
||||
inner.chunk_layout = None;
|
||||
}
|
||||
|
||||
/// Record that the given chunk coordinates are predicted to be accessed
|
||||
/// soon (bookkeeping only).
|
||||
///
|
||||
/// This does **not** prefetch or pre-decompress anything — it only
|
||||
/// checks whether each coordinate is already in the chunk index and
|
||||
/// updates access-pattern stats accordingly. Real prefetching (e.g.
|
||||
/// background pre-decompression) is not implemented.
|
||||
pub fn prefetch_hint(&self, next_coords: &[ChunkCoord]) {
|
||||
let inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
||||
if inner.index.is_none() {
|
||||
return;
|
||||
}
|
||||
drop(inner);
|
||||
// For each predicted coordinate, verify it exists in the index.
|
||||
// The index is already populated, so this is a no-op for known chunks.
|
||||
// The purpose is to signal intent — callers can pre-decompress if needed.
|
||||
// We touch the stats to record that prefetch hints were issued.
|
||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
||||
for coord in next_coords {
|
||||
let exists = inner
|
||||
.index
|
||||
.as_ref()
|
||||
.map(|idx| idx.contains_key(coord))
|
||||
.unwrap_or(false);
|
||||
if exists {
|
||||
inner.stats.sequential_count += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Return the current access pattern statistics.
|
||||
pub fn access_stats(&self) -> AccessStats {
|
||||
self.inner
|
||||
.lock()
|
||||
.unwrap_or_else(|e| e.into_inner())
|
||||
.stats
|
||||
.clone()
|
||||
self.lock().stats.clone()
|
||||
}
|
||||
|
||||
/// Update the sweep direction label in the access stats.
|
||||
pub fn set_sweep_direction(&self, direction: &'static str) {
|
||||
self.inner
|
||||
.lock()
|
||||
.unwrap_or_else(|e| e.into_inner())
|
||||
.stats
|
||||
.sweep_direction = Some(direction);
|
||||
self.lock().stats.sweep_direction = Some(direction);
|
||||
}
|
||||
|
||||
/// Number of decompressed chunks currently cached.
|
||||
/// Number of decompressed chunks currently cached (all datasets).
|
||||
pub fn cached_chunk_count(&self) -> usize {
|
||||
self.inner
|
||||
.lock()
|
||||
.unwrap_or_else(|e| e.into_inner())
|
||||
.slots
|
||||
.len()
|
||||
self.lock().slots.len()
|
||||
}
|
||||
|
||||
/// Total bytes of decompressed data currently cached.
|
||||
/// Total bytes of decompressed data currently cached (all datasets).
|
||||
pub fn cached_bytes(&self) -> usize {
|
||||
self.inner
|
||||
.lock()
|
||||
.unwrap_or_else(|e| e.into_inner())
|
||||
.current_bytes
|
||||
self.lock().current_bytes
|
||||
}
|
||||
|
||||
/// Number of datasets whose chunk index is currently kept.
|
||||
pub fn indexed_dataset_count(&self) -> usize {
|
||||
self.lock().datasets.len()
|
||||
}
|
||||
}
|
||||
|
||||
@@ -808,6 +967,92 @@ mod tests {
|
||||
assert_eq!(cache.cached_bytes(), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn datasets_sharing_coordinates_stay_separate() {
|
||||
let cache = ChunkCache::new();
|
||||
let a = vec![make_chunk(vec![0, 0], 0x100, 8)];
|
||||
let b = vec![make_chunk(vec![0, 0], 0x900, 8)];
|
||||
let got_a = cache.chunks_for::<()>(1, 1, || Ok(a.clone())).unwrap();
|
||||
let got_b = cache.chunks_for::<()>(2, 1, || Ok(b.clone())).unwrap();
|
||||
assert_eq!(got_a[0].address, 0x100);
|
||||
assert_eq!(got_b[0].address, 0x900);
|
||||
// Built once per dataset: a second lookup doesn't call the builder.
|
||||
let again = cache
|
||||
.chunks_for::<()>(1, 1, || panic!("index rebuilt"))
|
||||
.unwrap();
|
||||
assert_eq!(again[0].address, 0x100);
|
||||
|
||||
cache.put_decompressed_in(1, vec![0], vec![1; 4]);
|
||||
cache.put_decompressed_in(2, vec![0], vec![2; 4]);
|
||||
assert_eq!(
|
||||
cache.get_decompressed_in(1, &[0]).unwrap().as_slice(),
|
||||
&[1; 4]
|
||||
);
|
||||
assert_eq!(
|
||||
cache.get_decompressed_in(2, &[0]).unwrap().as_slice(),
|
||||
&[2; 4]
|
||||
);
|
||||
assert!(cache.get_decompressed_in(3, &[0]).is_none());
|
||||
assert_eq!(cache.cached_chunk_count(), 2);
|
||||
|
||||
// The bound-dataset methods see only the bound dataset.
|
||||
cache.ensure_dataset(2);
|
||||
assert_eq!(cache.lookup_index(&[0]).unwrap().address, 0x900);
|
||||
assert_eq!(cache.get_decompressed(&[0]).unwrap(), vec![2; 4]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn dataset_indexes_are_bounded() {
|
||||
let cache = ChunkCache::new();
|
||||
for addr in 0..(MAX_INDEXED_DATASETS as u64 + 10) {
|
||||
cache
|
||||
.chunks_for::<()>(addr, 1, || Ok(vec![make_chunk(vec![0], addr, 8)]))
|
||||
.unwrap();
|
||||
}
|
||||
assert_eq!(cache.indexed_dataset_count(), MAX_INDEXED_DATASETS);
|
||||
|
||||
// One huge index evicts the others but is itself kept.
|
||||
let huge: Vec<ChunkInfo> = (0..MAX_INDEXED_CHUNKS as u64)
|
||||
.map(|i| make_chunk(vec![i], i, 8))
|
||||
.collect();
|
||||
let got = cache.chunks_for::<()>(9999, 1, || Ok(huge)).unwrap();
|
||||
assert_eq!(got.len(), MAX_INDEXED_CHUNKS);
|
||||
assert_eq!(cache.indexed_dataset_count(), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn concurrent_readers_of_different_datasets_see_their_own_chunks() {
|
||||
let cache = std::sync::Arc::new(ChunkCache::with_capacity(1 << 20, 64));
|
||||
let handles: Vec<_> = (0..8u64)
|
||||
.map(|t| {
|
||||
let cache = std::sync::Arc::clone(&cache);
|
||||
std::thread::spawn(move || {
|
||||
for round in 0..500u64 {
|
||||
let addr = (t + round) % 16;
|
||||
let coord = vec![round % 4];
|
||||
let chunks = cache
|
||||
.chunks_for::<()>(addr, 1, || {
|
||||
Ok((0..4).map(|c| make_chunk(vec![c], addr, 8)).collect())
|
||||
})
|
||||
.unwrap();
|
||||
assert!(chunks.iter().all(|c| c.address == addr));
|
||||
let want = vec![addr as u8; 8];
|
||||
let got = match cache.get_decompressed_in(addr, &coord) {
|
||||
Some(hit) => hit.to_vec(),
|
||||
None => cache
|
||||
.put_decompressed_in(addr, coord, want.clone())
|
||||
.to_vec(),
|
||||
};
|
||||
assert_eq!(got, want);
|
||||
}
|
||||
})
|
||||
})
|
||||
.collect();
|
||||
for h in handles {
|
||||
h.join().unwrap();
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn duplicate_insert_is_noop() {
|
||||
let cache = ChunkCache::new();
|
||||
|
||||
@@ -0,0 +1,200 @@
|
||||
//! Chunk-index linearisation shared by the Fixed Array and Extensible Array
|
||||
//! chunk indexes (reader and writer).
|
||||
//!
|
||||
//! Both indexes store one element per chunk at a *linear* index, and the
|
||||
//! library derives that index from the chunk's scaled coordinates
|
||||
//! (`offset / chunk_dim`) using the dataset's **maximum** dimensions, not its
|
||||
//! current ones (`H5D__farray_idx_get_addr` / `H5D__earray_idx_get_addr`,
|
||||
//! via `layout->max_down_chunks`). A dataset whose current shape is smaller
|
||||
//! than its maxshape therefore has gaps in the index, and laying it out by the
|
||||
//! current shape puts every chunk after the first row in the wrong place.
|
||||
//!
|
||||
//! The Extensible Array adds one more step: its one unlimited dimension has no
|
||||
//! finite chunk count, so the library *swizzles* the coordinates to make that
|
||||
//! dimension the slowest-varying one (`H5VM_swizzle_coords`, which moves
|
||||
//! `coords[unlim_dim]` to the front and shifts the dimensions before it right
|
||||
//! by one) before linearising with `swizzled_max_down_chunks`. When the
|
||||
//! unlimited dimension is already dimension 0 no swizzle happens.
|
||||
|
||||
#[cfg(not(feature = "std"))]
|
||||
extern crate alloc;
|
||||
|
||||
#[cfg(not(feature = "std"))]
|
||||
use alloc::{vec, vec::Vec};
|
||||
|
||||
use crate::error::FormatError;
|
||||
|
||||
/// How a chunk index maps linear element indexes to chunk coordinates.
|
||||
#[derive(Debug, Clone)]
|
||||
pub(crate) struct ChunkGrid {
|
||||
/// Spatial chunk dimensions, in dataset order.
|
||||
chunk_dims: Vec<u64>,
|
||||
/// Chunks per dimension covering the *current* extent, in dataset order.
|
||||
cur_chunks: Vec<u64>,
|
||||
/// Dataset dimension stored at each linearisation position (slowest
|
||||
/// first). The identity except for a swizzled Extensible Array.
|
||||
order: Vec<usize>,
|
||||
/// Linear stride of each linearisation position.
|
||||
down: Vec<u64>,
|
||||
}
|
||||
|
||||
impl ChunkGrid {
|
||||
/// Grid for a Fixed Array index: row-major over the chunk counts of the
|
||||
/// maximum dimensions (`max_dims`, falling back to the current dimensions
|
||||
/// when the dataspace records none).
|
||||
pub(crate) fn fixed_array(
|
||||
cur_dims: &[u64],
|
||||
max_dims: Option<&[u64]>,
|
||||
chunk_dims: &[u64],
|
||||
) -> Result<Self, FormatError> {
|
||||
Self::build(cur_dims, max_dims, chunk_dims, None)
|
||||
}
|
||||
|
||||
/// Grid for an Extensible Array index: like the Fixed Array, but the
|
||||
/// unlimited dimension (the one whose maximum is `H5S_UNLIMITED`) is moved
|
||||
/// to the slowest-varying position first.
|
||||
pub(crate) fn extensible_array(
|
||||
cur_dims: &[u64],
|
||||
max_dims: Option<&[u64]>,
|
||||
chunk_dims: &[u64],
|
||||
) -> Result<Self, FormatError> {
|
||||
let unlim = max_dims.and_then(|m| m.iter().position(|&d| d == u64::MAX));
|
||||
Self::build(cur_dims, max_dims, chunk_dims, unlim)
|
||||
}
|
||||
|
||||
fn build(
|
||||
cur_dims: &[u64],
|
||||
max_dims: Option<&[u64]>,
|
||||
chunk_dims: &[u64],
|
||||
unlim: Option<usize>,
|
||||
) -> Result<Self, FormatError> {
|
||||
let rank = chunk_dims.len();
|
||||
if cur_dims.len() != rank || max_dims.is_some_and(|m| m.len() != rank) {
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"chunk index rank does not match the dataspace".into(),
|
||||
));
|
||||
}
|
||||
if chunk_dims.contains(&0) {
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"chunk dimension is zero".into(),
|
||||
));
|
||||
}
|
||||
let cur_chunks: Vec<u64> = cur_dims
|
||||
.iter()
|
||||
.zip(chunk_dims)
|
||||
.map(|(&d, &c)| d.div_ceil(c))
|
||||
.collect();
|
||||
// Chunk counts of the maximum extent. An unlimited dimension has no
|
||||
// finite count; it only ever sits in the slowest position, where its
|
||||
// count never enters a stride. A (corrupt) maximum smaller than the
|
||||
// current extent is widened so no allocated chunk becomes unreachable.
|
||||
let max_chunks: Vec<u64> = (0..rank)
|
||||
.map(|d| {
|
||||
let max = max_dims.map_or(cur_dims[d], |m| m[d]);
|
||||
if max == u64::MAX {
|
||||
u64::MAX
|
||||
} else {
|
||||
max.div_ceil(chunk_dims[d]).max(cur_chunks[d])
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
|
||||
let mut order: Vec<usize> = (0..rank).collect();
|
||||
if let Some(u) = unlim {
|
||||
order.remove(u);
|
||||
order.insert(0, u);
|
||||
}
|
||||
let mut down = vec![1u64; rank];
|
||||
for p in (0..rank.saturating_sub(1)).rev() {
|
||||
let next = max_chunks[order[p + 1]];
|
||||
if next == u64::MAX {
|
||||
// Only reachable with more than one unlimited dimension, which
|
||||
// neither index type can describe.
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"array chunk index with more than one unlimited dimension".into(),
|
||||
));
|
||||
}
|
||||
down[p] = down[p + 1].checked_mul(next).ok_or_else(|| {
|
||||
FormatError::Overflow("chunk index linear stride overflows u64".into())
|
||||
})?;
|
||||
}
|
||||
Ok(Self {
|
||||
chunk_dims: chunk_dims.to_vec(),
|
||||
cur_chunks,
|
||||
order,
|
||||
down,
|
||||
})
|
||||
}
|
||||
|
||||
/// Dataset-space offsets of the chunk stored at linear `index`, or `None`
|
||||
/// when that chunk lies outside the current extent (the index still has a
|
||||
/// slot for it; the library ignores such chunks on read).
|
||||
pub(crate) fn offsets(&self, index: u64) -> Option<Vec<u64>> {
|
||||
let rank = self.chunk_dims.len();
|
||||
let mut offsets = vec![0u64; rank];
|
||||
let mut rem = index;
|
||||
for p in 0..rank {
|
||||
let d = self.order[p];
|
||||
let scaled = rem / self.down[p];
|
||||
rem %= self.down[p];
|
||||
if scaled >= self.cur_chunks[d] {
|
||||
return None;
|
||||
}
|
||||
offsets[d] = scaled * self.chunk_dims[d];
|
||||
}
|
||||
Some(offsets)
|
||||
}
|
||||
|
||||
/// Linear index of the chunk with scaled coordinates `scaled`
|
||||
/// (`offset / chunk_dim` per dimension, in dataset order).
|
||||
pub(crate) fn linear_index(&self, scaled: &[u64]) -> u64 {
|
||||
self.order
|
||||
.iter()
|
||||
.zip(&self.down)
|
||||
.map(|(&d, &stride)| scaled[d] * stride)
|
||||
.sum()
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn fixed_array_uses_max_dims() {
|
||||
// shape (4, 6), chunks (2, 3), maxshape (20, 10): 10 x 4 chunk grid.
|
||||
let g = ChunkGrid::fixed_array(&[4, 6], Some(&[20, 10]), &[2, 3]).unwrap();
|
||||
assert_eq!(g.offsets(0), Some(vec![0, 0]));
|
||||
assert_eq!(g.offsets(1), Some(vec![0, 3]));
|
||||
assert_eq!(g.offsets(2), None); // column chunk 2 is beyond the extent
|
||||
assert_eq!(g.offsets(4), Some(vec![2, 0]));
|
||||
assert_eq!(g.offsets(5), Some(vec![2, 3]));
|
||||
assert_eq!(g.offsets(8), None); // row chunk 2 is beyond the extent
|
||||
assert_eq!(g.linear_index(&[1, 1]), 5);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn extensible_array_swizzles_unlimited_dim() {
|
||||
// maxshape (10, None): dim 1 is unlimited and becomes slowest.
|
||||
let g = ChunkGrid::extensible_array(&[4, 6], Some(&[10, u64::MAX]), &[2, 3]).unwrap();
|
||||
// max chunks of dim 0 = 5, so index = c1 * 5 + c0.
|
||||
assert_eq!(g.linear_index(&[1, 0]), 1);
|
||||
assert_eq!(g.linear_index(&[0, 1]), 5);
|
||||
assert_eq!(g.offsets(5), Some(vec![0, 3]));
|
||||
assert_eq!(g.offsets(6), Some(vec![2, 3]));
|
||||
assert_eq!(g.offsets(2), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn extensible_array_unlimited_first_is_row_major() {
|
||||
let g = ChunkGrid::extensible_array(&[4, 6], Some(&[u64::MAX, 30]), &[2, 3]).unwrap();
|
||||
// max chunks of dim 1 = 10.
|
||||
assert_eq!(g.linear_index(&[1, 1]), 11);
|
||||
assert_eq!(g.offsets(11), Some(vec![2, 3]));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rejects_two_unlimited_dims_after_the_first() {
|
||||
assert!(ChunkGrid::fixed_array(&[4, 6], Some(&[u64::MAX, u64::MAX]), &[2, 3]).is_err());
|
||||
}
|
||||
}
|
||||
@@ -15,7 +15,7 @@ use crate::datatype::Datatype;
|
||||
use crate::error::FormatError;
|
||||
use crate::extensible_array::{ExtensibleArrayHeader, read_extensible_array_chunks};
|
||||
use crate::filter_pipeline::FilterPipeline;
|
||||
use crate::filters::decompress_chunk;
|
||||
use crate::filters::{all_filters_skipped, decompress_chunk_masked};
|
||||
use crate::fixed_array::{FixedArrayHeader, read_fixed_array_chunks};
|
||||
#[cfg(feature = "std")]
|
||||
use std::sync::Arc;
|
||||
@@ -65,11 +65,13 @@ fn decompress_all_chunks(
|
||||
let raw_chunk = &file_data[c_addr..c_addr + size];
|
||||
|
||||
let decompressed = if let Some(pl) = pipeline {
|
||||
if chunk_info.filter_mask == 0 {
|
||||
decompress_chunk(raw_chunk, pl, chunk_total_bytes, element_size)?
|
||||
} else {
|
||||
raw_chunk.to_vec()
|
||||
}
|
||||
decompress_chunk_masked(
|
||||
raw_chunk,
|
||||
pl,
|
||||
chunk_total_bytes,
|
||||
element_size,
|
||||
chunk_info.filter_mask,
|
||||
)?
|
||||
} else {
|
||||
raw_chunk.to_vec()
|
||||
};
|
||||
@@ -223,6 +225,10 @@ pub fn collect_chunk_info(
|
||||
collect_chunk_info_inner(file_data, btree_address, ndims, offset_size, length_size, 0)
|
||||
}
|
||||
|
||||
/// Width of each chunk offset in a v1 chunk B-tree key, independent of the
|
||||
/// file's size-of-offsets.
|
||||
const CHUNK_KEY_OFFSET_SIZE: u8 = 8;
|
||||
|
||||
/// Maximum recursion depth for chunk B-tree traversal (malformed/cyclic data
|
||||
/// protection), matching `btree_v1.rs`'s `MAX_BTREE_DEPTH`.
|
||||
const MAX_CHUNK_BTREE_DEPTH: usize = 64;
|
||||
@@ -260,8 +266,14 @@ fn collect_chunk_info_inner(
|
||||
|
||||
let mut pos = offset + 8 + os * 2; // skip left/right sibling
|
||||
|
||||
// Key size: chunk_size(4) + filter_mask(4) + ndims * offset_size
|
||||
let key_size = 4 + 4 + ndims * os;
|
||||
// Key: chunk_size(4) + filter_mask(4) + one offset per dimension. The
|
||||
// offsets are always 8 bytes each — they are dataset coordinates, not file
|
||||
// addresses, so they do not follow the superblock's size-of-offsets (only
|
||||
// the sibling and child addresses do).
|
||||
let key_size = ndims
|
||||
.checked_mul(CHUNK_KEY_OFFSET_SIZE as usize)
|
||||
.and_then(|n| n.checked_add(8))
|
||||
.ok_or_else(|| FormatError::ChunkedReadError("chunk key too large".into()))?;
|
||||
|
||||
if node_level == 0 {
|
||||
// Leaf node: keys and children interleaved
|
||||
@@ -287,8 +299,8 @@ fn collect_chunk_info_inner(
|
||||
let mut offsets = Vec::with_capacity(ndims);
|
||||
let mut kp = pos + 8;
|
||||
for _ in 0..ndims {
|
||||
offsets.push(read_offset(file_data, kp, offset_size)?);
|
||||
kp += os;
|
||||
offsets.push(read_offset(file_data, kp, CHUNK_KEY_OFFSET_SIZE)?);
|
||||
kp += CHUNK_KEY_OFFSET_SIZE as usize;
|
||||
}
|
||||
pos += key_size;
|
||||
|
||||
@@ -507,6 +519,7 @@ pub fn list_chunks(
|
||||
addr_opt,
|
||||
single_filtered_size,
|
||||
single_filter_mask,
|
||||
unfiltered_edges,
|
||||
) = match layout {
|
||||
DataLayout::Chunked {
|
||||
chunk_dimensions,
|
||||
@@ -515,6 +528,7 @@ pub fn list_chunks(
|
||||
chunk_index_type,
|
||||
single_chunk_filtered_size,
|
||||
single_chunk_filter_mask,
|
||||
dont_filter_partial_edge_chunks,
|
||||
} => (
|
||||
chunk_dimensions,
|
||||
*version,
|
||||
@@ -522,6 +536,7 @@ pub fn list_chunks(
|
||||
*btree_address,
|
||||
*single_chunk_filtered_size,
|
||||
*single_chunk_filter_mask,
|
||||
*dont_filter_partial_edge_chunks,
|
||||
),
|
||||
_ => {
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
@@ -554,7 +569,7 @@ pub fn list_chunks(
|
||||
}
|
||||
|
||||
// Collect chunks based on version and index type
|
||||
let chunks = match (version, chunk_index_type) {
|
||||
let mut chunks = match (version, chunk_index_type) {
|
||||
(3, _) => {
|
||||
let ndims = chunk_dimensions.len(); // rank+1
|
||||
collect_chunk_info(file_data, addr, ndims, offset_size, length_size)?
|
||||
@@ -593,6 +608,7 @@ pub fn list_chunks(
|
||||
file_data,
|
||||
&header,
|
||||
&dataspace.dimensions,
|
||||
dataspace.max_dimensions.as_deref(),
|
||||
spatial_chunk_dims,
|
||||
elem_size as u32,
|
||||
offset_size,
|
||||
@@ -608,6 +624,7 @@ pub fn list_chunks(
|
||||
file_data,
|
||||
&header,
|
||||
&dataspace.dimensions,
|
||||
dataspace.max_dimensions.as_deref(),
|
||||
spatial_chunk_dims,
|
||||
elem_size as u32,
|
||||
offset_size,
|
||||
@@ -633,6 +650,23 @@ pub fn list_chunks(
|
||||
}
|
||||
};
|
||||
|
||||
// With "don't filter partial edge chunks", a chunk that extends past the
|
||||
// dataset's extent is stored raw while its filter mask still reads 0.
|
||||
// Mark every filter skipped so all read paths copy it as-is.
|
||||
if unfiltered_edges {
|
||||
for chunk in &mut chunks {
|
||||
let partial = chunk
|
||||
.offsets
|
||||
.iter()
|
||||
.zip(&chunk_dims)
|
||||
.zip(&ds_dims)
|
||||
.any(|((&off, &cd), &dd)| off.saturating_add(cd as u64) > dd as u64);
|
||||
if partial {
|
||||
chunk.filter_mask = u32::MAX;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Ok((chunks, chunk_dims))
|
||||
}
|
||||
|
||||
@@ -804,24 +838,20 @@ pub fn read_chunked_data_cached(
|
||||
)));
|
||||
}
|
||||
|
||||
// The per-file cache is shared across datasets; bind it to this one so a
|
||||
// different dataset's chunk index is never reused for this read.
|
||||
cache.ensure_dataset(addr);
|
||||
|
||||
// Populate chunk index on first access
|
||||
if !cache.has_index() {
|
||||
let (chunks, _) = list_chunks(
|
||||
// The per-file cache is shared across datasets (and threads); every
|
||||
// lookup is keyed by this dataset's chunk-index address, so another
|
||||
// dataset's index or chunks are never used for this read.
|
||||
let chunks = cache.chunks_for(addr, rank, || {
|
||||
list_chunks(
|
||||
file_data,
|
||||
layout,
|
||||
dataspace,
|
||||
elem_size,
|
||||
offset_size,
|
||||
length_size,
|
||||
)?;
|
||||
cache.populate_index(&chunks, rank);
|
||||
}
|
||||
|
||||
let chunks = cache.all_indexed_chunks().unwrap_or_default();
|
||||
)
|
||||
.map(|(chunks, _)| chunks)
|
||||
})?;
|
||||
|
||||
// Assemble output
|
||||
let total_bytes = checked_byte_len(dataspace.checked_num_elements()?, elem_size)?;
|
||||
@@ -876,10 +906,11 @@ pub fn read_chunked_data_cached(
|
||||
};
|
||||
|
||||
// Chunks stored as-is (no pipeline, or the filter mask says this chunk
|
||||
// skipped it) are copied straight from the file bytes: they are already in
|
||||
// memory, so routing them through a Vec and then an aligned cache buffer
|
||||
// was two extra copies of the whole dataset for nothing.
|
||||
let stored_raw = |c: &ChunkInfo| pipeline.is_none() || c.filter_mask != 0;
|
||||
// skipped every filter) are copied straight from the file bytes: they are
|
||||
// already in memory, so routing them through a Vec and then an aligned
|
||||
// cache buffer was two extra copies of the whole dataset for nothing.
|
||||
let stored_raw =
|
||||
|c: &ChunkInfo| pipeline.is_none_or(|pl| all_filters_skipped(pl, c.filter_mask));
|
||||
let mut misses: Vec<&ChunkInfo> = Vec::new();
|
||||
for chunk_info in &chunks {
|
||||
if stored_raw(chunk_info) {
|
||||
@@ -887,7 +918,7 @@ pub fn read_chunked_data_cached(
|
||||
continue;
|
||||
}
|
||||
let coord: Vec<u64> = chunk_info.offsets.iter().take(rank).copied().collect();
|
||||
match cache.get_decompressed_aligned(&coord) {
|
||||
match cache.get_decompressed_in(addr, &coord) {
|
||||
Some(cached) => place(&cached, chunk_info),
|
||||
None => misses.push(chunk_info),
|
||||
}
|
||||
@@ -901,7 +932,13 @@ pub fn read_chunked_data_cached(
|
||||
let cache_them = total_bytes <= cache.max_bytes();
|
||||
if let Some(pl) = pipeline {
|
||||
let decode = |c: &&ChunkInfo| -> Result<Vec<u8>, FormatError> {
|
||||
decompress_chunk(raw_bytes(c)?, pl, chunk_total_bytes, elem_size as u32)
|
||||
decompress_chunk_masked(
|
||||
raw_bytes(c)?,
|
||||
pl,
|
||||
chunk_total_bytes,
|
||||
elem_size as u32,
|
||||
c.filter_mask,
|
||||
)
|
||||
};
|
||||
for batch in misses.chunks(DECODE_BATCH) {
|
||||
#[cfg(feature = "parallel")]
|
||||
@@ -918,7 +955,7 @@ pub fn read_chunked_data_cached(
|
||||
let data = data?;
|
||||
if cache_them {
|
||||
let coord: Vec<u64> = chunk_info.offsets.iter().take(rank).copied().collect();
|
||||
let cached = cache.put_decompressed(coord, data);
|
||||
let cached = cache.put_decompressed_in(addr, coord, data);
|
||||
place(&cached, chunk_info);
|
||||
} else {
|
||||
place(&data, chunk_info);
|
||||
@@ -1122,24 +1159,20 @@ pub fn read_chunked_data_sweep(
|
||||
)));
|
||||
}
|
||||
|
||||
// The per-file cache is shared across datasets; bind it to this one so a
|
||||
// different dataset's chunk index is never reused for this read.
|
||||
cache.ensure_dataset(addr);
|
||||
|
||||
// Populate chunk index on first access
|
||||
if !cache.has_index() {
|
||||
let (chunks, _) = list_chunks(
|
||||
// The per-file cache is shared across datasets (and threads); every
|
||||
// lookup is keyed by this dataset's chunk-index address, so another
|
||||
// dataset's index or chunks are never used for this read.
|
||||
let chunks = cache.chunks_for(addr, rank, || {
|
||||
list_chunks(
|
||||
file_data,
|
||||
layout,
|
||||
dataspace,
|
||||
elem_size,
|
||||
offset_size,
|
||||
length_size,
|
||||
)?;
|
||||
cache.populate_index(&chunks, rank);
|
||||
}
|
||||
|
||||
let chunks = cache.all_indexed_chunks().unwrap_or_default();
|
||||
)
|
||||
.map(|(chunks, _)| chunks)
|
||||
})?;
|
||||
|
||||
// Assemble output
|
||||
let total_bytes = checked_byte_len(dataspace.checked_num_elements()?, elem_size)?;
|
||||
@@ -1170,12 +1203,12 @@ pub fn read_chunked_data_sweep(
|
||||
|
||||
// Issue prefetch hint for predicted next chunks
|
||||
if !sweep.predicted_next.is_empty() {
|
||||
cache.prefetch_hint(&sweep.predicted_next);
|
||||
cache.prefetch_hint_in(addr, &sweep.predicted_next);
|
||||
cache.set_sweep_direction(sweep.direction);
|
||||
}
|
||||
|
||||
// Try decompressed cache first
|
||||
let decompressed = if let Some(cached) = cache.get_decompressed_aligned(&coord) {
|
||||
let decompressed = if let Some(cached) = cache.get_decompressed_in(addr, &coord) {
|
||||
cached
|
||||
} else {
|
||||
// Decompress from file
|
||||
@@ -1184,15 +1217,17 @@ pub fn read_chunked_data_sweep(
|
||||
ensure_len(file_data, c_addr, size)?;
|
||||
let raw_chunk = &file_data[c_addr..c_addr + size];
|
||||
let dec = if let Some(pl) = pipeline {
|
||||
if chunk_info.filter_mask == 0 {
|
||||
decompress_chunk(raw_chunk, pl, chunk_total_bytes, elem_size as u32)?
|
||||
} else {
|
||||
raw_chunk.to_vec()
|
||||
}
|
||||
decompress_chunk_masked(
|
||||
raw_chunk,
|
||||
pl,
|
||||
chunk_total_bytes,
|
||||
elem_size as u32,
|
||||
chunk_info.filter_mask,
|
||||
)?
|
||||
} else {
|
||||
raw_chunk.to_vec()
|
||||
};
|
||||
cache.put_decompressed(coord, dec)
|
||||
cache.put_decompressed_in(addr, coord, dec)
|
||||
};
|
||||
|
||||
let chunk_offsets: Vec<usize> = chunk_info
|
||||
@@ -1276,48 +1311,34 @@ pub fn read_chunked_data_indexed(
|
||||
)));
|
||||
}
|
||||
|
||||
// The per-file cache is shared across datasets; bind it to this one so a
|
||||
// different dataset's chunk index is never reused for this read.
|
||||
cache.ensure_dataset(addr);
|
||||
|
||||
// Build chunk index on first access
|
||||
if !cache.has_chunk_index() {
|
||||
let (chunks, _) = list_chunks(
|
||||
// Chunk index and assembly plan for this dataset, built on first access
|
||||
// and kept per dataset (keyed by chunk-index address) in the shared cache.
|
||||
let plan = cache.chunk_layout_for(
|
||||
addr,
|
||||
rank,
|
||||
|| {
|
||||
list_chunks(
|
||||
file_data,
|
||||
layout,
|
||||
dataspace,
|
||||
elem_size,
|
||||
offset_size,
|
||||
length_size,
|
||||
)
|
||||
.map(|(chunks, _)| chunks)
|
||||
},
|
||||
&ds_dims,
|
||||
&chunk_dims,
|
||||
elem_size,
|
||||
)?;
|
||||
cache.populate_chunk_index(&chunks, rank);
|
||||
// Also populate the legacy index for compatibility
|
||||
if !cache.has_index() {
|
||||
cache.populate_index(&chunks, rank);
|
||||
}
|
||||
}
|
||||
|
||||
// Build chunk layout on first access
|
||||
if !cache.has_chunk_layout() {
|
||||
cache.populate_chunk_layout(&ds_dims, &chunk_dims, elem_size);
|
||||
}
|
||||
|
||||
// Get the layout info (mappings, output size, chunk total bytes)
|
||||
let (mappings_info, output_bytes, chunk_total_bytes) = cache
|
||||
.with_chunk_layout(|layout| {
|
||||
let info: Vec<_> = layout
|
||||
.mappings
|
||||
.iter()
|
||||
.map(|m| (m.coord.clone(), m.file_offset, m.file_size, m.filter_mask))
|
||||
.collect();
|
||||
(info, layout.output_bytes, layout.chunk_total_bytes)
|
||||
})
|
||||
.ok_or_else(|| FormatError::ChunkedReadError("chunk layout not available".into()))?;
|
||||
let chunk_total_bytes = plan.chunk_total_bytes;
|
||||
|
||||
// Decompress chunks (using LRU cache where possible)
|
||||
let mut chunk_buffers: Vec<Arc<CacheAlignedBuffer>> = Vec::with_capacity(mappings_info.len());
|
||||
for (coord, file_offset, file_size, filter_mask) in &mappings_info {
|
||||
if let Some(cached) = cache.get_decompressed_aligned(coord) {
|
||||
let mut chunk_buffers: Vec<Arc<CacheAlignedBuffer>> = Vec::with_capacity(plan.mappings.len());
|
||||
for m in &plan.mappings {
|
||||
let (coord, file_offset, file_size, filter_mask) =
|
||||
(&m.coord, &m.file_offset, &m.file_size, &m.filter_mask);
|
||||
if let Some(cached) = cache.get_decompressed_in(addr, coord) {
|
||||
chunk_buffers.push(cached);
|
||||
} else {
|
||||
let c_addr = *file_offset as usize;
|
||||
@@ -1325,26 +1346,26 @@ pub fn read_chunked_data_indexed(
|
||||
ensure_len(file_data, c_addr, size)?;
|
||||
let raw_chunk = &file_data[c_addr..c_addr + size];
|
||||
let decompressed = if let Some(pl) = pipeline {
|
||||
if *filter_mask == 0 {
|
||||
decompress_chunk(raw_chunk, pl, chunk_total_bytes, elem_size as u32)?
|
||||
} else {
|
||||
raw_chunk.to_vec()
|
||||
}
|
||||
decompress_chunk_masked(
|
||||
raw_chunk,
|
||||
pl,
|
||||
chunk_total_bytes,
|
||||
elem_size as u32,
|
||||
*filter_mask,
|
||||
)?
|
||||
} else {
|
||||
raw_chunk.to_vec()
|
||||
};
|
||||
let aligned = CacheAlignedBuffer::from_vec(decompressed);
|
||||
let arc = cache.put_decompressed_aligned(coord.clone(), aligned);
|
||||
let arc = cache.put_decompressed_aligned_in(addr, coord.clone(), aligned);
|
||||
chunk_buffers.push(arc);
|
||||
}
|
||||
}
|
||||
|
||||
// Assemble using pre-computed layout
|
||||
let mut output = vec![0u8; output_bytes];
|
||||
let mut output = vec![0u8; plan.output_bytes];
|
||||
let data_refs: Vec<&[u8]> = chunk_buffers.iter().map(|b| b.as_slice()).collect();
|
||||
cache.with_chunk_layout(|layout| {
|
||||
layout.assemble(&data_refs, &mut output);
|
||||
});
|
||||
plan.assemble(&data_refs, &mut output);
|
||||
|
||||
Ok(output)
|
||||
}
|
||||
@@ -1592,7 +1613,8 @@ mod tests {
|
||||
} else {
|
||||
0
|
||||
};
|
||||
write_offset(&mut buf, off, offset_size);
|
||||
// Key offsets are always 8 bytes (they are coordinates).
|
||||
write_offset(&mut buf, off, 8);
|
||||
}
|
||||
// Child: address
|
||||
write_offset(&mut buf, chunk.address, offset_size);
|
||||
@@ -1602,7 +1624,7 @@ mod tests {
|
||||
buf.extend_from_slice(&0u32.to_le_bytes()); // chunk_size
|
||||
buf.extend_from_slice(&0u32.to_le_bytes()); // filter_mask
|
||||
for _ in 0..ndims {
|
||||
write_offset(&mut buf, u64::MAX, offset_size);
|
||||
write_offset(&mut buf, u64::MAX, 8);
|
||||
}
|
||||
|
||||
buf
|
||||
@@ -1680,6 +1702,37 @@ mod tests {
|
||||
assert_eq!(result[2].address, 0x300);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn collect_chunks_with_four_byte_addresses() {
|
||||
// Sibling and child addresses are 4 bytes; the key offsets stay 8.
|
||||
let ndims = 3;
|
||||
let os: u8 = 4;
|
||||
let chunks = vec![
|
||||
ChunkInfo {
|
||||
chunk_size: 80,
|
||||
filter_mask: 2,
|
||||
offsets: vec![0, 5, 0],
|
||||
address: 0x1000,
|
||||
},
|
||||
ChunkInfo {
|
||||
chunk_size: 96,
|
||||
filter_mask: 0,
|
||||
offsets: vec![8, 10, 0],
|
||||
address: 0x2000,
|
||||
},
|
||||
];
|
||||
let btree = build_chunk_btree_leaf(&chunks, ndims, os);
|
||||
assert_eq!(btree.len(), 8 + 2 * 4 + 2 * (8 + 3 * 8 + 4) + (8 + 3 * 8));
|
||||
let result = collect_chunk_info(&btree, 0, ndims, os, os).unwrap();
|
||||
assert_eq!(result.len(), 2);
|
||||
for (got, want) in result.iter().zip(&chunks) {
|
||||
assert_eq!(got.offsets, want.offsets);
|
||||
assert_eq!(got.address, want.address);
|
||||
assert_eq!(got.chunk_size, want.chunk_size);
|
||||
assert_eq!(got.filter_mask, want.filter_mask);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn collect_empty_btree() {
|
||||
let ndims = 2;
|
||||
@@ -1774,6 +1827,7 @@ mod tests {
|
||||
chunk_index_type: None,
|
||||
single_chunk_filtered_size: None,
|
||||
single_chunk_filter_mask: None,
|
||||
dont_filter_partial_edge_chunks: false,
|
||||
};
|
||||
|
||||
let dataspace = Dataspace {
|
||||
@@ -1797,6 +1851,7 @@ mod tests {
|
||||
chunk_index_type: None,
|
||||
single_chunk_filtered_size: None,
|
||||
single_chunk_filter_mask: None,
|
||||
dont_filter_partial_edge_chunks: false,
|
||||
};
|
||||
let dataspace = Dataspace {
|
||||
space_type: DataspaceType::Simple,
|
||||
@@ -1954,6 +2009,7 @@ mod tests {
|
||||
chunk_index_type: None,
|
||||
single_chunk_filtered_size: None,
|
||||
single_chunk_filter_mask: None,
|
||||
dont_filter_partial_edge_chunks: false,
|
||||
};
|
||||
let dataspace = Dataspace {
|
||||
space_type: DataspaceType::Simple,
|
||||
@@ -2036,6 +2092,7 @@ mod tests {
|
||||
chunk_index_type: None,
|
||||
single_chunk_filtered_size: None,
|
||||
single_chunk_filter_mask: None,
|
||||
dont_filter_partial_edge_chunks: false,
|
||||
};
|
||||
let dataspace = Dataspace {
|
||||
space_type: DataspaceType::Simple,
|
||||
@@ -2198,6 +2255,7 @@ mod tests {
|
||||
chunk_index_type: Some(1),
|
||||
single_chunk_filtered_size: None,
|
||||
single_chunk_filter_mask: None,
|
||||
dont_filter_partial_edge_chunks: false,
|
||||
};
|
||||
let dataspace = Dataspace {
|
||||
space_type: DataspaceType::Simple,
|
||||
@@ -2227,12 +2285,12 @@ mod tests {
|
||||
let datatype = make_f64_type();
|
||||
let cache = ChunkCache::new();
|
||||
|
||||
assert!(!cache.has_index());
|
||||
assert_eq!(cache.indexed_dataset_count(), 0);
|
||||
let raw = read_chunked_data_cached(
|
||||
&file_data, &layout, &dataspace, &datatype, None, 8, 8, &cache,
|
||||
)
|
||||
.unwrap();
|
||||
assert!(cache.has_index());
|
||||
assert_eq!(cache.indexed_dataset_count(), 1);
|
||||
assert_eq!(raw.len(), 20 * 8);
|
||||
for i in 0..20 {
|
||||
let val = f64::from_le_bytes(raw[i * 8..(i + 1) * 8].try_into().unwrap());
|
||||
@@ -2254,7 +2312,7 @@ mod tests {
|
||||
&file_data, &layout, &dataspace, &datatype, None, 8, 8, &cache,
|
||||
)
|
||||
.unwrap();
|
||||
assert!(cache.has_index());
|
||||
assert_eq!(cache.indexed_dataset_count(), 1);
|
||||
assert_eq!(cache.cached_chunk_count(), 0);
|
||||
|
||||
// Second read — reuses the cached index
|
||||
@@ -2263,6 +2321,7 @@ mod tests {
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(raw1, raw2);
|
||||
assert_eq!(cache.indexed_dataset_count(), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
|
||||
@@ -4,15 +4,16 @@
|
||||
extern crate alloc;
|
||||
|
||||
#[cfg(not(feature = "std"))]
|
||||
use alloc::{vec, vec::Vec};
|
||||
use alloc::{format, vec, vec::Vec};
|
||||
|
||||
use crate::checksum::jenkins_lookup3;
|
||||
use crate::chunk_cache::{CACHE_LINE_SIZE, align_to_cache_line};
|
||||
use crate::chunk_grid::ChunkGrid;
|
||||
use crate::ea_writer;
|
||||
use crate::error::FormatError;
|
||||
use crate::filter_pipeline::{
|
||||
FILTER_DEFLATE, FILTER_FLETCHER32, FILTER_LZ4, FILTER_PCODEC, FILTER_SHUFFLE, FILTER_ZSTD,
|
||||
FilterDescription, FilterPipeline,
|
||||
FILTER_DEFLATE, FILTER_FLETCHER32, FILTER_LZ4, FILTER_PCODEC, FILTER_PCODEC_NAME,
|
||||
FILTER_SHUFFLE, FILTER_ZSTD, FilterDescription, FilterPipeline,
|
||||
};
|
||||
use crate::filters::compress_chunk;
|
||||
/// Round a file offset up to the next cache-line boundary.
|
||||
@@ -44,7 +45,8 @@ pub struct ChunkOptions {
|
||||
pub lz4: bool,
|
||||
/// Zstandard compression level (1-22), None = no zstd. Filter ID 32015.
|
||||
pub zstd_level: Option<u32>,
|
||||
/// Pcodec lossless numerical compression. Filter ID 32023.
|
||||
/// Pcodec lossless numerical compression. Private, unregistered filter
|
||||
/// ID [`FILTER_PCODEC`] (480): only clawhdf5 can read it.
|
||||
pub pcodec: bool,
|
||||
}
|
||||
|
||||
@@ -115,7 +117,7 @@ impl ChunkOptions {
|
||||
if self.pcodec {
|
||||
filters.push(FilterDescription {
|
||||
filter_id: FILTER_PCODEC,
|
||||
name: Some("pcodec".into()),
|
||||
name: Some(FILTER_PCODEC_NAME.into()),
|
||||
flags: 0,
|
||||
client_data: vec![element_size],
|
||||
});
|
||||
@@ -443,6 +445,27 @@ fn serialize_v4_fixed_array(
|
||||
element_size: u32,
|
||||
max_bits: u8,
|
||||
) -> Vec<u8> {
|
||||
let mut buf = layout_v4_chunked_prefix(chunk_dims, element_size);
|
||||
|
||||
// chunk index type = 3 (Fixed Array)
|
||||
buf.push(3);
|
||||
|
||||
// max_dblk_page_nelmts_bits — must match FAHD max_nelmts_bits
|
||||
buf.push(max_bits);
|
||||
|
||||
// Fixed Array header address
|
||||
match offset_size {
|
||||
4 => buf.extend_from_slice(&(fixed_array_address as u32).to_le_bytes()),
|
||||
8 => buf.extend_from_slice(&fixed_array_address.to_le_bytes()),
|
||||
_ => {}
|
||||
}
|
||||
|
||||
buf
|
||||
}
|
||||
|
||||
/// The part of a v4 chunked layout message before the chunk index type:
|
||||
/// version, class, flags and the chunk dimensions (plus the element size).
|
||||
fn layout_v4_chunked_prefix(chunk_dims: &[u32], element_size: u32) -> Vec<u8> {
|
||||
let mut buf = Vec::new();
|
||||
buf.push(4); // version
|
||||
buf.push(2); // class = chunked
|
||||
@@ -482,124 +505,142 @@ fn serialize_v4_fixed_array(
|
||||
4 => buf.extend_from_slice(&element_size.to_le_bytes()),
|
||||
_ => {}
|
||||
}
|
||||
|
||||
// chunk index type = 3 (Fixed Array)
|
||||
buf.push(3);
|
||||
|
||||
// max_dblk_page_nelmts_bits — must match FAHD max_nelmts_bits
|
||||
buf.push(max_bits);
|
||||
|
||||
// Fixed Array header address
|
||||
match offset_size {
|
||||
4 => buf.extend_from_slice(&(fixed_array_address as u32).to_le_bytes()),
|
||||
8 => buf.extend_from_slice(&fixed_array_address.to_le_bytes()),
|
||||
_ => {}
|
||||
}
|
||||
|
||||
buf
|
||||
}
|
||||
|
||||
/// log2 of the elements per Fixed Array data block page (the library's
|
||||
/// default, `H5D_FARRAY_MAX_DBLK_PAGE_NELMTS_BITS`).
|
||||
const FA_PAGE_BITS: u8 = 10;
|
||||
|
||||
pub(crate) fn push_addr(buf: &mut Vec<u8>, addr: u64, offset_size: u8) {
|
||||
match offset_size {
|
||||
4 => buf.extend_from_slice(&(addr as u32).to_le_bytes()),
|
||||
_ => buf.extend_from_slice(&addr.to_le_bytes()),
|
||||
}
|
||||
}
|
||||
|
||||
/// Width of the chunk-size field of a filtered chunk index element. Must
|
||||
/// match the library's `H5D_FARRAY_FILT_COMPUTE_CHUNK_SIZE_LEN` (the EA and
|
||||
/// B-tree v2 indexes use the same formula):
|
||||
/// `1 + ((log2(unfiltered chunk bytes) + 8) / 8)`, capped at 8.
|
||||
pub(crate) fn filtered_chunk_size_len(slots: &[Option<WrittenChunk>]) -> usize {
|
||||
let max_raw = slots
|
||||
.iter()
|
||||
.flatten()
|
||||
.map(|c| c.raw_size)
|
||||
.max()
|
||||
.unwrap_or(1);
|
||||
let log2_val = if max_raw <= 1 {
|
||||
0
|
||||
} else {
|
||||
63 - max_raw.leading_zeros()
|
||||
};
|
||||
(1 + ((log2_val + 8) / 8) as usize).min(8)
|
||||
}
|
||||
|
||||
/// Append one chunk index element: the chunk's address, plus its stored size
|
||||
/// and filter mask when the dataset is filtered. `None` is an unallocated
|
||||
/// chunk (undefined address, zero size and mask).
|
||||
pub(crate) fn push_index_element(
|
||||
buf: &mut Vec<u8>,
|
||||
slot: Option<&WrittenChunk>,
|
||||
offset_size: u8,
|
||||
chunk_size_bytes: Option<usize>,
|
||||
) {
|
||||
match slot {
|
||||
Some(c) => {
|
||||
push_addr(buf, c.address, offset_size);
|
||||
if let Some(n) = chunk_size_bytes {
|
||||
buf.extend_from_slice(&c.compressed_size.to_le_bytes()[..n]);
|
||||
buf.extend_from_slice(&c.filter_mask.to_le_bytes());
|
||||
}
|
||||
}
|
||||
None => {
|
||||
buf.extend(core::iter::repeat_n(0xFF, offset_size as usize));
|
||||
if let Some(n) = chunk_size_bytes {
|
||||
buf.extend(core::iter::repeat_n(0x00, n + 4));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Build a complete Fixed Array at a known absolute address.
|
||||
///
|
||||
/// `slots` holds one entry per element of the array, i.e. per chunk of the
|
||||
/// dataset's *maximum* extent in the order [`crate::chunk_grid`] defines;
|
||||
/// `None` marks a chunk that is not allocated. An array with more elements
|
||||
/// than fit in one page (`2^FA_PAGE_BITS`) gets a paged data block: a
|
||||
/// page-init bitmap after the prefix, then one checksummed page per
|
||||
/// `2^FA_PAGE_BITS` elements, the last one short (`H5FA__dblock_create`).
|
||||
pub fn build_fixed_array_at(
|
||||
chunks: &[WrittenChunk],
|
||||
slots: &[Option<WrittenChunk>],
|
||||
offset_size: u8,
|
||||
length_size: u8,
|
||||
has_filters: bool,
|
||||
fa_base_address: u64,
|
||||
) -> Vec<u8> {
|
||||
let os = offset_size as usize;
|
||||
let num_elements = chunks.len();
|
||||
|
||||
// For filtered chunks, compute chunk_size encoding width.
|
||||
// Must match the HDF5 C library's H5D_FARRAY_FILT_COMPUTE_CHUNK_SIZE_LEN macro:
|
||||
// chunk_size_len = 1 + ((H5VM_log2_gen(chunk.size) + 8) / 8)
|
||||
// where chunk.size is the unfiltered chunk size in bytes (product of all chunk dims).
|
||||
let chunk_size_bytes: usize = if has_filters {
|
||||
let max_raw = chunks.iter().map(|c| c.raw_size).max().unwrap_or(1);
|
||||
let log2_val = if max_raw <= 1 {
|
||||
0
|
||||
} else {
|
||||
63 - max_raw.leading_zeros()
|
||||
};
|
||||
let len = 1 + ((log2_val + 8) / 8) as usize;
|
||||
len.min(8)
|
||||
} else {
|
||||
0
|
||||
};
|
||||
|
||||
let elem_size = if has_filters {
|
||||
os + chunk_size_bytes + 4
|
||||
} else {
|
||||
os
|
||||
};
|
||||
let num_elements = slots.len();
|
||||
|
||||
let chunk_size_bytes = has_filters.then(|| filtered_chunk_size_len(slots));
|
||||
let elem_size = os + chunk_size_bytes.map_or(0, |n| n + 4);
|
||||
let client_id: u8 = if has_filters { 1 } else { 0 };
|
||||
|
||||
// FAHD total size
|
||||
let nelmts_field_size = length_size as usize;
|
||||
let fahd_total_size = 4 + 1 + 1 + 1 + 1 + nelmts_field_size + os + 4;
|
||||
let fahd_total_size = 4 + 1 + 1 + 1 + 1 + length_size as usize + os + 4;
|
||||
let fadb_address = fa_base_address + fahd_total_size as u64;
|
||||
|
||||
// Build FAHD
|
||||
let mut fahd = Vec::with_capacity(fahd_total_size);
|
||||
fahd.extend_from_slice(b"FAHD");
|
||||
fahd.push(0); // version
|
||||
fahd.push(client_id);
|
||||
fahd.push(elem_size as u8);
|
||||
|
||||
// max_nelmts_bits: use 10 as default (page_size = 1024), matching h5py convention
|
||||
let max_bits: u8 = 10;
|
||||
fahd.push(max_bits);
|
||||
|
||||
fahd.push(FA_PAGE_BITS);
|
||||
match length_size {
|
||||
4 => fahd.extend_from_slice(&(num_elements as u32).to_le_bytes()),
|
||||
8 => fahd.extend_from_slice(&(num_elements as u64).to_le_bytes()),
|
||||
_ => fahd.extend_from_slice(&(num_elements as u64).to_le_bytes()),
|
||||
}
|
||||
|
||||
match offset_size {
|
||||
4 => fahd.extend_from_slice(&(fadb_address as u32).to_le_bytes()),
|
||||
8 => fahd.extend_from_slice(&fadb_address.to_le_bytes()),
|
||||
_ => fahd.extend_from_slice(&fadb_address.to_le_bytes()),
|
||||
}
|
||||
|
||||
// Checksum
|
||||
push_addr(&mut fahd, fadb_address, offset_size);
|
||||
let checksum = jenkins_lookup3(&fahd);
|
||||
fahd.extend_from_slice(&checksum.to_le_bytes());
|
||||
|
||||
assert_eq!(fahd.len(), fahd_total_size);
|
||||
|
||||
// Build FADB
|
||||
// FADB prefix
|
||||
let mut fadb = Vec::new();
|
||||
fadb.extend_from_slice(b"FADB");
|
||||
fadb.push(0); // version
|
||||
fadb.push(client_id);
|
||||
push_addr(&mut fadb, fa_base_address, offset_size);
|
||||
|
||||
// header address
|
||||
match offset_size {
|
||||
4 => fadb.extend_from_slice(&(fa_base_address as u32).to_le_bytes()),
|
||||
8 => fadb.extend_from_slice(&fa_base_address.to_le_bytes()),
|
||||
_ => fadb.extend_from_slice(&fa_base_address.to_le_bytes()),
|
||||
let page_nelmts = 1usize << FA_PAGE_BITS;
|
||||
if num_elements <= page_nelmts {
|
||||
// Unpaged: the elements follow the prefix, one checksum over both.
|
||||
for slot in slots {
|
||||
push_index_element(&mut fadb, slot.as_ref(), offset_size, chunk_size_bytes);
|
||||
}
|
||||
|
||||
// Element data
|
||||
for chunk in chunks {
|
||||
match offset_size {
|
||||
4 => fadb.extend_from_slice(&(chunk.address as u32).to_le_bytes()),
|
||||
8 => fadb.extend_from_slice(&chunk.address.to_le_bytes()),
|
||||
_ => fadb.extend_from_slice(&chunk.address.to_le_bytes()),
|
||||
}
|
||||
if has_filters {
|
||||
// Write compressed size using chunk_size_bytes (variable width)
|
||||
let cs_bytes = chunk.compressed_size.to_le_bytes();
|
||||
fadb.extend_from_slice(&cs_bytes[..chunk_size_bytes]);
|
||||
fadb.extend_from_slice(&chunk.filter_mask.to_le_bytes());
|
||||
}
|
||||
}
|
||||
|
||||
// FADB checksum
|
||||
let fadb_checksum = jenkins_lookup3(&fadb);
|
||||
fadb.extend_from_slice(&fadb_checksum.to_le_bytes());
|
||||
} else {
|
||||
// Paged: every page is written, so every page-init bit is set
|
||||
// (MSB-first, as `H5VM_bit_set` packs them). The prefix and bitmap
|
||||
// share a checksum; each page carries its own.
|
||||
let npages = num_elements.div_ceil(page_nelmts);
|
||||
let mut bitmap = vec![0u8; npages.div_ceil(8)];
|
||||
for p in 0..npages {
|
||||
bitmap[p / 8] |= 0x80 >> (p % 8);
|
||||
}
|
||||
fadb.extend_from_slice(&bitmap);
|
||||
let prefix_checksum = jenkins_lookup3(&fadb);
|
||||
fadb.extend_from_slice(&prefix_checksum.to_le_bytes());
|
||||
for page in slots.chunks(page_nelmts) {
|
||||
let start = fadb.len();
|
||||
for slot in page {
|
||||
push_index_element(&mut fadb, slot.as_ref(), offset_size, chunk_size_bytes);
|
||||
}
|
||||
let page_checksum = jenkins_lookup3(&fadb[start..]);
|
||||
fadb.extend_from_slice(&page_checksum.to_le_bytes());
|
||||
}
|
||||
}
|
||||
|
||||
let mut combined = fahd;
|
||||
combined.extend_from_slice(&fadb);
|
||||
@@ -667,7 +708,8 @@ pub fn build_chunked_data_from_precompressed(
|
||||
pre: &PrecompressedChunks,
|
||||
base_address: u64,
|
||||
maxshape: Option<&[u64]>,
|
||||
) -> ChunkedDataResult {
|
||||
) -> Result<ChunkedDataResult, FormatError> {
|
||||
let index = ChunkIndexPlan::new(&pre.shape, maxshape, &pre.chunk_dims)?;
|
||||
let offset_size: u8 = 8;
|
||||
let length_size: u8 = 8;
|
||||
let num_chunks = pre.chunks.len();
|
||||
@@ -693,17 +735,18 @@ pub fn build_chunked_data_from_precompressed(
|
||||
}
|
||||
|
||||
let chunk_dims_u32: Vec<u32> = pre.chunk_dims.iter().map(|&d| d as u32).collect();
|
||||
let use_extensible = maxshape.is_some_and(|ms| ms.contains(&u64::MAX));
|
||||
|
||||
let aligned_idx = align_to_cache_line(data_buf.len());
|
||||
if aligned_idx > data_buf.len() {
|
||||
data_buf.resize(aligned_idx, 0u8);
|
||||
}
|
||||
|
||||
let layout_message = if use_extensible {
|
||||
let layout_message = match &index {
|
||||
ChunkIndexPlan::ExtensibleArray(grid) => {
|
||||
let ea_address = base_address + data_buf.len() as u64;
|
||||
let slots = index_slots(grid, &pre.shape, &pre.chunk_dims, &written_chunks, None)?;
|
||||
let ea_bytes = ea_writer::build_extensible_array_at(
|
||||
&written_chunks,
|
||||
&slots,
|
||||
offset_size,
|
||||
length_size,
|
||||
pre.has_filters,
|
||||
@@ -716,7 +759,8 @@ pub fn build_chunked_data_from_precompressed(
|
||||
offset_size,
|
||||
element_size as u32,
|
||||
)
|
||||
} else if num_chunks == 1 {
|
||||
}
|
||||
ChunkIndexPlan::SingleChunk => {
|
||||
let chunk_addr = written_chunks[0].address;
|
||||
let filtered_size = if pre.has_filters {
|
||||
Some(written_chunks[0].compressed_size)
|
||||
@@ -732,10 +776,18 @@ pub fn build_chunked_data_from_precompressed(
|
||||
offset_size,
|
||||
element_size as u32,
|
||||
)
|
||||
} else {
|
||||
}
|
||||
ChunkIndexPlan::FixedArray(grid, nslots) => {
|
||||
let fa_address = base_address + data_buf.len() as u64;
|
||||
let fa_bytes = build_fixed_array_at(
|
||||
let slots = index_slots(
|
||||
grid,
|
||||
&pre.shape,
|
||||
&pre.chunk_dims,
|
||||
&written_chunks,
|
||||
Some(*nslots),
|
||||
)?;
|
||||
let fa_bytes = build_fixed_array_at(
|
||||
&slots,
|
||||
offset_size,
|
||||
length_size,
|
||||
pre.has_filters,
|
||||
@@ -747,15 +799,263 @@ pub fn build_chunked_data_from_precompressed(
|
||||
fa_address,
|
||||
offset_size,
|
||||
element_size as u32,
|
||||
10, // max_nelmts_bits — matches h5py convention
|
||||
FA_PAGE_BITS,
|
||||
)
|
||||
}
|
||||
ChunkIndexPlan::BTreeV2 => {
|
||||
let bt_address = base_address + data_buf.len() as u64;
|
||||
let records: Vec<(Vec<u64>, &WrittenChunk)> = written_chunks
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(i, c)| (scaled_coords(&pre.shape, &pre.chunk_dims, i), c))
|
||||
.collect();
|
||||
let (bt_bytes, node_size) = build_btree_v2_chunk_index_at(
|
||||
pre.shape.len(),
|
||||
&records,
|
||||
offset_size,
|
||||
length_size,
|
||||
pre.has_filters,
|
||||
bt_address,
|
||||
)?;
|
||||
data_buf.extend_from_slice(&bt_bytes);
|
||||
serialize_v4_btree_v2(
|
||||
&chunk_dims_u32,
|
||||
bt_address,
|
||||
offset_size,
|
||||
element_size as u32,
|
||||
node_size,
|
||||
)
|
||||
}
|
||||
};
|
||||
|
||||
ChunkedDataResult {
|
||||
Ok(ChunkedDataResult {
|
||||
data_bytes: data_buf,
|
||||
layout_message,
|
||||
pipeline_message: pre.pipeline_message.clone(),
|
||||
})
|
||||
}
|
||||
|
||||
/// Most slots a Fixed Array index may have before we refuse to build it: its
|
||||
/// data block holds one element per chunk of the *maximum* extent, so a huge
|
||||
/// finite maxshape with small chunks would otherwise exhaust memory.
|
||||
const MAX_FIXED_ARRAY_SLOTS: u64 = 1 << 26;
|
||||
|
||||
/// Which chunk index a dataset gets, following the library's choice in
|
||||
/// `H5D__layout_set_latest_indexing`: version-2 B-tree for more than one
|
||||
/// unlimited dimension, Extensible Array for exactly one, Fixed Array for a
|
||||
/// finite maxshape, Single Chunk when the whole maximum extent is one chunk.
|
||||
enum ChunkIndexPlan {
|
||||
SingleChunk,
|
||||
/// The grid and the number of array elements (chunks of the max extent).
|
||||
FixedArray(ChunkGrid, usize),
|
||||
ExtensibleArray(ChunkGrid),
|
||||
BTreeV2,
|
||||
}
|
||||
|
||||
impl ChunkIndexPlan {
|
||||
fn new(
|
||||
shape: &[u64],
|
||||
maxshape: Option<&[u64]>,
|
||||
chunk_dims: &[u64],
|
||||
) -> Result<Self, FormatError> {
|
||||
let bad = |what: &str| FormatError::ChunkedReadError(format!("maxshape: {what}"));
|
||||
if let Some(ms) = maxshape {
|
||||
if ms.len() != shape.len() {
|
||||
return Err(bad("rank differs from the shape"));
|
||||
}
|
||||
if ms.iter().zip(shape).any(|(&m, &s)| m < s) {
|
||||
return Err(bad("smaller than the shape"));
|
||||
}
|
||||
}
|
||||
let max = maxshape.unwrap_or(shape);
|
||||
let nunlim = max.iter().filter(|&&d| d == u64::MAX).count();
|
||||
match nunlim {
|
||||
0 => {
|
||||
let nslots = max
|
||||
.iter()
|
||||
.zip(chunk_dims)
|
||||
.try_fold(1u64, |acc, (&m, &c)| acc.checked_mul(m.div_ceil(c.max(1))))
|
||||
.filter(|&n| n <= MAX_FIXED_ARRAY_SLOTS)
|
||||
.ok_or_else(|| {
|
||||
bad("too many chunks for a Fixed Array index; \
|
||||
use larger chunks or an unlimited dimension")
|
||||
})?;
|
||||
// A Single Chunk index needs that one chunk to exist; an
|
||||
// empty dataset gets an all-unallocated Fixed Array instead.
|
||||
let empty = shape.contains(&0);
|
||||
if nslots == 1 && !empty {
|
||||
Ok(Self::SingleChunk)
|
||||
} else {
|
||||
let grid = ChunkGrid::fixed_array(shape, Some(max), chunk_dims)?;
|
||||
Ok(Self::FixedArray(grid, nslots as usize))
|
||||
}
|
||||
}
|
||||
1 => Ok(Self::ExtensibleArray(ChunkGrid::extensible_array(
|
||||
shape,
|
||||
Some(max),
|
||||
chunk_dims,
|
||||
)?)),
|
||||
_ => Ok(Self::BTreeV2),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Place each written chunk at its linear index in `grid`. `chunks` are in
|
||||
/// row-major order over the chunks of the current extent (`split_into_chunks`).
|
||||
/// `len` fixes the slot count (Fixed Array); otherwise it is one past the
|
||||
/// highest index used.
|
||||
fn index_slots(
|
||||
grid: &ChunkGrid,
|
||||
shape: &[u64],
|
||||
chunk_dims: &[u64],
|
||||
chunks: &[WrittenChunk],
|
||||
len: Option<usize>,
|
||||
) -> Result<Vec<Option<WrittenChunk>>, FormatError> {
|
||||
let mut placed: Vec<(usize, &WrittenChunk)> = Vec::with_capacity(chunks.len());
|
||||
for (i, chunk) in chunks.iter().enumerate() {
|
||||
let scaled = scaled_coords(shape, chunk_dims, i);
|
||||
let idx = usize::try_from(grid.linear_index(&scaled))
|
||||
.map_err(|_| FormatError::Overflow("chunk index slot".into()))?;
|
||||
placed.push((idx, chunk));
|
||||
}
|
||||
let n = len.unwrap_or_else(|| placed.iter().map(|&(i, _)| i + 1).max().unwrap_or(0));
|
||||
let mut slots = vec![None; n];
|
||||
for (idx, chunk) in placed {
|
||||
*slots
|
||||
.get_mut(idx)
|
||||
.ok_or_else(|| FormatError::Overflow("chunk index slot".into()))? = Some(chunk.clone());
|
||||
}
|
||||
Ok(slots)
|
||||
}
|
||||
|
||||
/// Scaled coordinates (`offset / chunk_dim`) of the `i`-th chunk in the
|
||||
/// row-major order `split_into_chunks` produces over the current extent.
|
||||
fn scaled_coords(shape: &[u64], chunk_dims: &[u64], i: usize) -> Vec<u64> {
|
||||
let rank = shape.len();
|
||||
let mut scaled = vec![0u64; rank];
|
||||
let mut rem = i as u64;
|
||||
for d in (0..rank).rev() {
|
||||
let n = shape[d].div_ceil(chunk_dims[d]);
|
||||
scaled[d] = rem % n;
|
||||
rem /= n;
|
||||
}
|
||||
scaled
|
||||
}
|
||||
|
||||
/// Node size the library gives a chunk index B-tree (`H5D_BT2_NODE_SIZE`),
|
||||
/// with its split and merge percentages.
|
||||
const BT2_NODE_SIZE: u32 = 2048;
|
||||
const BT2_SPLIT_PERCENT: u8 = 100;
|
||||
const BT2_MERGE_PERCENT: u8 = 40;
|
||||
/// B-tree v2 record types for chunk indexes (`H5B2_CDSET_ID`,
|
||||
/// `H5B2_CDSET_FILT_ID`).
|
||||
const BT2_CHUNK_UNFILTERED: u8 = 10;
|
||||
const BT2_CHUNK_FILTERED: u8 = 11;
|
||||
|
||||
/// Build a version-2 B-tree chunk index (the library's index for datasets
|
||||
/// with more than one unlimited dimension) at a known absolute address.
|
||||
///
|
||||
/// `records` are `(scaled coordinates, chunk)` in lexicographic order of the
|
||||
/// coordinates, which is the order the library's comparator
|
||||
/// (`H5VM_vector_cmp_u`) keeps them in. The tree is a single leaf: the
|
||||
/// library's 2048-byte node when the records fit, otherwise a leaf node
|
||||
/// sized to hold them all (the root's record count is 16-bit, so at most
|
||||
/// 65535 chunks). Returns the bytes and the node size the layout message
|
||||
/// must record.
|
||||
fn build_btree_v2_chunk_index_at(
|
||||
rank: usize,
|
||||
records: &[(Vec<u64>, &WrittenChunk)],
|
||||
offset_size: u8,
|
||||
length_size: u8,
|
||||
has_filters: bool,
|
||||
base_address: u64,
|
||||
) -> Result<(Vec<u8>, u32), FormatError> {
|
||||
let os = offset_size as usize;
|
||||
let nrec = u16::try_from(records.len()).map_err(|_| {
|
||||
FormatError::ChunkedReadError(
|
||||
"more than 65535 chunks with more than one unlimited dimension: \
|
||||
use larger chunks"
|
||||
.into(),
|
||||
)
|
||||
})?;
|
||||
let chunk_size_bytes = has_filters.then(|| {
|
||||
let slots: Vec<Option<WrittenChunk>> =
|
||||
records.iter().map(|(_, c)| Some((*c).clone())).collect();
|
||||
filtered_chunk_size_len(&slots)
|
||||
});
|
||||
let record_size = os + chunk_size_bytes.map_or(0, |n| n + 4) + 8 * rank;
|
||||
// Leaf: signature, version, type, records, checksum.
|
||||
let leaf_len = 4 + 1 + 1 + records.len() * record_size + 4;
|
||||
let node_size = u32::try_from(leaf_len)
|
||||
.map_err(|_| FormatError::Overflow("B-tree v2 leaf size".into()))?
|
||||
.max(BT2_NODE_SIZE);
|
||||
let tree_type = if has_filters {
|
||||
BT2_CHUNK_FILTERED
|
||||
} else {
|
||||
BT2_CHUNK_UNFILTERED
|
||||
};
|
||||
|
||||
let hdr_len = 4 + 1 + 1 + 4 + 2 + 2 + 1 + 1 + os + 2 + length_size as usize + 4;
|
||||
let leaf_address = base_address + hdr_len as u64;
|
||||
|
||||
let mut out = Vec::with_capacity(hdr_len + node_size as usize);
|
||||
out.extend_from_slice(b"BTHD");
|
||||
out.push(0); // version
|
||||
out.push(tree_type);
|
||||
out.extend_from_slice(&node_size.to_le_bytes());
|
||||
out.extend_from_slice(&(record_size as u16).to_le_bytes());
|
||||
out.extend_from_slice(&0u16.to_le_bytes()); // depth
|
||||
out.push(BT2_SPLIT_PERCENT);
|
||||
out.push(BT2_MERGE_PERCENT);
|
||||
if records.is_empty() {
|
||||
out.extend(core::iter::repeat_n(0xFF, os));
|
||||
} else {
|
||||
push_addr(&mut out, leaf_address, offset_size);
|
||||
}
|
||||
out.extend_from_slice(&nrec.to_le_bytes());
|
||||
match length_size {
|
||||
4 => out.extend_from_slice(&(records.len() as u32).to_le_bytes()),
|
||||
_ => out.extend_from_slice(&(records.len() as u64).to_le_bytes()),
|
||||
}
|
||||
let sum = jenkins_lookup3(&out);
|
||||
out.extend_from_slice(&sum.to_le_bytes());
|
||||
debug_assert_eq!(out.len(), hdr_len);
|
||||
if records.is_empty() {
|
||||
return Ok((out, node_size));
|
||||
}
|
||||
|
||||
let leaf_start = out.len();
|
||||
out.extend_from_slice(b"BTLF");
|
||||
out.push(0); // version
|
||||
out.push(tree_type);
|
||||
for (scaled, chunk) in records {
|
||||
push_index_element(&mut out, Some(chunk), offset_size, chunk_size_bytes);
|
||||
for &c in scaled {
|
||||
out.extend_from_slice(&c.to_le_bytes());
|
||||
}
|
||||
}
|
||||
let sum = jenkins_lookup3(&out[leaf_start..]);
|
||||
out.extend_from_slice(&sum.to_le_bytes());
|
||||
// The library reads whole nodes; pad the leaf out to the node size.
|
||||
out.resize(leaf_start + node_size as usize, 0);
|
||||
Ok((out, node_size))
|
||||
}
|
||||
|
||||
/// Serialize a v4 layout message for a version-2 B-tree chunk index.
|
||||
fn serialize_v4_btree_v2(
|
||||
chunk_dims: &[u32],
|
||||
btree_address: u64,
|
||||
offset_size: u8,
|
||||
element_size: u32,
|
||||
node_size: u32,
|
||||
) -> Vec<u8> {
|
||||
let mut buf = layout_v4_chunked_prefix(chunk_dims, element_size);
|
||||
buf.push(5); // chunk index type = 5 (version-2 B-tree)
|
||||
buf.extend_from_slice(&node_size.to_le_bytes());
|
||||
buf.push(BT2_SPLIT_PERCENT);
|
||||
buf.push(BT2_MERGE_PERCENT);
|
||||
push_addr(&mut buf, btree_address, offset_size);
|
||||
buf
|
||||
}
|
||||
|
||||
/// Build chunked data with absolute addresses.
|
||||
@@ -790,11 +1090,7 @@ pub fn build_chunked_data_at_ext(
|
||||
maxshape: Option<&[u64]>,
|
||||
) -> Result<ChunkedDataResult, FormatError> {
|
||||
let pre = precompress_chunks(raw_data, shape, chunk_dims, element_size, options)?;
|
||||
Ok(build_chunked_data_from_precompressed(
|
||||
&pre,
|
||||
base_address,
|
||||
maxshape,
|
||||
))
|
||||
build_chunked_data_from_precompressed(&pre, base_address, maxshape)
|
||||
}
|
||||
|
||||
/// Write selected elements into an existing in-memory dataset buffer.
|
||||
@@ -1314,6 +1610,7 @@ mod tests {
|
||||
chunk_index_type,
|
||||
single_chunk_filtered_size,
|
||||
single_chunk_filter_mask,
|
||||
..
|
||||
} => {
|
||||
assert_eq!(version, 4);
|
||||
assert_eq!(chunk_index_type, Some(1));
|
||||
@@ -1382,7 +1679,8 @@ mod tests {
|
||||
filter_mask: 0,
|
||||
},
|
||||
];
|
||||
let fa = build_fixed_array_at(&chunks, 8, 8, false, 0x2000);
|
||||
let slots: Vec<_> = chunks.into_iter().map(Some).collect();
|
||||
let fa = build_fixed_array_at(&slots, 8, 8, false, 0x2000);
|
||||
// Should start with FAHD
|
||||
assert_eq!(&fa[0..4], b"FAHD");
|
||||
// FAHD size = 4+1+1+1+1+8+8+4 = 28
|
||||
@@ -1429,7 +1727,8 @@ mod tests {
|
||||
filter_mask: 0,
|
||||
},
|
||||
];
|
||||
let ea = ea_writer::build_extensible_array_at(&chunks, 8, 8, false, 0x2000);
|
||||
let slots: Vec<_> = chunks.into_iter().map(Some).collect();
|
||||
let ea = ea_writer::build_extensible_array_at(&slots, 8, 8, false, 0x2000);
|
||||
assert_eq!(&ea[0..4], b"EAHD");
|
||||
// Find EAIB after EAHD: 12 fixed + 6*8 stats + 8 addr + 4 checksum = 72
|
||||
let aehd_size = 4 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 6 * 8 + 8 + 4;
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
//! HDF5 Data Layout message parsing (message type 0x0008).
|
||||
|
||||
#[cfg(not(feature = "std"))]
|
||||
use alloc::{string::String, vec::Vec};
|
||||
use alloc::{format, string::String, vec::Vec};
|
||||
|
||||
#[cfg(feature = "std")]
|
||||
use std::string::String;
|
||||
@@ -45,7 +45,9 @@ pub enum DataLayout {
|
||||
chunk_dimensions: Vec<u32>,
|
||||
/// B-tree address, or `None` if undefined.
|
||||
btree_address: Option<u64>,
|
||||
/// Layout version (3 or 4).
|
||||
/// Layout version (3 or 4). Version 1/2 messages (HDF5 1.4/1.6-era)
|
||||
/// use the same version-1 B-tree chunk index as version 3 and are
|
||||
/// reported as 3.
|
||||
version: u8,
|
||||
/// Chunk index type (v4 only).
|
||||
chunk_index_type: Option<u8>,
|
||||
@@ -53,6 +55,11 @@ pub enum DataLayout {
|
||||
single_chunk_filtered_size: Option<u64>,
|
||||
/// Filter mask for v4 single chunk with filters.
|
||||
single_chunk_filter_mask: Option<u32>,
|
||||
/// Layout v4 flag bit 0 (`H5D_CHUNK_DONT_FILTER_PARTIAL_CHUNKS`):
|
||||
/// partial edge chunks — those extending past the dataset's current
|
||||
/// extent in some dimension — are stored without the filter pipeline,
|
||||
/// even though their filter mask is 0. Always `false` for v3.
|
||||
dont_filter_partial_edge_chunks: bool,
|
||||
},
|
||||
/// Virtual dataset layout (v4 only).
|
||||
Virtual {
|
||||
@@ -67,21 +74,33 @@ pub enum DataLayout {
|
||||
},
|
||||
}
|
||||
|
||||
/// Version-1 VDS mapping flag: the source file name is stored by an earlier
|
||||
/// entry, whose index follows in place of the name.
|
||||
const VDS_SOURCE_FILE_SHARED: u8 = 0x01;
|
||||
/// Version-1 VDS mapping flag: likewise for the source dataset name.
|
||||
const VDS_SOURCE_DSET_SHARED: u8 = 0x02;
|
||||
/// Version-1 VDS mapping flag: the source is in the virtual file itself
|
||||
/// (`"."`); no file name is stored.
|
||||
const VDS_SOURCE_SAME_FILE: u8 = 0x04;
|
||||
const VDS_ALL_FLAGS: u8 = VDS_SOURCE_FILE_SHARED | VDS_SOURCE_DSET_SHARED | VDS_SOURCE_SAME_FILE;
|
||||
|
||||
/// Parse VDS mappings from global-heap object data.
|
||||
///
|
||||
/// The global-heap block holding a VDS mapping list is laid out as
|
||||
/// (reverse-engineered and validated against HDF5 2.0):
|
||||
/// (`H5D__virtual_store_layout` / `H5D__virtual_load_layout` in libhdf5):
|
||||
///
|
||||
/// ```text
|
||||
/// version(1) · nused(length_size, LE) · entry[nused] · checksum(4)
|
||||
/// ```
|
||||
///
|
||||
/// Each entry is:
|
||||
/// - source file name — a null-terminated string in **block version 0**; in
|
||||
/// **block version 1** a same-file reference is encoded as a single `0x04`
|
||||
/// marker byte (the source file is the virtual file itself) in place of the
|
||||
/// name;
|
||||
/// - source dataset name (null-terminated string);
|
||||
/// - **block version 1 only:** a flags byte. `0x04`: the source is in the
|
||||
/// virtual file itself and no file name is stored; `0x01`/`0x02`: the
|
||||
/// source file/dataset name is that of an earlier entry, whose index
|
||||
/// (`length_size` bytes) is stored instead of the name. libhdf5 2.0 writes
|
||||
/// version 1 when the file's low version bound is 2.0 and it saves space;
|
||||
/// - source file name (null-terminated string, unless flagged above);
|
||||
/// - source dataset name (null-terminated string, unless flagged above);
|
||||
/// - source selection (serialized `H5S` dataspace selection — self-describing
|
||||
/// in length);
|
||||
/// - virtual selection (serialized `H5S` dataspace selection).
|
||||
@@ -107,7 +126,7 @@ pub fn parse_vds_mappings(
|
||||
// `nused` is untrusted; don't pre-allocate from it. Each entry consumes at
|
||||
// least a few bytes, so the loop is naturally bounded by the heap data and
|
||||
// a bogus `nused` simply errors out on the first short read.
|
||||
let mut mappings = Vec::new();
|
||||
let mut mappings: Vec<VdsMapping> = Vec::new();
|
||||
// Reads one self-describing selection at `pos`, returning its raw bytes and
|
||||
// advancing past it — bounds-checked so a corrupt selection can't overrun.
|
||||
let read_selection = |heap_data: &[u8], pos: &mut usize| -> Result<Vec<u8>, FormatError> {
|
||||
@@ -127,17 +146,57 @@ pub fn parse_vds_mappings(
|
||||
Ok(bytes)
|
||||
};
|
||||
|
||||
for _ in 0..nused {
|
||||
// Source file name (with the version-1 same-file marker handled).
|
||||
let source_file = if version >= 1 && heap_data.get(pos) == Some(&0x04) {
|
||||
if version > 1 {
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"unsupported VDS mapping block version".into(),
|
||||
));
|
||||
}
|
||||
for i in 0..nused {
|
||||
// Version 1 prefixes each entry with a flags byte; a name may then be
|
||||
// omitted (same file) or replaced by the index of an earlier entry
|
||||
// holding the same name (`H5D__virtual_load_layout`).
|
||||
let flags = if version >= 1 {
|
||||
let f = *heap_data.get(pos).ok_or(FormatError::UnexpectedEof {
|
||||
expected: pos + 1,
|
||||
available: heap_data.len(),
|
||||
})?;
|
||||
pos += 1;
|
||||
if f & !VDS_ALL_FLAGS != 0 {
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"unknown VDS mapping flags".into(),
|
||||
));
|
||||
}
|
||||
f
|
||||
} else {
|
||||
0
|
||||
};
|
||||
// Index of an earlier entry, for a shared name.
|
||||
let earlier = |pos: &mut usize| -> Result<usize, FormatError> {
|
||||
let idx = read_length(heap_data, *pos, length_size)?;
|
||||
*pos += ls;
|
||||
if idx >= i {
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"VDS mapping shares a name with a later entry".into(),
|
||||
));
|
||||
}
|
||||
Ok(idx as usize)
|
||||
};
|
||||
|
||||
let source_file = if flags & VDS_SOURCE_SAME_FILE != 0 {
|
||||
String::from(".")
|
||||
} else if flags & VDS_SOURCE_FILE_SHARED != 0 {
|
||||
let idx = earlier(&mut pos)?;
|
||||
mappings[idx].source_file.clone()
|
||||
} else {
|
||||
read_null_terminated_string(heap_data, &mut pos)?
|
||||
};
|
||||
|
||||
// Source dataset name.
|
||||
let source_dataset = read_null_terminated_string(heap_data, &mut pos)?;
|
||||
let source_dataset = if flags & VDS_SOURCE_DSET_SHARED != 0 {
|
||||
let idx = earlier(&mut pos)?;
|
||||
mappings[idx].source_dataset.clone()
|
||||
} else {
|
||||
read_null_terminated_string(heap_data, &mut pos)?
|
||||
};
|
||||
|
||||
// Source selection, then virtual selection (both self-describing length).
|
||||
let source_selection = read_selection(heap_data, &mut pos)?;
|
||||
@@ -256,6 +315,7 @@ impl DataLayout {
|
||||
let layout_class = data[1];
|
||||
|
||||
match version {
|
||||
1 | 2 => Self::parse_v1_v2(data, offset_size),
|
||||
3 => Self::parse_v3(data, layout_class, offset_size, length_size),
|
||||
// v5 (emitted by HDF5 1.14+/2.0 with `libver=latest`) uses the same
|
||||
// message structure as v4 — only the version number was bumped.
|
||||
@@ -264,6 +324,87 @@ impl DataLayout {
|
||||
}
|
||||
}
|
||||
|
||||
/// Layout message versions 1 and 2 (HDF5 before 1.6.3):
|
||||
///
|
||||
/// ```text
|
||||
/// version(1) · dimensionality(1) · layout class(1) · reserved(5)
|
||||
/// · address(offset_size) — contiguous and chunked only
|
||||
/// · dimension sizes(4 × dimensionality)
|
||||
/// · compact data size(4) · compact raw data — compact only
|
||||
/// ```
|
||||
///
|
||||
/// The dimension sizes are the dataset's (contiguous/compact) or the
|
||||
/// chunk's (chunked) extent plus a trailing element-size dimension, as in
|
||||
/// version 3's chunked form. libhdf5 ignores them for contiguous storage
|
||||
/// and sizes the data from the dataspace; the product of the stored
|
||||
/// dimensions is that same size, and a disagreement (a dimension that was
|
||||
/// truncated to 32 bits) is caught by the reader's size check rather than
|
||||
/// returning wrong data.
|
||||
fn parse_v1_v2(data: &[u8], offset_size: u8) -> Result<DataLayout, FormatError> {
|
||||
ensure_len(data, 0, 8)?;
|
||||
let dimensionality = data[1] as usize;
|
||||
let layout_class = data[2];
|
||||
// H5O_LAYOUT_NDIMS: 32 dataspace dimensions + the element-size one.
|
||||
if dimensionality > 33 {
|
||||
return Err(FormatError::Overflow(format!(
|
||||
"data layout dimensionality {dimensionality} exceeds 33"
|
||||
)));
|
||||
}
|
||||
let mut p = 8;
|
||||
let os = offset_size as usize;
|
||||
let address = match layout_class {
|
||||
1 | 2 => {
|
||||
ensure_len(data, p, os)?;
|
||||
let a = if is_undefined(data, p, offset_size) {
|
||||
None
|
||||
} else {
|
||||
Some(read_offset(data, p, offset_size)?)
|
||||
};
|
||||
p += os;
|
||||
a
|
||||
}
|
||||
0 => None,
|
||||
_ => return Err(FormatError::InvalidLayoutClass(layout_class)),
|
||||
};
|
||||
ensure_len(data, p, dimensionality * 4)?;
|
||||
let dims: Vec<u32> = data[p..p + dimensionality * 4]
|
||||
.as_chunks::<4>()
|
||||
.0
|
||||
.iter()
|
||||
.map(|c| u32::from_le_bytes(*c))
|
||||
.collect();
|
||||
p += dimensionality * 4;
|
||||
match layout_class {
|
||||
0 => {
|
||||
ensure_len(data, p, 4)?;
|
||||
let size =
|
||||
u32::from_le_bytes([data[p], data[p + 1], data[p + 2], data[p + 3]]) as usize;
|
||||
ensure_len(data, p + 4, size)?;
|
||||
Ok(DataLayout::Compact {
|
||||
data: data[p + 4..p + 4 + size].to_vec(),
|
||||
})
|
||||
}
|
||||
1 => {
|
||||
let size = dims
|
||||
.iter()
|
||||
.try_fold(1u64, |acc, &d| acc.checked_mul(d as u64))
|
||||
.ok_or_else(|| {
|
||||
FormatError::Overflow(format!("contiguous layout size {dims:?}"))
|
||||
})?;
|
||||
Ok(DataLayout::Contiguous { address, size })
|
||||
}
|
||||
_ => Ok(DataLayout::Chunked {
|
||||
chunk_dimensions: dims,
|
||||
btree_address: address,
|
||||
version: 3,
|
||||
chunk_index_type: None,
|
||||
single_chunk_filtered_size: None,
|
||||
single_chunk_filter_mask: None,
|
||||
dont_filter_partial_edge_chunks: false,
|
||||
}),
|
||||
}
|
||||
}
|
||||
|
||||
fn parse_v3(
|
||||
data: &[u8],
|
||||
layout_class: u8,
|
||||
@@ -322,6 +463,7 @@ impl DataLayout {
|
||||
chunk_index_type: None,
|
||||
single_chunk_filtered_size: None,
|
||||
single_chunk_filter_mask: None,
|
||||
dont_filter_partial_edge_chunks: false,
|
||||
})
|
||||
}
|
||||
_ => Err(FormatError::InvalidLayoutClass(layout_class)),
|
||||
@@ -505,6 +647,7 @@ impl DataLayout {
|
||||
chunk_index_type: Some(chunk_index_type),
|
||||
single_chunk_filtered_size,
|
||||
single_chunk_filter_mask,
|
||||
dont_filter_partial_edge_chunks: flags & 0x01 != 0,
|
||||
})
|
||||
}
|
||||
3 => {
|
||||
@@ -539,6 +682,108 @@ impl DataLayout {
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// Version 1/2 header: version, dimensionality, class, reserved(5).
|
||||
fn v1v2_header(version: u8, ndims: u8, class: u8) -> Vec<u8> {
|
||||
vec![version, ndims, class, 0, 0, 0, 0, 0]
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn v2_compact() {
|
||||
let mut buf = v1v2_header(2, 2, 0);
|
||||
// dims (3 elements of 2 bytes) — no address for compact
|
||||
buf.extend_from_slice(&3u32.to_le_bytes());
|
||||
buf.extend_from_slice(&2u32.to_le_bytes());
|
||||
buf.extend_from_slice(&6u32.to_le_bytes()); // compact size (u32 in v1/v2)
|
||||
buf.extend_from_slice(&[1, 0, 2, 0, 3, 0]);
|
||||
assert_eq!(
|
||||
DataLayout::parse(&buf, 8, 8).unwrap(),
|
||||
DataLayout::Compact {
|
||||
data: vec![1, 0, 2, 0, 3, 0]
|
||||
}
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn v1_contiguous_size_from_dimensions() {
|
||||
let mut buf = v1v2_header(1, 3, 1);
|
||||
buf.extend_from_slice(&0x800u32.to_le_bytes()); // 4-byte address
|
||||
for d in [10u32, 20, 4] {
|
||||
buf.extend_from_slice(&d.to_le_bytes());
|
||||
}
|
||||
assert_eq!(
|
||||
DataLayout::parse(&buf, 4, 4).unwrap(),
|
||||
DataLayout::Contiguous {
|
||||
address: Some(0x800),
|
||||
size: 800,
|
||||
}
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn v1_contiguous_undefined_address() {
|
||||
let mut buf = v1v2_header(1, 2, 1);
|
||||
buf.extend_from_slice(&[0xFF; 8]);
|
||||
buf.extend_from_slice(&5u32.to_le_bytes());
|
||||
buf.extend_from_slice(&8u32.to_le_bytes());
|
||||
assert_eq!(
|
||||
DataLayout::parse(&buf, 8, 8).unwrap(),
|
||||
DataLayout::Contiguous {
|
||||
address: None,
|
||||
size: 40,
|
||||
}
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn v1_chunked_maps_to_btree_v1_index() {
|
||||
let mut buf = v1v2_header(1, 3, 2);
|
||||
buf.extend_from_slice(&0x1234u64.to_le_bytes());
|
||||
for d in [50u32, 50, 4] {
|
||||
buf.extend_from_slice(&d.to_le_bytes());
|
||||
}
|
||||
assert_eq!(
|
||||
DataLayout::parse(&buf, 8, 8).unwrap(),
|
||||
DataLayout::Chunked {
|
||||
chunk_dimensions: vec![50, 50, 4],
|
||||
btree_address: Some(0x1234),
|
||||
version: 3,
|
||||
chunk_index_type: None,
|
||||
single_chunk_filtered_size: None,
|
||||
single_chunk_filter_mask: None,
|
||||
dont_filter_partial_edge_chunks: false,
|
||||
}
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn v1v2_rejects_bad_class_dimensionality_and_truncation() {
|
||||
assert_eq!(
|
||||
DataLayout::parse(&v1v2_header(1, 1, 3), 8, 8).unwrap_err(),
|
||||
FormatError::InvalidLayoutClass(3)
|
||||
);
|
||||
assert!(matches!(
|
||||
DataLayout::parse(&v1v2_header(2, 34, 1), 8, 8).unwrap_err(),
|
||||
FormatError::Overflow(_)
|
||||
));
|
||||
// Chunked, dims cut short.
|
||||
let mut buf = v1v2_header(1, 2, 2);
|
||||
buf.extend_from_slice(&0x10u64.to_le_bytes());
|
||||
buf.extend_from_slice(&7u32.to_le_bytes());
|
||||
assert!(matches!(
|
||||
DataLayout::parse(&buf, 8, 8).unwrap_err(),
|
||||
FormatError::UnexpectedEof { .. }
|
||||
));
|
||||
// Compact, raw data shorter than its declared size.
|
||||
let mut buf = v1v2_header(2, 1, 0);
|
||||
buf.extend_from_slice(&4u32.to_le_bytes());
|
||||
buf.extend_from_slice(&100u32.to_le_bytes());
|
||||
buf.extend_from_slice(&[0; 4]);
|
||||
assert!(matches!(
|
||||
DataLayout::parse(&buf, 8, 8).unwrap_err(),
|
||||
FormatError::UnexpectedEof { .. }
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn v3_compact() {
|
||||
let mut buf = vec![3u8, 0]; // version=3, class=0 (compact)
|
||||
@@ -602,6 +847,7 @@ mod tests {
|
||||
chunk_index_type: None,
|
||||
single_chunk_filtered_size: None,
|
||||
single_chunk_filter_mask: None,
|
||||
dont_filter_partial_edge_chunks: false,
|
||||
}
|
||||
);
|
||||
}
|
||||
@@ -679,10 +925,35 @@ mod tests {
|
||||
chunk_index_type: Some(1),
|
||||
single_chunk_filtered_size: None,
|
||||
single_chunk_filter_mask: None,
|
||||
dont_filter_partial_edge_chunks: false,
|
||||
}
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn v4_chunked_dont_filter_partial_edge_chunks_flag() {
|
||||
let mut buf = vec![4u8, 2]; // version=4, class=2
|
||||
buf.push(0x01); // flags bit 0 = don't filter partial edge chunks
|
||||
buf.push(2); // dimensionality=2
|
||||
buf.push(4); // dim_size_encoded_length=4
|
||||
buf.extend_from_slice(&5u32.to_le_bytes());
|
||||
buf.extend_from_slice(&4u32.to_le_bytes());
|
||||
buf.push(3); // Fixed Array
|
||||
buf.push(10); // max_dblk_page_nelmts_bits
|
||||
buf.extend_from_slice(&0x3000u64.to_le_bytes());
|
||||
match DataLayout::parse(&buf, 8, 8).unwrap() {
|
||||
DataLayout::Chunked {
|
||||
dont_filter_partial_edge_chunks,
|
||||
btree_address,
|
||||
..
|
||||
} => {
|
||||
assert!(dont_filter_partial_edge_chunks);
|
||||
assert_eq!(btree_address, Some(0x3000));
|
||||
}
|
||||
other => panic!("expected Chunked, got {other:?}"),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn v4_chunked_single_chunk_with_filters() {
|
||||
let mut buf = vec![4u8, 2]; // version=4, class=2
|
||||
@@ -705,6 +976,7 @@ mod tests {
|
||||
chunk_index_type: Some(1),
|
||||
single_chunk_filtered_size: Some(1024),
|
||||
single_chunk_filter_mask: Some(0),
|
||||
dont_filter_partial_edge_chunks: false,
|
||||
}
|
||||
);
|
||||
}
|
||||
@@ -815,6 +1087,62 @@ mod tests {
|
||||
assert_eq!(v1.iter_linear_1d(8).unwrap(), vec![4, 5, 6, 7]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_vds_mappings_v1_shared_names() {
|
||||
// Written by HDF5 2.0 (h5py, libver=("v200", "v200")) for three
|
||||
// mappings from `a_rather_long_source_file.h5:a_rather_long_dataset_name`
|
||||
// and one from the same file: the entries carry flags 0x00, 0x03, 0x03
|
||||
// and 0x06, so names after the first are stored as entry indices.
|
||||
let blob: &[u8] = &[
|
||||
0x01, 0x04, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x61, 0x5f, 0x72, 0x61,
|
||||
0x74, 0x68, 0x65, 0x72, 0x5f, 0x6c, 0x6f, 0x6e, 0x67, 0x5f, 0x73, 0x6f, 0x75, 0x72,
|
||||
0x63, 0x65, 0x5f, 0x66, 0x69, 0x6c, 0x65, 0x2e, 0x68, 0x35, 0x00, 0x61, 0x5f, 0x72,
|
||||
0x61, 0x74, 0x68, 0x65, 0x72, 0x5f, 0x6c, 0x6f, 0x6e, 0x67, 0x5f, 0x64, 0x61, 0x74,
|
||||
0x61, 0x73, 0x65, 0x74, 0x5f, 0x6e, 0x61, 0x6d, 0x65, 0x00, 0x02, 0x00, 0x00, 0x00,
|
||||
0x03, 0x00, 0x00, 0x00, 0x01, 0x02, 0x01, 0x00, 0x00, 0x00, 0x00, 0x00, 0x01, 0x00,
|
||||
0x01, 0x00, 0x04, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00, 0x00, 0x00, 0x01, 0x02,
|
||||
0x02, 0x00, 0x00, 0x00, 0x00, 0x00, 0x01, 0x00, 0x01, 0x00, 0x01, 0x00, 0x00, 0x00,
|
||||
0x01, 0x00, 0x01, 0x00, 0x04, 0x00, 0x03, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
|
||||
0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03,
|
||||
0x00, 0x00, 0x00, 0x01, 0x02, 0x01, 0x00, 0x00, 0x00, 0x04, 0x00, 0x01, 0x00, 0x01,
|
||||
0x00, 0x04, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00, 0x00, 0x00, 0x01, 0x02, 0x02,
|
||||
0x00, 0x00, 0x00, 0x01, 0x00, 0x01, 0x00, 0x01, 0x00, 0x01, 0x00, 0x00, 0x00, 0x01,
|
||||
0x00, 0x01, 0x00, 0x04, 0x00, 0x03, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
|
||||
0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00,
|
||||
0x00, 0x00, 0x01, 0x02, 0x01, 0x00, 0x00, 0x00, 0x08, 0x00, 0x01, 0x00, 0x01, 0x00,
|
||||
0x04, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00, 0x00, 0x00, 0x01, 0x02, 0x02, 0x00,
|
||||
0x00, 0x00, 0x02, 0x00, 0x01, 0x00, 0x01, 0x00, 0x01, 0x00, 0x00, 0x00, 0x01, 0x00,
|
||||
0x01, 0x00, 0x04, 0x00, 0x06, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x02,
|
||||
0x00, 0x00, 0x00, 0x03, 0x00, 0x00, 0x00, 0x01, 0x02, 0x01, 0x00, 0x00, 0x00, 0x00,
|
||||
0x00, 0x01, 0x00, 0x01, 0x00, 0x04, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00, 0x00,
|
||||
0x00, 0x01, 0x02, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00, 0x01, 0x00, 0x01, 0x00, 0x01,
|
||||
0x00, 0x00, 0x00, 0x01, 0x00, 0x01, 0x00, 0x04, 0x00, 0x8e, 0xa7, 0xea, 0x7a,
|
||||
];
|
||||
let mappings = parse_vds_mappings(blob, 8).unwrap();
|
||||
let names: Vec<(&str, &str)> = mappings
|
||||
.iter()
|
||||
.map(|m| (m.source_file.as_str(), m.source_dataset.as_str()))
|
||||
.collect();
|
||||
let (file, dset) = ("a_rather_long_source_file.h5", "a_rather_long_dataset_name");
|
||||
assert_eq!(
|
||||
names,
|
||||
vec![(file, dset), (file, dset), (file, dset), (".", dset)]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_vds_mappings_v1_forward_reference_is_error() {
|
||||
// Entry 0 claiming to share entry 0's file name must not index past
|
||||
// the entries decoded so far.
|
||||
let mut blob = vec![0x01u8, 1, 0, 0, 0, 0, 0, 0, 0, 0x01];
|
||||
blob.extend_from_slice(&[0u8; 8]);
|
||||
blob.extend_from_slice(b"d\0");
|
||||
assert!(parse_vds_mappings(&blob, 8).is_err());
|
||||
// Unknown flag bits are refused.
|
||||
let blob = [0x01u8, 1, 0, 0, 0, 0, 0, 0, 0, 0x08, b'd', 0];
|
||||
assert!(parse_vds_mappings(&blob, 8).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_vds_mappings_external_v0() {
|
||||
// Block version 0 with an explicit (external) source file name.
|
||||
|
||||
@@ -191,14 +191,9 @@ fn read_raw_data_full_impl(
|
||||
offset_size,
|
||||
length_size,
|
||||
),
|
||||
DataLayout::Virtual {
|
||||
global_heap_address,
|
||||
global_heap_index,
|
||||
..
|
||||
} => read_virtual_data(
|
||||
DataLayout::Virtual { .. } => read_virtual_data(
|
||||
file_data,
|
||||
*global_heap_address,
|
||||
*global_heap_index,
|
||||
layout,
|
||||
dataspace,
|
||||
datatype,
|
||||
offset_size,
|
||||
@@ -465,158 +460,54 @@ pub fn read_raw_data_selection(
|
||||
}
|
||||
}
|
||||
|
||||
/// Assemble a **Virtual Dataset (VDS)** from its source mappings.
|
||||
/// Assemble a **Virtual Dataset (VDS)** through the raw-read API, which has no
|
||||
/// access to the dataset's fill value message.
|
||||
///
|
||||
/// Supports virtual datasets of any rank. Same-file sources are read directly;
|
||||
/// **external-file** sources are read through the caller-supplied `resolver`,
|
||||
/// which maps a stored source file name to that file's bytes. Each mapping's
|
||||
/// selected source elements are scattered into the virtual buffer at the
|
||||
/// positions given by the virtual selection (both enumerated in row-major
|
||||
/// order, as HDF5 pairs them). Unmapped regions are left at the zero fill value.
|
||||
///
|
||||
/// A mapping whose external source file the resolver cannot supply (`None`) is
|
||||
/// skipped, leaving its region at fill — matching HDF5's tolerance of missing
|
||||
/// sources. An external source with no resolver at all is a hard error.
|
||||
/// Delegates to [`crate::vds::read_virtual_dataset`]. Because the fill value
|
||||
/// is unknown here, a virtual dataset with any element no mapping supplies
|
||||
/// (an unmapped region, or a missing source file or dataset) is an error
|
||||
/// rather than a guess at the fill value; so is one whose extent libhdf5
|
||||
/// would report differently from the stored dataspace (unlimited mappings).
|
||||
/// Use [`crate::vds::read_virtual_dataset`] to read those.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
fn read_virtual_data(
|
||||
file_data: &[u8],
|
||||
global_heap_address: Option<u64>,
|
||||
global_heap_index: u32,
|
||||
layout: &DataLayout,
|
||||
dataspace: &Dataspace,
|
||||
datatype: &Datatype,
|
||||
offset_size: u8,
|
||||
length_size: u8,
|
||||
resolver: Option<&VdsSourceResolver>,
|
||||
) -> Result<Vec<u8>, FormatError> {
|
||||
use crate::data_layout::parse_vds_mappings;
|
||||
use crate::global_heap::GlobalHeapCollection;
|
||||
use crate::selection::Selection;
|
||||
|
||||
let elem_size = datatype.type_size() as usize;
|
||||
let mut out = crate::chunked_read::alloc_output(crate::chunked_read::checked_byte_len(
|
||||
dataspace.checked_num_elements()?,
|
||||
elem_size,
|
||||
)?)?;
|
||||
|
||||
let virtual_dims = &dataspace.dimensions;
|
||||
|
||||
let addr = global_heap_address.ok_or_else(|| {
|
||||
FormatError::ChunkedReadError("virtual dataset has no mapping global heap".into())
|
||||
})?;
|
||||
let coll = GlobalHeapCollection::parse(file_data, addr as usize, length_size)?;
|
||||
let obj =
|
||||
coll.get_object(global_heap_index as u16)
|
||||
.ok_or(FormatError::GlobalHeapObjectNotFound {
|
||||
collection_address: addr,
|
||||
index: global_heap_index as u16,
|
||||
})?;
|
||||
let mappings = parse_vds_mappings(&obj.data, length_size)?;
|
||||
|
||||
for m in &mappings {
|
||||
let same_file = m.source_file.is_empty() || m.source_file == ".";
|
||||
|
||||
// Resolve the bytes of the file holding this source dataset.
|
||||
let external;
|
||||
let src_file_data: &[u8] = if same_file {
|
||||
file_data
|
||||
} else {
|
||||
let r = resolver.ok_or_else(|| {
|
||||
FormatError::ChunkedReadError(
|
||||
"external-file virtual dataset sources require a file resolver".into(),
|
||||
)
|
||||
})?;
|
||||
match r(&m.source_file) {
|
||||
Some(bytes) => {
|
||||
external = bytes;
|
||||
&external
|
||||
}
|
||||
// Source file unavailable: leave this region at fill value.
|
||||
None => continue,
|
||||
}
|
||||
};
|
||||
|
||||
let (vsel, _) = Selection::decode_serialized(&m.virtual_selection)?;
|
||||
let (ssel, _) = Selection::decode_serialized(&m.source_selection)?;
|
||||
|
||||
let (src_raw, src_dims) =
|
||||
read_named_dataset_raw(src_file_data, &m.source_dataset, offset_size, length_size)?;
|
||||
|
||||
let vidx = vsel.iter_linear(virtual_dims)?;
|
||||
let sidx = ssel.iter_linear(&src_dims)?;
|
||||
if vidx.len() != sidx.len() {
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"virtual/source selection element counts differ".into(),
|
||||
));
|
||||
}
|
||||
|
||||
for (&v, &s) in vidx.iter().zip(sidx.iter()) {
|
||||
let (vo, so) = (v as usize * elem_size, s as usize * elem_size);
|
||||
if vo + elem_size > out.len() || so + elem_size > src_raw.len() {
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"virtual dataset selection out of bounds".into(),
|
||||
));
|
||||
}
|
||||
out[vo..vo + elem_size].copy_from_slice(&src_raw[so..so + elem_size]);
|
||||
}
|
||||
}
|
||||
|
||||
Ok(out)
|
||||
}
|
||||
|
||||
/// Read a named dataset's raw (decoded) bytes and its dimensions, navigating
|
||||
/// from the superblock. Used to pull VDS source datasets out of the same file.
|
||||
fn read_named_dataset_raw(
|
||||
file_data: &[u8],
|
||||
path: &str,
|
||||
_offset_size: u8,
|
||||
_length_size: u8,
|
||||
) -> Result<(Vec<u8>, Vec<u64>), FormatError> {
|
||||
use crate::filter_pipeline::FilterPipeline;
|
||||
use crate::group_v2::resolve_path_any;
|
||||
use crate::message_type::MessageType;
|
||||
use crate::object_header::ObjectHeader;
|
||||
use crate::signature::find_signature;
|
||||
use crate::superblock::Superblock;
|
||||
|
||||
let sig = find_signature(file_data)?;
|
||||
let sb = Superblock::parse(file_data, sig)?;
|
||||
let addr = resolve_path_any(file_data, &sb, path)?;
|
||||
let hdr = ObjectHeader::parse(file_data, addr as usize, sb.offset_size, sb.length_size)?;
|
||||
|
||||
let find = |t: MessageType| hdr.messages.iter().find(|m| m.msg_type == t);
|
||||
let ds_msg = find(MessageType::Dataspace)
|
||||
.ok_or_else(|| FormatError::ChunkedReadError("VDS source has no dataspace".into()))?;
|
||||
let dataspace = Dataspace::parse(&ds_msg.data, sb.length_size)?;
|
||||
let dt_msg = find(MessageType::Datatype)
|
||||
.ok_or_else(|| FormatError::ChunkedReadError("VDS source has no datatype".into()))?;
|
||||
let (datatype, _) = Datatype::parse(&dt_msg.data)?;
|
||||
let dl_msg = find(MessageType::DataLayout)
|
||||
.ok_or_else(|| FormatError::ChunkedReadError("VDS source has no data layout".into()))?;
|
||||
let layout = DataLayout::parse(&dl_msg.data, sb.offset_size, sb.length_size)?;
|
||||
// A virtual dataset whose source is itself another virtual dataset could
|
||||
// form a cycle (A -> B -> A) and recurse into a stack overflow. Nested
|
||||
// virtual sources are exotic and unsupported, so stop here cleanly.
|
||||
if matches!(layout, DataLayout::Virtual { .. }) {
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"virtual dataset source is itself virtual (unsupported)".into(),
|
||||
));
|
||||
}
|
||||
let pipeline = find(MessageType::FilterPipeline)
|
||||
.map(|m| FilterPipeline::parse(&m.data))
|
||||
.transpose()?;
|
||||
|
||||
let raw = read_raw_data_full(
|
||||
let wrapped =
|
||||
resolver.map(|r| move |name: &str| -> Result<Option<Vec<u8>>, FormatError> { Ok(r(name)) });
|
||||
let wrapped_ref = wrapped.as_ref().map(|w| w as &crate::vds::VdsFileResolver);
|
||||
let v = crate::vds::read_virtual_dataset(
|
||||
file_data,
|
||||
&layout,
|
||||
&dataspace,
|
||||
&datatype,
|
||||
pipeline.as_ref(),
|
||||
sb.offset_size,
|
||||
sb.length_size,
|
||||
layout,
|
||||
dataspace,
|
||||
datatype,
|
||||
None,
|
||||
offset_size,
|
||||
length_size,
|
||||
wrapped_ref,
|
||||
)?;
|
||||
Ok((raw, dataspace.dimensions.clone()))
|
||||
if v.dims != dataspace.dimensions {
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"virtual dataset extent differs from its stored dataspace; \
|
||||
read it with vds::read_virtual_dataset"
|
||||
.into(),
|
||||
));
|
||||
}
|
||||
if v.unmapped > 0 {
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"virtual dataset has elements no source supplies, which read as its \
|
||||
fill value; read it with vds::read_virtual_dataset and the fill value"
|
||||
.into(),
|
||||
));
|
||||
}
|
||||
Ok(v.data)
|
||||
}
|
||||
|
||||
/// Extract selected elements from a full dataset buffer.
|
||||
pub fn extract_selection_from_buffer(
|
||||
full_data: &[u8],
|
||||
@@ -773,14 +664,7 @@ pub fn read_as_f64_zerocopy<'a>(raw: &'a [u8], datatype: &Datatype) -> Option<&'
|
||||
// Only native LE f64 is eligible
|
||||
#[cfg(target_endian = "little")]
|
||||
{
|
||||
if !matches!(
|
||||
datatype,
|
||||
Datatype::FloatingPoint {
|
||||
size: 8,
|
||||
byte_order: DatatypeByteOrder::LittleEndian,
|
||||
..
|
||||
}
|
||||
) {
|
||||
if !is_native_le_float(datatype, FloatFormat::Double) {
|
||||
return None;
|
||||
}
|
||||
if !raw.len().is_multiple_of(8) {
|
||||
@@ -809,14 +693,7 @@ pub fn read_as_f64_zerocopy<'a>(raw: &'a [u8], datatype: &Datatype) -> Option<&'
|
||||
pub fn read_as_f32_zerocopy<'a>(raw: &'a [u8], datatype: &Datatype) -> Option<&'a [f32]> {
|
||||
#[cfg(target_endian = "little")]
|
||||
{
|
||||
if !matches!(
|
||||
datatype,
|
||||
Datatype::FloatingPoint {
|
||||
size: 4,
|
||||
byte_order: DatatypeByteOrder::LittleEndian,
|
||||
..
|
||||
}
|
||||
) {
|
||||
if !is_native_le_float(datatype, FloatFormat::Single) {
|
||||
return None;
|
||||
}
|
||||
if !raw.len().is_multiple_of(4) {
|
||||
@@ -902,9 +779,9 @@ fn native_le_to_vec<T: Copy>(raw: &[u8], count: usize) -> Vec<T> {
|
||||
|
||||
/// Convert raw bytes to `f64` values.
|
||||
pub fn read_as_f64(raw: &[u8], datatype: &Datatype) -> Result<Vec<f64>, FormatError> {
|
||||
// Array datatypes (e.g. an array-typed compound member) are read as a flat
|
||||
// sequence of their base elements.
|
||||
if let Datatype::Array { base_type, .. } = datatype {
|
||||
// Array datatypes read as a flat sequence of their base elements, and
|
||||
// enumerations (h5py's bool among them) as their integer values.
|
||||
if let Datatype::Array { base_type, .. } | Datatype::Enumeration { base_type, .. } = datatype {
|
||||
return read_as_f64(raw, base_type);
|
||||
}
|
||||
ensure_numeric(datatype, "FloatingPoint or FixedPoint")?;
|
||||
@@ -919,20 +796,19 @@ pub fn read_as_f64(raw: &[u8], datatype: &Datatype) -> Result<Vec<f64>, FormatEr
|
||||
|
||||
// Fast path: native-endian f64 — single bulk memcpy
|
||||
#[cfg(target_endian = "little")]
|
||||
if matches!(
|
||||
datatype,
|
||||
Datatype::FloatingPoint {
|
||||
size: 8,
|
||||
byte_order: DatatypeByteOrder::LittleEndian,
|
||||
..
|
||||
}
|
||||
) {
|
||||
if is_native_le_float(datatype, FloatFormat::Double) {
|
||||
return Ok(native_le_to_vec::<f64>(raw, count));
|
||||
}
|
||||
|
||||
let order = get_byte_order(datatype);
|
||||
let mut result = Vec::with_capacity(count);
|
||||
|
||||
if let Datatype::FloatingPoint { .. } = datatype {
|
||||
let format = FloatFormat::of(datatype)?;
|
||||
for chunk in raw.chunks_exact(elem_size) {
|
||||
result.push(format.decode(chunk, &order));
|
||||
}
|
||||
return Ok(result);
|
||||
}
|
||||
for i in 0..count {
|
||||
let chunk = &raw[i * elem_size..(i + 1) * elem_size];
|
||||
let val = convert_to_f64(chunk, datatype, &order)?;
|
||||
@@ -947,18 +823,7 @@ fn convert_to_f64(
|
||||
order: &DatatypeByteOrder,
|
||||
) -> Result<f64, FormatError> {
|
||||
match dt {
|
||||
Datatype::FloatingPoint { size, .. } => match size {
|
||||
4 => {
|
||||
let v = read_f32_bytes(bytes, order);
|
||||
Ok(v as f64)
|
||||
}
|
||||
8 => Ok(read_f64_bytes(bytes, order)),
|
||||
2 => Ok(read_f16_bytes(bytes, order) as f64),
|
||||
_ => Err(FormatError::DataSizeMismatch {
|
||||
expected: 8,
|
||||
actual: *size as usize,
|
||||
}),
|
||||
},
|
||||
Datatype::FloatingPoint { .. } => Ok(FloatFormat::of(dt)?.decode(bytes, order)),
|
||||
Datatype::FixedPoint {
|
||||
size,
|
||||
signed,
|
||||
@@ -982,9 +847,83 @@ fn convert_to_f64(
|
||||
}
|
||||
}
|
||||
|
||||
/// One numeric element as stored, before conversion to the caller's type.
|
||||
#[derive(Debug, Clone, Copy, PartialEq)]
|
||||
enum Scalar {
|
||||
Signed(i64),
|
||||
Unsigned(u64),
|
||||
Float(f64),
|
||||
}
|
||||
|
||||
impl Scalar {
|
||||
// Every conversion follows libhdf5's default (hard) conversions: a value
|
||||
// outside the target type's range saturates to its minimum or maximum —
|
||||
// including a negative value read as unsigned, which reads as 0 — rather
|
||||
// than being truncated to its low bits. Floats truncate toward zero; NaN
|
||||
// converts to 0 (libhdf5 leaves that case to the C cast, whose result is
|
||||
// platform-dependent).
|
||||
|
||||
fn to_i64(self) -> i64 {
|
||||
match self {
|
||||
Scalar::Signed(v) => v,
|
||||
Scalar::Unsigned(v) => i64::try_from(v).unwrap_or(i64::MAX),
|
||||
Scalar::Float(v) => v as i64,
|
||||
}
|
||||
}
|
||||
|
||||
fn to_u64(self) -> u64 {
|
||||
match self {
|
||||
Scalar::Signed(v) => u64::try_from(v).unwrap_or(0),
|
||||
Scalar::Unsigned(v) => v,
|
||||
Scalar::Float(v) => v as u64,
|
||||
}
|
||||
}
|
||||
|
||||
fn to_i32(self) -> i32 {
|
||||
match self {
|
||||
Scalar::Signed(v) => v.clamp(i32::MIN.into(), i32::MAX.into()) as i32,
|
||||
Scalar::Unsigned(v) => i32::try_from(v).unwrap_or(i32::MAX),
|
||||
Scalar::Float(v) => v as i32,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Decode one element of a numeric datatype.
|
||||
fn decode_scalar(
|
||||
bytes: &[u8],
|
||||
dt: &Datatype,
|
||||
order: &DatatypeByteOrder,
|
||||
) -> Result<Scalar, FormatError> {
|
||||
match dt {
|
||||
Datatype::FixedPoint {
|
||||
size,
|
||||
signed,
|
||||
bit_offset,
|
||||
bit_precision,
|
||||
..
|
||||
} => {
|
||||
let full = read_unsigned_int(bytes, *size as usize, order);
|
||||
let (off, prec) = effective_bits(*size as usize, *bit_offset, *bit_precision);
|
||||
Ok(if *signed {
|
||||
Scalar::Signed(extract_signed(full, off, prec))
|
||||
} else {
|
||||
Scalar::Unsigned(extract_unsigned(full, off, prec))
|
||||
})
|
||||
}
|
||||
_ => convert_to_f64(bytes, dt, order).map(Scalar::Float),
|
||||
}
|
||||
}
|
||||
|
||||
/// Convert raw bytes to `i64` values.
|
||||
///
|
||||
/// Values are converted the way libhdf5 converts them: integers outside the
|
||||
/// target range saturate at its minimum or maximum (a negative value read as
|
||||
/// unsigned is 0), and floating-point data is truncated toward zero and
|
||||
/// saturated, with NaN read as 0.
|
||||
pub fn read_as_i64(raw: &[u8], datatype: &Datatype) -> Result<Vec<i64>, FormatError> {
|
||||
if let Datatype::Array { base_type, .. } = datatype {
|
||||
// Array datatypes read as a flat sequence of their base elements, and
|
||||
// enumerations (h5py's bool among them) as their integer values.
|
||||
if let Datatype::Array { base_type, .. } | Datatype::Enumeration { base_type, .. } = datatype {
|
||||
return read_as_i64(raw, base_type);
|
||||
}
|
||||
ensure_numeric(datatype, "FixedPoint (signed)")?;
|
||||
@@ -1014,19 +953,24 @@ pub fn read_as_i64(raw: &[u8], datatype: &Datatype) -> Result<Vec<i64>, FormatEr
|
||||
}
|
||||
|
||||
let order = get_byte_order(datatype);
|
||||
let (off, prec) = fixed_bits(datatype);
|
||||
let mut result = Vec::with_capacity(count);
|
||||
for i in 0..count {
|
||||
let chunk = &raw[i * elem_size..(i + 1) * elem_size];
|
||||
let full = read_unsigned_int(chunk, elem_size, &order);
|
||||
result.push(extract_signed(full, off, prec));
|
||||
result.push(decode_scalar(chunk, datatype, &order)?.to_i64());
|
||||
}
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
/// Convert raw bytes to `u64` values.
|
||||
///
|
||||
/// Values are converted the way libhdf5 converts them: integers outside the
|
||||
/// target range saturate at its minimum or maximum (a negative value read as
|
||||
/// unsigned is 0), and floating-point data is truncated toward zero and
|
||||
/// saturated, with NaN read as 0.
|
||||
pub fn read_as_u64(raw: &[u8], datatype: &Datatype) -> Result<Vec<u64>, FormatError> {
|
||||
if let Datatype::Array { base_type, .. } = datatype {
|
||||
// Array datatypes read as a flat sequence of their base elements, and
|
||||
// enumerations (h5py's bool among them) as their integer values.
|
||||
if let Datatype::Array { base_type, .. } | Datatype::Enumeration { base_type, .. } = datatype {
|
||||
return read_as_u64(raw, base_type);
|
||||
}
|
||||
ensure_numeric(datatype, "FixedPoint (unsigned)")?;
|
||||
@@ -1039,19 +983,19 @@ pub fn read_as_u64(raw: &[u8], datatype: &Datatype) -> Result<Vec<u64>, FormatEr
|
||||
}
|
||||
let count = raw.len() / elem_size;
|
||||
let order = get_byte_order(datatype);
|
||||
let (off, prec) = fixed_bits(datatype);
|
||||
let mut result = Vec::with_capacity(count);
|
||||
for i in 0..count {
|
||||
let chunk = &raw[i * elem_size..(i + 1) * elem_size];
|
||||
let full = read_unsigned_int(chunk, elem_size, &order);
|
||||
result.push(extract_unsigned(full, off, prec));
|
||||
result.push(decode_scalar(chunk, datatype, &order)?.to_u64());
|
||||
}
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
/// Convert raw bytes to `f32` values.
|
||||
pub fn read_as_f32(raw: &[u8], datatype: &Datatype) -> Result<Vec<f32>, FormatError> {
|
||||
if let Datatype::Array { base_type, .. } = datatype {
|
||||
// Array datatypes read as a flat sequence of their base elements, and
|
||||
// enumerations (h5py's bool among them) as their integer values.
|
||||
if let Datatype::Array { base_type, .. } | Datatype::Enumeration { base_type, .. } = datatype {
|
||||
return read_as_f32(raw, base_type);
|
||||
}
|
||||
ensure_numeric(datatype, "FloatingPoint")?;
|
||||
@@ -1066,25 +1010,11 @@ pub fn read_as_f32(raw: &[u8], datatype: &Datatype) -> Result<Vec<f32>, FormatEr
|
||||
|
||||
// Fast path: native-endian f32 — single bulk memcpy
|
||||
#[cfg(target_endian = "little")]
|
||||
if matches!(
|
||||
datatype,
|
||||
Datatype::FloatingPoint {
|
||||
size: 4,
|
||||
byte_order: DatatypeByteOrder::LittleEndian,
|
||||
..
|
||||
}
|
||||
) {
|
||||
if is_native_le_float(datatype, FloatFormat::Single) {
|
||||
return Ok(native_le_to_vec::<f32>(raw, count));
|
||||
}
|
||||
// Little-endian half precision (numpy float16): widen directly.
|
||||
if matches!(
|
||||
datatype,
|
||||
Datatype::FloatingPoint {
|
||||
size: 2,
|
||||
byte_order: DatatypeByteOrder::LittleEndian,
|
||||
..
|
||||
}
|
||||
) {
|
||||
// Little-endian IEEE half precision (numpy float16): widen directly.
|
||||
if is_native_le_float(datatype, FloatFormat::Half) {
|
||||
let (halves, _) = raw[..count * 2].as_chunks::<2>();
|
||||
return Ok(halves
|
||||
.iter()
|
||||
@@ -1094,18 +1024,22 @@ pub fn read_as_f32(raw: &[u8], datatype: &Datatype) -> Result<Vec<f32>, FormatEr
|
||||
|
||||
let order = get_byte_order(datatype);
|
||||
let mut result = Vec::with_capacity(count);
|
||||
if let Datatype::FloatingPoint { .. } = datatype {
|
||||
let format = FloatFormat::of(datatype)?;
|
||||
for chunk in raw.chunks_exact(elem_size) {
|
||||
result.push(match format {
|
||||
FloatFormat::Single => read_f32_bytes(chunk, &order),
|
||||
FloatFormat::Half => read_f16_bytes(chunk, &order),
|
||||
// Double rounds; every other supported layout (bfloat16, FP8)
|
||||
// is exact in f32.
|
||||
_ => format.decode(chunk, &order) as f32,
|
||||
});
|
||||
}
|
||||
return Ok(result);
|
||||
}
|
||||
for i in 0..count {
|
||||
let chunk = &raw[i * elem_size..(i + 1) * elem_size];
|
||||
match datatype {
|
||||
Datatype::FloatingPoint { size: 4, .. } => {
|
||||
result.push(read_f32_bytes(chunk, &order));
|
||||
}
|
||||
Datatype::FloatingPoint { size: 8, .. } => {
|
||||
result.push(read_f64_bytes(chunk, &order) as f32);
|
||||
}
|
||||
Datatype::FloatingPoint { size: 2, .. } => {
|
||||
result.push(read_f16_bytes(chunk, &order));
|
||||
}
|
||||
Datatype::FixedPoint {
|
||||
signed: true,
|
||||
size,
|
||||
@@ -1140,8 +1074,15 @@ pub fn read_as_f32(raw: &[u8], datatype: &Datatype) -> Result<Vec<f32>, FormatEr
|
||||
}
|
||||
|
||||
/// Convert raw bytes to `i32` values.
|
||||
///
|
||||
/// Values are converted the way libhdf5 converts them: integers outside the
|
||||
/// target range saturate at its minimum or maximum (a negative value read as
|
||||
/// unsigned is 0), and floating-point data is truncated toward zero and
|
||||
/// saturated, with NaN read as 0.
|
||||
pub fn read_as_i32(raw: &[u8], datatype: &Datatype) -> Result<Vec<i32>, FormatError> {
|
||||
if let Datatype::Array { base_type, .. } = datatype {
|
||||
// Array datatypes read as a flat sequence of their base elements, and
|
||||
// enumerations (h5py's bool among them) as their integer values.
|
||||
if let Datatype::Array { base_type, .. } | Datatype::Enumeration { base_type, .. } = datatype {
|
||||
return read_as_i32(raw, base_type);
|
||||
}
|
||||
ensure_numeric(datatype, "FixedPoint")?;
|
||||
@@ -1162,6 +1103,7 @@ pub fn read_as_i32(raw: &[u8], datatype: &Datatype) -> Result<Vec<i32>, FormatEr
|
||||
datatype,
|
||||
Datatype::FixedPoint {
|
||||
byte_order: DatatypeByteOrder::LittleEndian,
|
||||
signed: true,
|
||||
..
|
||||
}
|
||||
)
|
||||
@@ -1170,12 +1112,10 @@ pub fn read_as_i32(raw: &[u8], datatype: &Datatype) -> Result<Vec<i32>, FormatEr
|
||||
}
|
||||
|
||||
let order = get_byte_order(datatype);
|
||||
let (off, prec) = fixed_bits(datatype);
|
||||
let mut result = Vec::with_capacity(count);
|
||||
for i in 0..count {
|
||||
let chunk = &raw[i * elem_size..(i + 1) * elem_size];
|
||||
let full = read_unsigned_int(chunk, elem_size, &order);
|
||||
result.push(extract_signed(full, off, prec) as i32);
|
||||
result.push(decode_scalar(chunk, datatype, &order)?.to_i32());
|
||||
}
|
||||
Ok(result)
|
||||
}
|
||||
@@ -1616,6 +1556,174 @@ fn reorder_bytes(bytes: &[u8], order: &DatatypeByteOrder) -> [u8; 8] {
|
||||
buf
|
||||
}
|
||||
|
||||
/// How the bits of a floating-point datatype are laid out, read from the
|
||||
/// datatype message's fields rather than assumed from its size (a 2-byte
|
||||
/// float may be IEEE half or bfloat16).
|
||||
#[derive(Debug, Clone, Copy, PartialEq)]
|
||||
enum FloatFormat {
|
||||
/// IEEE-754 binary16.
|
||||
Half,
|
||||
/// IEEE-754 binary32.
|
||||
Single,
|
||||
/// IEEE-754 binary64.
|
||||
Double,
|
||||
/// Any other IEEE-style layout (implied leading mantissa bit, all-ones
|
||||
/// exponent for infinity/NaN) whose values are all exact in `f64`:
|
||||
/// bfloat16, the FP8 formats, and similar.
|
||||
Other(FloatLayout),
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq)]
|
||||
struct FloatLayout {
|
||||
exponent_location: u32,
|
||||
exponent_size: u32,
|
||||
mantissa_location: u32,
|
||||
mantissa_size: u32,
|
||||
exponent_bias: u32,
|
||||
}
|
||||
|
||||
impl FloatFormat {
|
||||
fn of(dt: &Datatype) -> Result<FloatFormat, FormatError> {
|
||||
let Datatype::FloatingPoint {
|
||||
size,
|
||||
exponent_location,
|
||||
exponent_size,
|
||||
mantissa_location,
|
||||
mantissa_size,
|
||||
exponent_bias,
|
||||
..
|
||||
} = dt
|
||||
else {
|
||||
return Err(FormatError::TypeMismatch {
|
||||
expected: "FloatingPoint",
|
||||
actual: datatype_name(dt),
|
||||
});
|
||||
};
|
||||
let layout = FloatLayout {
|
||||
exponent_location: u32::from(*exponent_location),
|
||||
exponent_size: u32::from(*exponent_size),
|
||||
mantissa_location: u32::from(*mantissa_location),
|
||||
mantissa_size: u32::from(*mantissa_size),
|
||||
exponent_bias: *exponent_bias,
|
||||
};
|
||||
let fields = (
|
||||
layout.exponent_location,
|
||||
layout.exponent_size,
|
||||
layout.mantissa_location,
|
||||
layout.mantissa_size,
|
||||
layout.exponent_bias,
|
||||
);
|
||||
let bits = size.saturating_mul(8);
|
||||
// The sign bit is not kept in `Datatype`; every standard layout has it
|
||||
// directly above the exponent, with the mantissa below.
|
||||
let well_formed = layout.exponent_size > 0
|
||||
&& layout.mantissa_size > 0
|
||||
&& layout.mantissa_location + layout.mantissa_size <= layout.exponent_location
|
||||
&& layout.exponent_location + layout.exponent_size < bits;
|
||||
match (size, fields) {
|
||||
(2, (10, 5, 0, 10, 15)) => Ok(FloatFormat::Half),
|
||||
(4, (23, 8, 0, 23, 127)) => Ok(FloatFormat::Single),
|
||||
(8, (52, 11, 0, 52, 1023)) => Ok(FloatFormat::Double),
|
||||
_ if well_formed
|
||||
&& *size <= 8
|
||||
&& layout.exponent_size <= 11
|
||||
&& layout.mantissa_size <= 52 =>
|
||||
{
|
||||
Ok(FloatFormat::Other(layout))
|
||||
}
|
||||
// Fields that cannot describe any float (e.g. left zeroed by a
|
||||
// hand-built datatype): fall back to the IEEE type of that size.
|
||||
(2, _) if !well_formed => Ok(FloatFormat::Half),
|
||||
(4, _) if !well_formed => Ok(FloatFormat::Single),
|
||||
(8, _) if !well_formed => Ok(FloatFormat::Double),
|
||||
// x87 80-bit extended, binary128, ...: not representable in f64.
|
||||
_ => Err(FormatError::TypeMismatch {
|
||||
expected: "floating point of at most 64 bits (IEEE-style layout)",
|
||||
actual: "FloatingPoint",
|
||||
}),
|
||||
}
|
||||
}
|
||||
|
||||
fn decode(self, bytes: &[u8], order: &DatatypeByteOrder) -> f64 {
|
||||
match self {
|
||||
FloatFormat::Half => f64::from(read_f16_bytes(bytes, order)),
|
||||
FloatFormat::Single => f64::from(read_f32_bytes(bytes, order)),
|
||||
FloatFormat::Double => read_f64_bytes(bytes, order),
|
||||
FloatFormat::Other(layout) => {
|
||||
layout.decode(read_unsigned_int(bytes, bytes.len(), order))
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl FloatLayout {
|
||||
/// Decode the value held in the low `size * 8` bits of `bits`.
|
||||
fn decode(self, bits: u64) -> f64 {
|
||||
let field = |location: u32, size: u32| (bits >> location) & ((1u64 << size) - 1);
|
||||
let exponent = field(self.exponent_location, self.exponent_size);
|
||||
let mantissa = field(self.mantissa_location, self.mantissa_size);
|
||||
let negative = field(self.exponent_location + self.exponent_size, 1) == 1;
|
||||
let max_exponent = (1u64 << self.exponent_size) - 1;
|
||||
let magnitude = if exponent == max_exponent {
|
||||
if mantissa == 0 {
|
||||
f64::INFINITY
|
||||
} else {
|
||||
f64::NAN
|
||||
}
|
||||
} else {
|
||||
let bias = i64::from(self.exponent_bias);
|
||||
let msize = i64::from(self.mantissa_size);
|
||||
// value = significand * 2^power, with an implied leading 1 unless
|
||||
// the number is subnormal (exponent field 0).
|
||||
let (significand, power) = if exponent == 0 {
|
||||
(mantissa, 1 - bias - msize)
|
||||
} else {
|
||||
(
|
||||
mantissa | (1u64 << self.mantissa_size),
|
||||
exponent as i64 - bias - msize,
|
||||
)
|
||||
};
|
||||
scale_by_pow2(significand as f64, power)
|
||||
};
|
||||
if negative { -magnitude } else { magnitude }
|
||||
}
|
||||
}
|
||||
|
||||
/// `x * 2^power` without `std` (no `powi`/`libm`). `x` is a non-negative
|
||||
/// integer below 2^53, so it is exact.
|
||||
fn scale_by_pow2(x: f64, power: i64) -> f64 {
|
||||
if x == 0.0 || power < -1200 {
|
||||
return 0.0;
|
||||
}
|
||||
if power > 1100 {
|
||||
return f64::INFINITY;
|
||||
}
|
||||
let pow2 = |p: i64| f64::from_bits(((p + 1023) as u64) << 52);
|
||||
let mut x = x;
|
||||
let mut power = power;
|
||||
while power > 1023 {
|
||||
x *= pow2(1023);
|
||||
power -= 1023;
|
||||
}
|
||||
while power < -1022 {
|
||||
x *= pow2(-1022);
|
||||
power += 1022;
|
||||
}
|
||||
x * pow2(power)
|
||||
}
|
||||
|
||||
/// Whether `datatype` is the little-endian IEEE float `format`, whose bytes
|
||||
/// can be copied straight into native values on a little-endian target.
|
||||
fn is_native_le_float(datatype: &Datatype, format: FloatFormat) -> bool {
|
||||
matches!(
|
||||
datatype,
|
||||
Datatype::FloatingPoint {
|
||||
byte_order: DatatypeByteOrder::LittleEndian,
|
||||
..
|
||||
}
|
||||
) && FloatFormat::of(datatype).is_ok_and(|f| f == format)
|
||||
}
|
||||
|
||||
fn read_f64_bytes(bytes: &[u8], order: &DatatypeByteOrder) -> f64 {
|
||||
let buf = reorder_bytes(bytes, order);
|
||||
f64::from_le_bytes(buf)
|
||||
@@ -1666,20 +1774,6 @@ fn effective_bits(size: usize, bit_offset: u16, bit_precision: u16) -> (u32, u32
|
||||
(bit_offset as u32, prec)
|
||||
}
|
||||
|
||||
/// `(bit_offset, bit_precision)` for a fixed-point datatype, full width for
|
||||
/// other types.
|
||||
fn fixed_bits(datatype: &Datatype) -> (u32, u32) {
|
||||
match datatype {
|
||||
Datatype::FixedPoint {
|
||||
size,
|
||||
bit_offset,
|
||||
bit_precision,
|
||||
..
|
||||
} => effective_bits(*size as usize, *bit_offset, *bit_precision),
|
||||
_ => (0, 0),
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether a datatype occupies its full storage width (bit offset 0, precision
|
||||
/// == size·8), in which case the bulk-copy fast read paths apply. Non
|
||||
/// fixed-point types are treated as full width.
|
||||
@@ -1892,6 +1986,67 @@ mod tests {
|
||||
assert_eq!(read_as_u64(&raw, &dt).unwrap(), vec![4095, 1, 2048]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn float_to_int_truncates_and_saturates() {
|
||||
// Values libhdf5 hands to an undefined C cast: NaN reads as 0 and
|
||||
// exactly 2^63 saturates instead of wrapping to i64::MIN.
|
||||
let dt = make_f64_le_type();
|
||||
let vals = [f64::NAN, 2f64.powi(63), -2.5, 2.0f64.powi(64)];
|
||||
let raw: Vec<u8> = vals.iter().flat_map(|v| v.to_le_bytes()).collect();
|
||||
assert_eq!(
|
||||
read_as_i64(&raw, &dt).unwrap(),
|
||||
vec![0, i64::MAX, -2, i64::MAX]
|
||||
);
|
||||
assert_eq!(
|
||||
read_as_u64(&raw, &dt).unwrap(),
|
||||
vec![0, 1 << 63, 0, u64::MAX]
|
||||
);
|
||||
assert_eq!(
|
||||
read_as_i32(&raw, &dt).unwrap(),
|
||||
vec![0, i32::MAX, -2, i32::MAX]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn bfloat16_and_fp8_decode_by_fields() {
|
||||
// bfloat16 is a 2-byte float that is not IEEE half.
|
||||
let bf16 = Datatype::FloatingPoint {
|
||||
size: 2,
|
||||
byte_order: DatatypeByteOrder::LittleEndian,
|
||||
bit_offset: 0,
|
||||
bit_precision: 16,
|
||||
exponent_location: 7,
|
||||
exponent_size: 8,
|
||||
mantissa_location: 0,
|
||||
mantissa_size: 7,
|
||||
exponent_bias: 127,
|
||||
};
|
||||
let raw: Vec<u8> = [0x3FC0u16, 0xC010, 0x7F80, 0x0001]
|
||||
.iter()
|
||||
.flat_map(|v| v.to_le_bytes())
|
||||
.collect();
|
||||
let got = read_as_f64(&raw, &bf16).unwrap();
|
||||
assert_eq!(&got[..3], &[1.5, -2.25, f64::INFINITY]);
|
||||
assert_eq!(got[3], 2f64.powi(-133)); // smallest subnormal
|
||||
assert_eq!(read_as_f32(&raw, &bf16).unwrap()[..2], [1.5, -2.25]);
|
||||
|
||||
// FP8 E4M3: 1, -1, 2, 0, NaN (IEEE-style, as libhdf5 treats it).
|
||||
let e4m3 = Datatype::FloatingPoint {
|
||||
size: 1,
|
||||
byte_order: DatatypeByteOrder::LittleEndian,
|
||||
bit_offset: 0,
|
||||
bit_precision: 8,
|
||||
exponent_location: 3,
|
||||
exponent_size: 4,
|
||||
mantissa_location: 0,
|
||||
mantissa_size: 3,
|
||||
exponent_bias: 7,
|
||||
};
|
||||
let got = read_as_f64(&[0x38, 0xB8, 0x40, 0x00, 0x7E], &e4m3).unwrap();
|
||||
assert_eq!(&got[..4], &[1.0, -1.0, 2.0, 0.0]);
|
||||
assert!(got[4].is_nan());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn full_width_signed_unchanged() {
|
||||
// Regression: full-width 32-bit signed must be unaffected.
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
//! for compound, enumeration, variable-length, and array types.
|
||||
|
||||
#[cfg(not(feature = "std"))]
|
||||
use alloc::{boxed::Box, string::String, vec, vec::Vec};
|
||||
use alloc::{boxed::Box, format, string::String, vec, vec::Vec};
|
||||
|
||||
use byteorder::{ByteOrder, LittleEndian};
|
||||
|
||||
@@ -137,6 +137,17 @@ pub enum Datatype {
|
||||
},
|
||||
}
|
||||
|
||||
/// Longest opaque tag that can be stored: its NUL-padded length must fit
|
||||
/// the 8-bit length in the datatype's class bits.
|
||||
pub const MAX_OPAQUE_TAG_LEN: usize = 248;
|
||||
|
||||
/// An opaque tag up to (not including) its first NUL.
|
||||
fn opaque_tag_text(tag: &[u8]) -> &[u8] {
|
||||
tag.iter()
|
||||
.position(|&b| b == 0)
|
||||
.map_or(tag, |end| &tag[..end])
|
||||
}
|
||||
|
||||
fn ensure_len(data: &[u8], offset: usize, needed: usize) -> Result<(), FormatError> {
|
||||
match offset.checked_add(needed) {
|
||||
Some(end) if end <= data.len() => Ok(()),
|
||||
@@ -361,7 +372,10 @@ impl Datatype {
|
||||
// Opaque
|
||||
let tag_len = bf0 as usize;
|
||||
ensure_len(data, pos, tag_len)?;
|
||||
let tag = data[pos..pos + tag_len].to_vec();
|
||||
// The stored tag is NUL-padded to a multiple of 8 bytes; the
|
||||
// tag itself ends at the first NUL (libhdf5 reads it with
|
||||
// `strndup`).
|
||||
let tag = opaque_tag_text(&data[pos..pos + tag_len]).to_vec();
|
||||
// Tags are padded to multiple of 8 bytes
|
||||
let padded = (tag_len + 7) & !7;
|
||||
let pos = 8 + padded; // from start of properties
|
||||
@@ -409,13 +423,44 @@ impl Datatype {
|
||||
ensure_len(data, pos, 4)?;
|
||||
let byte_offset = LittleEndian::read_u32(&data[pos..pos + 4]) as u64;
|
||||
pos += 4;
|
||||
// v1 members can be fixed-size arrays of the member
|
||||
// type (libhdf5 builds an array type from these
|
||||
// fields; the permutation is ignored, as libhdf5
|
||||
// does). Skipping them read a `[4] i32` member as
|
||||
// one `i32`.
|
||||
let mut array_dims = Vec::new();
|
||||
if version == 1 {
|
||||
ensure_len(data, pos, 28)?;
|
||||
let ndims = data[pos] as usize;
|
||||
// libhdf5 refuses more than four dimensions and,
|
||||
// when building the array type, a zero-sized one.
|
||||
let zero_dim = (0..ndims.min(4)).any(|j| {
|
||||
let at = pos + 12 + 4 * j;
|
||||
LittleEndian::read_u32(&data[at..at + 4]) == 0
|
||||
});
|
||||
if ndims > 4 || zero_dim {
|
||||
return Err(FormatError::InvalidDatatypeVersion {
|
||||
class: class_id,
|
||||
version,
|
||||
});
|
||||
}
|
||||
array_dims = (0..ndims)
|
||||
.map(|j| {
|
||||
let at = pos + 12 + 4 * j;
|
||||
LittleEndian::read_u32(&data[at..at + 4])
|
||||
})
|
||||
.collect();
|
||||
pos += 28;
|
||||
}
|
||||
let (member_dt, consumed) =
|
||||
let (mut member_dt, consumed) =
|
||||
Self::parse_with_depth(&data[pos..], depth + 1)?;
|
||||
pos += consumed;
|
||||
if !array_dims.is_empty() {
|
||||
member_dt = Datatype::Array {
|
||||
base_type: Box::new(member_dt),
|
||||
dimensions: array_dims,
|
||||
};
|
||||
}
|
||||
members.push(CompoundMember {
|
||||
name,
|
||||
byte_offset,
|
||||
@@ -767,7 +812,77 @@ impl Datatype {
|
||||
buf.extend_from_slice(&base_type.serialize());
|
||||
buf
|
||||
}
|
||||
_ => Vec::new(),
|
||||
Datatype::Time {
|
||||
size,
|
||||
bit_precision,
|
||||
} => {
|
||||
// Byte order is not modelled for time types; write little-endian.
|
||||
let mut buf = Self::build_header(2, 1, [0, 0, 0], *size);
|
||||
buf.extend_from_slice(&bit_precision.to_le_bytes());
|
||||
buf
|
||||
}
|
||||
Datatype::BitField {
|
||||
size,
|
||||
byte_order,
|
||||
bit_offset,
|
||||
bit_precision,
|
||||
} => {
|
||||
let bf0 = u8::from(matches!(byte_order, DatatypeByteOrder::BigEndian));
|
||||
let mut buf = Self::build_header(4, 1, [bf0, 0, 0], *size);
|
||||
buf.extend_from_slice(&bit_offset.to_le_bytes());
|
||||
buf.extend_from_slice(&bit_precision.to_le_bytes());
|
||||
buf
|
||||
}
|
||||
Datatype::Opaque { size, tag } => {
|
||||
// The tag is stored NUL-padded to a multiple of 8 bytes and the
|
||||
// padded length goes in the class bits, as libhdf5 writes it.
|
||||
// A tag longer than MAX_OPAQUE_TAG_LEN cannot be encoded;
|
||||
// `check_encodable` rejects it before a file is written.
|
||||
let tag = opaque_tag_text(tag);
|
||||
let tag = &tag[..tag.len().min(MAX_OPAQUE_TAG_LEN)];
|
||||
let padded = tag.len().div_ceil(8) * 8;
|
||||
let mut buf = Self::build_header(5, 1, [padded as u8, 0, 0], *size);
|
||||
buf.extend_from_slice(tag);
|
||||
buf.resize(8 + padded, 0);
|
||||
buf
|
||||
}
|
||||
Datatype::Reference { size, ref_type } => {
|
||||
// Legacy references are datatype version 1; the H5T_STD_REF
|
||||
// kinds only exist from version 4, which also carries their
|
||||
// encoding version (1) in the high nibble.
|
||||
let (version, bf0) = match ref_type {
|
||||
ReferenceType::Object => (1, 0),
|
||||
ReferenceType::DatasetRegion => (1, 1),
|
||||
ReferenceType::Object2 => (4, 0x12),
|
||||
ReferenceType::DatasetRegion2 => (4, 0x13),
|
||||
ReferenceType::Attribute => (4, 0x14),
|
||||
};
|
||||
Self::build_header(7, version, [bf0, 0, 0], *size)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Check that this datatype can be written: every part of it has an
|
||||
/// on-disk encoding. [`Self::serialize`] cannot report errors, so the
|
||||
/// writer calls this first.
|
||||
pub fn check_encodable(&self) -> Result<(), FormatError> {
|
||||
match self {
|
||||
Datatype::Opaque { tag, .. } if opaque_tag_text(tag).len() > MAX_OPAQUE_TAG_LEN => {
|
||||
Err(FormatError::SerializationError(format!(
|
||||
"opaque tag is {} bytes; at most {MAX_OPAQUE_TAG_LEN} can be stored",
|
||||
opaque_tag_text(tag).len()
|
||||
)))
|
||||
}
|
||||
Datatype::String { size: 0, .. } => Err(FormatError::SerializationError(
|
||||
"fixed-length string datatype of size 0 (libhdf5 requires at least 1 byte)".into(),
|
||||
)),
|
||||
Datatype::Compound { members, .. } => members
|
||||
.iter()
|
||||
.try_for_each(|m| m.datatype.check_encodable()),
|
||||
Datatype::Enumeration { base_type, .. }
|
||||
| Datatype::VariableLength { base_type, .. }
|
||||
| Datatype::Array { base_type, .. } => base_type.check_encodable(),
|
||||
_ => Ok(()),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1252,6 +1367,64 @@ mod tests {
|
||||
assert_xyid_compound(dt);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_compound_v1_member_array_fields() {
|
||||
// HDF5 1.6 wrote array members of a v1 compound through the legacy
|
||||
// per-member fields (as in libhdf5's tools/test/testfiles/
|
||||
// tcompound.h5 `type2`: `int_array` [4] i32, `float_array` [5][6]
|
||||
// f32). They used to be skipped, reading each member as a scalar.
|
||||
let i32le: [u8; 12] = [
|
||||
0x10, 0x08, 0x00, 0x00, 0x04, 0x00, 0x00, 0x00, 0x00, 0x00, 0x20, 0x00,
|
||||
];
|
||||
let mut b = vec![0x16, 0x02, 0x00, 0x00, 0x88, 0x00, 0x00, 0x00];
|
||||
for (name, offset, dims) in [
|
||||
(&b"int_array"[..], 0u32, &[4u32][..]),
|
||||
(&b"xy"[..], 16, &[5u32, 6][..]),
|
||||
] {
|
||||
let mut padded = name.to_vec();
|
||||
padded.resize((name.len() + 1 + 7) & !7, 0);
|
||||
b.extend_from_slice(&padded);
|
||||
b.extend_from_slice(&offset.to_le_bytes());
|
||||
b.push(dims.len() as u8);
|
||||
b.extend_from_slice(&[0u8; 3 + 4 + 4]); // reserved, permutation, reserved
|
||||
for j in 0..4 {
|
||||
b.extend_from_slice(&dims.get(j).copied().unwrap_or(0).to_le_bytes());
|
||||
}
|
||||
b.extend_from_slice(&i32le);
|
||||
}
|
||||
let (dt, consumed) = Datatype::parse(&b).unwrap();
|
||||
assert_eq!(consumed, b.len());
|
||||
let Datatype::Compound { members, .. } = dt else {
|
||||
panic!("expected Compound, got {dt:?}");
|
||||
};
|
||||
let got: Vec<(&str, u64, u32, Option<Vec<u32>>)> = members
|
||||
.iter()
|
||||
.map(|m| {
|
||||
let dims = match &m.datatype {
|
||||
Datatype::Array { dimensions, .. } => Some(dimensions.clone()),
|
||||
_ => None,
|
||||
};
|
||||
(m.name.as_str(), m.byte_offset, m.datatype.type_size(), dims)
|
||||
})
|
||||
.collect();
|
||||
assert_eq!(
|
||||
got,
|
||||
vec![
|
||||
("int_array", 0, 16, Some(vec![4])),
|
||||
("xy", 16, 120, Some(vec![5, 6])),
|
||||
]
|
||||
);
|
||||
|
||||
// More than four dimensions cannot be encoded, and libhdf5 refuses a
|
||||
// zero-sized dimension (a fuzzed tcompound.h5, cve-2024-32616.h5).
|
||||
let mut bad = b.clone();
|
||||
bad[8 + 16 + 4] = 5;
|
||||
assert!(Datatype::parse(&bad).is_err());
|
||||
let mut bad = b.clone();
|
||||
bad[8 + 16 + 4] = 2; // [4, 0]
|
||||
assert!(Datatype::parse(&bad).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_compound_v1_truncated_is_error_not_panic() {
|
||||
let bytes = compound_v1_bytes();
|
||||
@@ -1625,6 +1798,122 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
fn hex(s: &str) -> Vec<u8> {
|
||||
(0..s.len())
|
||||
.step_by(2)
|
||||
.map(|i| u8::from_str_radix(&s[i..i + 2], 16).unwrap())
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// `serialize` used to return an empty message for these four classes,
|
||||
/// which libhdf5 rejects ("ran off end of input buffer while decoding").
|
||||
/// Expected bytes are libhdf5's own encoding (HDF5 2.0 `H5Tencode`, or the
|
||||
/// datatype message of an HDF5 2.0 file for `H5T_STD_REF`).
|
||||
#[test]
|
||||
fn serialize_matches_libhdf5_for_time_bitfield_opaque_reference() {
|
||||
let cases = [
|
||||
(
|
||||
Datatype::Reference {
|
||||
size: 8,
|
||||
ref_type: ReferenceType::Object,
|
||||
},
|
||||
"1700000008000000",
|
||||
),
|
||||
(
|
||||
Datatype::Reference {
|
||||
size: 12,
|
||||
ref_type: ReferenceType::DatasetRegion,
|
||||
},
|
||||
"170100000c000000",
|
||||
),
|
||||
(
|
||||
Datatype::Reference {
|
||||
size: 18,
|
||||
ref_type: ReferenceType::Object2,
|
||||
},
|
||||
"4712000012000000",
|
||||
),
|
||||
(
|
||||
Datatype::BitField {
|
||||
size: 1,
|
||||
byte_order: DatatypeByteOrder::LittleEndian,
|
||||
bit_offset: 0,
|
||||
bit_precision: 8,
|
||||
},
|
||||
"140000000100000000000800",
|
||||
),
|
||||
(
|
||||
Datatype::BitField {
|
||||
size: 2,
|
||||
byte_order: DatatypeByteOrder::BigEndian,
|
||||
bit_offset: 0,
|
||||
bit_precision: 16,
|
||||
},
|
||||
"140100000200000000001000",
|
||||
),
|
||||
(
|
||||
Datatype::Opaque {
|
||||
size: 4,
|
||||
tag: b"mytag".to_vec(),
|
||||
},
|
||||
"15080000040000006d79746167000000",
|
||||
),
|
||||
(
|
||||
Datatype::Opaque {
|
||||
size: 4,
|
||||
tag: b"12345678".to_vec(),
|
||||
},
|
||||
"15080000040000003132333435363738",
|
||||
),
|
||||
(
|
||||
Datatype::Opaque {
|
||||
size: 4,
|
||||
tag: vec![],
|
||||
},
|
||||
"1500000004000000",
|
||||
),
|
||||
(
|
||||
Datatype::Time {
|
||||
size: 4,
|
||||
bit_precision: 32,
|
||||
},
|
||||
"12000000040000002000",
|
||||
),
|
||||
];
|
||||
for (dt, expected) in cases {
|
||||
let bytes = dt.serialize();
|
||||
assert_eq!(bytes, hex(expected), "{dt:?}");
|
||||
let (parsed, consumed) = Datatype::parse(&bytes).unwrap();
|
||||
assert_eq!(parsed, dt);
|
||||
assert_eq!(consumed, bytes.len());
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn opaque_tag_padding_is_not_part_of_the_tag() {
|
||||
// libhdf5 pads "mytag" to 8 bytes; parsing must not return the NULs,
|
||||
// or copying the type would grow the tag.
|
||||
let (dt, _) = Datatype::parse(&hex("15080000040000006d79746167000000")).unwrap();
|
||||
assert_eq!(
|
||||
dt,
|
||||
Datatype::Opaque {
|
||||
size: 4,
|
||||
tag: b"mytag".to_vec()
|
||||
}
|
||||
);
|
||||
let long = Datatype::Opaque {
|
||||
size: 1,
|
||||
tag: vec![b'x'; MAX_OPAQUE_TAG_LEN + 1],
|
||||
};
|
||||
assert!(long.check_encodable().is_err());
|
||||
let ok = Datatype::Opaque {
|
||||
size: 1,
|
||||
tag: vec![b'x'; MAX_OPAQUE_TAG_LEN],
|
||||
};
|
||||
assert!(ok.check_encodable().is_ok());
|
||||
assert_eq!(Datatype::parse(&ok.serialize()).unwrap().0, ok);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_error_invalid_reference_type() {
|
||||
let buf = build_dt_header(7, 1, [5, 0, 0], 8);
|
||||
|
||||
@@ -7,7 +7,7 @@ extern crate alloc;
|
||||
use alloc::{vec, vec::Vec};
|
||||
|
||||
use crate::checksum::jenkins_lookup3;
|
||||
use crate::chunked_write::WrittenChunk;
|
||||
use crate::chunked_write::{WrittenChunk, filtered_chunk_size_len, push_addr, push_index_element};
|
||||
|
||||
/// Serialize a v4 Extensible Array layout message.
|
||||
pub(crate) fn serialize_v4_extensible_array(
|
||||
@@ -58,11 +58,11 @@ pub(crate) fn serialize_v4_extensible_array(
|
||||
buf.push(4);
|
||||
|
||||
// EA creation parameters (must match AEHD and HDF5 C library defaults)
|
||||
buf.push(32); // max_nelmts_bits
|
||||
buf.push(4); // idx_blk_elmts
|
||||
buf.push(4); // super_blk_min_data_ptrs
|
||||
buf.push(16); // data_blk_min_elmts
|
||||
buf.push(10); // max_dblk_page_nelmts_bits
|
||||
buf.push(MAX_NELMTS_BITS);
|
||||
buf.push(IDX_BLK_ELMTS);
|
||||
buf.push(SUP_BLK_MIN_DATA_PTRS);
|
||||
buf.push(DATA_BLK_MIN_ELMTS);
|
||||
buf.push(MAX_DBLK_PAGE_NELMTS_BITS);
|
||||
|
||||
// EA header address
|
||||
match offset_size {
|
||||
@@ -74,304 +74,281 @@ pub(crate) fn serialize_v4_extensible_array(
|
||||
buf
|
||||
}
|
||||
|
||||
// EA creation parameters — the HDF5 library's defaults for chunk indexes
|
||||
// (`H5D_EARRAY_*`); the layout message above and the header must agree.
|
||||
const MAX_NELMTS_BITS: u8 = 32;
|
||||
const IDX_BLK_ELMTS: u8 = 4;
|
||||
const SUP_BLK_MIN_DATA_PTRS: u8 = 4;
|
||||
const DATA_BLK_MIN_ELMTS: u8 = 16;
|
||||
const MAX_DBLK_PAGE_NELMTS_BITS: u8 = 10;
|
||||
|
||||
/// One data block of the array: its first element (relative to the end of
|
||||
/// the index block's own elements), element count, and address when it is
|
||||
/// allocated.
|
||||
struct DataBlock {
|
||||
start: usize,
|
||||
nelmts: usize,
|
||||
addr: Option<u64>,
|
||||
}
|
||||
|
||||
/// Build a complete Extensible Array at a known absolute address.
|
||||
///
|
||||
/// For simplicity, we put all elements inline in the index block when the
|
||||
/// number of chunks is small (up to idx_blk_elmts), otherwise use inline +
|
||||
/// direct data blocks.
|
||||
/// `slots[i]` is the element at linear index `i` (see `chunk_grid`); `None`
|
||||
/// marks an unallocated chunk. The first `IDX_BLK_ELMTS` elements live in
|
||||
/// the index block, the rest in data blocks grouped by super block level
|
||||
/// exactly as `H5EA__hdr_init` sizes them: level `u` has `2^(u/2)` data
|
||||
/// blocks of `DATA_BLK_MIN_ELMTS * 2^ceil(u/2)` elements. The data blocks of
|
||||
/// the first levels are addressed straight from the index block; later
|
||||
/// levels go through a super block (EASB). Data blocks larger than a page
|
||||
/// (`2^MAX_DBLK_PAGE_NELMTS_BITS` elements) are paged, with their page-init
|
||||
/// bits kept in the owning super block. Only blocks holding a defined element
|
||||
/// are allocated; the rest keep the undefined address, as in a file the
|
||||
/// library wrote.
|
||||
pub fn build_extensible_array_at(
|
||||
chunks: &[WrittenChunk],
|
||||
slots: &[Option<WrittenChunk>],
|
||||
offset_size: u8,
|
||||
length_size: u8,
|
||||
has_filters: bool,
|
||||
ea_base_address: u64,
|
||||
) -> Vec<u8> {
|
||||
let os = offset_size as usize;
|
||||
let num_elements = chunks.len();
|
||||
|
||||
// Compute element encoding size (same logic as Fixed Array)
|
||||
let chunk_size_bytes: usize = if has_filters {
|
||||
let max_raw = chunks.iter().map(|c| c.raw_size).max().unwrap_or(1);
|
||||
let log2_val = if max_raw <= 1 {
|
||||
0
|
||||
} else {
|
||||
63 - max_raw.leading_zeros()
|
||||
};
|
||||
let len = 1 + ((log2_val + 8) / 8) as usize;
|
||||
len.min(8)
|
||||
} else {
|
||||
0
|
||||
};
|
||||
|
||||
let elem_size = if has_filters {
|
||||
os + chunk_size_bytes + 4
|
||||
} else {
|
||||
os
|
||||
};
|
||||
|
||||
let chunk_size_bytes = has_filters.then(|| filtered_chunk_size_len(slots));
|
||||
let elem_size = os + chunk_size_bytes.map_or(0, |n| n + 4);
|
||||
let client_id: u8 = if has_filters { 1 } else { 0 };
|
||||
let arr_off_size = (MAX_NELMTS_BITS as usize).div_ceil(8);
|
||||
let page_nelmts = 1usize << MAX_DBLK_PAGE_NELMTS_BITS;
|
||||
let idx_blk = IDX_BLK_ELMTS as usize;
|
||||
|
||||
// EA creation parameters — must match HDF5 C library defaults exactly
|
||||
let max_nelmts_bits: u8 = 32;
|
||||
let idx_blk_elmts: u8 = 4;
|
||||
let min_dblk_nelmts: u8 = 16;
|
||||
let super_blk_min_nelmts: u8 = 4;
|
||||
let max_dblk_nelmts_bits: u8 = 10;
|
||||
// Elements past the last defined one are never realised
|
||||
// (`max_idx_set` is one past the highest index ever set).
|
||||
let max_idx_set = slots.iter().rposition(Option::is_some).map_or(0, |i| i + 1);
|
||||
let slots = &slots[..max_idx_set];
|
||||
let defined_in = |start: usize, n: usize| -> bool {
|
||||
let lo = idx_blk.saturating_add(start).min(slots.len());
|
||||
let hi = idx_blk
|
||||
.saturating_add(start)
|
||||
.saturating_add(n)
|
||||
.min(slots.len());
|
||||
slots[lo..hi].iter().any(Option::is_some)
|
||||
};
|
||||
|
||||
// EAHD size: fixed(12) + 6 stats(6*length_size) + addr(offset_size) + checksum(4)
|
||||
// Super block levels: (ndblks, dblk_nelmts, first element).
|
||||
let log2_dmin = (DATA_BLK_MIN_ELMTS as u32).trailing_zeros() as usize;
|
||||
let nsblks = 1 + MAX_NELMTS_BITS as usize - log2_dmin;
|
||||
let ndblk_addrs = 2 * (SUP_BLK_MIN_DATA_PTRS as usize - 1);
|
||||
let mut levels: Vec<(usize, usize, usize)> = Vec::with_capacity(nsblks);
|
||||
let mut start = 0usize;
|
||||
for u in 0..nsblks {
|
||||
let ndblks = 1usize << (u / 2);
|
||||
let nelmts = (DATA_BLK_MIN_ELMTS as usize) << u.div_ceil(2);
|
||||
levels.push((ndblks, nelmts, start));
|
||||
// Saturate: on 32-bit targets the last levels only need to compare
|
||||
// as "beyond the end".
|
||||
start = start.saturating_add(ndblks.saturating_mul(nelmts));
|
||||
}
|
||||
// Levels whose data blocks the index block addresses directly.
|
||||
let mut direct_levels = 0;
|
||||
let mut n = 0;
|
||||
while n < ndblk_addrs {
|
||||
n += levels[direct_levels].0;
|
||||
direct_levels += 1;
|
||||
}
|
||||
let nsblk_addrs = nsblks - direct_levels;
|
||||
|
||||
let dblk_size = |nelmts: usize| -> usize {
|
||||
let prefix = 4 + 1 + 1 + os + arr_off_size + 4;
|
||||
if nelmts > page_nelmts {
|
||||
prefix + (nelmts / page_nelmts) * (page_nelmts * elem_size + 4)
|
||||
} else {
|
||||
prefix + nelmts * elem_size
|
||||
}
|
||||
};
|
||||
let sblk_bitmap_len = |ndblks: usize, nelmts: usize| -> usize {
|
||||
if nelmts > page_nelmts {
|
||||
ndblks * (nelmts / page_nelmts).div_ceil(8)
|
||||
} else {
|
||||
0
|
||||
}
|
||||
};
|
||||
|
||||
// Plan addresses: header, index block, the direct data blocks, then each
|
||||
// allocated super block followed by its allocated data blocks.
|
||||
let aehd_size = 4 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 6 * length_size as usize + os + 4;
|
||||
let aeib_address = ea_base_address + aehd_size as u64;
|
||||
let aeib_size = 4 + 1 + 1 + os + idx_blk * elem_size + ndblk_addrs * os + nsblk_addrs * os + 4;
|
||||
let mut cursor = aeib_address + aeib_size as u64;
|
||||
|
||||
// Determine how many elements go inline vs data blocks
|
||||
let n_inline = (idx_blk_elmts as usize).min(num_elements);
|
||||
let remaining_after_inline = num_elements.saturating_sub(n_inline);
|
||||
let mut ndata_blks = 0u64;
|
||||
let mut data_blk_size = 0u64;
|
||||
let mut nsuper_blks = 0u64;
|
||||
let mut super_blk_size = 0u64;
|
||||
let mut realized = idx_blk as u64;
|
||||
|
||||
// Compute super block layout per HDF5 spec
|
||||
let sblk_min = super_blk_min_nelmts as usize;
|
||||
let log2_dblk_min = if min_dblk_nelmts <= 1 {
|
||||
0
|
||||
} else {
|
||||
(min_dblk_nelmts as u32).trailing_zeros() as usize
|
||||
};
|
||||
let nsblks = (max_nelmts_bits as usize).saturating_sub(log2_dblk_min) + 1;
|
||||
|
||||
// Direct data block addresses (from super blocks 0..sblk_min-1)
|
||||
let mut dblk_sizes: Vec<usize> = Vec::new();
|
||||
for sblk_idx in 0..sblk_min.min(nsblks) {
|
||||
let ndblks = 1usize << (sblk_idx / 2);
|
||||
let dblk_nelmts = (min_dblk_nelmts as usize) * (1 << sblk_idx.div_ceil(2));
|
||||
for _ in 0..ndblks {
|
||||
dblk_sizes.push(dblk_nelmts);
|
||||
let mut plan_dblk = |cursor: &mut u64, start: usize, nelmts: usize| -> DataBlock {
|
||||
let addr = defined_in(start, nelmts).then(|| {
|
||||
let a = *cursor;
|
||||
let size = dblk_size(nelmts) as u64;
|
||||
*cursor += size;
|
||||
ndata_blks += 1;
|
||||
data_blk_size += size;
|
||||
realized += nelmts as u64;
|
||||
a
|
||||
});
|
||||
DataBlock {
|
||||
start,
|
||||
nelmts,
|
||||
addr,
|
||||
}
|
||||
}
|
||||
let n_direct_dblks = dblk_sizes.len();
|
||||
|
||||
// Super block addresses (for super blocks sblk_min..nsblks-1)
|
||||
let n_sblk_addrs = nsblks.saturating_sub(sblk_min);
|
||||
|
||||
// EAIB size
|
||||
let aeib_size = 4
|
||||
+ 1
|
||||
+ 1
|
||||
+ os
|
||||
+ idx_blk_elmts as usize * elem_size
|
||||
+ n_direct_dblks * os
|
||||
+ n_sblk_addrs * os
|
||||
+ 4;
|
||||
|
||||
// Build AEHD
|
||||
let mut aehd = Vec::with_capacity(aehd_size);
|
||||
aehd.extend_from_slice(b"EAHD");
|
||||
aehd.push(0); // version
|
||||
aehd.push(client_id);
|
||||
aehd.push(elem_size as u8);
|
||||
aehd.push(max_nelmts_bits);
|
||||
aehd.push(idx_blk_elmts);
|
||||
aehd.push(min_dblk_nelmts);
|
||||
aehd.push(super_blk_min_nelmts);
|
||||
aehd.push(max_dblk_nelmts_bits);
|
||||
|
||||
// Count data blocks that will have chunks
|
||||
let n_active_dblks: u64 = if remaining_after_inline > 0 {
|
||||
let mut count = 0u64;
|
||||
let mut ci = n_inline;
|
||||
for &sz in &dblk_sizes {
|
||||
if ci < num_elements {
|
||||
count += 1;
|
||||
ci += sz;
|
||||
}
|
||||
}
|
||||
count
|
||||
} else {
|
||||
0
|
||||
};
|
||||
let blk_off_size = (max_nelmts_bits as usize).div_ceil(8);
|
||||
let aedb_header_overhead = 4 + 1 + 1 + os + blk_off_size + 4;
|
||||
let data_blk_total_size: u64 = if remaining_after_inline > 0 {
|
||||
let mut total = 0u64;
|
||||
let mut ci = n_inline;
|
||||
for &sz in &dblk_sizes {
|
||||
if ci < num_elements {
|
||||
total += (aedb_header_overhead + sz * elem_size) as u64;
|
||||
ci += sz;
|
||||
}
|
||||
}
|
||||
total
|
||||
} else {
|
||||
0
|
||||
};
|
||||
let max_idx_set: u64 = if remaining_after_inline > 0 {
|
||||
let mut max_set = idx_blk_elmts as u64;
|
||||
let mut ci = n_inline;
|
||||
for &sz in &dblk_sizes {
|
||||
if ci < num_elements {
|
||||
max_set += sz as u64;
|
||||
ci += sz;
|
||||
}
|
||||
}
|
||||
max_set
|
||||
} else {
|
||||
idx_blk_elmts as u64
|
||||
};
|
||||
|
||||
let mut direct: Vec<DataBlock> = Vec::with_capacity(ndblk_addrs);
|
||||
for &(ndblks, nelmts, first) in &levels[..direct_levels] {
|
||||
for k in 0..ndblks {
|
||||
direct.push(plan_dblk(&mut cursor, first + k * nelmts, nelmts));
|
||||
}
|
||||
}
|
||||
// (super block address, level, its data blocks)
|
||||
let mut supers: Vec<(Option<u64>, usize, Vec<DataBlock>)> = Vec::with_capacity(nsblk_addrs);
|
||||
for (u, &(ndblks, nelmts, first)) in levels.iter().enumerate().skip(direct_levels) {
|
||||
if !defined_in(first, ndblks.saturating_mul(nelmts)) {
|
||||
supers.push((None, u, Vec::new()));
|
||||
continue;
|
||||
}
|
||||
let sb_size =
|
||||
4 + 1 + 1 + os + arr_off_size + sblk_bitmap_len(ndblks, nelmts) + ndblks * os + 4;
|
||||
let sb_addr = cursor;
|
||||
cursor += sb_size as u64;
|
||||
nsuper_blks += 1;
|
||||
super_blk_size += sb_size as u64;
|
||||
let dblks = (0..ndblks)
|
||||
.map(|k| plan_dblk(&mut cursor, first + k * nelmts, nelmts))
|
||||
.collect();
|
||||
supers.push((Some(sb_addr), u, dblks));
|
||||
}
|
||||
|
||||
let slot = |i: usize| slots.get(i).and_then(Option::as_ref);
|
||||
let write_length = |buf: &mut Vec<u8>, val: u64| match length_size {
|
||||
4 => buf.extend_from_slice(&(val as u32).to_le_bytes()),
|
||||
_ => buf.extend_from_slice(&val.to_le_bytes()),
|
||||
};
|
||||
let write_addr = |buf: &mut Vec<u8>, val: u64| match offset_size {
|
||||
4 => buf.extend_from_slice(&(val as u32).to_le_bytes()),
|
||||
_ => buf.extend_from_slice(&val.to_le_bytes()),
|
||||
let write_addr_opt = |buf: &mut Vec<u8>, addr: Option<u64>| match addr {
|
||||
Some(a) => push_addr(buf, a, offset_size),
|
||||
None => buf.extend(core::iter::repeat_n(0xFF, os)),
|
||||
};
|
||||
let block_prefix = |buf: &mut Vec<u8>, sig: &[u8; 4], block_off: usize| {
|
||||
buf.extend_from_slice(sig);
|
||||
buf.push(0); // version
|
||||
buf.push(client_id);
|
||||
push_addr(buf, ea_base_address, offset_size);
|
||||
buf.extend_from_slice(&(block_off as u64).to_le_bytes()[..arr_off_size]);
|
||||
};
|
||||
// Serialise one data block (paged or not) onto `out`.
|
||||
let write_dblk = |out: &mut Vec<u8>, db: &DataBlock| {
|
||||
let at = out.len();
|
||||
block_prefix(out, b"EADB", db.start);
|
||||
let first = idx_blk + db.start;
|
||||
if db.nelmts > page_nelmts {
|
||||
// Paged: the prefix carries only its own checksum; each page
|
||||
// follows with one of its own.
|
||||
let sum = jenkins_lookup3(&out[at..]);
|
||||
out.extend_from_slice(&sum.to_le_bytes());
|
||||
for p in 0..db.nelmts / page_nelmts {
|
||||
let page_at = out.len();
|
||||
for e in 0..page_nelmts {
|
||||
let i = first + p * page_nelmts + e;
|
||||
push_index_element(out, slot(i), offset_size, chunk_size_bytes);
|
||||
}
|
||||
let sum = jenkins_lookup3(&out[page_at..]);
|
||||
out.extend_from_slice(&sum.to_le_bytes());
|
||||
}
|
||||
} else {
|
||||
for i in first..first + db.nelmts {
|
||||
push_index_element(out, slot(i), offset_size, chunk_size_bytes);
|
||||
}
|
||||
let sum = jenkins_lookup3(&out[at..]);
|
||||
out.extend_from_slice(&sum.to_le_bytes());
|
||||
}
|
||||
debug_assert_eq!(out.len() - at, dblk_size(db.nelmts));
|
||||
};
|
||||
|
||||
write_length(&mut aehd, 0);
|
||||
write_length(&mut aehd, 0);
|
||||
write_length(&mut aehd, n_active_dblks);
|
||||
write_length(&mut aehd, data_blk_total_size);
|
||||
write_length(&mut aehd, num_elements as u64);
|
||||
write_length(&mut aehd, max_idx_set);
|
||||
// Header (EAHD). The six statistics are, in order: super blocks, their
|
||||
// bytes, data blocks, their bytes, max index set, elements realised.
|
||||
let mut out = Vec::with_capacity((cursor - ea_base_address) as usize);
|
||||
out.extend_from_slice(b"EAHD");
|
||||
out.push(0); // version
|
||||
out.push(client_id);
|
||||
out.push(elem_size as u8);
|
||||
out.push(MAX_NELMTS_BITS);
|
||||
out.push(IDX_BLK_ELMTS);
|
||||
out.push(DATA_BLK_MIN_ELMTS);
|
||||
out.push(SUP_BLK_MIN_DATA_PTRS);
|
||||
out.push(MAX_DBLK_PAGE_NELMTS_BITS);
|
||||
write_length(&mut out, nsuper_blks);
|
||||
write_length(&mut out, super_blk_size);
|
||||
write_length(&mut out, ndata_blks);
|
||||
write_length(&mut out, data_blk_size);
|
||||
write_length(&mut out, max_idx_set as u64);
|
||||
write_length(&mut out, realized);
|
||||
push_addr(&mut out, aeib_address, offset_size);
|
||||
let sum = jenkins_lookup3(&out);
|
||||
out.extend_from_slice(&sum.to_le_bytes());
|
||||
debug_assert_eq!(out.len(), aehd_size);
|
||||
|
||||
write_addr(&mut aehd, aeib_address);
|
||||
|
||||
let aehd_checksum = jenkins_lookup3(&aehd);
|
||||
aehd.extend_from_slice(&aehd_checksum.to_le_bytes());
|
||||
debug_assert_eq!(aehd.len(), aehd_size);
|
||||
|
||||
// Build AEIB
|
||||
let mut aeib = Vec::with_capacity(aeib_size);
|
||||
aeib.extend_from_slice(b"EAIB");
|
||||
aeib.push(0);
|
||||
aeib.push(client_id);
|
||||
|
||||
match offset_size {
|
||||
4 => aeib.extend_from_slice(&(ea_base_address as u32).to_le_bytes()),
|
||||
8 => aeib.extend_from_slice(&ea_base_address.to_le_bytes()),
|
||||
_ => aeib.extend_from_slice(&ea_base_address.to_le_bytes()),
|
||||
// Index block (EAIB): inline elements, data block and super block
|
||||
// addresses.
|
||||
let ib_start = out.len();
|
||||
out.extend_from_slice(b"EAIB");
|
||||
out.push(0);
|
||||
out.push(client_id);
|
||||
push_addr(&mut out, ea_base_address, offset_size);
|
||||
for i in 0..idx_blk {
|
||||
push_index_element(&mut out, slot(i), offset_size, chunk_size_bytes);
|
||||
}
|
||||
|
||||
// Inline elements
|
||||
#[allow(clippy::needless_range_loop)]
|
||||
for i in 0..idx_blk_elmts as usize {
|
||||
if i < n_inline {
|
||||
write_chunk_element(
|
||||
&mut aeib,
|
||||
&chunks[i],
|
||||
offset_size,
|
||||
has_filters,
|
||||
chunk_size_bytes,
|
||||
);
|
||||
} else {
|
||||
write_undefined_element(&mut aeib, offset_size, has_filters, chunk_size_bytes);
|
||||
for db in &direct {
|
||||
write_addr_opt(&mut out, db.addr);
|
||||
}
|
||||
for (sb_addr, _, _) in &supers {
|
||||
write_addr_opt(&mut out, *sb_addr);
|
||||
}
|
||||
let sum = jenkins_lookup3(&out[ib_start..]);
|
||||
out.extend_from_slice(&sum.to_le_bytes());
|
||||
debug_assert_eq!(out.len() - ib_start, aeib_size);
|
||||
|
||||
// Data block addresses + build data blocks
|
||||
let mut data_blocks_buf = Vec::new();
|
||||
let dblks_base = aeib_address + aeib_size as u64;
|
||||
let mut dblk_cursor = dblks_base;
|
||||
let mut chunk_idx = n_inline;
|
||||
|
||||
for &nelmts in &dblk_sizes {
|
||||
if chunk_idx >= num_elements {
|
||||
match offset_size {
|
||||
4 => aeib.extend_from_slice(&u32::MAX.to_le_bytes()),
|
||||
8 => aeib.extend_from_slice(&u64::MAX.to_le_bytes()),
|
||||
_ => aeib.extend_from_slice(&u64::MAX.to_le_bytes()),
|
||||
for db in direct.iter().filter(|d| d.addr.is_some()) {
|
||||
write_dblk(&mut out, db);
|
||||
}
|
||||
for (sb_addr, u, dblks) in &supers {
|
||||
if sb_addr.is_none() {
|
||||
continue;
|
||||
}
|
||||
|
||||
match offset_size {
|
||||
4 => aeib.extend_from_slice(&(dblk_cursor as u32).to_le_bytes()),
|
||||
8 => aeib.extend_from_slice(&dblk_cursor.to_le_bytes()),
|
||||
_ => aeib.extend_from_slice(&dblk_cursor.to_le_bytes()),
|
||||
}
|
||||
|
||||
// Build EADB
|
||||
let mut aedb = Vec::new();
|
||||
aedb.extend_from_slice(b"EADB");
|
||||
aedb.push(0);
|
||||
aedb.push(client_id);
|
||||
match offset_size {
|
||||
4 => aedb.extend_from_slice(&(ea_base_address as u32).to_le_bytes()),
|
||||
8 => aedb.extend_from_slice(&ea_base_address.to_le_bytes()),
|
||||
_ => aedb.extend_from_slice(&ea_base_address.to_le_bytes()),
|
||||
}
|
||||
|
||||
let blk_off_size = (max_nelmts_bits as usize).div_ceil(8);
|
||||
let blk_off_val = (chunk_idx - n_inline) as u64;
|
||||
aedb.extend_from_slice(&blk_off_val.to_le_bytes()[..blk_off_size]);
|
||||
|
||||
for slot in 0..nelmts {
|
||||
if chunk_idx + slot < num_elements {
|
||||
write_chunk_element(
|
||||
&mut aedb,
|
||||
&chunks[chunk_idx + slot],
|
||||
offset_size,
|
||||
has_filters,
|
||||
chunk_size_bytes,
|
||||
);
|
||||
} else {
|
||||
write_undefined_element(&mut aedb, offset_size, has_filters, chunk_size_bytes);
|
||||
let (ndblks, nelmts, first) = levels[*u];
|
||||
let sb_start = out.len();
|
||||
block_prefix(&mut out, b"EASB", first);
|
||||
if nelmts > page_nelmts {
|
||||
// Page-init bits, `npages` per data block, packed MSB-first
|
||||
// (`H5VM_bit_set`): every page of an allocated data block is
|
||||
// written.
|
||||
let npages = nelmts / page_nelmts;
|
||||
let mut bitmap = vec![0u8; sblk_bitmap_len(ndblks, nelmts)];
|
||||
for (k, db) in dblks.iter().enumerate() {
|
||||
if db.addr.is_some() {
|
||||
for p in 0..npages {
|
||||
let bit = k * npages + p;
|
||||
bitmap[bit / 8] |= 0x80 >> (bit % 8);
|
||||
}
|
||||
}
|
||||
|
||||
let aedb_checksum = jenkins_lookup3(&aedb);
|
||||
aedb.extend_from_slice(&aedb_checksum.to_le_bytes());
|
||||
|
||||
dblk_cursor += aedb.len() as u64;
|
||||
data_blocks_buf.extend_from_slice(&aedb);
|
||||
chunk_idx += nelmts;
|
||||
}
|
||||
|
||||
// Super block addresses (all undefined)
|
||||
for _ in 0..n_sblk_addrs {
|
||||
match offset_size {
|
||||
4 => aeib.extend_from_slice(&u32::MAX.to_le_bytes()),
|
||||
8 => aeib.extend_from_slice(&u64::MAX.to_le_bytes()),
|
||||
_ => aeib.extend_from_slice(&u64::MAX.to_le_bytes()),
|
||||
out.extend_from_slice(&bitmap);
|
||||
}
|
||||
for db in dblks {
|
||||
write_addr_opt(&mut out, db.addr);
|
||||
}
|
||||
let sum = jenkins_lookup3(&out[sb_start..]);
|
||||
out.extend_from_slice(&sum.to_le_bytes());
|
||||
for db in dblks.iter().filter(|d| d.addr.is_some()) {
|
||||
write_dblk(&mut out, db);
|
||||
}
|
||||
}
|
||||
|
||||
let aeib_checksum = jenkins_lookup3(&aeib);
|
||||
aeib.extend_from_slice(&aeib_checksum.to_le_bytes());
|
||||
debug_assert_eq!(aeib.len(), aeib_size);
|
||||
|
||||
let mut combined = aehd;
|
||||
combined.extend_from_slice(&aeib);
|
||||
combined.extend_from_slice(&data_blocks_buf);
|
||||
combined
|
||||
}
|
||||
|
||||
fn write_chunk_element(
|
||||
buf: &mut Vec<u8>,
|
||||
chunk: &WrittenChunk,
|
||||
offset_size: u8,
|
||||
has_filters: bool,
|
||||
chunk_size_bytes: usize,
|
||||
) {
|
||||
match offset_size {
|
||||
4 => buf.extend_from_slice(&(chunk.address as u32).to_le_bytes()),
|
||||
8 => buf.extend_from_slice(&chunk.address.to_le_bytes()),
|
||||
_ => buf.extend_from_slice(&chunk.address.to_le_bytes()),
|
||||
}
|
||||
if has_filters {
|
||||
let cs_bytes = chunk.compressed_size.to_le_bytes();
|
||||
buf.extend_from_slice(&cs_bytes[..chunk_size_bytes]);
|
||||
buf.extend_from_slice(&chunk.filter_mask.to_le_bytes());
|
||||
}
|
||||
}
|
||||
|
||||
fn write_undefined_element(
|
||||
buf: &mut Vec<u8>,
|
||||
offset_size: u8,
|
||||
has_filters: bool,
|
||||
chunk_size_bytes: usize,
|
||||
) {
|
||||
let os = offset_size as usize;
|
||||
// Use extend with repeat to avoid heap-allocating a temporary Vec on each call.
|
||||
buf.extend(core::iter::repeat_n(0xFF, os));
|
||||
if has_filters {
|
||||
buf.extend(core::iter::repeat_n(0x00, chunk_size_bytes));
|
||||
buf.extend_from_slice(&0u32.to_le_bytes());
|
||||
}
|
||||
debug_assert_eq!(out.len() as u64, cursor - ea_base_address);
|
||||
out
|
||||
}
|
||||
|
||||
@@ -80,6 +80,9 @@ pub enum FormatError {
|
||||
InvalidLocalHeapSignature,
|
||||
/// Invalid local heap version.
|
||||
InvalidLocalHeapVersion(u8),
|
||||
/// A local heap's free list points outside its data segment (libhdf5:
|
||||
/// "bad heap free list").
|
||||
InvalidLocalHeapFreeList,
|
||||
/// Invalid B-tree v1 signature.
|
||||
InvalidBTreeSignature,
|
||||
/// Invalid B-tree node type.
|
||||
@@ -117,6 +120,14 @@ pub enum FormatError {
|
||||
/// A message is marked shared but was parsed without access to the file,
|
||||
/// so the reference to the real message could not be followed.
|
||||
UnresolvedSharedMessage,
|
||||
/// A shared-message reference points at an object header that holds no
|
||||
/// (unshared) message of the referenced type (raw message type id).
|
||||
SharedMessageTargetMissing(u16),
|
||||
/// A superblock was parsed at a non-zero offset of the buffer (the file
|
||||
/// has a user block of this many bytes). HDF5 addresses are relative to
|
||||
/// the superblock, so the buffer must start there: see
|
||||
/// `signature::split_user_block`.
|
||||
UserBlockNotStripped(u64),
|
||||
/// A selection does not fit the dataset it was applied to (wrong rank, or
|
||||
/// it reaches past a dimension's extent).
|
||||
SelectionOutOfBounds(String),
|
||||
@@ -270,6 +281,9 @@ impl fmt::Display for FormatError {
|
||||
FormatError::InvalidLocalHeapSignature => {
|
||||
write!(f, "invalid local heap signature")
|
||||
}
|
||||
FormatError::InvalidLocalHeapFreeList => {
|
||||
write!(f, "bad local heap free list")
|
||||
}
|
||||
FormatError::InvalidLocalHeapVersion(v) => {
|
||||
write!(f, "invalid local heap version: {v}")
|
||||
}
|
||||
@@ -339,6 +353,16 @@ impl fmt::Display for FormatError {
|
||||
FormatError::SelectionOutOfBounds(msg) => {
|
||||
write!(f, "selection out of bounds: {msg}")
|
||||
}
|
||||
FormatError::UserBlockNotStripped(n) => write!(
|
||||
f,
|
||||
"file has a {n}-byte user block: parse the bytes from the superblock on \
|
||||
(signature::split_user_block)"
|
||||
),
|
||||
FormatError::SharedMessageTargetMissing(t) => write!(
|
||||
f,
|
||||
"shared message reference points at an object header with no message of type \
|
||||
{t:#06x}"
|
||||
),
|
||||
FormatError::UnresolvedSharedMessage => write!(
|
||||
f,
|
||||
"message is shared but no file data was available to resolve it"
|
||||
|
||||
@@ -9,6 +9,7 @@ extern crate alloc;
|
||||
#[cfg(not(feature = "std"))]
|
||||
use alloc::{format, vec, vec::Vec};
|
||||
|
||||
use crate::chunk_grid::ChunkGrid;
|
||||
use crate::chunked_read::ChunkInfo;
|
||||
use crate::error::FormatError;
|
||||
|
||||
@@ -203,8 +204,7 @@ fn read_element(
|
||||
offset_size: u8,
|
||||
chunk_byte_size: u64,
|
||||
linear_index: usize,
|
||||
num_chunks_per_dim: &[u64],
|
||||
chunk_dimensions: &[u32],
|
||||
grid: &ChunkGrid,
|
||||
) -> Result<(Option<ChunkInfo>, usize), FormatError> {
|
||||
let os = offset_size as usize;
|
||||
|
||||
@@ -220,7 +220,10 @@ fn read_element(
|
||||
return Ok((None, os));
|
||||
}
|
||||
let address = read_offset(data, pos, offset_size)?;
|
||||
let offsets = index_to_chunk_offsets(linear_index, num_chunks_per_dim, chunk_dimensions);
|
||||
// A slot beyond the current extent is ignored, as the library does.
|
||||
let Some(offsets) = grid.offsets(linear_index as u64) else {
|
||||
return Ok((None, os));
|
||||
};
|
||||
Ok((
|
||||
Some(ChunkInfo {
|
||||
chunk_size: chunk_byte_size as u32,
|
||||
@@ -261,7 +264,9 @@ fn read_element(
|
||||
data[fm_off + 2],
|
||||
data[fm_off + 3],
|
||||
]);
|
||||
let offsets = index_to_chunk_offsets(linear_index, num_chunks_per_dim, chunk_dimensions);
|
||||
let Some(offsets) = grid.offsets(linear_index as u64) else {
|
||||
return Ok((None, elem_total));
|
||||
};
|
||||
Ok((
|
||||
Some(ChunkInfo {
|
||||
chunk_size: chunk_size as u32,
|
||||
@@ -274,27 +279,6 @@ fn read_element(
|
||||
}
|
||||
}
|
||||
|
||||
/// Convert a linear chunk index to N-dimensional chunk offsets in dataset space.
|
||||
fn index_to_chunk_offsets(
|
||||
index: usize,
|
||||
num_chunks_per_dim: &[u64],
|
||||
chunk_dimensions: &[u32],
|
||||
) -> Vec<u64> {
|
||||
let rank = num_chunks_per_dim.len();
|
||||
let mut offsets = vec![0u64; rank];
|
||||
let mut remaining = index as u64;
|
||||
for d in (0..rank).rev() {
|
||||
let nchunks = num_chunks_per_dim[d];
|
||||
if nchunks == 0 {
|
||||
continue;
|
||||
}
|
||||
let chunk_idx = remaining % nchunks;
|
||||
remaining /= nchunks;
|
||||
offsets[d] = chunk_idx * chunk_dimensions[d] as u64;
|
||||
}
|
||||
offsets
|
||||
}
|
||||
|
||||
/// Collect elements from a data block at the given offset.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
/// Layout of super block `u`, per the HDF5 spec: the number of data blocks it
|
||||
@@ -339,8 +323,7 @@ fn read_data_block_elements(
|
||||
offset_size: u8,
|
||||
chunk_byte_size: u64,
|
||||
start_index: usize,
|
||||
num_chunks_per_dim: &[u64],
|
||||
chunk_dimensions: &[u32],
|
||||
grid: &ChunkGrid,
|
||||
page_init: &[u8],
|
||||
first_page: usize,
|
||||
) -> Result<Vec<ChunkInfo>, FormatError> {
|
||||
@@ -376,8 +359,7 @@ fn read_data_block_elements(
|
||||
offset_size,
|
||||
chunk_byte_size,
|
||||
first_index + i,
|
||||
num_chunks_per_dim,
|
||||
chunk_dimensions,
|
||||
grid,
|
||||
)?;
|
||||
if let Some(ci) = info {
|
||||
chunks.push(ci);
|
||||
@@ -449,25 +431,19 @@ pub fn read_extensible_array_chunks(
|
||||
file_data: &[u8],
|
||||
header: &ExtensibleArrayHeader,
|
||||
dataset_dims: &[u64],
|
||||
max_dims: Option<&[u64]>,
|
||||
chunk_dimensions: &[u32],
|
||||
element_size: u32,
|
||||
offset_size: u8,
|
||||
_length_size: u8,
|
||||
) -> Result<Vec<ChunkInfo>, FormatError> {
|
||||
let rank = chunk_dimensions.len();
|
||||
let os = offset_size as usize;
|
||||
|
||||
let mut num_chunks_per_dim = Vec::with_capacity(rank);
|
||||
for d in 0..rank {
|
||||
let ch_dim = chunk_dimensions[d] as u64;
|
||||
if ch_dim == 0 {
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"chunk dimension is zero".into(),
|
||||
));
|
||||
}
|
||||
let ds_dim = dataset_dims[d];
|
||||
num_chunks_per_dim.push(ds_dim.div_ceil(ch_dim));
|
||||
}
|
||||
// Linear indexes follow the maximum dimensions, with the unlimited
|
||||
// dimension swizzled to the slowest position (see `chunk_grid`).
|
||||
let dims_u64: Vec<u64> = chunk_dimensions.iter().map(|&d| d as u64).collect();
|
||||
let grid = ChunkGrid::extensible_array(dataset_dims, max_dims, &dims_u64)?;
|
||||
let grid = &grid;
|
||||
|
||||
let chunk_byte_size: u64 =
|
||||
chunk_dimensions.iter().map(|&d| d as u64).product::<u64>() * element_size as u64;
|
||||
@@ -557,8 +533,7 @@ pub fn read_extensible_array_chunks(
|
||||
offset_size,
|
||||
chunk_byte_size,
|
||||
i,
|
||||
&num_chunks_per_dim,
|
||||
chunk_dimensions,
|
||||
grid,
|
||||
)?;
|
||||
if let Some(ci) = info {
|
||||
chunks.push(ci);
|
||||
@@ -594,8 +569,7 @@ pub fn read_extensible_array_chunks(
|
||||
offset_size,
|
||||
chunk_byte_size,
|
||||
global_index,
|
||||
&num_chunks_per_dim,
|
||||
chunk_dimensions,
|
||||
grid,
|
||||
&[],
|
||||
0,
|
||||
)?);
|
||||
@@ -625,8 +599,7 @@ pub fn read_extensible_array_chunks(
|
||||
offset_size,
|
||||
chunk_byte_size,
|
||||
global_index,
|
||||
&num_chunks_per_dim,
|
||||
chunk_dimensions,
|
||||
grid,
|
||||
)?);
|
||||
}
|
||||
global_index =
|
||||
@@ -653,8 +626,7 @@ fn read_super_block(
|
||||
offset_size: u8,
|
||||
chunk_byte_size: u64,
|
||||
start_index: usize,
|
||||
num_chunks_per_dim: &[u64],
|
||||
chunk_dimensions: &[u32],
|
||||
grid: &ChunkGrid,
|
||||
) -> Result<Vec<ChunkInfo>, FormatError> {
|
||||
let os = offset_size as usize;
|
||||
let sb_header_size = 4 + 1 + 1 + os + arr_off_size(header);
|
||||
@@ -710,8 +682,7 @@ fn read_super_block(
|
||||
offset_size,
|
||||
chunk_byte_size,
|
||||
global_idx,
|
||||
num_chunks_per_dim,
|
||||
chunk_dimensions,
|
||||
grid,
|
||||
bitmap,
|
||||
i * npages,
|
||||
)?);
|
||||
@@ -735,35 +706,18 @@ mod tests {
|
||||
}
|
||||
#[test]
|
||||
fn index_to_offsets_1d() {
|
||||
let num_chunks = vec![5u64];
|
||||
let chunk_dims = vec![20u32];
|
||||
assert_eq!(index_to_chunk_offsets(0, &num_chunks, &chunk_dims), vec![0]);
|
||||
assert_eq!(
|
||||
index_to_chunk_offsets(1, &num_chunks, &chunk_dims),
|
||||
vec![20]
|
||||
);
|
||||
assert_eq!(
|
||||
index_to_chunk_offsets(4, &num_chunks, &chunk_dims),
|
||||
vec![80]
|
||||
);
|
||||
let g = ChunkGrid::fixed_array(&[100], None, &[20]).unwrap();
|
||||
assert_eq!(g.offsets(0).unwrap(), vec![0]);
|
||||
assert_eq!(g.offsets(1).unwrap(), vec![20]);
|
||||
assert_eq!(g.offsets(4).unwrap(), vec![80]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn index_to_offsets_2d() {
|
||||
let num_chunks = vec![3u64, 2];
|
||||
let chunk_dims = vec![4u32, 3];
|
||||
assert_eq!(
|
||||
index_to_chunk_offsets(0, &num_chunks, &chunk_dims),
|
||||
vec![0, 0]
|
||||
);
|
||||
assert_eq!(
|
||||
index_to_chunk_offsets(1, &num_chunks, &chunk_dims),
|
||||
vec![0, 3]
|
||||
);
|
||||
assert_eq!(
|
||||
index_to_chunk_offsets(2, &num_chunks, &chunk_dims),
|
||||
vec![4, 0]
|
||||
);
|
||||
let g = ChunkGrid::fixed_array(&[10, 6], None, &[4, 3]).unwrap();
|
||||
assert_eq!(g.offsets(0).unwrap(), vec![0, 0]);
|
||||
assert_eq!(g.offsets(1).unwrap(), vec![0, 3]);
|
||||
assert_eq!(g.offsets(2).unwrap(), vec![4, 0]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -830,7 +784,7 @@ mod tests {
|
||||
index_block_address: (usize::MAX - 4) as u64,
|
||||
};
|
||||
let buf = vec![0u8; 64];
|
||||
let r = read_extensible_array_chunks(&buf, &header, &[100], &[20], 8, 8, 8);
|
||||
let r = read_extensible_array_chunks(&buf, &header, &[100], None, &[20], 8, 8, 8);
|
||||
assert!(r.is_err());
|
||||
}
|
||||
|
||||
@@ -913,8 +867,16 @@ mod tests {
|
||||
let header = ExtensibleArrayHeader::parse(&file_data, aehd_offset, os, ls).unwrap();
|
||||
let ds_dims = vec![40u64]; // 2 chunks × 20 elements
|
||||
let chunk_dims = vec![20u32];
|
||||
let chunks =
|
||||
read_extensible_array_chunks(&file_data, &header, &ds_dims, &chunk_dims, 8, os, ls)
|
||||
let chunks = read_extensible_array_chunks(
|
||||
&file_data,
|
||||
&header,
|
||||
&ds_dims,
|
||||
None,
|
||||
&chunk_dims,
|
||||
8,
|
||||
os,
|
||||
ls,
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(chunks.len(), 2);
|
||||
@@ -1023,8 +985,16 @@ mod tests {
|
||||
let header = ExtensibleArrayHeader::parse(&file_data, aehd_offset, os, ls).unwrap();
|
||||
let ds_dims = vec![40u64];
|
||||
let chunk_dims = vec![10u32];
|
||||
let chunks =
|
||||
read_extensible_array_chunks(&file_data, &header, &ds_dims, &chunk_dims, 8, os, ls)
|
||||
let chunks = read_extensible_array_chunks(
|
||||
&file_data,
|
||||
&header,
|
||||
&ds_dims,
|
||||
None,
|
||||
&chunk_dims,
|
||||
8,
|
||||
os,
|
||||
ls,
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(chunks.len(), 4);
|
||||
@@ -1047,10 +1017,8 @@ mod tests {
|
||||
#[test]
|
||||
fn read_element_unallocated() {
|
||||
let data = vec![0xFFu8; 16];
|
||||
let num_chunks = vec![5u64];
|
||||
let chunk_dims = vec![10u32];
|
||||
let (info, consumed) =
|
||||
read_element(&data, 0, 0, 8, 8, 80, 0, &num_chunks, &chunk_dims).unwrap();
|
||||
let grid = ChunkGrid::fixed_array(&[50], None, &[10]).unwrap();
|
||||
let (info, consumed) = read_element(&data, 0, 0, 8, 8, 80, 0, &grid).unwrap();
|
||||
assert!(info.is_none());
|
||||
assert_eq!(consumed, 8);
|
||||
}
|
||||
@@ -1069,20 +1037,9 @@ mod tests {
|
||||
// Filter mask
|
||||
data[12..16].copy_from_slice(&0u32.to_le_bytes());
|
||||
|
||||
let num_chunks = vec![5u64];
|
||||
let chunk_dims = vec![10u32];
|
||||
let (info, consumed) = read_element(
|
||||
&data,
|
||||
0,
|
||||
1,
|
||||
elem_size as u8,
|
||||
os,
|
||||
80,
|
||||
2,
|
||||
&num_chunks,
|
||||
&chunk_dims,
|
||||
)
|
||||
.unwrap();
|
||||
let grid = ChunkGrid::fixed_array(&[50], None, &[10]).unwrap();
|
||||
let (info, consumed) =
|
||||
read_element(&data, 0, 1, elem_size as u8, os, 80, 2, &grid).unwrap();
|
||||
let ci = info.unwrap();
|
||||
assert_eq!(ci.address, 0x2000);
|
||||
assert_eq!(ci.chunk_size, 120);
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
//! link messages, contiguous datasets, inline and dense attributes.
|
||||
|
||||
#[cfg(not(feature = "std"))]
|
||||
use alloc::{string::String, string::ToString, vec, vec::Vec};
|
||||
use alloc::{format, string::String, string::ToString, vec, vec::Vec};
|
||||
|
||||
use crate::attribute::AttributeMessage;
|
||||
use crate::chunked_write::{
|
||||
@@ -19,7 +19,7 @@ use crate::metadata_index::{DatasetMetadata, MetadataBlock, MetadataIndex};
|
||||
use crate::object_header_writer::ObjectHeaderWriter;
|
||||
use crate::superblock::Superblock;
|
||||
use crate::type_builders::{
|
||||
DatasetBuilder, FillTime, FinishedGroup, GroupBuilder, build_attr_message,
|
||||
DatasetBuilder, FinishedGroup, GroupBuilder, build_attr_message, fill_value_message,
|
||||
};
|
||||
|
||||
// Re-export public types that moved to type_builders for API compatibility.
|
||||
@@ -33,6 +33,49 @@ pub(crate) const OFFSET_SIZE: u8 = 8;
|
||||
pub(crate) const LENGTH_SIZE: u8 = 8;
|
||||
const SUPERBLOCK_SIZE: usize = 48;
|
||||
|
||||
/// Largest raw data a compact dataset can hold: the layout message (version,
|
||||
/// class, 2-byte size, data) must fit an object header message, whose size
|
||||
/// field is 2 bytes. Bigger "compact" requests fall back to contiguous storage.
|
||||
const MAX_COMPACT_DATA_SIZE: usize = crate::object_header_writer::MAX_MESSAGE_SIZE - 4;
|
||||
|
||||
/// libhdf5's bounds on a file space page size (`H5F_FILE_SPACE_PAGE_SIZE_MIN`
|
||||
/// and `_MAX`).
|
||||
const MIN_FILE_SPACE_PAGE_SIZE: u32 = 512;
|
||||
const MAX_FILE_SPACE_PAGE_SIZE: u32 = 1024 * 1024 * 1024;
|
||||
|
||||
/// Superblock extension object header for a file using the paged file-space
|
||||
/// strategy: a single File Space Info message (0x0017), as libhdf5 writes it
|
||||
/// for `fs_strategy="page"` without persisted free space.
|
||||
fn build_paged_superblock_extension(page_size: u32) -> Result<Vec<u8>, FormatError> {
|
||||
let mut fsinfo = Vec::new();
|
||||
fsinfo.push(1); // version
|
||||
fsinfo.push(1); // strategy: H5F_FSPACE_STRATEGY_PAGE
|
||||
fsinfo.push(0); // persisting free space: no
|
||||
write_length(&mut fsinfo, 1, LENGTH_SIZE); // free-space section threshold
|
||||
write_length(&mut fsinfo, u64::from(page_size), LENGTH_SIZE);
|
||||
fsinfo.extend_from_slice(&0u16.to_le_bytes()); // page end metadata threshold
|
||||
write_undef_offset(&mut fsinfo, OFFSET_SIZE); // EOA before free-space info
|
||||
let mut w = ObjectHeaderWriter::new();
|
||||
// Flags as libhdf5 sets them: bit 2 (never share) and bit 4 (mark if
|
||||
// unknown). Not constant: libhdf5 rewrites the message when it closes a
|
||||
// file it opened for writing.
|
||||
w.add_message_with_flags(MessageType::Unknown(0x0017), fsinfo, 0x14);
|
||||
w.serialize()
|
||||
}
|
||||
|
||||
/// A group or dataset name must be one path component: not empty, not ".",
|
||||
/// and without '/'. `FileWriter` writes a root group plus one level of
|
||||
/// groups, and cannot create intermediate groups for a path.
|
||||
fn check_link_name(name: &str) -> Result<(), FormatError> {
|
||||
if name.is_empty() || name == "." || name.contains('/') {
|
||||
return Err(FormatError::SerializationError(format!(
|
||||
"invalid object name {name:?}: names must be a single path component \
|
||||
(FileWriter does not create nested groups)"
|
||||
)));
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Threshold for switching from compact (inline) to dense attribute storage.
|
||||
const DENSE_ATTR_THRESHOLD: usize = 8;
|
||||
|
||||
@@ -50,12 +93,12 @@ pub(crate) fn build_chunked_dataset_oh(
|
||||
pipeline_message: Option<&[u8]>,
|
||||
attrs: &[AttributeMessage],
|
||||
dense_blob: Option<&DenseAttrBlob>,
|
||||
fill_time: FillTime,
|
||||
) -> Vec<u8> {
|
||||
fill_message: &[u8],
|
||||
) -> Result<Vec<u8>, FormatError> {
|
||||
let mut w = ObjectHeaderWriter::new();
|
||||
w.add_message_with_flags(MessageType::Datatype, dt.serialize(), 0x01);
|
||||
w.add_message(MessageType::Dataspace, ds.serialize(LENGTH_SIZE));
|
||||
w.add_message_with_flags(MessageType::FillValue, vec![3, fill_time.to_byte()], 0x01);
|
||||
w.add_message_with_flags(MessageType::FillValue, fill_message.to_vec(), 0x01);
|
||||
w.add_message(MessageType::DataLayout, layout_message.to_vec());
|
||||
if let Some(pm) = pipeline_message {
|
||||
w.add_message(MessageType::FilterPipeline, pm.to_vec());
|
||||
@@ -77,12 +120,12 @@ pub(crate) fn build_dataset_oh(
|
||||
data_size: u64,
|
||||
attrs: &[AttributeMessage],
|
||||
dense_blob: Option<&DenseAttrBlob>,
|
||||
fill_time: FillTime,
|
||||
) -> Vec<u8> {
|
||||
fill_message: &[u8],
|
||||
) -> Result<Vec<u8>, FormatError> {
|
||||
let mut w = ObjectHeaderWriter::new();
|
||||
w.add_message_with_flags(MessageType::Datatype, dt.serialize(), 0x01);
|
||||
w.add_message(MessageType::Dataspace, ds.serialize(LENGTH_SIZE));
|
||||
w.add_message_with_flags(MessageType::FillValue, vec![3, fill_time.to_byte()], 0x01);
|
||||
w.add_message_with_flags(MessageType::FillValue, fill_message.to_vec(), 0x01);
|
||||
let mut dl = Vec::new();
|
||||
dl.push(4); // version
|
||||
dl.push(1); // class = contiguous
|
||||
@@ -112,12 +155,12 @@ pub(crate) fn build_compact_dataset_oh(
|
||||
data: &[u8],
|
||||
attrs: &[AttributeMessage],
|
||||
dense_blob: Option<&DenseAttrBlob>,
|
||||
fill_time: FillTime,
|
||||
) -> Vec<u8> {
|
||||
fill_message: &[u8],
|
||||
) -> Result<Vec<u8>, FormatError> {
|
||||
let mut w = ObjectHeaderWriter::new();
|
||||
w.add_message_with_flags(MessageType::Datatype, dt.serialize(), 0x01);
|
||||
w.add_message(MessageType::Dataspace, ds.serialize(LENGTH_SIZE));
|
||||
w.add_message_with_flags(MessageType::FillValue, vec![3, fill_time.to_byte()], 0x01);
|
||||
w.add_message_with_flags(MessageType::FillValue, fill_message.to_vec(), 0x01);
|
||||
// Compact layout message: version=4, class=0, u16 size, inline data
|
||||
let mut dl = Vec::new();
|
||||
dl.push(4); // version
|
||||
@@ -140,7 +183,7 @@ pub(crate) fn build_group_oh(
|
||||
dense_link_info: Option<&[u8]>,
|
||||
attrs: &[AttributeMessage],
|
||||
dense_blob: Option<&DenseAttrBlob>,
|
||||
) -> Vec<u8> {
|
||||
) -> Result<Vec<u8>, FormatError> {
|
||||
let mut w = ObjectHeaderWriter::new();
|
||||
if let Some(li) = dense_link_info {
|
||||
// Dense link storage: a LinkInfo pointing at the fractal heap + name
|
||||
@@ -902,12 +945,12 @@ pub(crate) fn build_vds_dataset_oh(
|
||||
global_heap_addr: u64,
|
||||
attrs: &[AttributeMessage],
|
||||
dense_blob: Option<&DenseAttrBlob>,
|
||||
fill_time: FillTime,
|
||||
) -> Vec<u8> {
|
||||
fill_message: &[u8],
|
||||
) -> Result<Vec<u8>, FormatError> {
|
||||
let mut w = ObjectHeaderWriter::new();
|
||||
w.add_message_with_flags(MessageType::Datatype, dt.serialize(), 0x01);
|
||||
w.add_message(MessageType::Dataspace, ds.serialize(LENGTH_SIZE));
|
||||
w.add_message_with_flags(MessageType::FillValue, vec![3, fill_time.to_byte()], 0x01);
|
||||
w.add_message_with_flags(MessageType::FillValue, fill_message.to_vec(), 0x01);
|
||||
// VDS layout message: version=4, class=3, global_heap_address(8), global_heap_index=1(4)
|
||||
let mut dl = Vec::new();
|
||||
dl.push(4u8); // version
|
||||
@@ -956,7 +999,9 @@ pub struct FileWriter {
|
||||
alignment_threshold: usize,
|
||||
/// Global alignment boundary in bytes (0 = disabled).
|
||||
alignment_bytes: usize,
|
||||
/// Page size for page-buffer mode. When set, a v4 superblock is written.
|
||||
/// File space page size. When set, the file uses libhdf5's paged
|
||||
/// file-space strategy (a File Space Info message in the superblock
|
||||
/// extension).
|
||||
page_size: Option<u32>,
|
||||
}
|
||||
|
||||
@@ -988,9 +1033,16 @@ impl FileWriter {
|
||||
self
|
||||
}
|
||||
|
||||
/// Enable page-buffer mode with the given page size. Writing this causes
|
||||
/// the file to be written with a v4 superblock (page_size field) instead
|
||||
/// of the default v3.
|
||||
/// Write the file with libhdf5's *paged* file-space strategy and the given
|
||||
/// page size, as `H5Pset_file_space_strategy(H5F_FSPACE_STRATEGY_PAGE)` +
|
||||
/// `H5Pset_file_space_page_size` (h5py: `fs_strategy="page"`,
|
||||
/// `fs_page_size=...`) do: a v3 superblock with an extension holding a
|
||||
/// File Space Info message, and the file padded to a whole number of
|
||||
/// pages. Readers with a page buffer can then fetch metadata page by page.
|
||||
///
|
||||
/// `page_size` must be between 512 bytes and 1 GiB (libhdf5's limits);
|
||||
/// [`Self::finish`] fails otherwise. This used to write a "version 4"
|
||||
/// superblock, which does not exist and no HDF5 library can open.
|
||||
pub fn with_page_size(&mut self, page_size: u32) -> &mut Self {
|
||||
self.page_size = Some(page_size);
|
||||
self
|
||||
@@ -1015,6 +1067,14 @@ impl FileWriter {
|
||||
|
||||
pub fn finish(self) -> Result<Vec<u8>, FormatError> {
|
||||
let page_size = self.page_size;
|
||||
if let Some(ps) = page_size
|
||||
&& !(MIN_FILE_SPACE_PAGE_SIZE..=MAX_FILE_SPACE_PAGE_SIZE).contains(&ps)
|
||||
{
|
||||
return Err(FormatError::SerializationError(format!(
|
||||
"file space page size {ps} is outside libhdf5's \
|
||||
{MIN_FILE_SPACE_PAGE_SIZE}..={MAX_FILE_SPACE_PAGE_SIZE} bytes"
|
||||
)));
|
||||
}
|
||||
struct DsFlat {
|
||||
name: String,
|
||||
dt: Datatype,
|
||||
@@ -1023,7 +1083,8 @@ impl FileWriter {
|
||||
attrs: Vec<AttributeMessage>,
|
||||
chunk_options: ChunkOptions,
|
||||
maxshape: Option<Vec<u64>>,
|
||||
fill_time: FillTime,
|
||||
/// Serialized Fill Value message.
|
||||
fill_message: Vec<u8>,
|
||||
compact: bool,
|
||||
alignment: usize,
|
||||
/// VDS source mappings (set for Virtual datasets).
|
||||
@@ -1073,6 +1134,7 @@ impl FileWriter {
|
||||
};
|
||||
attrs.extend(p.build_attrs(&raw));
|
||||
}
|
||||
let fill_message = fill_value_message(db.fill_time, db.fill_value.as_deref(), &dt)?;
|
||||
Ok(DsFlat {
|
||||
name: db.name,
|
||||
dt,
|
||||
@@ -1081,13 +1143,26 @@ impl FileWriter {
|
||||
attrs,
|
||||
chunk_options: db.chunk_options,
|
||||
maxshape: db.maxshape,
|
||||
fill_time: db.fill_time,
|
||||
fill_message,
|
||||
compact: db.compact,
|
||||
alignment: db.alignment,
|
||||
virtual_sources: db.virtual_sources,
|
||||
})
|
||||
};
|
||||
|
||||
// Every name becomes a single link in its parent group. The writer
|
||||
// has no nested groups, so a path like "a/b" would be stored as one
|
||||
// link literally named "a/b" — which no HDF5 reader can resolve.
|
||||
let root_names = self.root_datasets.iter().map(|d| d.name.as_str());
|
||||
let group_names = self.groups.iter().flat_map(|g| {
|
||||
core::iter::once(g.name.as_str())
|
||||
.chain(g.datasets.iter().map(|d| d.name.as_str()))
|
||||
.chain(g.external_links.iter().map(|l| l.0.as_str()))
|
||||
});
|
||||
for name in root_names.chain(group_names) {
|
||||
check_link_name(name)?;
|
||||
}
|
||||
|
||||
let mut all_ds: Vec<DsFlat> = Vec::new();
|
||||
let mut groups: Vec<GrpFlat> = Vec::new();
|
||||
let mut root_ds_indices: Vec<usize> = Vec::new();
|
||||
@@ -1120,17 +1195,35 @@ impl FileWriter {
|
||||
root_attrs.push(build_attr_message(n, v));
|
||||
}
|
||||
|
||||
// Every datatype must have an on-disk encoding before anything is laid
|
||||
// out: `Datatype::serialize` itself cannot report a failure.
|
||||
let group_attrs = groups.iter().flat_map(|g| &g.attrs);
|
||||
let ds_attrs = all_ds.iter().flat_map(|d| &d.attrs);
|
||||
for a in root_attrs.iter().chain(group_attrs).chain(ds_attrs) {
|
||||
a.datatype.check_encodable()?;
|
||||
}
|
||||
for d in &all_ds {
|
||||
d.dt.check_encodable()?;
|
||||
}
|
||||
|
||||
let is_vds: Vec<bool> = all_ds.iter().map(|d| d.virtual_sources.is_some()).collect();
|
||||
let is_chunked: Vec<bool> = all_ds
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(i, d)| !is_vds[i] && (d.chunk_options.is_chunked() || d.maxshape.is_some()))
|
||||
.map(|(i, d)| {
|
||||
// Only a dataset that can grow needs chunks; a maxshape equal
|
||||
// to the shape is as fixed as no maxshape at all.
|
||||
let resizable = d.maxshape.as_ref().is_some_and(|m| *m != d.ds.dimensions);
|
||||
!is_vds[i] && (d.chunk_options.is_chunked() || resizable)
|
||||
})
|
||||
.collect();
|
||||
// Determine which datasets use compact storage
|
||||
let is_compact: Vec<bool> = all_ds
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(i, d)| !is_vds[i] && !is_chunked[i] && d.compact && d.raw.len() <= 65535)
|
||||
.map(|(i, d)| {
|
||||
!is_vds[i] && !is_chunked[i] && d.compact && d.raw.len() <= MAX_COMPACT_DATA_SIZE
|
||||
})
|
||||
.collect();
|
||||
let root_dense = root_attrs.len() > DENSE_ATTR_THRESHOLD;
|
||||
let group_dense: Vec<bool> = groups
|
||||
@@ -1169,9 +1262,9 @@ impl FileWriter {
|
||||
}
|
||||
let attr_blob = group_dense[gi].then(|| build_dense_attrs(&g.attrs, 0));
|
||||
let dl = group_links_dense[gi].then_some(dummy_link_info.as_slice());
|
||||
build_group_oh(&dummy_links, dl, &g.attrs, attr_blob.as_ref()).len()
|
||||
build_group_oh(&dummy_links, dl, &g.attrs, attr_blob.as_ref()).map(|oh| oh.len())
|
||||
})
|
||||
.collect();
|
||||
.collect::<Result<_, _>>()?;
|
||||
|
||||
let root_dummy_links: Vec<LinkMessage> = {
|
||||
let mut links = Vec::new();
|
||||
@@ -1186,7 +1279,7 @@ impl FileWriter {
|
||||
let root_oh_size = {
|
||||
let attr_blob = root_dense.then(|| build_dense_attrs(&root_attrs, 0));
|
||||
let dl = root_links_dense.then_some(dummy_link_info.as_slice());
|
||||
build_group_oh(&root_dummy_links, dl, &root_attrs, attr_blob.as_ref()).len()
|
||||
build_group_oh(&root_dummy_links, dl, &root_attrs, attr_blob.as_ref())?.len()
|
||||
};
|
||||
|
||||
struct DataBlob {
|
||||
@@ -1214,8 +1307,8 @@ impl FileWriter {
|
||||
0, // dummy address
|
||||
&d.attrs,
|
||||
dense_blob.as_ref(),
|
||||
d.fill_time,
|
||||
);
|
||||
&d.fill_message,
|
||||
)?;
|
||||
// Global heap blob size is address-independent; compute it now
|
||||
// so pass 2 can place it correctly.
|
||||
let vds_mappings = d.virtual_sources.as_deref().unwrap_or(&[]);
|
||||
@@ -1244,7 +1337,7 @@ impl FileWriter {
|
||||
&pre,
|
||||
dummy_cursor,
|
||||
d.maxshape.as_deref(),
|
||||
);
|
||||
)?;
|
||||
dummy_cursor += result.data_bytes.len() as u64;
|
||||
let dense_blob = if ds_dense[i] {
|
||||
Some(build_dense_attrs(&d.attrs, 0))
|
||||
@@ -1258,8 +1351,8 @@ impl FileWriter {
|
||||
result.pipeline_message.as_deref(),
|
||||
&d.attrs,
|
||||
dense_blob.as_ref(),
|
||||
d.fill_time,
|
||||
);
|
||||
&d.fill_message,
|
||||
)?;
|
||||
dummy_blobs.push(DataBlob {
|
||||
data: result.data_bytes,
|
||||
oh_bytes: oh,
|
||||
@@ -1277,8 +1370,8 @@ impl FileWriter {
|
||||
&d.raw,
|
||||
&d.attrs,
|
||||
dense_blob.as_ref(),
|
||||
d.fill_time,
|
||||
);
|
||||
&d.fill_message,
|
||||
)?;
|
||||
dummy_blobs.push(DataBlob {
|
||||
data: vec![],
|
||||
oh_bytes: oh,
|
||||
@@ -1297,8 +1390,8 @@ impl FileWriter {
|
||||
d.raw.len() as u64,
|
||||
&d.attrs,
|
||||
dense_blob.as_ref(),
|
||||
d.fill_time,
|
||||
);
|
||||
&d.fill_message,
|
||||
)?;
|
||||
dummy_blobs.push(DataBlob {
|
||||
data: d.raw.clone(),
|
||||
oh_bytes: oh,
|
||||
@@ -1310,12 +1403,12 @@ impl FileWriter {
|
||||
let actual_ds_oh_sizes: Vec<usize> = dummy_blobs.iter().map(|b| b.oh_bytes.len()).collect();
|
||||
|
||||
// Pass 2: compute real addresses
|
||||
// v4 superblocks add a 4-byte page_size field before the checksum.
|
||||
let superblock_size = if page_size.is_some() {
|
||||
SUPERBLOCK_SIZE + 4
|
||||
} else {
|
||||
SUPERBLOCK_SIZE
|
||||
};
|
||||
// A paged file carries its File Space Info in a superblock extension
|
||||
// object header, placed right after the superblock.
|
||||
let sb_ext = page_size
|
||||
.map(build_paged_superblock_extension)
|
||||
.transpose()?;
|
||||
let superblock_size = SUPERBLOCK_SIZE + sb_ext.as_ref().map_or(0, Vec::len);
|
||||
let root_group_addr = superblock_size as u64;
|
||||
let mut cursor2 = superblock_size + root_oh_size;
|
||||
|
||||
@@ -1406,8 +1499,8 @@ impl FileWriter {
|
||||
heap_addr,
|
||||
&d.attrs,
|
||||
ds_dense_blobs[i].as_ref(),
|
||||
d.fill_time,
|
||||
);
|
||||
&d.fill_message,
|
||||
)?;
|
||||
ds_blobs2.push(DataBlob {
|
||||
data: gcol_bytes.clone(),
|
||||
oh_bytes: oh,
|
||||
@@ -1424,7 +1517,7 @@ impl FileWriter {
|
||||
.expect("chunked dataset missing precompressed cache"),
|
||||
base_address,
|
||||
d.maxshape.as_deref(),
|
||||
);
|
||||
)?;
|
||||
cursor2 += result.data_bytes.len();
|
||||
let oh = build_chunked_dataset_oh(
|
||||
&d.dt,
|
||||
@@ -1433,8 +1526,8 @@ impl FileWriter {
|
||||
result.pipeline_message.as_deref(),
|
||||
&d.attrs,
|
||||
ds_dense_blobs[i].as_ref(),
|
||||
d.fill_time,
|
||||
);
|
||||
&d.fill_message,
|
||||
)?;
|
||||
ds_blobs2.push(DataBlob {
|
||||
data: result.data_bytes,
|
||||
oh_bytes: oh,
|
||||
@@ -1448,8 +1541,8 @@ impl FileWriter {
|
||||
&d.raw,
|
||||
&d.attrs,
|
||||
ds_dense_blobs[i].as_ref(),
|
||||
d.fill_time,
|
||||
);
|
||||
&d.fill_message,
|
||||
)?;
|
||||
ds_blobs2.push(DataBlob {
|
||||
data: vec![],
|
||||
oh_bytes: oh,
|
||||
@@ -1473,8 +1566,8 @@ impl FileWriter {
|
||||
d.raw.len() as u64,
|
||||
&d.attrs,
|
||||
ds_dense_blobs[i].as_ref(),
|
||||
d.fill_time,
|
||||
);
|
||||
&d.fill_message,
|
||||
)?;
|
||||
let mut data = vec![0u8; padding];
|
||||
data.extend_from_slice(&d.raw);
|
||||
cursor2 += d.raw.len();
|
||||
@@ -1489,11 +1582,16 @@ impl FileWriter {
|
||||
let actual_ds_oh_sizes2: Vec<usize> = ds_blobs2.iter().map(|b| b.oh_bytes.len()).collect();
|
||||
debug_assert_eq!(actual_ds_oh_sizes, actual_ds_oh_sizes2);
|
||||
|
||||
// libhdf5 ends a paged file on a page boundary.
|
||||
let data_end = cursor2;
|
||||
if let Some(ps) = page_size {
|
||||
cursor2 = cursor2.next_multiple_of(ps as usize);
|
||||
}
|
||||
let eof_addr2 = cursor2 as u64;
|
||||
let mut buf = Vec::with_capacity(cursor2);
|
||||
|
||||
let sb = Superblock {
|
||||
version: if page_size.is_some() { 4 } else { 3 },
|
||||
version: 3,
|
||||
offset_size: OFFSET_SIZE,
|
||||
length_size: LENGTH_SIZE,
|
||||
base_address: 0,
|
||||
@@ -1505,11 +1603,18 @@ impl FileWriter {
|
||||
free_space_address: None,
|
||||
driver_info_address: None,
|
||||
consistency_flags: 0,
|
||||
superblock_extension_address: Some(u64::MAX),
|
||||
superblock_extension_address: Some(if sb_ext.is_some() {
|
||||
SUPERBLOCK_SIZE as u64
|
||||
} else {
|
||||
u64::MAX
|
||||
}),
|
||||
checksum: None,
|
||||
page_size,
|
||||
page_size: None,
|
||||
};
|
||||
buf.extend_from_slice(&sb.serialize());
|
||||
if let Some(ref ext) = sb_ext {
|
||||
buf.extend_from_slice(ext);
|
||||
}
|
||||
|
||||
// Root group OH
|
||||
let mut root_links: Vec<LinkMessage> = Vec::new();
|
||||
@@ -1530,7 +1635,7 @@ impl FileWriter {
|
||||
root_dl,
|
||||
&root_attrs,
|
||||
root_dense_blob.as_ref(),
|
||||
));
|
||||
)?);
|
||||
if let Some(ref b) = root_link_blob {
|
||||
buf.extend_from_slice(&b.blob);
|
||||
}
|
||||
@@ -1555,7 +1660,7 @@ impl FileWriter {
|
||||
dl,
|
||||
&g.attrs,
|
||||
group_dense_blobs[gi].as_ref(),
|
||||
));
|
||||
)?);
|
||||
if let Some(ref b) = link_blob {
|
||||
buf.extend_from_slice(&b.blob);
|
||||
}
|
||||
@@ -1577,7 +1682,8 @@ impl FileWriter {
|
||||
buf.extend_from_slice(&blob.data);
|
||||
}
|
||||
|
||||
debug_assert_eq!(buf.len(), cursor2);
|
||||
debug_assert_eq!(buf.len(), data_end);
|
||||
buf.resize(cursor2, 0);
|
||||
Ok(buf)
|
||||
}
|
||||
}
|
||||
@@ -1841,6 +1947,12 @@ mod tests {
|
||||
root_block_address: 0,
|
||||
current_rows_in_root_indirect_block: 0,
|
||||
managed_objects_count: 0,
|
||||
huge_btree_address: u64::MAX,
|
||||
filter_pipeline: None,
|
||||
root_direct_block_filtered_size: 0,
|
||||
root_direct_block_filter_mask: 0,
|
||||
offset_size: 8,
|
||||
length_size: 8,
|
||||
};
|
||||
let (off, len) = fh.decode_managed_id(&id).unwrap();
|
||||
assert_eq!(off, 100);
|
||||
@@ -2151,7 +2263,8 @@ mod tests {
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn file_writer_v4_superblock() {
|
||||
fn file_writer_paged_file_uses_v3_superblock_and_fsinfo_extension() {
|
||||
// This used to write superblock "version 4", which does not exist.
|
||||
let mut fw = FileWriter::new();
|
||||
fw.with_page_size(4096);
|
||||
fw.create_dataset("data").with_f64_data(&[1.0, 2.0]);
|
||||
@@ -2159,8 +2272,31 @@ mod tests {
|
||||
|
||||
let sig = signature::find_signature(&bytes).unwrap();
|
||||
let sb = Superblock::parse(&bytes, sig).unwrap();
|
||||
assert_eq!(sb.version, 4, "expected superblock v4");
|
||||
assert_eq!(sb.page_size, Some(4096));
|
||||
assert_eq!(sb.version, 3);
|
||||
assert_eq!(sb.superblock_extension_address, Some(48));
|
||||
assert_eq!(bytes.len() % 4096, 0);
|
||||
assert_eq!(sb.eof_address, bytes.len() as u64);
|
||||
let ext = ObjectHeader::parse(&bytes, 48, 8, 8).unwrap();
|
||||
let fsinfo = &ext.messages[0];
|
||||
assert_eq!(fsinfo.msg_type, MessageType::Unknown(0x0017));
|
||||
// Byte-for-byte what HDF5 2.0 writes for fs_strategy="page",
|
||||
// fs_page_size=4096.
|
||||
let mut expected = vec![1u8, 1, 0, 1, 0, 0, 0, 0, 0, 0, 0];
|
||||
expected.extend_from_slice(&4096u64.to_le_bytes());
|
||||
expected.extend_from_slice(&[0, 0]);
|
||||
expected.extend_from_slice(&[0xff; 8]);
|
||||
assert_eq!(fsinfo.data, expected);
|
||||
assert_eq!(fsinfo.flags, 0x14);
|
||||
assert_eq!(read_dataset_f64(&bytes, "data"), vec![1.0, 2.0]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn file_writer_rejects_page_sizes_libhdf5_would() {
|
||||
for ps in [0u32, 511, MAX_FILE_SPACE_PAGE_SIZE + 1] {
|
||||
let mut fw = FileWriter::new();
|
||||
fw.with_page_size(ps);
|
||||
assert!(fw.finish().is_err(), "page size {ps}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
|
||||
@@ -98,15 +98,50 @@ pub fn parse_fill_value(msg: &HeaderMessage) -> Result<Option<Vec<u8>>, FormatEr
|
||||
|
||||
/// The fill value that applies to a dataset given its header messages. The new
|
||||
/// message wins over the old one when both are present.
|
||||
///
|
||||
/// A *shared* fill value message holds only a reference to the real message,
|
||||
/// which cannot be followed without the file: this returns
|
||||
/// [`FormatError::UnresolvedSharedMessage`] for one (it used to answer "zeros").
|
||||
/// Use [`dataset_fill_value_in`] when the file bytes are at hand.
|
||||
pub fn dataset_fill_value(messages: &[HeaderMessage]) -> Result<Option<Vec<u8>>, FormatError> {
|
||||
fill_value_from(messages, |_| Err(FormatError::UnresolvedSharedMessage))
|
||||
}
|
||||
|
||||
/// [`dataset_fill_value`] for a dataset in `file_data`, following a shared
|
||||
/// fill value message to where it lives: another object header, or the
|
||||
/// file's shared-message (SOHM) heap, as libhdf5 writes it when the file has
|
||||
/// a SOHM index for fill values.
|
||||
pub fn dataset_fill_value_in(
|
||||
file_data: &[u8],
|
||||
messages: &[HeaderMessage],
|
||||
offset_size: u8,
|
||||
length_size: u8,
|
||||
) -> Result<Option<Vec<u8>>, FormatError> {
|
||||
fill_value_from(messages, |msg| {
|
||||
crate::shared_message::message_data_with_sohm(file_data, msg, offset_size, length_size)
|
||||
.map(|data| data.into_owned())
|
||||
})
|
||||
}
|
||||
|
||||
fn fill_value_from(
|
||||
messages: &[HeaderMessage],
|
||||
resolve_shared: impl Fn(&HeaderMessage) -> Result<Vec<u8>, FormatError>,
|
||||
) -> Result<Option<Vec<u8>>, FormatError> {
|
||||
for wanted in [MessageType::FillValue, MessageType::FillValueOld] {
|
||||
if let Some(msg) = messages.iter().find(|m| m.msg_type == wanted) {
|
||||
if crate::shared_message::is_shared(msg.flags) {
|
||||
// A shared fill value is legal but vanishingly rare; treat it
|
||||
// as the default rather than misparsing the reference.
|
||||
return Ok(None);
|
||||
}
|
||||
if let Some(value) = parse_fill_value(msg)? {
|
||||
let value = if crate::shared_message::is_shared(msg.flags) {
|
||||
let data = resolve_shared(msg)?;
|
||||
parse_fill_value(&HeaderMessage {
|
||||
msg_type: msg.msg_type,
|
||||
size: data.len(),
|
||||
flags: msg.flags & !0x02,
|
||||
creation_order: msg.creation_order,
|
||||
data,
|
||||
})?
|
||||
} else {
|
||||
parse_fill_value(msg)?
|
||||
};
|
||||
if let Some(value) = value {
|
||||
return Ok(Some(value));
|
||||
}
|
||||
}
|
||||
@@ -174,7 +209,7 @@ pub fn read_full_with_fill<E: From<FormatError>>(
|
||||
{
|
||||
return Err(FormatError::ExternalDataFilesUnsupported.into());
|
||||
}
|
||||
let fill = dataset_fill_value(messages)?;
|
||||
let fill = dataset_fill_value_in(file_data, messages, offset_size, length_size)?;
|
||||
if !has_storage(layout) {
|
||||
return Ok(filled_dataset(dataspace, elem_size, fill.as_deref())?);
|
||||
}
|
||||
|
||||
@@ -19,8 +19,23 @@ pub const FILTER_SCALEOFFSET: u16 = 6;
|
||||
pub const FILTER_LZ4: u16 = 32004;
|
||||
/// Zstandard compression.
|
||||
pub const FILTER_ZSTD: u16 = 32015;
|
||||
/// Pcodec lossless numerical codec (clawhdf5 internal; not yet HDF5-registered).
|
||||
pub const FILTER_PCODEC: u16 = 32023;
|
||||
/// Pcodec lossless numerical codec — a **private, unregistered** clawhdf5
|
||||
/// filter. Pcodec has no ID in the HDF Group's filter registry (checked
|
||||
/// 2026-09-25, `hdf5_plugins/docs/RegisteredFilterPlugins.md`), so it uses an
|
||||
/// ID from the registry's testing/private range (256–511). No libhdf5 plugin
|
||||
/// decodes it: h5py/libhdf5 report the filter as unavailable. Only clawhdf5
|
||||
/// (with the `pcodec` feature) reads these datasets.
|
||||
pub const FILTER_PCODEC: u16 = 480;
|
||||
/// Filter name written with [`FILTER_PCODEC`].
|
||||
pub const FILTER_PCODEC_NAME: &str = "pcodec (clawhdf5 private)";
|
||||
/// The ID clawhdf5 up to 2.7.0 wrote pcodec under. It is registered to
|
||||
/// Granular BitRound (GBR), whose decode is a pass-through, so libhdf5 with
|
||||
/// that plugin would have returned the compressed bytes as data. Read as
|
||||
/// pcodec only when the filter is named exactly [`FILTER_PCODEC_LEGACY_NAME`],
|
||||
/// the name those versions wrote; never written.
|
||||
pub const FILTER_PCODEC_LEGACY: u16 = 32023;
|
||||
/// The filter name clawhdf5 up to 2.7.0 wrote with [`FILTER_PCODEC_LEGACY`].
|
||||
pub const FILTER_PCODEC_LEGACY_NAME: &str = "pcodec";
|
||||
|
||||
/// Description of a single filter in a pipeline.
|
||||
#[derive(Debug, Clone, PartialEq)]
|
||||
|
||||
@@ -8,8 +8,9 @@ use alloc::{boxed::Box, vec, vec::Vec};
|
||||
|
||||
use crate::error::FormatError;
|
||||
use crate::filter_pipeline::{
|
||||
FILTER_DEFLATE, FILTER_FLETCHER32, FILTER_LZ4, FILTER_NBIT, FILTER_PCODEC, FILTER_SCALEOFFSET,
|
||||
FILTER_SHUFFLE, FILTER_SZIP, FILTER_ZSTD, FilterPipeline,
|
||||
FILTER_DEFLATE, FILTER_FLETCHER32, FILTER_LZ4, FILTER_NBIT, FILTER_PCODEC,
|
||||
FILTER_PCODEC_LEGACY, FILTER_PCODEC_LEGACY_NAME, FILTER_SCALEOFFSET, FILTER_SHUFFLE,
|
||||
FILTER_SZIP, FILTER_ZSTD, FilterPipeline,
|
||||
};
|
||||
|
||||
/// Absolute ceiling on a single decompressed chunk's output size, used only
|
||||
@@ -19,33 +20,104 @@ pub(crate) const MAX_DECOMPRESS_SIZE: usize = 256 * 1024 * 1024;
|
||||
|
||||
/// Apply a filter pipeline to decompress a chunk.
|
||||
/// Filters are applied in REVERSE order for decompression.
|
||||
///
|
||||
/// Equivalent to [`decompress_chunk_masked`] with a filter mask of 0 (every
|
||||
/// filter was applied when the chunk was written).
|
||||
pub fn decompress_chunk(
|
||||
compressed: &[u8],
|
||||
pipeline: &FilterPipeline,
|
||||
chunk_size: usize,
|
||||
element_size: u32,
|
||||
) -> Result<Vec<u8>, FormatError> {
|
||||
let mut data = compressed.to_vec();
|
||||
decompress_chunk_masked(compressed, pipeline, chunk_size, element_size, 0)
|
||||
}
|
||||
|
||||
for filter in pipeline.filters.iter().rev() {
|
||||
/// Upper bound on the output of filter `filter_id` applied (in the write
|
||||
/// direction) to `input` bytes. 0 means "unknown" and stays unknown.
|
||||
///
|
||||
/// Shuffle preserves the size and Fletcher32 appends a 4-byte checksum. Any
|
||||
/// other filter is a codec whose output can exceed its input on
|
||||
/// incompressible data (deflate's stored blocks, LZ4's and zstd's literal
|
||||
/// runs, codec headers); `n + n/8 + 64` covers every supported codec's worst
|
||||
/// case while still bounding a decompression bomb to a small multiple of the
|
||||
/// chunk.
|
||||
fn filter_output_bound(filter_id: u16, input: usize) -> usize {
|
||||
if input == 0 {
|
||||
return 0;
|
||||
}
|
||||
match filter_id {
|
||||
FILTER_SHUFFLE => input,
|
||||
FILTER_FLETCHER32 => input.saturating_add(4),
|
||||
_ => input.saturating_add(input / 8).saturating_add(64),
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether bit `index` of a chunk's filter mask says filter `index` was
|
||||
/// skipped when the chunk was written.
|
||||
fn filter_skipped(filter_mask: u32, index: usize) -> bool {
|
||||
index < 32 && filter_mask & (1u32 << index) != 0
|
||||
}
|
||||
|
||||
/// Whether `filter_mask` says none of `pipeline`'s filters were applied, so
|
||||
/// the stored bytes are the chunk itself.
|
||||
pub fn all_filters_skipped(pipeline: &FilterPipeline, filter_mask: u32) -> bool {
|
||||
(0..pipeline.filters.len()).all(|i| filter_skipped(filter_mask, i))
|
||||
}
|
||||
|
||||
/// Decompress a chunk whose filter mask is `filter_mask`: bit *i* set means
|
||||
/// filter *i* of the pipeline was not applied when the chunk was written (an
|
||||
/// optional filter that declined, or a direct chunk write), so only that
|
||||
/// filter is skipped here; the others are still undone, in reverse order.
|
||||
///
|
||||
/// `chunk_size` is the chunk's decoded size (0 if unknown). Each stage's
|
||||
/// output is capped at what the filters before it (in write order) can have
|
||||
/// produced from `chunk_size` bytes — e.g. a Fletcher32 checksum placed
|
||||
/// before deflate (NetCDF-4's ordering) makes deflate's output 4 bytes
|
||||
/// larger than the chunk — so the decompression-bomb limit stays tight
|
||||
/// without rejecting valid pipelines.
|
||||
pub fn decompress_chunk_masked(
|
||||
compressed: &[u8],
|
||||
pipeline: &FilterPipeline,
|
||||
chunk_size: usize,
|
||||
element_size: u32,
|
||||
filter_mask: u32,
|
||||
) -> Result<Vec<u8>, FormatError> {
|
||||
// bounds[i]: the most bytes that entered filter i on the write side, and
|
||||
// so the most that undoing filter i may produce.
|
||||
let mut bounds = Vec::with_capacity(pipeline.filters.len());
|
||||
let mut size = chunk_size;
|
||||
for (i, filter) in pipeline.filters.iter().enumerate() {
|
||||
bounds.push(size);
|
||||
if !filter_skipped(filter_mask, i) {
|
||||
size = filter_output_bound(filter.filter_id, size);
|
||||
}
|
||||
}
|
||||
|
||||
let mut data = compressed.to_vec();
|
||||
for (i, filter) in pipeline.filters.iter().enumerate().rev() {
|
||||
if filter_skipped(filter_mask, i) {
|
||||
continue;
|
||||
}
|
||||
let bound = bounds[i];
|
||||
data = match filter.filter_id {
|
||||
FILTER_SHUFFLE => shuffle_decompress(&data, element_size as usize)?,
|
||||
// `chunk_size` is the expected decompressed size (shuffle/fletcher32
|
||||
// are size-preserving, so it bounds these too); pass it so these
|
||||
// decoders can't be forced into unbounded allocation by a hostile
|
||||
// or corrupted compressed payload.
|
||||
FILTER_DEFLATE => deflate_decompress(&data, chunk_size)?,
|
||||
FILTER_LZ4 => lz4_decompress(&data, chunk_size)?,
|
||||
FILTER_ZSTD => zstd_decompress(&data, chunk_size)?,
|
||||
// `bound` caps the decoded size so these decoders can't be forced
|
||||
// into unbounded allocation by a hostile or corrupted payload.
|
||||
FILTER_DEFLATE => deflate_decompress(&data, bound)?,
|
||||
FILTER_LZ4 => lz4_decompress(&data, bound)?,
|
||||
FILTER_ZSTD => zstd_decompress(&data, bound)?,
|
||||
FILTER_FLETCHER32 => fletcher32_verify(&data)?,
|
||||
FILTER_PCODEC => pcodec_decompress(&data, element_size as usize, chunk_size)?,
|
||||
// `chunk_size` is the expected decompressed size; pass it so these
|
||||
// decoders can reject an element count that would over-allocate.
|
||||
FILTER_SCALEOFFSET => scaleoffset_decompress(&data, &filter.client_data, chunk_size)?,
|
||||
FILTER_NBIT => nbit_decompress(&data, &filter.client_data, chunk_size)?,
|
||||
FILTER_SZIP => {
|
||||
crate::filters_szip::szip_decompress(&data, &filter.client_data, chunk_size)?
|
||||
FILTER_PCODEC => pcodec_decompress(&data, element_size as usize, bound)?,
|
||||
// Pcodec chunks written by clawhdf5 <= 2.7.0 under the ID registered
|
||||
// to Granular BitRound; recognised by the name those versions wrote.
|
||||
FILTER_PCODEC_LEGACY if filter.name.as_deref() == Some(FILTER_PCODEC_LEGACY_NAME) => {
|
||||
pcodec_decompress(&data, element_size as usize, bound)?
|
||||
}
|
||||
// These decoders also reject an element count that would
|
||||
// over-allocate past `bound`.
|
||||
FILTER_SCALEOFFSET => scaleoffset_decompress(&data, &filter.client_data, bound)?,
|
||||
FILTER_NBIT => nbit_decompress(&data, &filter.client_data, bound)?,
|
||||
FILTER_SZIP => crate::filters_szip::szip_decompress(&data, &filter.client_data, bound)?,
|
||||
other => return Err(FormatError::UnsupportedFilter(other)),
|
||||
};
|
||||
}
|
||||
@@ -69,7 +141,7 @@ pub fn compress_chunk(
|
||||
let level = filter.client_data.first().copied().unwrap_or(6);
|
||||
deflate_compress(&result, level)?
|
||||
}
|
||||
FILTER_LZ4 => lz4_compress(&result)?,
|
||||
FILTER_LZ4 => lz4_compress(&result, &filter.client_data)?,
|
||||
FILTER_ZSTD => {
|
||||
let level = filter.client_data.first().copied().unwrap_or(3);
|
||||
zstd_compress(&result, level)?
|
||||
@@ -240,8 +312,20 @@ fn scaleoffset_decompress(
|
||||
fill_value
|
||||
} else if is_escale {
|
||||
minval + code as f64 * powi_f64(2.0, scale_factor)
|
||||
} else if elem_size == 4 {
|
||||
// H5Z_scaleoffset_modify_3/4 for `float`: the code is
|
||||
// read as an `int` and everything is single precision,
|
||||
// `(float)code / powf(10, D) + min`. Doing it in f64 and
|
||||
// rounding once at the end is off by 1 ULP at times.
|
||||
let d = if scale_factor >= 0 {
|
||||
powi_f64(10.0, scale_factor) as f32
|
||||
} else {
|
||||
minval + code as f64 / powi_f64(10.0, scale_factor)
|
||||
1.0 / powi_f64(10.0, -scale_factor) as f32
|
||||
};
|
||||
((code as u32 as i32) as f32 / d + minval as f32) as f64
|
||||
} else {
|
||||
// ... and for `double`: `(double)(long)code / pow(10, D) + min`.
|
||||
(code as i64) as f64 / powi_f64(10.0, scale_factor) + minval
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
@@ -379,6 +463,9 @@ enum NbitNode {
|
||||
count: usize,
|
||||
base_size: usize,
|
||||
},
|
||||
/// `H5Z_NBIT_NOOPTYPE`: a field N-Bit does not reduce (enum, string,
|
||||
/// opaque, ...), stored as all `size` bytes, 8 bits each.
|
||||
Noop { size: usize },
|
||||
}
|
||||
|
||||
impl NbitNode {
|
||||
@@ -389,6 +476,7 @@ impl NbitNode {
|
||||
NbitNode::Array {
|
||||
count, base_size, ..
|
||||
} => count * base_size,
|
||||
NbitNode::Noop { size } => *size,
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -408,6 +496,7 @@ fn parse_nbit_node(cd: &[u32], idx: &mut usize, depth: u32) -> Result<NbitNode,
|
||||
const ATOMIC: u32 = 1;
|
||||
const ARRAY: u32 = 2;
|
||||
const COMPOUND: u32 = 3;
|
||||
const NOOPTYPE: u32 = 4;
|
||||
if depth > NBIT_MAX_DEPTH {
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"nbit: type tree nested too deeply".into(),
|
||||
@@ -483,8 +572,17 @@ fn parse_nbit_node(cd: &[u32], idx: &mut usize, depth: u32) -> Result<NbitNode,
|
||||
members,
|
||||
})
|
||||
}
|
||||
// Class 4 is H5Z_NBIT_NOOPTYPE (members copied verbatim) — not seen in
|
||||
// practice for the supported leaf types and left unsupported.
|
||||
NOOPTYPE => {
|
||||
// class, size
|
||||
let size = nbit_cd(cd, *idx + 1)? as usize;
|
||||
*idx += 2;
|
||||
if size == 0 {
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"nbit: invalid no-op type size".into(),
|
||||
));
|
||||
}
|
||||
Ok(NbitNode::Noop { size })
|
||||
}
|
||||
_ => Err(FormatError::UnsupportedFilter(FILTER_NBIT)),
|
||||
}
|
||||
}
|
||||
@@ -549,6 +647,11 @@ fn decode_nbit_node(
|
||||
decode_nbit_node(bnode, br, elem, base + i * base_size)?;
|
||||
}
|
||||
}
|
||||
NbitNode::Noop { size } => {
|
||||
for slot in &mut elem[base..base + size] {
|
||||
*slot = br.read(8)? as u8;
|
||||
}
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
@@ -561,17 +664,28 @@ fn decode_nbit_node(
|
||||
/// type tree — atomic (`[1, size, order, precision, offset]`), array
|
||||
/// (`[2, total_size, <base>]`) and compound
|
||||
/// (`[3, total_size, nmembers, (offset, <node>)*]`) — preceded by
|
||||
/// `[nparms, flag, nelmts]`. Decompression walks the tree once per element,
|
||||
/// `[nparms, need_not_compress, nelmts]`; when `need_not_compress` is set
|
||||
/// (every field already uses its full width, e.g. a 32-bit int of precision
|
||||
/// 32) libhdf5 stores the data unchanged and so do we. Fields N-Bit cannot
|
||||
/// reduce (enums, strings, ...) are no-op nodes (`[4, size]`) copied whole.
|
||||
/// Decompression walks the tree once per element,
|
||||
/// placing each field's bits at its byte/bit offset in a zero-filled element
|
||||
/// (HDF5's canonical reduced-precision layout). Sign-extension of reduced
|
||||
/// precision signed integers is the datatype reader's job. Atomic floats are
|
||||
/// encoded as full-precision atomics and handled transparently.
|
||||
/// precision signed integers is the datatype reader's job, and so is
|
||||
/// converting a reduced-precision float (its own sign/exponent/mantissa
|
||||
/// layout, e.g. `le_data.h5`'s 20-bit `Nbit_float_data_*`) to IEEE: the
|
||||
/// filter's output is the file type's bytes, as libhdf5's is before type
|
||||
/// conversion.
|
||||
fn nbit_decompress(data: &[u8], cd: &[u32], expected_bytes: usize) -> Result<Vec<u8>, FormatError> {
|
||||
if cd.len() < 4 {
|
||||
if cd.len() < 3 {
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"nbit: missing filter client data".into(),
|
||||
));
|
||||
}
|
||||
// H5Z__filter_nbit: `if (cd_values[1]) HGOTO_DONE(*buf_size)`.
|
||||
if cd[1] != 0 {
|
||||
return Ok(data.to_vec());
|
||||
}
|
||||
let nelmts = cd[2] as usize;
|
||||
let mut idx = 3;
|
||||
let root = parse_nbit_node(cd, &mut idx, 0)?;
|
||||
@@ -813,12 +927,32 @@ fn deflate_compress(_data: &[u8], _level: u32) -> Result<Vec<u8>, FormatError> {
|
||||
Err(FormatError::UnsupportedFilter(FILTER_DEFLATE))
|
||||
}
|
||||
|
||||
/// Decompress LZ4 data. Format: 4 bytes LE original size + LZ4 block data.
|
||||
/// Default LZ4 block size of the registered HDF5 LZ4 filter (`H5Zlz4.c`,
|
||||
/// `DEFAULT_BLOCK_SIZE`): 1 GiB, so an HDF5 chunk is normally one block.
|
||||
#[cfg(feature = "lz4")]
|
||||
const LZ4_DEFAULT_BLOCK_SIZE: usize = 1 << 30;
|
||||
|
||||
/// Decompress an LZ4 (filter 32004) chunk.
|
||||
///
|
||||
/// The 4-byte "original size" header is part of the attacker-controlled
|
||||
/// compressed payload itself, so it is bounded against `expected_bytes` (the
|
||||
/// pipeline's declared chunk size) before being used to size the output
|
||||
/// allocation — otherwise a crafted 4-byte value can request up to ~4 GiB.
|
||||
/// Two framings are read:
|
||||
///
|
||||
/// * The registered HDF5 LZ4 filter format (`H5Zlz4.c`, what libhdf5 +
|
||||
/// hdf5plugin write, and what clawhdf5 writes after 2.7.0): an 8-byte
|
||||
/// big-endian total decompressed size, a 4-byte big-endian block size, then
|
||||
/// per block a 4-byte big-endian compressed length followed by the block. A
|
||||
/// block whose compressed length equals its decompressed length is stored
|
||||
/// raw.
|
||||
/// * The legacy clawhdf5 framing (up to 2.7.0): a 4-byte little-endian size
|
||||
/// followed by one raw LZ4 block. libhdf5 cannot read it.
|
||||
///
|
||||
/// They are told apart unambiguously: an HDF5 chunk is smaller than 4 GiB, so
|
||||
/// the registered format's big-endian `u64` size always starts with four zero
|
||||
/// bytes and the whole chunk is at least 12 bytes; a legacy chunk starts with
|
||||
/// four zero bytes only when it is empty, and is then 5 bytes long.
|
||||
///
|
||||
/// Every size read from the payload is bounded against `expected_bytes` (the
|
||||
/// pipeline's declared chunk size) before it sizes an allocation, so a crafted
|
||||
/// header cannot request gigabytes.
|
||||
#[cfg(feature = "lz4")]
|
||||
fn lz4_decompress(data: &[u8], expected_bytes: usize) -> Result<Vec<u8>, FormatError> {
|
||||
if data.len() < 4 {
|
||||
@@ -826,38 +960,112 @@ fn lz4_decompress(data: &[u8], expected_bytes: usize) -> Result<Vec<u8>, FormatE
|
||||
"lz4: data too short".into(),
|
||||
));
|
||||
}
|
||||
let orig_size = u32::from_le_bytes([data[0], data[1], data[2], data[3]]) as usize;
|
||||
if expected_bytes != 0 && orig_size > expected_bytes {
|
||||
let check_size = |size: usize| -> Result<(), FormatError> {
|
||||
if expected_bytes != 0 && size > expected_bytes {
|
||||
return Err(FormatError::DecompressionError(
|
||||
"lz4: declared size exceeds chunk size".into(),
|
||||
));
|
||||
}
|
||||
if orig_size > MAX_DECOMPRESS_SIZE {
|
||||
if size > MAX_DECOMPRESS_SIZE {
|
||||
return Err(FormatError::DecompressionError(
|
||||
"lz4: declared size exceeds limit".into(),
|
||||
));
|
||||
}
|
||||
Ok(())
|
||||
};
|
||||
if data.len() >= 12 && data[..4] == [0, 0, 0, 0] {
|
||||
return lz4_decompress_hdf5(data, check_size);
|
||||
}
|
||||
// Legacy clawhdf5 framing: 4-byte LE size + one LZ4 block.
|
||||
let orig_size = u32::from_le_bytes([data[0], data[1], data[2], data[3]]) as usize;
|
||||
check_size(orig_size)?;
|
||||
lz4_flex::block::decompress(&data[4..], orig_size)
|
||||
.map_err(|e| FormatError::DecompressionError(format!("lz4: {e}")))
|
||||
}
|
||||
|
||||
/// Decode the registered HDF5 LZ4 framing (see [`lz4_decompress`]).
|
||||
#[cfg(feature = "lz4")]
|
||||
fn lz4_decompress_hdf5(
|
||||
data: &[u8],
|
||||
check_size: impl Fn(usize) -> Result<(), FormatError>,
|
||||
) -> Result<Vec<u8>, FormatError> {
|
||||
let err = |m: &str| FormatError::DecompressionError(format!("lz4: {m}"));
|
||||
let be32 = |b: &[u8]| u32::from_be_bytes([b[0], b[1], b[2], b[3]]) as usize;
|
||||
// The first four bytes are zero (checked by the caller), so the size is
|
||||
// the low 32 bits of the big-endian u64.
|
||||
let orig_size = be32(&data[4..8]);
|
||||
check_size(orig_size)?;
|
||||
let block_size = be32(&data[8..12]).min(orig_size);
|
||||
if block_size == 0 && orig_size != 0 {
|
||||
return Err(err("zero block size"));
|
||||
}
|
||||
let mut out = vec![0u8; orig_size];
|
||||
let mut pos = 12usize;
|
||||
let mut done = 0usize;
|
||||
while done < orig_size {
|
||||
let this_block = block_size.min(orig_size - done);
|
||||
let comp_len = be32(
|
||||
data.get(pos..pos + 4)
|
||||
.ok_or_else(|| err("truncated block header"))?,
|
||||
);
|
||||
pos += 4;
|
||||
let block = data
|
||||
.get(pos..pos.saturating_add(comp_len))
|
||||
.ok_or_else(|| err("truncated block"))?;
|
||||
let dst = &mut out[done..done + this_block];
|
||||
if comp_len == this_block {
|
||||
dst.copy_from_slice(block);
|
||||
} else {
|
||||
let n = lz4_flex::block::decompress_into(block, dst)
|
||||
.map_err(|e| FormatError::DecompressionError(format!("lz4: {e}")))?;
|
||||
if n != this_block {
|
||||
return Err(err("block decompressed to the wrong size"));
|
||||
}
|
||||
}
|
||||
pos += comp_len;
|
||||
done += this_block;
|
||||
}
|
||||
Ok(out)
|
||||
}
|
||||
|
||||
#[cfg(not(feature = "lz4"))]
|
||||
fn lz4_decompress(_data: &[u8], _expected_bytes: usize) -> Result<Vec<u8>, FormatError> {
|
||||
Err(FormatError::UnsupportedFilter(FILTER_LZ4))
|
||||
}
|
||||
|
||||
/// Compress data with LZ4 block format. Format: 4 bytes LE original size + LZ4 block data.
|
||||
/// Compress data in the registered HDF5 LZ4 filter format (see
|
||||
/// [`lz4_decompress`]), so libhdf5 with the LZ4 plugin (e.g. hdf5plugin) can
|
||||
/// read it. `cd[0]`, when present and non-zero, is the block size in bytes,
|
||||
/// as in `H5Zlz4.c`; otherwise the 1 GiB default applies.
|
||||
#[cfg(feature = "lz4")]
|
||||
fn lz4_compress(data: &[u8]) -> Result<Vec<u8>, FormatError> {
|
||||
let compressed = lz4_flex::block::compress(data);
|
||||
let mut result = Vec::with_capacity(4 + compressed.len());
|
||||
result.extend_from_slice(&(data.len() as u32).to_le_bytes());
|
||||
fn lz4_compress(data: &[u8], cd: &[u32]) -> Result<Vec<u8>, FormatError> {
|
||||
let block_size = match cd.first() {
|
||||
Some(&b) if b != 0 => b as usize,
|
||||
_ => LZ4_DEFAULT_BLOCK_SIZE,
|
||||
}
|
||||
.min(data.len());
|
||||
let mut result = Vec::with_capacity(16 + data.len() / 2);
|
||||
result.extend_from_slice(&(data.len() as u64).to_be_bytes());
|
||||
result.extend_from_slice(&(block_size as u32).to_be_bytes());
|
||||
if block_size == 0 {
|
||||
return Ok(result);
|
||||
}
|
||||
for block in data.chunks(block_size) {
|
||||
let compressed = lz4_flex::block::compress(block);
|
||||
if compressed.len() >= block.len() {
|
||||
// Incompressible: stored raw, marked by length == block length.
|
||||
result.extend_from_slice(&(block.len() as u32).to_be_bytes());
|
||||
result.extend_from_slice(block);
|
||||
} else {
|
||||
result.extend_from_slice(&(compressed.len() as u32).to_be_bytes());
|
||||
result.extend_from_slice(&compressed);
|
||||
}
|
||||
}
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
#[cfg(not(feature = "lz4"))]
|
||||
fn lz4_compress(_data: &[u8]) -> Result<Vec<u8>, FormatError> {
|
||||
fn lz4_compress(_data: &[u8], _cd: &[u32]) -> Result<Vec<u8>, FormatError> {
|
||||
Err(FormatError::UnsupportedFilter(FILTER_LZ4))
|
||||
}
|
||||
|
||||
@@ -894,10 +1102,14 @@ fn zstd_decompress(_data: &[u8], _expected_bytes: usize) -> Result<Vec<u8>, Form
|
||||
Err(FormatError::UnsupportedFilter(FILTER_ZSTD))
|
||||
}
|
||||
|
||||
/// Compress data with zstd.
|
||||
/// Compress data with zstd as one frame whose header records the content
|
||||
/// size. The registered HDF5 Zstandard filter (`H5Zzstd.c`, used by
|
||||
/// libhdf5 + hdf5plugin) sizes its output buffer from
|
||||
/// `ZSTD_getFrameContentSize` and fails on a frame without it, which is what
|
||||
/// the streaming encoder (`zstd::encode_all`) produced.
|
||||
#[cfg(feature = "zstd")]
|
||||
fn zstd_compress(data: &[u8], level: u32) -> Result<Vec<u8>, FormatError> {
|
||||
zstd::encode_all(data, level as i32)
|
||||
zstd::bulk::compress(data, level as i32)
|
||||
.map_err(|e| FormatError::CompressionError(format!("zstd: {e}")))
|
||||
}
|
||||
|
||||
@@ -913,13 +1125,13 @@ fn shuffle_decompress(data: &[u8], element_size: usize) -> Result<Vec<u8>, Forma
|
||||
if element_size <= 1 {
|
||||
return Ok(data.to_vec());
|
||||
}
|
||||
if !data.len().is_multiple_of(element_size) {
|
||||
return Err(FormatError::FilterError(
|
||||
"shuffle: data length not a multiple of element size".into(),
|
||||
));
|
||||
}
|
||||
// Like libhdf5, only whole elements are shuffled; trailing bytes (e.g. a
|
||||
// Fletcher32 checksum appended before the shuffle) are stored as-is.
|
||||
let whole = data.len() - data.len() % element_size;
|
||||
let (data, tail) = data.split_at(whole);
|
||||
let num_elements = data.len() / element_size;
|
||||
let mut result = vec![0u8; data.len()];
|
||||
let mut result = vec![0u8; whole];
|
||||
result.reserve_exact(tail.len());
|
||||
|
||||
// The shuffled stream is `element_size` byte planes of `num_elements`
|
||||
// bytes each; un-shuffling interleaves them. This is on the read path of
|
||||
@@ -950,6 +1162,7 @@ fn shuffle_decompress(data: &[u8], element_size: usize) -> Result<Vec<u8>, Forma
|
||||
}
|
||||
}
|
||||
}
|
||||
result.extend_from_slice(tail);
|
||||
|
||||
Ok(result)
|
||||
}
|
||||
@@ -965,19 +1178,19 @@ fn shuffle_compress(data: &[u8], element_size: usize) -> Result<Vec<u8>, FormatE
|
||||
if element_size <= 1 {
|
||||
return Ok(data.to_vec());
|
||||
}
|
||||
if !data.len().is_multiple_of(element_size) {
|
||||
return Err(FormatError::FilterError(
|
||||
"shuffle: data length not a multiple of element size".into(),
|
||||
));
|
||||
}
|
||||
// Trailing bytes that don't make a whole element are left in place, as
|
||||
// libhdf5 does.
|
||||
let whole = data.len() - data.len() % element_size;
|
||||
let (data, tail) = data.split_at(whole);
|
||||
let num_elements = data.len() / element_size;
|
||||
let mut result = vec![0u8; data.len()];
|
||||
let mut result = vec![0u8; whole];
|
||||
|
||||
match element_size {
|
||||
4 => shuffle_compress_4(data, num_elements, &mut result),
|
||||
8 => shuffle_compress_general(data, num_elements, element_size, &mut result),
|
||||
_ => shuffle_compress_general(data, num_elements, element_size, &mut result),
|
||||
}
|
||||
result.extend_from_slice(tail);
|
||||
|
||||
Ok(result)
|
||||
}
|
||||
@@ -1413,6 +1626,110 @@ mod tests {
|
||||
assert_eq!(decompressed, data);
|
||||
}
|
||||
|
||||
fn filter(filter_id: u16) -> FilterDescription {
|
||||
FilterDescription {
|
||||
filter_id,
|
||||
name: None,
|
||||
flags: 0,
|
||||
client_data: vec![],
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
#[cfg(feature = "deflate")]
|
||||
fn filter_mask_skips_only_the_masked_filters() {
|
||||
let pipeline = FilterPipeline {
|
||||
version: 2,
|
||||
filters: vec![filter(FILTER_SHUFFLE), filter(FILTER_DEFLATE)],
|
||||
};
|
||||
let only_shuffle = FilterPipeline {
|
||||
version: 2,
|
||||
filters: vec![filter(FILTER_SHUFFLE)],
|
||||
};
|
||||
let only_deflate = FilterPipeline {
|
||||
version: 2,
|
||||
filters: vec![filter(FILTER_DEFLATE)],
|
||||
};
|
||||
let data: Vec<u8> = (0..200).map(|i| (i * 7 % 256) as u8).collect();
|
||||
let n = data.len();
|
||||
|
||||
let shuffled = compress_chunk(&data, &only_shuffle, 8).unwrap();
|
||||
assert_ne!(shuffled, data);
|
||||
let deflated = compress_chunk(&data, &only_deflate, 8).unwrap();
|
||||
|
||||
// Bit 1: deflate skipped, shuffle still undone.
|
||||
assert_eq!(
|
||||
decompress_chunk_masked(&shuffled, &pipeline, n, 8, 0b10).unwrap(),
|
||||
data
|
||||
);
|
||||
// Bit 0: shuffle skipped, deflate still undone.
|
||||
assert_eq!(
|
||||
decompress_chunk_masked(&deflated, &pipeline, n, 8, 0b01).unwrap(),
|
||||
data
|
||||
);
|
||||
// Both bits (and bits past the pipeline): stored as-is.
|
||||
assert_eq!(
|
||||
decompress_chunk_masked(&data, &pipeline, n, 8, u32::MAX).unwrap(),
|
||||
data
|
||||
);
|
||||
assert!(all_filters_skipped(&pipeline, 0b11));
|
||||
assert!(!all_filters_skipped(&pipeline, 0b10));
|
||||
// An unsupported filter is fine when the chunk skipped it.
|
||||
let unknown = FilterPipeline {
|
||||
version: 2,
|
||||
filters: vec![filter(32000), filter(FILTER_DEFLATE)],
|
||||
};
|
||||
assert_eq!(
|
||||
decompress_chunk_masked(&deflated, &unknown, n, 8, 0b01).unwrap(),
|
||||
data
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
#[cfg(feature = "deflate")]
|
||||
fn fletcher32_ahead_of_deflate_stays_bounded() {
|
||||
// NetCDF-4 order: the checksum is appended before shuffle and deflate,
|
||||
// so deflate decodes chunk + 4 bytes.
|
||||
let pipeline = FilterPipeline {
|
||||
version: 2,
|
||||
filters: vec![
|
||||
filter(FILTER_FLETCHER32),
|
||||
filter(FILTER_SHUFFLE),
|
||||
filter(FILTER_DEFLATE),
|
||||
],
|
||||
};
|
||||
let data: Vec<u8> = (0..400).map(|i| (i * 13 % 251) as u8).collect();
|
||||
let n = data.len();
|
||||
let stored = compress_chunk(&data, &pipeline, 8).unwrap();
|
||||
assert_eq!(decompress_chunk(&stored, &pipeline, n, 8).unwrap(), data);
|
||||
|
||||
// The cap still bites: a stream that inflates past chunk + 4 bytes
|
||||
// is rejected rather than allocated.
|
||||
let only_deflate = FilterPipeline {
|
||||
version: 2,
|
||||
filters: vec![filter(FILTER_DEFLATE)],
|
||||
};
|
||||
let oversized = compress_chunk(&vec![0u8; n + 5], &only_deflate, 8).unwrap();
|
||||
let err = decompress_chunk(&oversized, &pipeline, n, 8).unwrap_err();
|
||||
assert!(
|
||||
matches!(err, FormatError::DecompressionError(_)),
|
||||
"expected a size-limit error, got {err:?}"
|
||||
);
|
||||
// A bomb is still stopped near the chunk size.
|
||||
let bomb = compress_chunk(&vec![0u8; 64 * n], &only_deflate, 8).unwrap();
|
||||
assert!(decompress_chunk(&bomb, &pipeline, n, 8).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn shuffle_leaves_a_partial_trailing_element_in_place() {
|
||||
// libhdf5 shuffles whole elements and copies the remainder as-is.
|
||||
let data: Vec<u8> = (0..20).collect();
|
||||
let shuffled = shuffle_compress(&data, 8).unwrap();
|
||||
assert_eq!(&shuffled[16..], &data[16..]);
|
||||
assert_eq!(&shuffled[..4], &[0, 8, 1, 9]);
|
||||
assert_eq!(shuffle_decompress(&shuffled, 8).unwrap(), data);
|
||||
}
|
||||
|
||||
#[test]
|
||||
#[cfg(feature = "deflate")]
|
||||
fn pipeline_compress_decompress_roundtrip() {
|
||||
@@ -1480,11 +1797,317 @@ mod tests {
|
||||
|
||||
// --- LZ4 tests ---
|
||||
|
||||
fn unhex(s: &str) -> Vec<u8> {
|
||||
(0..s.len())
|
||||
.step_by(2)
|
||||
.map(|i| u8::from_str_radix(&s[i..i + 2], 16).unwrap())
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn one_filter(filter_id: u16, client_data: Vec<u32>) -> FilterDescription {
|
||||
FilterDescription {
|
||||
filter_id,
|
||||
name: None,
|
||||
flags: 1,
|
||||
client_data,
|
||||
}
|
||||
}
|
||||
|
||||
/// Fletcher32 before a compressor (libhdf5 applies filters in pipeline
|
||||
/// order, so the compressor sees chunk + checksum): deflate's output is 4
|
||||
/// bytes over the chunk size, which we rejected as "deflate: output
|
||||
/// exceeds size limit". Chunks from h5py/libhdf5, values from h5py.
|
||||
#[test]
|
||||
#[cfg(feature = "deflate")]
|
||||
fn fletcher32_before_deflate_decodes() {
|
||||
// h5py: set_fletcher32(); set_deflate(4); i32 0..100, chunks of 10.
|
||||
let pipeline = FilterPipeline {
|
||||
version: 2,
|
||||
filters: vec![
|
||||
one_filter(FILTER_FLETCHER32, vec![]),
|
||||
one_filter(FILTER_DEFLATE, vec![4]),
|
||||
],
|
||||
};
|
||||
let raw =
|
||||
unhex("785e936360609007620520560462252056066215205605623520560762c6483e3d00234501f0");
|
||||
let want: Vec<u8> = (30..40i32).flat_map(i32::to_le_bytes).collect();
|
||||
assert_eq!(decompress_chunk(&raw, &pipeline, 40, 4).unwrap(), want);
|
||||
|
||||
// And our own writer's round trip through the same pipeline order.
|
||||
let data: Vec<u8> = (0..400u32).map(|i| (i % 13) as u8).collect();
|
||||
let pipeline = FilterPipeline {
|
||||
version: 2,
|
||||
filters: vec![
|
||||
one_filter(FILTER_SHUFFLE, vec![4]),
|
||||
one_filter(FILTER_FLETCHER32, vec![]),
|
||||
one_filter(FILTER_DEFLATE, vec![9]),
|
||||
],
|
||||
};
|
||||
let c = compress_chunk(&data, &pipeline, 4).unwrap();
|
||||
assert_eq!(decompress_chunk(&c, &pipeline, 400, 4).unwrap(), data);
|
||||
}
|
||||
|
||||
/// `le_data.h5` scale-offset (D-scale, D = 3, fill -2.2) chunks, decoded
|
||||
/// bit for bit as libhdf5 does: single-precision arithmetic for `float`
|
||||
/// (we computed in f64 and rounded once, which was 1 ULP off for e.g.
|
||||
/// 1.6663333: `694ad53f` instead of `6a4ad53f`), double for `double`.
|
||||
#[test]
|
||||
fn scaleoffset_float_dscale_matches_libhdf5_bits() {
|
||||
let file: &[u8] = include_bytes!("../tests/fixtures/filters/le_data.h5");
|
||||
let cd = |size: u32, order: u32, fill_lo: u32, fill_hi: u32| {
|
||||
let mut cd = vec![0, 3, 12, 1, size, 0, order, 1, fill_lo, fill_hi];
|
||||
cd.resize(20, 0);
|
||||
cd
|
||||
};
|
||||
let f32_le = cd(4, 0, 0xC00C_CCCD, 0);
|
||||
let f32_be = cd(4, 1, 0xC00C_CCCD, 0);
|
||||
let f64_le = cd(8, 0, 2576980378, 3221330329);
|
||||
#[rustfmt::skip]
|
||||
let cases: [(usize, usize, &[u32], &str); 6] = [
|
||||
(2816, 38, &f32_le, "abaaaa3ed2942a3fec0a803fd2942a3fec0a803fabaaaa3fec0a803fabaaaa3f694ad53fabaaaa3f694ad53f76050040"),
|
||||
(2854, 38, &f32_le, "abaaaa3f6a4ad53f760500406a4ad53f7605004056551540760500405655154034a52a405655154034a52a4076054040"),
|
||||
(712, 38, &f32_be, "3eaaaaab3f2a94d23f800aec3f2a94d23f800aec3faaaaab3f800aec3faaaaab3fd54a693faaaaab3fd54a6940000576"),
|
||||
(750, 38, &f32_be, "3faaaaab3fd54a6a400005763fd54a6a40000576401555564000057640155556402aa53440155556402aa53440400576"),
|
||||
(2050, 38, &f64_le, concat!(
|
||||
"555555555555d53fb9d75c489a52e53fce3e7c865d01f03fb9d75c489a52e53fce3e7c865d01f03f555555555555f53f",
|
||||
"ce3e7c865d01f03f555555555555f53fdc6b2e244da9fa3f555555555555f53fdc6b2e244da9fa3f671f3ec3ae000040")),
|
||||
(2088, 38, &f64_le, concat!(
|
||||
"555555555555f53fdc6b2e244da9fa3f671f3ec3ae000040dc6b2e244da9fa3f671f3ec3ae000040aaaaaaaaaaaa0240",
|
||||
"671f3ec3ae000040aaaaaaaaaaaa0240ee351792a6540540aaaaaaaaaaaa0240ee351792a6540540671f3ec3ae000840")),
|
||||
];
|
||||
for (off, len, cd, want) in cases {
|
||||
let pipeline = FilterPipeline {
|
||||
version: 2,
|
||||
filters: vec![one_filter(FILTER_SCALEOFFSET, cd.to_vec())],
|
||||
};
|
||||
let want = unhex(want);
|
||||
let got =
|
||||
decompress_chunk(&file[off..off + len], &pipeline, want.len(), cd[4]).unwrap();
|
||||
assert_eq!(got, want, "chunk at {off}");
|
||||
}
|
||||
}
|
||||
|
||||
/// `le_data.h5` `/Nbit_float_data_{le,be}` chunk (0,0): a 20-bit float
|
||||
/// (offset 7) packed by N-Bit. The filter must reproduce libhdf5's
|
||||
/// decoded bytes in the *file* datatype (h5py `DatasetID.read` with the
|
||||
/// file type as memory type, so no conversion); converting that custom
|
||||
/// float layout to IEEE is the datatype reader's job, not the filter's.
|
||||
#[test]
|
||||
fn nbit_float_matches_libhdf5_file_type_bytes() {
|
||||
let file: &[u8] = include_bytes!("../tests/fixtures/filters/le_data.h5");
|
||||
let cases = [
|
||||
(
|
||||
55952,
|
||||
0,
|
||||
"8055d5018055e5010000f0018055e5010000f0018055f5010000f0018055f50180aafa018055f50180aafa0100000002",
|
||||
),
|
||||
(
|
||||
56076,
|
||||
1,
|
||||
"01d5558001e5558001f0000001e5558001f0000001f5558001f0000001f5558001faaa8001f5558001faaa8002000000",
|
||||
),
|
||||
];
|
||||
for (off, order, want) in cases {
|
||||
let pipeline = FilterPipeline {
|
||||
version: 2,
|
||||
filters: vec![one_filter(FILTER_NBIT, vec![8, 0, 12, 1, 4, order, 20, 7])],
|
||||
};
|
||||
let got = decompress_chunk(&file[off..off + 31], &pipeline, 48, 4).unwrap();
|
||||
assert_eq!(got, unhex(want), "byte order {order}");
|
||||
}
|
||||
}
|
||||
|
||||
/// libhdf5 sets `cd_values[1]` ("need not compress") when every field is
|
||||
/// already full width and then stores the data unchanged; we unpacked it
|
||||
/// anyway and failed with "nbit: packed data too short".
|
||||
#[test]
|
||||
fn nbit_need_not_compress_is_passthrough() {
|
||||
let data: Vec<u8> = (0..200u32).map(|i| (i * 7) as u8).collect();
|
||||
let pipeline = FilterPipeline {
|
||||
version: 2,
|
||||
filters: vec![one_filter(FILTER_NBIT, vec![8, 1, 50, 1, 4, 0, 32, 0])],
|
||||
};
|
||||
assert_eq!(decompress_chunk(&data, &pipeline, 200, 4).unwrap(), data);
|
||||
// A top-level type N-Bit has no parameters for (e.g. an enum) carries only
|
||||
// [nparms, need_not_compress, nelmts].
|
||||
let pipeline = FilterPipeline {
|
||||
version: 2,
|
||||
filters: vec![one_filter(FILTER_NBIT, vec![3, 1, 50])],
|
||||
};
|
||||
assert_eq!(decompress_chunk(&data, &pipeline, 200, 4).unwrap(), data);
|
||||
}
|
||||
|
||||
/// `tfilters.h5` `/all` chunk (0,0): shuffle, szip, deflate, fletcher32
|
||||
/// and a pass-through N-Bit in one pipeline; values from h5py.
|
||||
#[test]
|
||||
#[cfg(feature = "szip")]
|
||||
fn nbit_in_multi_filter_pipeline_matches_libhdf5() {
|
||||
let raw = unhex(concat!(
|
||||
"785e3bc1c0c030cb6517ff039e556de1576c0f300b152c771070641170640b99ba879f8167d5cce82f760ccc4285d71b",
|
||||
"20c2afc226329f60d60ab5636a3fc1905499c3c0a1d0c4a1d05c6a13d7f88271aaf6725fe7170c86363f1858c0ca0104",
|
||||
"bf1e95c75f3eeb",
|
||||
));
|
||||
let want = unhex(concat!(
|
||||
"00000000010000000200000003000000040000000a0000000b0000000c0000000d0000000e0000001400000015000000",
|
||||
"1600000017000000180000001e0000001f00000020000000210000002200000028000000290000002a0000002b000000",
|
||||
"2c00000032000000330000003400000035000000360000003c0000003d0000003e0000003f0000004000000046000000",
|
||||
"4700000048000000490000004a00000050000000510000005200000053000000540000005a0000005b0000005c000000",
|
||||
"5d0000005e000000",
|
||||
));
|
||||
let pipeline = FilterPipeline {
|
||||
version: 2,
|
||||
filters: vec![
|
||||
one_filter(FILTER_SHUFFLE, vec![4]),
|
||||
one_filter(FILTER_SZIP, vec![141, 4, 32, 5]),
|
||||
one_filter(FILTER_DEFLATE, vec![5]),
|
||||
one_filter(FILTER_FLETCHER32, vec![]),
|
||||
one_filter(FILTER_NBIT, vec![8, 1, 50, 1, 4, 0, 32, 0]),
|
||||
],
|
||||
};
|
||||
assert_eq!(decompress_chunk(&raw, &pipeline, 200, 4).unwrap(), want);
|
||||
}
|
||||
|
||||
/// `h5repack_nested_8bit_enum_deflated.h5` `/tracks/1/trace` chunk 0: a
|
||||
/// 376-byte compound whose `u1` enum member N-Bit stores whole as a
|
||||
/// no-op type (class 4), then deflate. Was `UnsupportedFilter(5)`.
|
||||
/// Expected bytes: libhdf5's decode in the file datatype.
|
||||
#[test]
|
||||
#[cfg(feature = "deflate")]
|
||||
fn nbit_compound_with_enum_member_matches_libhdf5() {
|
||||
#[rustfmt::skip]
|
||||
let cd: Vec<u32> = vec![
|
||||
251, 0, 1, 3, 376, 38,
|
||||
0, 1, 4, 0, 32, 0,
|
||||
8, 2, 96, 1, 8, 0, 64, 0,
|
||||
104, 1, 8, 0, 64, 0, 112, 1, 8, 0, 64, 0, 120, 1, 8, 0, 64, 0,
|
||||
128, 1, 8, 0, 64, 0, 136, 1, 8, 0, 64, 0, 144, 1, 8, 0, 64, 0,
|
||||
152, 1, 8, 0, 64, 0,
|
||||
160, 2, 24, 1, 8, 0, 64, 0, 184, 2, 24, 1, 8, 0, 64, 0,
|
||||
208, 2, 16, 1, 8, 0, 64, 0, 224, 2, 32, 1, 8, 0, 64, 0,
|
||||
256, 2, 16, 1, 4, 0, 32, 0, 272, 2, 16, 1, 4, 0, 32, 0,
|
||||
288, 2, 32, 1, 8, 0, 64, 0, 320, 2, 16, 1, 8, 0, 64, 0,
|
||||
336, 2, 8, 1, 4, 0, 32, 0,
|
||||
344, 4, 1,
|
||||
346, 1, 2, 0, 16, 0, 348, 1, 4, 0, 32, 0, 352, 1, 1, 0, 4, 0,
|
||||
354, 1, 2, 0, 16, 0, 356, 1, 1, 0, 4, 0, 357, 1, 1, 0, 4, 0,
|
||||
358, 1, 1, 0, 4, 0, 359, 1, 1, 0, 4, 0, 360, 1, 1, 0, 4, 0,
|
||||
361, 1, 1, 0, 4, 0, 362, 1, 1, 0, 4, 0, 364, 1, 1, 0, 4, 0,
|
||||
363, 1, 1, 0, 4, 0, 365, 1, 1, 0, 4, 0, 366, 1, 1, 0, 4, 0,
|
||||
367, 1, 1, 0, 4, 0, 368, 1, 1, 0, 4, 0, 369, 1, 1, 0, 4, 0,
|
||||
370, 1, 1, 0, 4, 0,
|
||||
];
|
||||
assert_eq!(cd.len(), 251);
|
||||
let raw = unhex(concat!(
|
||||
"780163606078c930c880c3879573a60b2d7883eeac06a880835c47bda16cda7987a0c59ec9f74c4af6ff677e5df16445",
|
||||
"adfd8493ce3ba592ddedab2a3bee87dd0faaff00d1808b66606006fa9d999b818125014837fd4703fba1fa71d150e720",
|
||||
"512caa4073dc59212206500946060600604c37fc",
|
||||
));
|
||||
let want = unhex(concat!(
|
||||
"e90000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000",
|
||||
"000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000",
|
||||
"000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000",
|
||||
"0000000000000000eca012979ca9f040000000000000000000000000000000000000000000000080cf661d317f881e40",
|
||||
"7434de6349a352407da8e478eb03ffbf47631ab943c9903f52df56df88797a3f000000000000f07f000000000000f07f",
|
||||
"000000000000f07f000000000000f07fe90300000b0300006004000082030000ffffffffffffffffffffffffffffffff",
|
||||
"000000000000f0bf000000000000f0bf000000000000f0bf000000000000f0bf00000000000000000000000000000000",
|
||||
"25040000470300000500000000000000030000000000000000000000000000000100000000000000",
|
||||
));
|
||||
let pipeline = FilterPipeline {
|
||||
version: 2,
|
||||
filters: vec![
|
||||
one_filter(FILTER_NBIT, cd),
|
||||
one_filter(FILTER_DEFLATE, vec![1]),
|
||||
],
|
||||
};
|
||||
assert_eq!(decompress_chunk(&raw, &pipeline, 376, 376).unwrap(), want);
|
||||
}
|
||||
|
||||
/// Chunk (0,0) of `/DS1` in the HDF Group's `h5ex_d_lz4.h5` example,
|
||||
/// written by libhdf5's registered LZ4 plugin with a 3-byte block size
|
||||
/// (so it has many blocks, some stored raw). Byte range from h5py's
|
||||
/// `get_chunk_info`; values are `i*j - j` (i32 LE), as h5py reads them.
|
||||
#[test]
|
||||
#[cfg(feature = "lz4")]
|
||||
fn lz4_reads_registered_hdf5_format() {
|
||||
let file: &[u8] = include_bytes!("../tests/fixtures/filters/h5ex_d_lz4.h5");
|
||||
let chunk = &file[4016..4016 + 312];
|
||||
let pipeline = FilterPipeline {
|
||||
version: 2,
|
||||
filters: vec![FilterDescription {
|
||||
filter_id: FILTER_LZ4,
|
||||
name: None,
|
||||
flags: 1,
|
||||
client_data: vec![3],
|
||||
}],
|
||||
};
|
||||
let out = decompress_chunk(chunk, &pipeline, 4 * 8 * 4, 4).unwrap();
|
||||
let expected: Vec<u8> = (0..4i32)
|
||||
.flat_map(|i| (0..8i32).map(move |j| i * j - j))
|
||||
.flat_map(i32::to_le_bytes)
|
||||
.collect();
|
||||
assert_eq!(out, expected);
|
||||
}
|
||||
|
||||
#[test]
|
||||
#[cfg(feature = "lz4")]
|
||||
fn lz4_writes_registered_hdf5_format() {
|
||||
// Compressible data, one block: 8-byte BE size, 4-byte BE block size,
|
||||
// 4-byte BE compressed length, block.
|
||||
let data = vec![7u8; 1000];
|
||||
let c = lz4_compress(&data, &[]).unwrap();
|
||||
assert_eq!(&c[0..8], &1000u64.to_be_bytes());
|
||||
assert_eq!(&c[8..12], &1000u32.to_be_bytes());
|
||||
let len = u32::from_be_bytes(c[12..16].try_into().unwrap()) as usize;
|
||||
assert_eq!(c.len(), 16 + len);
|
||||
assert!(len < 1000);
|
||||
assert_eq!(lz4_decompress(&c, 1000).unwrap(), data);
|
||||
|
||||
// Several blocks, incompressible ones stored raw (length == block).
|
||||
let data: Vec<u8> = (0..10u8).collect();
|
||||
let c = lz4_compress(&data, &[3]).unwrap();
|
||||
assert_eq!(&c[8..12], &3u32.to_be_bytes());
|
||||
assert_eq!(&c[12..16], &3u32.to_be_bytes());
|
||||
assert_eq!(&c[16..19], &[0, 1, 2]);
|
||||
assert_eq!(c.len(), 12 + 3 * (4 + 3) + (4 + 1));
|
||||
assert_eq!(lz4_decompress(&c, 10).unwrap(), data);
|
||||
|
||||
let c = lz4_compress(&[], &[]).unwrap();
|
||||
assert_eq!(lz4_decompress(&c, 0).unwrap(), Vec::<u8>::new());
|
||||
}
|
||||
|
||||
/// Chunks written by clawhdf5 up to 2.7.0 (4-byte LE size + one LZ4
|
||||
/// block) must stay readable.
|
||||
#[test]
|
||||
#[cfg(feature = "lz4")]
|
||||
fn lz4_reads_legacy_clawhdf5_format() {
|
||||
for data in [vec![], vec![5u8; 300], (0..=255u8).collect::<Vec<u8>>()] {
|
||||
let mut legacy = (data.len() as u32).to_le_bytes().to_vec();
|
||||
legacy.extend_from_slice(&lz4_flex::block::compress(&data));
|
||||
assert_eq!(lz4_decompress(&legacy, data.len()).unwrap(), data);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
#[cfg(feature = "lz4")]
|
||||
fn lz4_registered_format_rejects_hostile_sizes() {
|
||||
// Declared total larger than the chunk.
|
||||
let mut c = 1000u64.to_be_bytes().to_vec();
|
||||
c.extend_from_slice(&1000u32.to_be_bytes());
|
||||
c.extend_from_slice(&[0u8; 8]);
|
||||
assert!(lz4_decompress(&c, 64).is_err());
|
||||
// Truncated block.
|
||||
let mut c = 16u64.to_be_bytes().to_vec();
|
||||
c.extend_from_slice(&16u32.to_be_bytes());
|
||||
c.extend_from_slice(&16u32.to_be_bytes());
|
||||
c.extend_from_slice(&[1u8; 4]);
|
||||
assert!(lz4_decompress(&c, 16).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
#[cfg(feature = "lz4")]
|
||||
fn lz4_compress_decompress_roundtrip() {
|
||||
let data: Vec<u8> = (0..256).map(|i| (i % 256) as u8).collect();
|
||||
let compressed = lz4_compress(&data).unwrap();
|
||||
let compressed = lz4_compress(&data, &[]).unwrap();
|
||||
let decompressed = lz4_decompress(&compressed, data.len()).unwrap();
|
||||
assert_eq!(decompressed, data);
|
||||
}
|
||||
@@ -1535,6 +2158,23 @@ mod tests {
|
||||
|
||||
// --- Zstd tests ---
|
||||
|
||||
/// libhdf5's zstd plugin needs the frame content size to size its
|
||||
/// output; frames without it fail to decode there.
|
||||
#[test]
|
||||
#[cfg(feature = "zstd")]
|
||||
fn zstd_frames_record_content_size() {
|
||||
for n in [0usize, 1, 200, 100_000] {
|
||||
let data: Vec<u8> = (0..n).map(|i| (i % 7) as u8).collect();
|
||||
let c = zstd_compress(&data, 3).unwrap();
|
||||
assert_eq!(
|
||||
zstd::zstd_safe::get_frame_content_size(&c).unwrap(),
|
||||
Some(n as u64),
|
||||
"{n} bytes"
|
||||
);
|
||||
assert_eq!(zstd_decompress(&c, n).unwrap(), data);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
#[cfg(feature = "zstd")]
|
||||
fn zstd_compress_decompress_roundtrip() {
|
||||
@@ -1996,6 +2636,57 @@ mod tests {
|
||||
assert!(pcodec_decompress(&compressed, 4, 16).is_err());
|
||||
}
|
||||
|
||||
/// Pcodec is written under the private ID 480, not 32023 (registered to
|
||||
/// Granular BitRound, whose pass-through decode would hand libhdf5 users
|
||||
/// the compressed bytes as data). Chunks under 32023 are read as pcodec
|
||||
/// only with the name clawhdf5 <= 2.7.0 wrote.
|
||||
#[test]
|
||||
#[cfg(feature = "pcodec")]
|
||||
fn pcodec_uses_private_id_and_reads_legacy_32023() {
|
||||
use crate::chunked_write::ChunkOptions;
|
||||
let opts = ChunkOptions {
|
||||
pcodec: true,
|
||||
..Default::default()
|
||||
};
|
||||
let pl = opts.build_pipeline(8).unwrap();
|
||||
let f = pl.filters.iter().find(|f| f.filter_id == 480).unwrap();
|
||||
assert_eq!(
|
||||
f.name.as_deref(),
|
||||
Some(crate::filter_pipeline::FILTER_PCODEC_NAME)
|
||||
);
|
||||
assert!(pl.filters.iter().all(|f| f.filter_id != 32023));
|
||||
|
||||
let data: Vec<f64> = (0..100).map(|i| i as f64 * 0.25).collect();
|
||||
let raw: Vec<u8> = data.iter().flat_map(|x| x.to_le_bytes()).collect();
|
||||
let compressed = pcodec_compress(&raw, 8).unwrap();
|
||||
let pipeline = |id: u16, name: Option<&str>| FilterPipeline {
|
||||
version: 2,
|
||||
filters: vec![FilterDescription {
|
||||
filter_id: id,
|
||||
name: name.map(Into::into),
|
||||
flags: 0,
|
||||
client_data: vec![8],
|
||||
}],
|
||||
};
|
||||
let legacy = pipeline(32023, Some("pcodec"));
|
||||
assert_eq!(
|
||||
decompress_chunk(&compressed, &legacy, raw.len(), 8).unwrap(),
|
||||
raw
|
||||
);
|
||||
let current = pipeline(480, None);
|
||||
assert_eq!(
|
||||
decompress_chunk(&compressed, ¤t, raw.len(), 8).unwrap(),
|
||||
raw
|
||||
);
|
||||
// A real Granular BitRound filter is not pcodec.
|
||||
for name in [None, Some("Granular BitRound")] {
|
||||
assert!(matches!(
|
||||
decompress_chunk(&compressed, &pipeline(32023, name), raw.len(), 8),
|
||||
Err(FormatError::UnsupportedFilter(32023))
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
#[cfg(feature = "lz4")]
|
||||
fn decompress_chunk_rejects_hostile_lz4_size_via_public_entrypoint() {
|
||||
|
||||
@@ -1,19 +1,39 @@
|
||||
//! SZIP (libaec Adaptive Entropy Coding) decompression.
|
||||
//!
|
||||
//! Gated by the `szip` feature which links against the system libaec library.
|
||||
//!
|
||||
//! libhdf5's SZIP filter (`H5Zszip.c`) prefixes each chunk with its
|
||||
//! uncompressed size and hands the rest to szlib's `SZ_BufftoBuffDecompress`.
|
||||
//! libaec implements that call (`sz_compat.c`) on top of `aec_buffer_decode`
|
||||
//! with some reshaping — 32/64-bit samples are coded as byte planes of 8-bit
|
||||
//! samples, and scanlines that are not a whole number of blocks are padded —
|
||||
//! which [`szip_decompress`] reproduces so its output matches libhdf5's.
|
||||
|
||||
#[cfg(not(feature = "std"))]
|
||||
use alloc::vec::Vec;
|
||||
|
||||
use crate::error::FormatError;
|
||||
|
||||
/// Decompress SZIP-compressed data using libaec.
|
||||
/// `SZ_MSB_OPTION_MASK`: samples are big-endian.
|
||||
#[cfg(feature = "szip")]
|
||||
const SZ_MSB_OPTION_MASK: u32 = 16;
|
||||
/// `SZ_NN_OPTION_MASK`: nearest-neighbour preprocessing.
|
||||
#[cfg(feature = "szip")]
|
||||
const SZ_NN_OPTION_MASK: u32 = 32;
|
||||
|
||||
/// Decompress one SZIP-filtered chunk.
|
||||
///
|
||||
/// `cd` is the HDF5 SZIP filter client data (matches `H5Z_SZIP_PARM_*` indices):
|
||||
/// cd[0] = options mask (`H5_SZIP_NN_OPTION_MASK = 0x20` enables NN preprocessing)
|
||||
/// cd[1] = pixels per block (H5Z_SZIP_PARM_PPB; 8, 10, 16, or 32)
|
||||
/// cd[2] = bits per sample (H5Z_SZIP_PARM_BPP; element bit width)
|
||||
/// cd[3] = pixels per scan line (H5Z_SZIP_PARM_PPS; informational only)
|
||||
/// `cd` is the HDF5 SZIP filter client data (`H5Z_SZIP_PARM_*` indices):
|
||||
/// cd[0] = options mask (`SZ_*_OPTION_MASK`: 16 = MSB byte order,
|
||||
/// 32 = nearest-neighbour preprocessing; K13/EC/LSB/RAW bits carry
|
||||
/// no decoding information for libaec)
|
||||
/// cd[1] = pixels per block
|
||||
/// cd[2] = bits per pixel (sample precision, rounded up to 32 or 64 above
|
||||
/// 24 by libhdf5)
|
||||
/// cd[3] = pixels per scanline
|
||||
///
|
||||
/// The chunk is a 4-byte little-endian uncompressed size followed by the
|
||||
/// szlib stream.
|
||||
pub(crate) fn szip_decompress(
|
||||
_data: &[u8],
|
||||
_cd: &[u32],
|
||||
@@ -33,62 +53,174 @@ pub(crate) fn szip_decompress(
|
||||
|
||||
#[cfg(feature = "szip")]
|
||||
fn szip_decode_impl(data: &[u8], cd: &[u32], chunk_size: usize) -> Result<Vec<u8>, FormatError> {
|
||||
if cd.len() < 3 {
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"szip: missing client data".into(),
|
||||
));
|
||||
let err = |m: &str| FormatError::ChunkedReadError(format!("szip: {m}"));
|
||||
if cd.len() < 4 {
|
||||
return Err(err("missing client data"));
|
||||
}
|
||||
let options = cd[0];
|
||||
let pixels_per_block = cd[1];
|
||||
let bits_per_sample = cd[2]; // H5Z_SZIP_PARM_BPP
|
||||
if bits_per_sample == 0 || bits_per_sample > 32 {
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"szip: invalid bits per sample".into(),
|
||||
));
|
||||
let pixels_per_block = cd[1] as usize;
|
||||
let bits_per_pixel = cd[2];
|
||||
let pixels_per_scanline = cd[3] as usize;
|
||||
if !(1..=32).contains(&bits_per_pixel) && bits_per_pixel != 64 {
|
||||
return Err(err("invalid bits per sample"));
|
||||
}
|
||||
if chunk_size == 0 {
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"szip: unknown output size".into(),
|
||||
));
|
||||
if pixels_per_block == 0 || pixels_per_scanline == 0 {
|
||||
return Err(err("invalid block or scanline size"));
|
||||
}
|
||||
if data.is_empty() {
|
||||
return Err(FormatError::ChunkedReadError("szip: empty input".into()));
|
||||
if data.len() < 4 {
|
||||
return Err(err("chunk too short"));
|
||||
}
|
||||
// H5Zszip.c: UINT32DECODE of the uncompressed size, then the stream.
|
||||
let dest_len = u32::from_le_bytes([data[0], data[1], data[2], data[3]]) as usize;
|
||||
let limit = if chunk_size != 0 {
|
||||
chunk_size
|
||||
} else {
|
||||
crate::filters::MAX_DECOMPRESS_SIZE
|
||||
};
|
||||
if dest_len > limit {
|
||||
return Err(err("declared size exceeds chunk size"));
|
||||
}
|
||||
let stream = &data[4..];
|
||||
|
||||
// Map HDF5 option mask to libaec flags.
|
||||
// HDF5 always stores SZIP data in MSB order, so AEC_DATA_MSB is unconditional.
|
||||
// H5_SZIP_NN_OPTION_MASK (0x20): NN differential preprocessing.
|
||||
let mut flags: u32 = libaec_sys::AEC_DATA_MSB;
|
||||
if options & 0x20 != 0 {
|
||||
// --- libaec sz_compat.c: SZ_BufftoBuffDecompress ---
|
||||
let rsi = pixels_per_scanline.div_ceil(pixels_per_block);
|
||||
let mut flags = 0;
|
||||
if options & SZ_MSB_OPTION_MASK != 0 {
|
||||
flags |= libaec_sys::AEC_DATA_MSB;
|
||||
}
|
||||
if options & SZ_NN_OPTION_MASK != 0 {
|
||||
flags |= libaec_sys::AEC_DATA_PREPROCESS;
|
||||
}
|
||||
let pad_scanline = !pixels_per_scanline.is_multiple_of(pixels_per_block);
|
||||
let deinterleave = bits_per_pixel == 32 || bits_per_pixel == 64;
|
||||
let bits_per_sample = if deinterleave { 8 } else { bits_per_pixel };
|
||||
let pixel_size = match bits_per_sample {
|
||||
17.. => 4,
|
||||
9.. => 2,
|
||||
_ => 1,
|
||||
};
|
||||
let scanlines = (dest_len / pixel_size).div_ceil(pixels_per_scanline);
|
||||
let buf_size = if pad_scanline {
|
||||
rsi.checked_mul(pixels_per_block)
|
||||
.and_then(|n| n.checked_mul(pixel_size))
|
||||
.and_then(|n| n.checked_mul(scanlines))
|
||||
.filter(|&n| n <= crate::filters::MAX_DECOMPRESS_SIZE.max(limit))
|
||||
.ok_or_else(|| err("scanline padding too large"))?
|
||||
} else {
|
||||
dest_len
|
||||
};
|
||||
|
||||
let mut out = vec![0u8; chunk_size];
|
||||
let mut buf = vec![0u8; buf_size];
|
||||
let mut strm = libaec_sys::AecStream::zeroed();
|
||||
strm.next_in = data.as_ptr();
|
||||
strm.avail_in = data.len();
|
||||
strm.next_out = out.as_mut_ptr();
|
||||
strm.avail_out = chunk_size;
|
||||
strm.next_in = stream.as_ptr();
|
||||
strm.avail_in = stream.len();
|
||||
strm.next_out = buf.as_mut_ptr();
|
||||
strm.avail_out = buf_size;
|
||||
strm.bits_per_sample = bits_per_sample;
|
||||
strm.block_size = pixels_per_block;
|
||||
strm.rsi = 128; // HDF5 default: 128 blocks per reference sample interval
|
||||
strm.block_size = pixels_per_block as u32;
|
||||
strm.rsi = rsi as u32;
|
||||
strm.flags = flags;
|
||||
|
||||
// SAFETY: next_in/avail_in and next_out/avail_out describe live buffers
|
||||
// (`stream` and `buf`) that outlive the call.
|
||||
let result = unsafe { libaec_sys::aec_buffer_decode(&mut strm) };
|
||||
if result != 0 {
|
||||
return Err(FormatError::DecompressionError(format!(
|
||||
"szip: libaec error {result}"
|
||||
)));
|
||||
}
|
||||
let decoded_len = chunk_size - strm.avail_out;
|
||||
out.truncate(decoded_len);
|
||||
let mut total_out = strm.total_out;
|
||||
if pad_scanline {
|
||||
let line = pixels_per_scanline * pixel_size;
|
||||
let padded_line = rsi * pixels_per_block * pixel_size;
|
||||
// remove_padding: compact each padded line down to `line` bytes.
|
||||
let mut i = line;
|
||||
let mut j = padded_line;
|
||||
while j < total_out {
|
||||
let end = (j + line).min(buf.len());
|
||||
buf.copy_within(j..end, i);
|
||||
i += line;
|
||||
j += padded_line;
|
||||
}
|
||||
total_out = scanlines * line;
|
||||
}
|
||||
if total_out < dest_len {
|
||||
return Err(err("stream decoded to fewer bytes than declared"));
|
||||
}
|
||||
buf.truncate(dest_len);
|
||||
if deinterleave {
|
||||
// deinterleave_buffer: byte planes back into words.
|
||||
let w = (bits_per_pixel / 8) as usize;
|
||||
let n = dest_len / w;
|
||||
let mut out = vec![0u8; dest_len];
|
||||
for i in 0..n {
|
||||
for j in 0..w {
|
||||
out[i * w + j] = buf[j * n + i];
|
||||
}
|
||||
}
|
||||
Ok(out)
|
||||
} else {
|
||||
Ok(buf)
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[cfg(feature = "szip")]
|
||||
fn unhex(s: &str) -> Vec<u8> {
|
||||
(0..s.len())
|
||||
.step_by(2)
|
||||
.map(|i| u8::from_str_radix(&s[i..i + 2], 16).unwrap())
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// SZIP chunks written by libhdf5, decoded exactly as libhdf5 decodes
|
||||
/// them. Each case: fixture, chunk byte offset and size (from h5py's
|
||||
/// `get_chunk_info`), the filter's cd_values, and the chunk's values as
|
||||
/// h5py reads them (file byte order, hex). Before the fix every one of
|
||||
/// these came back as garbage or zeros (or "invalid bits per sample" for
|
||||
/// 64-bit): the 4-byte size prefix was fed to libaec, 32/64-bit samples
|
||||
/// were not de-interleaved from byte planes, the reference sample
|
||||
/// interval was fixed at 128 instead of derived from the scanline, padded
|
||||
/// scanlines were not unpadded, and LE data was decoded as MSB.
|
||||
#[cfg(feature = "szip")]
|
||||
#[test]
|
||||
fn szip_decodes_libhdf5_chunks_exactly() {
|
||||
/// (name, file, chunk offset, chunk size, cd_values, decoded hex)
|
||||
type Case<'a> = (&'a str, &'a [u8], usize, usize, [u32; 4], &'a str);
|
||||
let noencoder: &[u8] = include_bytes!("../tests/fixtures/filters/noencoder.h5");
|
||||
let le_data: &[u8] = include_bytes!("../tests/fixtures/filters/le_data.h5");
|
||||
let h5py: &[u8] = include_bytes!("../tests/fixtures/filters/szip_h5py.h5");
|
||||
#[rustfmt::skip]
|
||||
let cases: &[Case] = &[
|
||||
// <i4, 10 px/scanline over 4 px/block: padded scanlines + byte planes.
|
||||
("noencoder /noencoder_szip_dset.h5", noencoder, 6040, 16, [168, 4, 32, 10],
|
||||
"00000000010000000200000003000000040000000500000006000000070000000800000009000000"),
|
||||
// <f4, LSB + NN.
|
||||
("le_data /Szip_float_data_le", le_data, 55224, 48, [169, 4, 32, 12],
|
||||
"abaaaa3eabaa2a3f0000803fabaa2a3f0000803fabaaaa3f0000803fabaaaa3f5555d53fabaaaa3f5555d53f00000040"),
|
||||
// >f4, MSB + NN.
|
||||
("le_data /Szip_float_data_be", le_data, 55396, 48, [177, 4, 32, 12],
|
||||
"3eaaaaab3f2aaaab3f8000003f2aaaab3f8000003faaaaab3f8000003faaaaab3fd555553faaaaab3fd5555540000000"),
|
||||
// <f8 (64-bit), NN.
|
||||
("szip_h5py /f8", h5py, 4016, 100, [169, 8, 64, 10],
|
||||
"00000000000008c000000000000008c000000000000008c000000000000008c000000000000004c000000000000004c000000000000004c000000000000004c000000000000000c000000000000000c000000000000000c000000000000000c0000000000000f8bf000000000000f8bf000000000000f8bf000000000000f8bf000000000000f0bf000000000000f0bf000000000000f0bf000000000000f0bf000000000000e0bf000000000000e0bf000000000000e0bf000000000000e0bf0000000000000000000000000000000000000000000000000000000000000000000000000000e03f000000000000e03f000000000000e03f000000000000e03f000000000000f03f000000000000f03f000000000000f03f000000000000f03f000000000000f83f000000000000f83f000000000000f83f000000000000f83f"),
|
||||
// <i8 (64-bit), entropy coding without NN.
|
||||
("szip_h5py /i8", h5py, 4188, 53, [141, 4, 64, 10],
|
||||
"000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000300000000000000030000000000000003000000000000000300000000000000030000000000000003000000000000000300000000000000030000000000000006000000000000000600000000000000060000000000000006000000000000000600000000000000060000000000000006000000000000000600000000000000090000000000000009000000000000000900000000000000090000000000000009000000000000000900000000000000090000000000000009000000000000000c000000000000000c000000000000000c000000000000000c000000000000000c000000000000000c000000000000000c000000000000000c00000000000000"),
|
||||
// <u2, 35 px/scanline over 8 px/block: padded scanlines, 16-bit samples.
|
||||
("szip_h5py /u2", h5py, 4308, 43, [169, 8, 16, 35],
|
||||
"00000000000000006100610061006100c200c200c200c20023012301230123018401840184018401e501e501e501e5014602460246024602a702a702a702a702080308030803"),
|
||||
];
|
||||
for (name, file, off, len, cd, want) in cases {
|
||||
let want = unhex(want);
|
||||
let got = szip_decompress(&file[*off..off + len], cd, want.len())
|
||||
.unwrap_or_else(|e| panic!("{name}: {e:?}"));
|
||||
assert_eq!(got, want, "{name}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn szip_disabled_returns_unsupported() {
|
||||
#[cfg(not(feature = "szip"))]
|
||||
@@ -132,6 +264,8 @@ mod tests {
|
||||
assert_eq!(rc, 0, "aec_buffer_encode failed: {rc}");
|
||||
let enc_len = encoded.len() - enc.avail_out;
|
||||
encoded.truncate(enc_len);
|
||||
// H5Zszip.c prefixes the stream with the uncompressed size.
|
||||
encoded.splice(0..0, (original.len() as u32).to_le_bytes());
|
||||
|
||||
// Decode through our public interface.
|
||||
// cd[0]=0 (no NN bit 0x20), cd[1]=8 (ppb), cd[2]=8 (bpp), cd[3]=1024 (pps).
|
||||
@@ -163,6 +297,8 @@ mod tests {
|
||||
assert_eq!(rc, 0, "aec_buffer_encode with NN failed: {rc}");
|
||||
let enc_len = encoded.len() - enc.avail_out;
|
||||
encoded.truncate(enc_len);
|
||||
// H5Zszip.c prefixes the stream with the uncompressed size.
|
||||
encoded.splice(0..0, (original.len() as u32).to_le_bytes());
|
||||
|
||||
// cd[0] = 0x20 (H5_SZIP_NN_OPTION_MASK) → decoder must set AEC_DATA_PREPROCESS.
|
||||
let cd = [0x20u32, 8, 8, 1024];
|
||||
|
||||
@@ -6,6 +6,7 @@ extern crate alloc;
|
||||
#[cfg(not(feature = "std"))]
|
||||
use alloc::{format, vec, vec::Vec};
|
||||
|
||||
use crate::chunk_grid::ChunkGrid;
|
||||
use crate::chunked_read::ChunkInfo;
|
||||
use crate::error::FormatError;
|
||||
|
||||
@@ -151,13 +152,13 @@ pub fn read_fixed_array_chunks(
|
||||
file_data: &[u8],
|
||||
header: &FixedArrayHeader,
|
||||
dataset_dims: &[u64],
|
||||
max_dims: Option<&[u64]>,
|
||||
chunk_dimensions: &[u32],
|
||||
element_size: u32,
|
||||
offset_size: u8,
|
||||
_length_size: u8,
|
||||
) -> Result<Vec<ChunkInfo>, FormatError> {
|
||||
let db_offset = header.data_block_address as usize;
|
||||
let rank = chunk_dimensions.len();
|
||||
|
||||
// Parse data block header: FADB(4) + version(1) + client_id(1) + header_address(offset_size)
|
||||
let db_header_size = 4 + 1 + 1 + offset_size as usize;
|
||||
@@ -198,19 +199,10 @@ pub fn read_fixed_array_chunks(
|
||||
))
|
||||
};
|
||||
|
||||
// Compute chunk offsets based on index.
|
||||
// Chunks are stored in row-major order within the dataset space.
|
||||
let mut num_chunks_per_dim = Vec::with_capacity(rank);
|
||||
for d_idx in 0..rank {
|
||||
let ch_dim = chunk_dimensions[d_idx] as u64;
|
||||
if ch_dim == 0 {
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"chunk dimension is zero".into(),
|
||||
));
|
||||
}
|
||||
let ds_dim = dataset_dims[d_idx];
|
||||
num_chunks_per_dim.push(ds_dim.div_ceil(ch_dim));
|
||||
}
|
||||
// The index is laid out over the chunk grid of the *maximum* dimensions
|
||||
// (row-major), so a dataset smaller than its maxshape has gaps.
|
||||
let dims_u64: Vec<u64> = chunk_dimensions.iter().map(|&d| d as u64).collect();
|
||||
let grid = ChunkGrid::fixed_array(dataset_dims, max_dims, &dims_u64)?;
|
||||
|
||||
let chunk_byte_size: u64 =
|
||||
chunk_dimensions.iter().map(|&d| d as u64).product::<u64>() * element_size as u64;
|
||||
@@ -226,7 +218,11 @@ pub fn read_fixed_array_chunks(
|
||||
header.element_size,
|
||||
chunk_byte_size,
|
||||
)? {
|
||||
let offsets = index_to_chunk_offsets(i, &num_chunks_per_dim, chunk_dimensions);
|
||||
// A slot beyond the current extent is ignored, as the
|
||||
// library does.
|
||||
let Some(offsets) = grid.offsets(i as u64) else {
|
||||
return Ok(());
|
||||
};
|
||||
chunks.push(ChunkInfo {
|
||||
chunk_size,
|
||||
filter_mask,
|
||||
@@ -367,27 +363,6 @@ fn parse_fa_element(
|
||||
}
|
||||
}
|
||||
|
||||
/// Convert a linear chunk index to N-dimensional chunk offsets in dataset space.
|
||||
fn index_to_chunk_offsets(
|
||||
index: usize,
|
||||
num_chunks_per_dim: &[u64],
|
||||
chunk_dimensions: &[u32],
|
||||
) -> Vec<u64> {
|
||||
let rank = num_chunks_per_dim.len();
|
||||
let mut offsets = vec![0u64; rank];
|
||||
let mut remaining = index as u64;
|
||||
for d in (0..rank).rev() {
|
||||
let nchunks = num_chunks_per_dim[d];
|
||||
if nchunks == 0 {
|
||||
continue;
|
||||
}
|
||||
let chunk_idx = remaining % nchunks;
|
||||
remaining /= nchunks;
|
||||
offsets[d] = chunk_idx * chunk_dimensions[d] as u64;
|
||||
}
|
||||
offsets
|
||||
}
|
||||
|
||||
/// Read a variable-length little-endian unsigned integer.
|
||||
fn read_variable_length(data: &[u8], size: usize) -> Result<u64, FormatError> {
|
||||
if size > 8 || data.len() < size {
|
||||
@@ -416,44 +391,21 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn index_to_offsets_1d() {
|
||||
let num_chunks = vec![5u64];
|
||||
let chunk_dims = vec![20u32];
|
||||
assert_eq!(index_to_chunk_offsets(0, &num_chunks, &chunk_dims), vec![0]);
|
||||
assert_eq!(
|
||||
index_to_chunk_offsets(1, &num_chunks, &chunk_dims),
|
||||
vec![20]
|
||||
);
|
||||
assert_eq!(
|
||||
index_to_chunk_offsets(4, &num_chunks, &chunk_dims),
|
||||
vec![80]
|
||||
);
|
||||
let g = ChunkGrid::fixed_array(&[100], None, &[20]).unwrap();
|
||||
assert_eq!(g.offsets(0).unwrap(), vec![0]);
|
||||
assert_eq!(g.offsets(1).unwrap(), vec![20]);
|
||||
assert_eq!(g.offsets(4).unwrap(), vec![80]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn index_to_offsets_2d() {
|
||||
// 10x6 dataset with 4x3 chunks => ceil(10/4)=3, ceil(6/3)=2 => 6 chunks
|
||||
let num_chunks = vec![3u64, 2];
|
||||
let chunk_dims = vec![4u32, 3];
|
||||
assert_eq!(
|
||||
index_to_chunk_offsets(0, &num_chunks, &chunk_dims),
|
||||
vec![0, 0]
|
||||
);
|
||||
assert_eq!(
|
||||
index_to_chunk_offsets(1, &num_chunks, &chunk_dims),
|
||||
vec![0, 3]
|
||||
);
|
||||
assert_eq!(
|
||||
index_to_chunk_offsets(2, &num_chunks, &chunk_dims),
|
||||
vec![4, 0]
|
||||
);
|
||||
assert_eq!(
|
||||
index_to_chunk_offsets(3, &num_chunks, &chunk_dims),
|
||||
vec![4, 3]
|
||||
);
|
||||
assert_eq!(
|
||||
index_to_chunk_offsets(5, &num_chunks, &chunk_dims),
|
||||
vec![8, 3]
|
||||
);
|
||||
let g = ChunkGrid::fixed_array(&[10, 6], None, &[4, 3]).unwrap();
|
||||
assert_eq!(g.offsets(0).unwrap(), vec![0, 0]);
|
||||
assert_eq!(g.offsets(1).unwrap(), vec![0, 3]);
|
||||
assert_eq!(g.offsets(2).unwrap(), vec![4, 0]);
|
||||
assert_eq!(g.offsets(3).unwrap(), vec![4, 3]);
|
||||
assert_eq!(g.offsets(5).unwrap(), vec![8, 3]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -517,7 +469,7 @@ mod tests {
|
||||
|
||||
let read = |f: &[u8], fahd: usize| -> Result<Vec<ChunkInfo>, FormatError> {
|
||||
let h = FixedArrayHeader::parse(f, fahd, 8, 8)?;
|
||||
read_fixed_array_chunks(f, &h, &[60], &[20], 8, 8, 8)
|
||||
read_fixed_array_chunks(f, &h, &[60], None, &[20], 8, 8, 8)
|
||||
};
|
||||
|
||||
let (clean, fahd) = build();
|
||||
@@ -562,7 +514,7 @@ mod tests {
|
||||
let db = 0x100usize;
|
||||
buf[db..db + 4].copy_from_slice(b"FADB");
|
||||
let header = FixedArrayHeader::parse(&buf, fahd, 8, 8).unwrap();
|
||||
let r = read_fixed_array_chunks(&buf, &header, &[100], &[20], 8, 8, 8);
|
||||
let r = read_fixed_array_chunks(&buf, &header, &[100], None, &[20], 8, 8, 8);
|
||||
assert!(r.is_err());
|
||||
}
|
||||
|
||||
@@ -579,7 +531,7 @@ mod tests {
|
||||
stamp_checksum(&mut buf, fahd, fahd + 24);
|
||||
buf[0x80..0x84].copy_from_slice(b"FADB");
|
||||
let header = FixedArrayHeader::parse(&buf, fahd, 8, 8).unwrap();
|
||||
let r = read_fixed_array_chunks(&buf, &header, &[100], &[20], 8, 8, 8);
|
||||
let r = read_fixed_array_chunks(&buf, &header, &[100], None, &[20], 8, 8, 8);
|
||||
assert!(r.is_err());
|
||||
}
|
||||
|
||||
@@ -602,7 +554,7 @@ mod tests {
|
||||
data_block_address: (usize::MAX - 4) as u64,
|
||||
};
|
||||
let buf = vec![0u8; 64];
|
||||
let r = read_fixed_array_chunks(&buf, &header, &[100], &[20], 8, 8, 8);
|
||||
let r = read_fixed_array_chunks(&buf, &header, &[100], None, &[20], 8, 8, 8);
|
||||
assert!(r.is_err());
|
||||
}
|
||||
|
||||
@@ -664,6 +616,7 @@ mod tests {
|
||||
&file_data,
|
||||
&header,
|
||||
&ds_dims,
|
||||
None,
|
||||
&chunk_dims,
|
||||
8,
|
||||
offset_size,
|
||||
@@ -740,6 +693,7 @@ mod tests {
|
||||
&file_data,
|
||||
&header,
|
||||
&ds_dims,
|
||||
None,
|
||||
&chunk_dims,
|
||||
8,
|
||||
offset_size,
|
||||
@@ -840,6 +794,7 @@ mod tests {
|
||||
&file_data,
|
||||
&header,
|
||||
&ds_dims,
|
||||
None,
|
||||
&chunk_dims,
|
||||
8,
|
||||
offset_size,
|
||||
|
||||
@@ -1,12 +1,14 @@
|
||||
//! HDF5 Fractal Heap parsing for v2 group link storage.
|
||||
|
||||
#[cfg(not(feature = "std"))]
|
||||
use alloc::vec::Vec;
|
||||
use alloc::{format, vec::Vec};
|
||||
|
||||
#[cfg(feature = "checksum")]
|
||||
use byteorder::{ByteOrder, LittleEndian};
|
||||
|
||||
use crate::btree_v2::{BTreeV2Header, collect_btree_v2_records};
|
||||
use crate::error::FormatError;
|
||||
use crate::filter_pipeline::FilterPipeline;
|
||||
|
||||
/// Parsed fractal heap header (signature "FRHP").
|
||||
#[derive(Debug, Clone)]
|
||||
@@ -33,6 +35,23 @@ pub struct FractalHeapHeader {
|
||||
pub current_rows_in_root_indirect_block: u16,
|
||||
/// Total number of managed objects.
|
||||
pub managed_objects_count: u64,
|
||||
/// Address of the v2 B-tree indexing "huge" objects (undefined address
|
||||
/// when the heap has none). Huge objects are those larger than
|
||||
/// `max_managed_object_size`; they live outside the heap's blocks.
|
||||
pub huge_btree_address: u64,
|
||||
/// The heap's I/O filter pipeline, if it has one. It applies to managed
|
||||
/// direct blocks and to huge objects.
|
||||
pub filter_pipeline: Option<FilterPipeline>,
|
||||
/// Stored (filtered) size of the root direct block; meaningful only when
|
||||
/// the heap is filtered and its root is a direct block.
|
||||
pub root_direct_block_filtered_size: u64,
|
||||
/// Filter mask of the root direct block (bit *i* set = filter *i*
|
||||
/// skipped); meaningful only when the heap is filtered.
|
||||
pub root_direct_block_filter_mask: u32,
|
||||
/// Size of addresses in the file ("Size of Offsets").
|
||||
pub offset_size: u8,
|
||||
/// Size of lengths in the file ("Size of Lengths").
|
||||
pub length_size: u8,
|
||||
}
|
||||
|
||||
fn read_offset(data: &[u8], pos: usize, size: u8) -> Result<u64, FormatError> {
|
||||
@@ -79,6 +98,38 @@ fn is_undefined(val: u64, offset_size: u8) -> bool {
|
||||
}
|
||||
}
|
||||
|
||||
/// Little-endian unsigned integer of up to 8 bytes.
|
||||
fn le_uint(bytes: &[u8]) -> u64 {
|
||||
bytes
|
||||
.iter()
|
||||
.take(8)
|
||||
.enumerate()
|
||||
.fold(0u64, |acc, (i, &b)| acc | (u64::from(b) << (i * 8)))
|
||||
}
|
||||
|
||||
fn heap_error(msg: &str) -> FormatError {
|
||||
FormatError::ChunkedReadError(format!("fractal heap: {msg}"))
|
||||
}
|
||||
|
||||
/// Heap ID type, from bits 4-5 of an ID's first byte (libhdf5's
|
||||
/// `H5HF_ID_TYPE_MASK`, 0x30); bits 6-7 are the ID version, which must be 0.
|
||||
const HEAP_ID_MANAGED: u8 = 0;
|
||||
const HEAP_ID_HUGE: u8 = 1;
|
||||
const HEAP_ID_TINY: u8 = 2;
|
||||
|
||||
/// The type (0 managed, 1 huge, 2 tiny) of a heap ID from its first byte,
|
||||
/// refusing an ID version other than 0.
|
||||
fn heap_id_type(first: u8) -> Result<u8, FormatError> {
|
||||
if first >> 6 != 0 {
|
||||
return Err(heap_error("unsupported heap ID version"));
|
||||
}
|
||||
Ok((first >> 4) & 0x03)
|
||||
}
|
||||
|
||||
/// v2 B-tree record types indexing a heap's huge objects.
|
||||
const BTREE_HUGE_INDIRECT: u8 = 1;
|
||||
const BTREE_HUGE_INDIRECT_FILTERED: u8 = 2;
|
||||
|
||||
impl FractalHeapHeader {
|
||||
/// Parse a fractal heap header at the given offset.
|
||||
pub fn parse(
|
||||
@@ -122,11 +173,17 @@ impl FractalHeapHeader {
|
||||
]);
|
||||
pos += 4;
|
||||
|
||||
// Skip several fixed fields: next_huge_object_id(ls), btree_huge_objects_address(os),
|
||||
// free_space_managed_blocks(ls), managed_block_free_space_manager_address(os),
|
||||
// next_huge_object_id (length_size)
|
||||
ensure_len(file_data, pos, ls)?;
|
||||
pos += ls;
|
||||
// btree_huge_objects_address (offset_size)
|
||||
let huge_btree_address = read_offset(file_data, pos, offset_size)?;
|
||||
pos += os;
|
||||
|
||||
// Skip: free_space_managed_blocks(ls), managed_block_free_space_manager_address(os),
|
||||
// managed_space_in_heap(ls), allocated_managed_space_in_heap(ls),
|
||||
// direct_block_allocation_iterator_offset(ls)
|
||||
let skip_size = 5 * ls + 2 * os;
|
||||
let skip_size = 4 * ls + os;
|
||||
ensure_len(file_data, pos, skip_size)?;
|
||||
pos += skip_size;
|
||||
|
||||
@@ -134,14 +191,9 @@ impl FractalHeapHeader {
|
||||
let managed_objects_count = read_offset(file_data, pos, length_size)?;
|
||||
pos += ls;
|
||||
|
||||
// huge_objects_size (length_size)
|
||||
pos += ls;
|
||||
// huge_objects_count (length_size)
|
||||
pos += ls;
|
||||
// tiny_objects_size (length_size)
|
||||
pos += ls;
|
||||
// tiny_objects_count (length_size)
|
||||
pos += ls;
|
||||
// huge_objects_size, huge_objects_count, tiny_objects_size,
|
||||
// tiny_objects_count (length_size each)
|
||||
pos += 4 * ls;
|
||||
|
||||
// table_width (2)
|
||||
ensure_len(file_data, pos, 2)?;
|
||||
@@ -175,16 +227,28 @@ impl FractalHeapHeader {
|
||||
ensure_len(file_data, pos, 2)?;
|
||||
let current_rows_in_root_indirect_block =
|
||||
u16::from_le_bytes([file_data[pos], file_data[pos + 1]]);
|
||||
#[allow(unused_variables, unused_mut, unused_assignments)]
|
||||
let mut pos = pos + 2;
|
||||
pos += 2;
|
||||
|
||||
// Skip IO filter encoded info if present
|
||||
// With I/O filters: root direct block's filtered size (length_size),
|
||||
// its filter mask (4), then the encoded filter pipeline message.
|
||||
let mut filter_pipeline = None;
|
||||
let mut root_direct_block_filtered_size = 0;
|
||||
let mut root_direct_block_filter_mask = 0;
|
||||
if io_filter_encoded_length > 0 {
|
||||
// root_block_filter_info_size (length_size) + filter_mask (4)
|
||||
#[allow(unused_assignments)]
|
||||
{
|
||||
pos += ls + 4;
|
||||
}
|
||||
root_direct_block_filtered_size = read_offset(file_data, pos, length_size)?;
|
||||
pos += ls;
|
||||
ensure_len(file_data, pos, 4)?;
|
||||
root_direct_block_filter_mask = u32::from_le_bytes([
|
||||
file_data[pos],
|
||||
file_data[pos + 1],
|
||||
file_data[pos + 2],
|
||||
file_data[pos + 3],
|
||||
]);
|
||||
pos += 4;
|
||||
let n = io_filter_encoded_length as usize;
|
||||
ensure_len(file_data, pos, n)?;
|
||||
filter_pipeline = Some(FilterPipeline::parse(&file_data[pos..pos + n])?);
|
||||
pos += n;
|
||||
}
|
||||
|
||||
// Validate header checksum
|
||||
@@ -200,6 +264,8 @@ impl FractalHeapHeader {
|
||||
});
|
||||
}
|
||||
}
|
||||
#[cfg(not(feature = "checksum"))]
|
||||
let _ = pos;
|
||||
|
||||
Ok(FractalHeapHeader {
|
||||
heap_id_length,
|
||||
@@ -213,13 +279,19 @@ impl FractalHeapHeader {
|
||||
root_block_address,
|
||||
current_rows_in_root_indirect_block,
|
||||
managed_objects_count,
|
||||
huge_btree_address,
|
||||
filter_pipeline,
|
||||
root_direct_block_filtered_size,
|
||||
root_direct_block_filter_mask,
|
||||
offset_size,
|
||||
length_size,
|
||||
})
|
||||
}
|
||||
|
||||
/// Decode a managed heap ID into (offset_in_heap, object_length).
|
||||
///
|
||||
/// The heap ID layout for managed objects (type 0):
|
||||
/// - Byte 0: bits 6-7 = type (0), bits 4-5 = version (0), bits 0-3 = reserved
|
||||
/// - Byte 0: bits 6-7 = version (0), bits 4-5 = type (0), bits 0-3 = reserved
|
||||
/// - Bytes 1+: offset (max_heap_size bits, LE) then length (remaining bits, LE)
|
||||
pub fn decode_managed_id(&self, id_bytes: &[u8]) -> Result<(u64, u64), FormatError> {
|
||||
if id_bytes.is_empty() {
|
||||
@@ -229,8 +301,8 @@ impl FractalHeapHeader {
|
||||
});
|
||||
}
|
||||
|
||||
let id_type = (id_bytes[0] >> 6) & 0x03;
|
||||
if id_type != 0 {
|
||||
let id_type = heap_id_type(id_bytes[0])?;
|
||||
if id_type != HEAP_ID_MANAGED {
|
||||
return Err(FormatError::InvalidHeapIdType(id_type));
|
||||
}
|
||||
|
||||
@@ -269,12 +341,183 @@ impl FractalHeapHeader {
|
||||
Ok((heap_offset, length_val))
|
||||
}
|
||||
|
||||
/// Read a managed object from the heap given its raw heap ID bytes.
|
||||
/// Read any object from the heap given its raw heap ID bytes: managed
|
||||
/// (stored in the heap's blocks), huge (stored outside them, found
|
||||
/// directly from the ID or through the huge-object v2 B-tree, optionally
|
||||
/// filtered) or tiny (stored in the ID itself).
|
||||
///
|
||||
/// Despite its name this accepts every ID type; `offset_size` must match
|
||||
/// the one the header was parsed with.
|
||||
pub fn read_managed_object(
|
||||
&self,
|
||||
file_data: &[u8],
|
||||
id_bytes: &[u8],
|
||||
offset_size: u8,
|
||||
) -> Result<Vec<u8>, FormatError> {
|
||||
let Some(&first) = id_bytes.first() else {
|
||||
return Err(FormatError::UnexpectedEof {
|
||||
expected: 1,
|
||||
available: 0,
|
||||
});
|
||||
};
|
||||
match heap_id_type(first)? {
|
||||
HEAP_ID_MANAGED => self.read_heap_managed(file_data, id_bytes, offset_size),
|
||||
HEAP_ID_HUGE => self.read_huge_object(file_data, id_bytes),
|
||||
HEAP_ID_TINY => self.read_tiny_object(id_bytes),
|
||||
other => Err(FormatError::InvalidHeapIdType(other)),
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether a huge object's ID holds its address and length directly
|
||||
/// (libhdf5 does this when they fit in the ID), rather than a key into
|
||||
/// the huge-object B-tree.
|
||||
fn huge_ids_direct(&self) -> bool {
|
||||
let room = usize::from(self.heap_id_length).saturating_sub(1);
|
||||
let os = usize::from(self.offset_size);
|
||||
let ls = usize::from(self.length_size);
|
||||
if self.filter_pipeline.is_some() {
|
||||
room >= os + ls + 4 + ls
|
||||
} else {
|
||||
room >= os + ls
|
||||
}
|
||||
}
|
||||
|
||||
/// Read a huge object (heap ID type 1).
|
||||
fn read_huge_object(&self, file_data: &[u8], id: &[u8]) -> Result<Vec<u8>, FormatError> {
|
||||
let os = usize::from(self.offset_size);
|
||||
let ls = usize::from(self.length_size);
|
||||
// (address, stored length, filter mask, decoded length); the last two
|
||||
// only matter for a filtered heap.
|
||||
let (addr, stored_len, mask, mem_len) = if self.huge_ids_direct() {
|
||||
let body = &id[1..];
|
||||
let need = if self.filter_pipeline.is_some() {
|
||||
os + ls + 4 + ls
|
||||
} else {
|
||||
os + ls
|
||||
};
|
||||
ensure_len(body, 0, need)?;
|
||||
let addr = le_uint(&body[..os]);
|
||||
let len = le_uint(&body[os..os + ls]);
|
||||
if self.filter_pipeline.is_some() {
|
||||
let mask = u32::from_le_bytes([
|
||||
body[os + ls],
|
||||
body[os + ls + 1],
|
||||
body[os + ls + 2],
|
||||
body[os + ls + 3],
|
||||
]);
|
||||
let mem = le_uint(&body[os + ls + 4..os + ls + 4 + ls]);
|
||||
(addr, len, mask, mem)
|
||||
} else {
|
||||
(addr, len, 0, len)
|
||||
}
|
||||
} else {
|
||||
let key_len = (usize::from(self.heap_id_length).saturating_sub(1)).min(8);
|
||||
ensure_len(id, 1, key_len)?;
|
||||
let key = le_uint(&id[1..1 + key_len]);
|
||||
self.find_huge_record(file_data, key)?
|
||||
};
|
||||
|
||||
let start = usize::try_from(addr).map_err(|_| heap_error("huge object address"))?;
|
||||
let len = usize::try_from(stored_len).map_err(|_| heap_error("huge object length"))?;
|
||||
ensure_len(file_data, start, len)?;
|
||||
let stored = &file_data[start..start + len];
|
||||
match &self.filter_pipeline {
|
||||
None => Ok(stored.to_vec()),
|
||||
Some(pipeline) => {
|
||||
let mem = usize::try_from(mem_len).map_err(|_| heap_error("huge object size"))?;
|
||||
let out = crate::filters::decompress_chunk_masked(stored, pipeline, mem, 1, mask)?;
|
||||
if out.len() != mem {
|
||||
return Err(heap_error("filtered huge object decoded to the wrong size"));
|
||||
}
|
||||
Ok(out)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Look up huge object `key` in the huge-object v2 B-tree, returning
|
||||
/// (address, stored length, filter mask, decoded length).
|
||||
fn find_huge_record(
|
||||
&self,
|
||||
file_data: &[u8],
|
||||
key: u64,
|
||||
) -> Result<(u64, u64, u32, u64), FormatError> {
|
||||
if is_undefined(self.huge_btree_address, self.offset_size) {
|
||||
return Err(heap_error(
|
||||
"huge object ID but the heap has no huge-object index",
|
||||
));
|
||||
}
|
||||
let hdr = BTreeV2Header::parse(
|
||||
file_data,
|
||||
self.huge_btree_address as usize,
|
||||
self.offset_size,
|
||||
self.length_size,
|
||||
)?;
|
||||
let os = usize::from(self.offset_size);
|
||||
let ls = usize::from(self.length_size);
|
||||
let filtered = self.filter_pipeline.is_some();
|
||||
let (expected_type, rec_len) = if filtered {
|
||||
(BTREE_HUGE_INDIRECT_FILTERED, os + ls + 4 + ls + ls)
|
||||
} else {
|
||||
(BTREE_HUGE_INDIRECT, os + ls + ls)
|
||||
};
|
||||
if hdr.tree_type != expected_type || usize::from(hdr.record_size) < rec_len {
|
||||
return Err(heap_error("unexpected huge-object B-tree record type"));
|
||||
}
|
||||
let records =
|
||||
collect_btree_v2_records(file_data, &hdr, self.offset_size, self.length_size)?;
|
||||
for rec in &records {
|
||||
let d = &rec.data;
|
||||
if d.len() < rec_len {
|
||||
continue;
|
||||
}
|
||||
let addr = le_uint(&d[..os]);
|
||||
let len = le_uint(&d[os..os + ls]);
|
||||
if filtered {
|
||||
let mask = u32::from_le_bytes([
|
||||
d[os + ls],
|
||||
d[os + ls + 1],
|
||||
d[os + ls + 2],
|
||||
d[os + ls + 3],
|
||||
]);
|
||||
let mem = le_uint(&d[os + ls + 4..os + 2 * ls + 4]);
|
||||
let id = le_uint(&d[os + 2 * ls + 4..os + 3 * ls + 4]);
|
||||
if id == key {
|
||||
return Ok((addr, len, mask, mem));
|
||||
}
|
||||
} else {
|
||||
let id = le_uint(&d[os + ls..os + 2 * ls]);
|
||||
if id == key {
|
||||
return Ok((addr, len, 0, len));
|
||||
}
|
||||
}
|
||||
}
|
||||
Err(heap_error("huge object not found in its B-tree"))
|
||||
}
|
||||
|
||||
/// Read a tiny object (heap ID type 2), stored in the ID itself.
|
||||
fn read_tiny_object(&self, id: &[u8]) -> Result<Vec<u8>, FormatError> {
|
||||
// libhdf5 uses a one-byte length (low 4 bits of byte 0) unless the ID
|
||||
// is long enough to need 12 bits, which then borrow byte 1.
|
||||
let extended = usize::from(self.heap_id_length).saturating_sub(1) > 17;
|
||||
let (len, start) = if extended {
|
||||
ensure_len(id, 0, 2)?;
|
||||
(
|
||||
((usize::from(id[0] & 0x0F)) << 8 | usize::from(id[1])) + 1,
|
||||
2,
|
||||
)
|
||||
} else {
|
||||
(usize::from(id[0] & 0x0F) + 1, 1)
|
||||
};
|
||||
ensure_len(id, start, len)?;
|
||||
Ok(id[start..start + len].to_vec())
|
||||
}
|
||||
|
||||
/// Read a managed object (heap ID type 0).
|
||||
fn read_heap_managed(
|
||||
&self,
|
||||
file_data: &[u8],
|
||||
id_bytes: &[u8],
|
||||
offset_size: u8,
|
||||
) -> Result<Vec<u8>, FormatError> {
|
||||
let (heap_offset, obj_len) = self.decode_managed_id(id_bytes)?;
|
||||
|
||||
@@ -289,12 +532,15 @@ impl FractalHeapHeader {
|
||||
// Root is a direct block
|
||||
self.read_from_direct_block(
|
||||
file_data,
|
||||
self.root_block_address as usize,
|
||||
self.starting_block_size,
|
||||
0, // block offset in heap = 0 for root
|
||||
DirectBlock {
|
||||
addr: self.root_block_address as usize,
|
||||
size: self.starting_block_size,
|
||||
heap_offset: 0,
|
||||
filtered_size: self.root_direct_block_filtered_size,
|
||||
filter_mask: self.root_direct_block_filter_mask,
|
||||
},
|
||||
heap_offset,
|
||||
obj_len as usize,
|
||||
offset_size,
|
||||
)
|
||||
} else {
|
||||
// Root is an indirect block — limit recursion to 64 levels
|
||||
@@ -313,27 +559,41 @@ impl FractalHeapHeader {
|
||||
|
||||
/// Read an object from a direct block.
|
||||
///
|
||||
/// The heap offset is relative to the start of the block (including its header),
|
||||
/// so we just add it to the block address minus the block's heap offset.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
/// The heap offset is relative to the start of the block (including its
|
||||
/// header), so we just add it to the block address minus the block's heap
|
||||
/// offset. A filtered heap stores each direct block (header included)
|
||||
/// through its filter pipeline, so the block is decoded first.
|
||||
fn read_from_direct_block(
|
||||
&self,
|
||||
file_data: &[u8],
|
||||
block_addr: usize,
|
||||
_block_size: u64,
|
||||
block_heap_offset: u64,
|
||||
block: DirectBlock,
|
||||
target_offset: u64,
|
||||
length: usize,
|
||||
_offset_size: u8,
|
||||
) -> Result<Vec<u8>, FormatError> {
|
||||
if target_offset < block_heap_offset {
|
||||
if target_offset < block.heap_offset {
|
||||
return Err(FormatError::UnexpectedEof {
|
||||
expected: block_heap_offset as usize,
|
||||
expected: block.heap_offset as usize,
|
||||
available: target_offset as usize,
|
||||
});
|
||||
}
|
||||
let local_offset = (target_offset - block_heap_offset) as usize;
|
||||
let pos = block_addr
|
||||
let local_offset = (target_offset - block.heap_offset) as usize;
|
||||
if let Some(pipeline) = &self.filter_pipeline {
|
||||
let stored_len = usize::try_from(block.filtered_size)
|
||||
.map_err(|_| heap_error("direct block size"))?;
|
||||
let size = usize::try_from(block.size).map_err(|_| heap_error("direct block size"))?;
|
||||
ensure_len(file_data, block.addr, stored_len)?;
|
||||
let decoded = crate::filters::decompress_chunk_masked(
|
||||
&file_data[block.addr..block.addr + stored_len],
|
||||
pipeline,
|
||||
size,
|
||||
1,
|
||||
block.filter_mask,
|
||||
)?;
|
||||
ensure_len(&decoded, local_offset, length)?;
|
||||
return Ok(decoded[local_offset..local_offset + length].to_vec());
|
||||
}
|
||||
let pos = block
|
||||
.addr
|
||||
.checked_add(local_offset)
|
||||
.ok_or(FormatError::UnexpectedEof {
|
||||
expected: usize::MAX,
|
||||
@@ -371,19 +631,13 @@ impl FractalHeapHeader {
|
||||
let iblock_header = 5 + offset_size as usize + block_offset_bytes;
|
||||
let mut pos = iblock_addr + iblock_header;
|
||||
|
||||
// Compute block sizes for each row using the doubling table
|
||||
let tw = self.table_width as u64;
|
||||
|
||||
let nrows_usize = nrows as usize;
|
||||
|
||||
// Build table of (block_size, heap_offset) for each child entry
|
||||
let mut current_heap_offset = iblock_heap_offset;
|
||||
|
||||
// Rows below max_direct_rows hold direct blocks; rows at/above hold
|
||||
// child indirect blocks. (NOT the FRHP "starting rows" field.)
|
||||
let start_indirect = self.max_direct_rows();
|
||||
|
||||
// Read child addresses for direct block rows
|
||||
let max_direct_rows = nrows_usize.min(start_indirect);
|
||||
|
||||
for row in 0..max_direct_rows {
|
||||
@@ -393,48 +647,66 @@ impl FractalHeapHeader {
|
||||
let child_addr = read_offset(file_data, pos, offset_size)?;
|
||||
pos += offset_size as usize;
|
||||
|
||||
if self.io_filter_encoded_length > 0 {
|
||||
// filtered_size(length_size) + filter_mask(4)
|
||||
// Skip for now - we don't handle filtered direct blocks in fractal heaps
|
||||
pos += 4; // filter_mask - simplified
|
||||
}
|
||||
// A filtered heap stores each direct block's filtered size
|
||||
// (length_size) and filter mask (4) after its address.
|
||||
let (filtered_size, filter_mask) = if self.filter_pipeline.is_some() {
|
||||
let size = read_offset(file_data, pos, self.length_size)?;
|
||||
pos += usize::from(self.length_size);
|
||||
ensure_len(file_data, pos, 4)?;
|
||||
let mask = u32::from_le_bytes([
|
||||
file_data[pos],
|
||||
file_data[pos + 1],
|
||||
file_data[pos + 2],
|
||||
file_data[pos + 3],
|
||||
]);
|
||||
pos += 4;
|
||||
(size, mask)
|
||||
} else {
|
||||
(0, 0)
|
||||
};
|
||||
|
||||
if !is_undefined(child_addr, offset_size) {
|
||||
let block_end = current_heap_offset + block_size;
|
||||
if target_offset >= current_heap_offset && target_offset < block_end {
|
||||
let block_end = current_heap_offset.saturating_add(block_size);
|
||||
if !is_undefined(child_addr, offset_size)
|
||||
&& target_offset >= current_heap_offset
|
||||
&& target_offset < block_end
|
||||
{
|
||||
return self.read_from_direct_block(
|
||||
file_data,
|
||||
child_addr as usize,
|
||||
block_size,
|
||||
current_heap_offset,
|
||||
DirectBlock {
|
||||
addr: child_addr as usize,
|
||||
size: block_size,
|
||||
heap_offset: current_heap_offset,
|
||||
filtered_size,
|
||||
filter_mask,
|
||||
},
|
||||
target_offset,
|
||||
length,
|
||||
offset_size,
|
||||
);
|
||||
}
|
||||
}
|
||||
current_heap_offset += block_size;
|
||||
current_heap_offset = block_end;
|
||||
}
|
||||
}
|
||||
|
||||
// If we have indirect block rows
|
||||
// Rows at and above `start_indirect` hold child indirect blocks. A
|
||||
// child in row r spans exactly that row's block size of heap space,
|
||||
// so it has as many rows as a table of that total size needs.
|
||||
for row in start_indirect..nrows_usize {
|
||||
let _block_size = self.block_size_for_row(row);
|
||||
let child_nrows = row - start_indirect + 1;
|
||||
let child_space = self.block_size_for_row(row);
|
||||
let child_nrows = self.rows_for_size(child_space);
|
||||
|
||||
for _col in 0..tw {
|
||||
let child_addr = read_offset(file_data, pos, offset_size)?;
|
||||
pos += offset_size as usize;
|
||||
|
||||
if !is_undefined(child_addr, offset_size) {
|
||||
// Calculate total heap space covered by this indirect block child
|
||||
let total_child_space = self.indirect_block_heap_size(child_nrows);
|
||||
let block_end = current_heap_offset + total_child_space;
|
||||
if target_offset >= current_heap_offset && target_offset < block_end {
|
||||
let block_end = current_heap_offset.saturating_add(child_space);
|
||||
if !is_undefined(child_addr, offset_size)
|
||||
&& target_offset >= current_heap_offset
|
||||
&& target_offset < block_end
|
||||
{
|
||||
return self.read_from_indirect_block(
|
||||
file_data,
|
||||
child_addr as usize,
|
||||
child_nrows as u16,
|
||||
child_nrows,
|
||||
current_heap_offset,
|
||||
target_offset,
|
||||
length,
|
||||
@@ -442,11 +714,7 @@ impl FractalHeapHeader {
|
||||
depth_remaining - 1,
|
||||
);
|
||||
}
|
||||
current_heap_offset += total_child_space;
|
||||
} else {
|
||||
let total_child_space = self.indirect_block_heap_size(child_nrows);
|
||||
current_heap_offset += total_child_space;
|
||||
}
|
||||
current_heap_offset = block_end;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -475,25 +743,34 @@ impl FractalHeapHeader {
|
||||
log2 + 2
|
||||
}
|
||||
|
||||
/// Rows an indirect block needs to span `size` bytes of heap space:
|
||||
/// `log2(size) - log2(starting_block_size * table_width) + 1`, as
|
||||
/// libhdf5's `H5HF__dtable_size_to_rows`.
|
||||
fn rows_for_size(&self, size: u64) -> u16 {
|
||||
let log2 = |v: u64| 63u32.saturating_sub(v.max(1).leading_zeros());
|
||||
let first_row_bits = log2(self.starting_block_size) + log2(u64::from(self.table_width));
|
||||
(log2(size).saturating_sub(first_row_bits) + 1) as u16
|
||||
}
|
||||
|
||||
/// Get block size for a given row in the doubling table.
|
||||
fn block_size_for_row(&self, row: usize) -> u64 {
|
||||
let sbs = self.starting_block_size;
|
||||
if row <= 1 {
|
||||
sbs
|
||||
} else {
|
||||
sbs * (1u64 << (row - 1))
|
||||
sbs.saturating_mul(1u64.checked_shl((row - 1) as u32).unwrap_or(u64::MAX))
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Total heap space covered by an indirect block with the given number of rows.
|
||||
fn indirect_block_heap_size(&self, nrows: usize) -> u64 {
|
||||
let tw = self.table_width as u64;
|
||||
let mut total = 0u64;
|
||||
for row in 0..nrows {
|
||||
total += self.block_size_for_row(row) * tw;
|
||||
}
|
||||
total
|
||||
}
|
||||
/// A managed direct block's location, extent and (for a filtered heap) its
|
||||
/// stored size and filter mask.
|
||||
struct DirectBlock {
|
||||
addr: usize,
|
||||
size: u64,
|
||||
heap_offset: u64,
|
||||
filtered_size: u64,
|
||||
filter_mask: u32,
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
@@ -641,7 +918,7 @@ mod tests {
|
||||
let hdr = FractalHeapHeader::parse(&file_data, 0, 8, 8).unwrap();
|
||||
|
||||
// Build a managed heap ID:
|
||||
// byte 0: type=0 (bits 6-7 = 00), version=0 (bits 4-5), reserved (bits 0-3)
|
||||
// byte 0: version=0 (bits 6-7), type=0 (bits 4-5), reserved (bits 0-3)
|
||||
// bytes 1-6: offset (max_heap_size=16 bits) then length (remaining bits)
|
||||
// For offset=0, length=13:
|
||||
// payload = offset | (length << 16) = 0 | (13 << 16) = 0x000D0000
|
||||
@@ -705,9 +982,46 @@ mod tests {
|
||||
fn invalid_heap_id_type() {
|
||||
let (file_data, _) = build_simple_heap(8, 8);
|
||||
let hdr = FractalHeapHeader::parse(&file_data, 0, 8, 8).unwrap();
|
||||
// Type = 1 (tiny) in bits 6-7
|
||||
let id = vec![0x40u8, 0, 0, 0, 0, 0, 0]; // bit 6 set = type 1
|
||||
// Type = 1 (huge) in bits 4-5 is not a managed ID
|
||||
let id = vec![0x10u8, 0, 0, 0, 0, 0, 0];
|
||||
let err = hdr.decode_managed_id(&id).unwrap_err();
|
||||
assert_eq!(err, FormatError::InvalidHeapIdType(1));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tiny_object_is_read_from_the_id() {
|
||||
let (file_data, _) = build_simple_heap(8, 8);
|
||||
let hdr = FractalHeapHeader::parse(&file_data, 0, 8, 8).unwrap();
|
||||
// Type 2 (0x20), length - 1 in the low 4 bits, data after.
|
||||
let id = [0x20 | 2, b'a', b'b', b'c', 0, 0, 0];
|
||||
assert_eq!(hdr.read_managed_object(&file_data, &id, 8).unwrap(), b"abc");
|
||||
// A length running past the ID is an error, not a short read.
|
||||
let id = [0x20 | 9, b'a', b'b', b'c', 0, 0, 0];
|
||||
assert!(hdr.read_managed_object(&file_data, &id, 8).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn huge_object_with_a_direct_id() {
|
||||
// With IDs long enough for an address and a length, libhdf5 stores
|
||||
// huge objects' location in the ID instead of the huge-object B-tree.
|
||||
let (mut file_data, _) = build_simple_heap(8, 8);
|
||||
let mut hdr = FractalHeapHeader::parse(&file_data, 0, 8, 8).unwrap();
|
||||
hdr.heap_id_length = 17;
|
||||
file_data[900..905].copy_from_slice(b"huge!");
|
||||
let mut id = vec![0x10u8];
|
||||
id.extend_from_slice(&900u64.to_le_bytes());
|
||||
id.extend_from_slice(&5u64.to_le_bytes());
|
||||
assert_eq!(
|
||||
hdr.read_managed_object(&file_data, &id, 8).unwrap(),
|
||||
b"huge!"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unknown_heap_id_version_is_refused() {
|
||||
let (file_data, _) = build_simple_heap(8, 8);
|
||||
let hdr = FractalHeapHeader::parse(&file_data, 0, 8, 8).unwrap();
|
||||
let id = [0x40u8, 0, 0, 0, 0, 0, 0];
|
||||
assert!(hdr.read_managed_object(&file_data, &id, 8).is_err());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -45,9 +45,16 @@ pub fn resolve_v1_group_entries(
|
||||
)?;
|
||||
|
||||
let mut entries = Vec::new();
|
||||
let mut heap_checked = false;
|
||||
for snod_addr in snod_addrs {
|
||||
let snod = SymbolTableNode::parse(file_data, snod_addr as usize, offset_size)?;
|
||||
for entry in &snod.entries {
|
||||
// Like libhdf5, look at the heap's free list only once a name is
|
||||
// needed: an empty group with a damaged heap still lists.
|
||||
if !heap_checked {
|
||||
heap.validate_free_list(file_data, length_size)?;
|
||||
heap_checked = true;
|
||||
}
|
||||
let name = heap.read_string(file_data, entry.link_name_offset)?;
|
||||
entries.push(GroupEntry {
|
||||
name,
|
||||
@@ -73,6 +80,53 @@ pub fn find_v1_soft_link(
|
||||
offset_size: u8,
|
||||
length_size: u8,
|
||||
) -> Result<Option<String>, FormatError> {
|
||||
let mut found = None;
|
||||
for_each_v1_soft_link(
|
||||
file_data,
|
||||
sym_table_msg,
|
||||
offset_size,
|
||||
length_size,
|
||||
|link_name| link_name == name,
|
||||
|_, target| {
|
||||
found = Some(target);
|
||||
false
|
||||
},
|
||||
)?;
|
||||
Ok(found)
|
||||
}
|
||||
|
||||
/// Every soft link in a v1 group, as `(name, target path)`.
|
||||
pub fn v1_soft_links(
|
||||
file_data: &[u8],
|
||||
sym_table_msg: &SymbolTableMessage,
|
||||
offset_size: u8,
|
||||
length_size: u8,
|
||||
) -> Result<Vec<(String, String)>, FormatError> {
|
||||
let mut links = Vec::new();
|
||||
for_each_v1_soft_link(
|
||||
file_data,
|
||||
sym_table_msg,
|
||||
offset_size,
|
||||
length_size,
|
||||
|_| true,
|
||||
|name, target| {
|
||||
links.push((String::from(name), target));
|
||||
true
|
||||
},
|
||||
)?;
|
||||
Ok(links)
|
||||
}
|
||||
|
||||
/// Visit the soft links of a v1 group whose name passes `wanted`, with their
|
||||
/// target paths, until `visit` returns false.
|
||||
fn for_each_v1_soft_link(
|
||||
file_data: &[u8],
|
||||
sym_table_msg: &SymbolTableMessage,
|
||||
offset_size: u8,
|
||||
length_size: u8,
|
||||
wanted: impl Fn(&str) -> bool,
|
||||
mut visit: impl FnMut(&str, String) -> bool,
|
||||
) -> Result<(), FormatError> {
|
||||
let heap = LocalHeap::parse(
|
||||
file_data,
|
||||
sym_table_msg.local_heap_address as usize,
|
||||
@@ -85,13 +139,19 @@ pub fn find_v1_soft_link(
|
||||
offset_size,
|
||||
length_size,
|
||||
)?;
|
||||
let mut heap_checked = false;
|
||||
for snod_addr in snod_addrs {
|
||||
let snod = SymbolTableNode::parse(file_data, snod_addr as usize, offset_size)?;
|
||||
for entry in &snod.entries {
|
||||
if entry.cache_type != CACHE_TYPE_SOFT_LINK {
|
||||
continue;
|
||||
}
|
||||
if heap.read_string(file_data, entry.link_name_offset)? != name {
|
||||
if !heap_checked {
|
||||
heap.validate_free_list(file_data, length_size)?;
|
||||
heap_checked = true;
|
||||
}
|
||||
let name = heap.read_string(file_data, entry.link_name_offset)?;
|
||||
if !wanted(&name) {
|
||||
continue;
|
||||
}
|
||||
let value_offset = u32::from_le_bytes([
|
||||
@@ -100,12 +160,19 @@ pub fn find_v1_soft_link(
|
||||
entry.scratch_pad[2],
|
||||
entry.scratch_pad[3],
|
||||
]);
|
||||
return heap
|
||||
.read_string(file_data, u64::from(value_offset))
|
||||
.map(Some);
|
||||
let target = heap.read_string(file_data, u64::from(value_offset))?;
|
||||
if !visit(&name, target) {
|
||||
return Ok(());
|
||||
}
|
||||
}
|
||||
Ok(None)
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Whether a v1 symbol-table entry is a soft link (no object header of its
|
||||
/// own; its target path is in the local heap).
|
||||
pub fn is_v1_soft_link(entry: &GroupEntry) -> bool {
|
||||
entry.cache_type == CACHE_TYPE_SOFT_LINK
|
||||
}
|
||||
|
||||
/// Extract the SymbolTableMessage from an object header's messages.
|
||||
|
||||
@@ -38,6 +38,24 @@ pub fn resolve_v2_group_entries(
|
||||
}
|
||||
}
|
||||
|
||||
/// First user-defined link type (HDF5 reserves 2-63; 64 is external).
|
||||
const FIRST_USER_DEFINED_LINK_TYPE: u8 = 65;
|
||||
|
||||
/// Parse a Link message, or `None` for a user-defined link (type 65-255).
|
||||
///
|
||||
/// A user-defined link's target is only meaningful to the application that
|
||||
/// registered its class, so, like libhdf5 without that class, we cannot
|
||||
/// follow it. Leaving it out lets the rest of the group be listed and
|
||||
/// resolved instead of one such link failing the whole group; reserved
|
||||
/// types (2-63) are still an error.
|
||||
fn parse_link(data: &[u8], offset_size: u8) -> Result<Option<LinkMessage>, FormatError> {
|
||||
match LinkMessage::parse(data, offset_size) {
|
||||
Ok(link) => Ok(Some(link)),
|
||||
Err(FormatError::InvalidLinkType(t)) if t >= FIRST_USER_DEFINED_LINK_TYPE => Ok(None),
|
||||
Err(e) => Err(e),
|
||||
}
|
||||
}
|
||||
|
||||
/// Extract link entries from Link messages directly in the object header (compact storage).
|
||||
fn resolve_compact_entries(
|
||||
object_header: &ObjectHeader,
|
||||
@@ -46,7 +64,9 @@ fn resolve_compact_entries(
|
||||
let mut entries = Vec::new();
|
||||
for msg in &object_header.messages {
|
||||
if msg.msg_type == MessageType::Link {
|
||||
let link = LinkMessage::parse(&msg.data, offset_size)?;
|
||||
let Some(link) = parse_link(&msg.data, offset_size)? else {
|
||||
continue;
|
||||
};
|
||||
if let LinkTarget::Hard {
|
||||
object_header_address,
|
||||
} = link.link_target
|
||||
@@ -98,7 +118,9 @@ fn for_each_dense_link(
|
||||
|
||||
// Read managed object from fractal heap
|
||||
let link_data = fh.read_managed_object(file_data, id_bytes, offset_size)?;
|
||||
visit(LinkMessage::parse(&link_data, offset_size)?);
|
||||
if let Some(link) = parse_link(&link_data, offset_size)? {
|
||||
visit(link);
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
@@ -178,7 +200,9 @@ fn find_symbolic_link(
|
||||
} else {
|
||||
for msg in &object_header.messages {
|
||||
if msg.msg_type == MessageType::Link {
|
||||
let link = LinkMessage::parse(&msg.data, offset_size)?;
|
||||
let Some(link) = parse_link(&msg.data, offset_size)? else {
|
||||
continue;
|
||||
};
|
||||
if link.name == name && is_symbolic(&link.link_target) {
|
||||
found = Some(link.link_target);
|
||||
}
|
||||
@@ -231,32 +255,135 @@ pub fn resolve_path_any(
|
||||
superblock: &Superblock,
|
||||
path: &str,
|
||||
) -> Result<u64, FormatError> {
|
||||
resolve_path_following_links(file_data, superblock, path, 0)
|
||||
resolve_path_following_links(
|
||||
file_data,
|
||||
superblock,
|
||||
superblock.root_group_address,
|
||||
path,
|
||||
0,
|
||||
)
|
||||
}
|
||||
|
||||
/// Resolve `path` relative to the group at `group_address` (an absolute path
|
||||
/// starts at the root group instead), following soft links. This is how a
|
||||
/// relative soft link's target is resolved: from the group holding the link.
|
||||
pub fn resolve_path_from(
|
||||
file_data: &[u8],
|
||||
superblock: &Superblock,
|
||||
group_address: u64,
|
||||
path: &str,
|
||||
) -> Result<u64, FormatError> {
|
||||
let start = if path.starts_with('/') {
|
||||
superblock.root_group_address
|
||||
} else {
|
||||
group_address
|
||||
};
|
||||
resolve_path_following_links(file_data, superblock, start, path, 0)
|
||||
}
|
||||
|
||||
/// The children of the group at `group_address` that can be opened, as h5py
|
||||
/// lists them: hard links, and soft links resolved to the object they point
|
||||
/// at (under the soft link's own name). Links that cannot be followed are
|
||||
/// left out rather than failing the listing — a dangling or cyclic soft link
|
||||
/// (h5py lists its name but cannot open it), an external link (another
|
||||
/// file), and a user-defined link. An object header that is not a group has
|
||||
/// no children.
|
||||
///
|
||||
/// Any other error, such as a corrupt structure met while resolving a soft
|
||||
/// link, is returned.
|
||||
pub fn resolve_group_children(
|
||||
file_data: &[u8],
|
||||
superblock: &Superblock,
|
||||
group_address: u64,
|
||||
) -> Result<Vec<GroupEntry>, FormatError> {
|
||||
let os = superblock.offset_size;
|
||||
let ls = superblock.length_size;
|
||||
let header = ObjectHeader::parse(file_data, group_address as usize, os, ls)?;
|
||||
|
||||
let mut entries = Vec::new();
|
||||
let mut soft = Vec::new();
|
||||
if is_v1_group(&header) {
|
||||
let sym_msg = header
|
||||
.messages
|
||||
.iter()
|
||||
.find(|m| m.msg_type == MessageType::SymbolTable)
|
||||
.ok_or_else(|| FormatError::PathNotFound(String::from("no symbol table message")))?;
|
||||
let stm = SymbolTableMessage::parse(&sym_msg.data, os)?;
|
||||
let all = group_v1::resolve_v1_group_entries(file_data, &stm, os, ls)?;
|
||||
if all.iter().any(group_v1::is_v1_soft_link) {
|
||||
soft = group_v1::v1_soft_links(file_data, &stm, os, ls)?;
|
||||
}
|
||||
entries.extend(all.into_iter().filter(|e| !group_v1::is_v1_soft_link(e)));
|
||||
} else if is_v2_group(&header) {
|
||||
let mut visit = |link: LinkMessage| match link.link_target {
|
||||
LinkTarget::Hard {
|
||||
object_header_address,
|
||||
} => entries.push(GroupEntry {
|
||||
name: link.name,
|
||||
object_header_address,
|
||||
cache_type: 0,
|
||||
}),
|
||||
LinkTarget::Soft { target_path } => soft.push((link.name, target_path)),
|
||||
LinkTarget::External { .. } => {}
|
||||
};
|
||||
let link_info = find_link_info(&header, os)?;
|
||||
if let Some(fh_addr) = link_info.fractal_heap_address {
|
||||
for_each_dense_link(file_data, &link_info, fh_addr, os, ls, visit)?;
|
||||
} else {
|
||||
for msg in &header.messages {
|
||||
if msg.msg_type == MessageType::Link
|
||||
&& let Some(link) = parse_link(&msg.data, os)?
|
||||
{
|
||||
visit(link);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (name, target) in soft {
|
||||
match resolve_path_from(file_data, superblock, group_address, &target) {
|
||||
Ok(object_header_address) => entries.push(GroupEntry {
|
||||
name,
|
||||
object_header_address,
|
||||
cache_type: 0,
|
||||
}),
|
||||
// Dangling, cyclic, or ending in another file: not openable here.
|
||||
Err(
|
||||
FormatError::PathNotFound(_)
|
||||
| FormatError::NestingDepthExceeded
|
||||
| FormatError::ExternalLinkUnsupported { .. },
|
||||
) => {}
|
||||
Err(e) => return Err(e),
|
||||
}
|
||||
}
|
||||
Ok(entries)
|
||||
}
|
||||
|
||||
/// Soft links followed while resolving one path. Guards against link cycles
|
||||
/// (`a -> b -> a`), which are legal to create.
|
||||
const MAX_SOFT_LINK_DEPTH: u8 = 16;
|
||||
|
||||
/// Walk `path` from the group at `start`, following soft links.
|
||||
fn resolve_path_following_links(
|
||||
file_data: &[u8],
|
||||
superblock: &Superblock,
|
||||
start: u64,
|
||||
path: &str,
|
||||
depth: u8,
|
||||
) -> Result<u64, FormatError> {
|
||||
let components: Vec<&str> = path.split('/').filter(|s| !s.is_empty()).collect();
|
||||
let components: Vec<&str> = path
|
||||
.split('/')
|
||||
.filter(|s| !s.is_empty() && *s != ".")
|
||||
.collect();
|
||||
if components.is_empty() {
|
||||
return Ok(superblock.root_group_address);
|
||||
return Ok(start);
|
||||
}
|
||||
|
||||
let os = superblock.offset_size;
|
||||
let ls = superblock.length_size;
|
||||
|
||||
let root_header =
|
||||
ObjectHeader::parse(file_data, superblock.root_group_address as usize, os, ls)?;
|
||||
|
||||
let mut current_addr = superblock.root_group_address;
|
||||
let mut current_header = root_header;
|
||||
let mut current_addr = start;
|
||||
let mut current_header = ObjectHeader::parse(file_data, start as usize, os, ls)?;
|
||||
|
||||
for (i, component) in components.iter().enumerate() {
|
||||
let entries = resolve_group_entries(file_data, ¤t_header, os, ls)?;
|
||||
@@ -280,20 +407,17 @@ fn resolve_path_following_links(
|
||||
}
|
||||
// A relative target is relative to the group holding
|
||||
// the link; then the rest of the original path.
|
||||
let mut full = String::new();
|
||||
if !target_path.starts_with('/') {
|
||||
for parent in &components[..i] {
|
||||
full.push('/');
|
||||
full.push_str(parent);
|
||||
}
|
||||
}
|
||||
full.push('/');
|
||||
full.push_str(&target_path);
|
||||
let from = if target_path.starts_with('/') {
|
||||
superblock.root_group_address
|
||||
} else {
|
||||
current_addr
|
||||
};
|
||||
let mut full = target_path;
|
||||
for rest in &components[i + 1..] {
|
||||
full.push('/');
|
||||
full.push_str(rest);
|
||||
}
|
||||
resolve_path_following_links(file_data, superblock, &full, depth + 1)
|
||||
resolve_path_following_links(file_data, superblock, from, &full, depth + 1)
|
||||
}
|
||||
Some(LinkTarget::External {
|
||||
filename,
|
||||
|
||||
@@ -26,12 +26,13 @@
|
||||
//! use clawhdf5_format::{signature, superblock, object_header, group_v2,
|
||||
//! datatype, dataspace, data_layout, data_read, message_type::MessageType};
|
||||
//!
|
||||
//! let file_data = std::fs::read("output.h5").unwrap();
|
||||
//! let sig = signature::find_signature(&file_data).unwrap();
|
||||
//! let sb = superblock::Superblock::parse(&file_data, sig).unwrap();
|
||||
//! let addr = group_v2::resolve_path_any(&file_data, &sb, "data").unwrap();
|
||||
//! let bytes = std::fs::read("output.h5").unwrap();
|
||||
//! // Addresses are relative to the superblock: skip any user block.
|
||||
//! let (_user_block, file_data) = signature::split_user_block(&bytes).unwrap();
|
||||
//! let sb = superblock::Superblock::parse(file_data, 0).unwrap();
|
||||
//! let addr = group_v2::resolve_path_any(file_data, &sb, "data").unwrap();
|
||||
//! let hdr = object_header::ObjectHeader::parse(
|
||||
//! &file_data, addr as usize, sb.offset_size, sb.length_size).unwrap();
|
||||
//! file_data, addr as usize, sb.offset_size, sb.length_size).unwrap();
|
||||
//! ```
|
||||
//!
|
||||
//! # Features
|
||||
@@ -54,6 +55,7 @@ pub mod btree_v1;
|
||||
pub mod btree_v2;
|
||||
pub mod checksum;
|
||||
pub mod chunk_cache;
|
||||
mod chunk_grid;
|
||||
pub mod chunk_index;
|
||||
pub mod chunked_read;
|
||||
pub mod chunked_write;
|
||||
@@ -99,6 +101,7 @@ pub mod signature;
|
||||
pub mod superblock;
|
||||
pub mod symbol_table;
|
||||
pub mod type_builders;
|
||||
pub mod vds;
|
||||
pub mod vl_data;
|
||||
|
||||
#[cfg(feature = "provenance")]
|
||||
|
||||
@@ -87,6 +87,57 @@ impl LocalHeap {
|
||||
})
|
||||
}
|
||||
|
||||
/// Walk the free list the way libhdf5 does when it loads a heap's data
|
||||
/// (`H5HL__fl_deserialize`), rejecting a heap whose free list points
|
||||
/// outside the data segment. libhdf5 refuses such a heap ("bad heap free
|
||||
/// list"), and names read from it would be garbage.
|
||||
///
|
||||
/// libhdf5 only loads a heap when it needs a name from it (an empty
|
||||
/// group's broken heap goes unnoticed), so call this before the first
|
||||
/// [`Self::read_string`], not on parse.
|
||||
///
|
||||
/// The end of the list is `H5HL_FREE_NULL` (1); an all-ones value (the
|
||||
/// undefined address) is accepted as "no free list" too.
|
||||
pub fn validate_free_list(&self, file_data: &[u8], length_size: u8) -> Result<(), FormatError> {
|
||||
const FREE_NULL: u64 = 1;
|
||||
let ls = length_size as usize;
|
||||
let undefined = if ls >= 8 {
|
||||
u64::MAX
|
||||
} else {
|
||||
(1u64 << (8 * ls)) - 1
|
||||
};
|
||||
let size = self.data_segment_size;
|
||||
let seg = self.data_segment_address;
|
||||
let mut next = self.free_list_head_offset;
|
||||
// Each free block holds two lengths, so a list longer than this
|
||||
// revisits a block: a cycle.
|
||||
let max_blocks = size / (2 * ls as u64) + 1;
|
||||
let mut walked = 0u64;
|
||||
while next != FREE_NULL && next != undefined {
|
||||
if next >= size || walked >= max_blocks {
|
||||
return Err(FormatError::InvalidLocalHeapFreeList);
|
||||
}
|
||||
walked += 1;
|
||||
let at = seg
|
||||
.checked_add(next)
|
||||
.and_then(|a| usize::try_from(a).ok())
|
||||
.ok_or(FormatError::InvalidLocalHeapFreeList)?;
|
||||
let block_offset = next;
|
||||
next = read_offset(file_data, at, length_size)?;
|
||||
if next == 0 {
|
||||
return Err(FormatError::InvalidLocalHeapFreeList);
|
||||
}
|
||||
let block_size = read_offset(file_data, at + ls, length_size)?;
|
||||
if block_offset
|
||||
.checked_add(block_size)
|
||||
.is_none_or(|end| end > size)
|
||||
{
|
||||
return Err(FormatError::InvalidLocalHeapFreeList);
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Read a null-terminated string from the heap's data segment at the given byte offset.
|
||||
pub fn read_string(&self, file_data: &[u8], string_offset: u64) -> Result<String, FormatError> {
|
||||
let seg_addr = self.data_segment_address as usize;
|
||||
@@ -162,8 +213,8 @@ mod tests {
|
||||
// data_segment_size
|
||||
write_val(&mut file, pos, data_seg_size as u64, length_size);
|
||||
pos += length_size as usize;
|
||||
// free_list_head_offset
|
||||
write_val(&mut file, pos, 0xFFFFFFFF, length_size);
|
||||
// free_list_head_offset: H5HL_FREE_NULL (no free space)
|
||||
write_val(&mut file, pos, 1, length_size);
|
||||
pos += length_size as usize;
|
||||
// data_segment_address
|
||||
write_val(&mut file, pos, data_seg_offset as u64, offset_size);
|
||||
@@ -243,6 +294,50 @@ mod tests {
|
||||
assert_eq!(s, "test");
|
||||
}
|
||||
|
||||
/// Heap with data segment `[a, b, c, 0-padding]` whose free list starts
|
||||
/// at `head` and has one block `(next, size)` at offset 8.
|
||||
fn heap_with_free_block(head: u64, next: u64, size: u64) -> Vec<u8> {
|
||||
let mut file = build_heap_file(0, 100, &["abcdefg"], 8, 8);
|
||||
file.resize(200, 0);
|
||||
write_val(&mut file, 8, 32, 8); // data segment size
|
||||
write_val(&mut file, 16, head, 8);
|
||||
write_val(&mut file, 108, next, 8);
|
||||
write_val(&mut file, 116, size, 8);
|
||||
file
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn free_list_inside_the_segment_is_accepted() {
|
||||
let file = heap_with_free_block(8, 1, 24);
|
||||
let heap = LocalHeap::parse(&file, 0, 8, 8).unwrap();
|
||||
heap.validate_free_list(&file, 8).unwrap();
|
||||
assert_eq!(heap.read_string(&file, 0).unwrap(), "abcdefg");
|
||||
// An all-ones head is "no free list" too.
|
||||
let file = heap_with_free_block(u64::MAX, 0, 0);
|
||||
let heap = LocalHeap::parse(&file, 0, 8, 8).unwrap();
|
||||
assert!(heap.validate_free_list(&file, 8).is_ok());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn bad_free_list_is_rejected_like_libhdf5() {
|
||||
for (head, next, size, why) in [
|
||||
(40, 1, 8, "head past the segment"),
|
||||
(8, 1, 25, "block runs past the segment"),
|
||||
(8, 0, 8, "next offset of zero"),
|
||||
(8, 8, 8, "cycle"),
|
||||
(8, 999, 8, "next past the segment"),
|
||||
] {
|
||||
let file = heap_with_free_block(head, next, size);
|
||||
// The header itself parses; the free list is checked on use.
|
||||
let heap = LocalHeap::parse(&file, 0, 8, 8).unwrap();
|
||||
assert_eq!(
|
||||
heap.validate_free_list(&file, 8).unwrap_err(),
|
||||
FormatError::InvalidLocalHeapFreeList,
|
||||
"{why}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn invalid_version() {
|
||||
let mut file = build_heap_file(0, 100, &["x"], 8, 8);
|
||||
|
||||
@@ -146,12 +146,7 @@ impl ObjectHeader {
|
||||
ensure_len(data, pos, msg_data_size)?;
|
||||
let msg_type = MessageType::from_u16(msg_type_raw);
|
||||
|
||||
// Check if unknown + must-understand (bit 3 of msg_flags)
|
||||
if let MessageType::Unknown(id) = msg_type
|
||||
&& msg_flags & 0x08 != 0
|
||||
{
|
||||
return Err(FormatError::UnsupportedMessage(id));
|
||||
}
|
||||
check_unknown_message(msg_type, msg_flags)?;
|
||||
|
||||
if msg_type != MessageType::Nil {
|
||||
messages.push(HeaderMessage {
|
||||
@@ -229,11 +224,7 @@ impl ObjectHeader {
|
||||
|
||||
let msg_type = MessageType::from_u16(msg_type_raw);
|
||||
|
||||
if let MessageType::Unknown(id) = msg_type
|
||||
&& msg_flags & 0x08 != 0
|
||||
{
|
||||
return Err(FormatError::UnsupportedMessage(id));
|
||||
}
|
||||
check_unknown_message(msg_type, msg_flags)?;
|
||||
|
||||
if msg_type != MessageType::Nil {
|
||||
messages.push(HeaderMessage {
|
||||
@@ -424,11 +415,7 @@ impl ObjectHeader {
|
||||
|
||||
let msg_type = MessageType::from_u16(msg_type_raw);
|
||||
|
||||
if let MessageType::Unknown(id) = msg_type
|
||||
&& msg_flags & 0x08 != 0
|
||||
{
|
||||
return Err(FormatError::UnsupportedMessage(id));
|
||||
}
|
||||
check_unknown_message(msg_type, msg_flags)?;
|
||||
|
||||
let msg_data = data[pos..pos + msg_data_size].to_vec();
|
||||
|
||||
@@ -509,6 +496,24 @@ impl ObjectHeader {
|
||||
}
|
||||
}
|
||||
|
||||
/// Header message flag bit 7: fail if the message is unknown, always.
|
||||
const MSG_FLAG_FAIL_IF_UNKNOWN_ALWAYS: u8 = 0x80;
|
||||
|
||||
/// Refuse an unknown message the file says no reader may skip.
|
||||
///
|
||||
/// The parser only ever reads, so bit 3 (fail only when opened for writing)
|
||||
/// is ignored, as libhdf5 ignores it for a read-only open; bit 7 fails
|
||||
/// regardless of access mode. This had the two the wrong way round, failing
|
||||
/// objects libhdf5 reads and reading ones it refuses (`tbogus.h5`).
|
||||
fn check_unknown_message(msg_type: MessageType, msg_flags: u8) -> Result<(), FormatError> {
|
||||
match msg_type {
|
||||
MessageType::Unknown(id) if msg_flags & MSG_FLAG_FAIL_IF_UNKNOWN_ALWAYS != 0 => {
|
||||
Err(FormatError::UnsupportedMessage(id))
|
||||
}
|
||||
_ => Ok(()),
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
@@ -632,14 +637,38 @@ mod tests {
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_v1_unknown_must_understand_errors() {
|
||||
// Bit 3 of msg_flags = must understand
|
||||
let messages = [(0x00FFu16, &[0xAA][..], 0x08u8)];
|
||||
fn parse_v1_unknown_fail_always_errors() {
|
||||
// Bit 7 of msg_flags = fail if unknown, whatever the access mode.
|
||||
let messages = [(0x00FFu16, &[0xAA][..], 0x80u8)];
|
||||
let data = build_v1_header(&messages, 8, 8);
|
||||
let err = ObjectHeader::parse(&data, 0, 8, 8).unwrap_err();
|
||||
assert_eq!(err, FormatError::UnsupportedMessage(0x00FF));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_v1_unknown_fail_on_write_is_ignored_when_reading() {
|
||||
// Bit 3 = fail if unknown *and the file is opened for writing*. This
|
||||
// parser only reads, so libhdf5 (read-only) opens such an object and
|
||||
// so must we. Bits 4/5 (mark if unknown / was unknown) never fail.
|
||||
for flags in [0x08u8, 0x10, 0x20, 0x38] {
|
||||
let messages = [(0x00FFu16, &[0xAA][..], flags)];
|
||||
let data = build_v1_header(&messages, 8, 8);
|
||||
let hdr = ObjectHeader::parse(&data, 0, 8, 8).unwrap();
|
||||
assert_eq!(hdr.messages[0].msg_type, MessageType::Unknown(0x00FF));
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_v2_unknown_message_flags() {
|
||||
let data = build_v2_header(0x00, &[(0xF0, &[1, 2], 0x08)], None);
|
||||
assert!(ObjectHeader::parse(&data, 0, 8, 8).is_ok());
|
||||
let data = build_v2_header(0x00, &[(0xF0, &[1, 2], 0x80)], None);
|
||||
assert_eq!(
|
||||
ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(),
|
||||
FormatError::UnsupportedMessage(0xF0)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_v2_no_timestamps_one_message() {
|
||||
let data = build_v2_header(0x00, &[(0x01, &[10, 20], 0)], None);
|
||||
|
||||
@@ -1,11 +1,17 @@
|
||||
//! Object header writer for v2 format.
|
||||
|
||||
#[cfg(not(feature = "std"))]
|
||||
use alloc::vec::Vec;
|
||||
use alloc::{format, vec::Vec};
|
||||
|
||||
use crate::checksum::jenkins_lookup3;
|
||||
use crate::error::FormatError;
|
||||
use crate::message_type::MessageType;
|
||||
|
||||
/// Largest message payload a v2 object header can describe: the per-message
|
||||
/// size field is 2 bytes. A bigger message cannot be encoded at all — writing
|
||||
/// its size truncated to 16 bits produced files libhdf5 refuses.
|
||||
pub const MAX_MESSAGE_SIZE: usize = u16::MAX as usize;
|
||||
|
||||
/// Writer for v2 object headers with proper checksums.
|
||||
pub struct ObjectHeaderWriter {
|
||||
messages: Vec<(MessageType, Vec<u8>, u8)>, // (type, data, msg_flags)
|
||||
@@ -30,7 +36,22 @@ impl ObjectHeaderWriter {
|
||||
}
|
||||
|
||||
/// Serialize the complete v2 object header (OHDR + messages + checksum).
|
||||
pub fn serialize(&self) -> Vec<u8> {
|
||||
///
|
||||
/// Fails with [`FormatError::SerializationError`] when a message is larger
|
||||
/// than [`MAX_MESSAGE_SIZE`] (e.g. an attribute over ~64 KiB, which would
|
||||
/// need dense attribute storage), rather than writing a corrupt header.
|
||||
pub fn serialize(&self) -> Result<Vec<u8>, FormatError> {
|
||||
if let Some((msg_type, data, _)) = self
|
||||
.messages
|
||||
.iter()
|
||||
.find(|(_, data, _)| data.len() > MAX_MESSAGE_SIZE)
|
||||
{
|
||||
return Err(FormatError::SerializationError(format!(
|
||||
"{msg_type:?} message is {} bytes; an object header message holds at most \
|
||||
{MAX_MESSAGE_SIZE} bytes",
|
||||
data.len()
|
||||
)));
|
||||
}
|
||||
// Calculate total message bytes: each message has type(1) + size(2) + flags(1) + data
|
||||
let msg_bytes_total: usize = self
|
||||
.messages
|
||||
@@ -80,7 +101,7 @@ impl ObjectHeaderWriter {
|
||||
let checksum = jenkins_lookup3(&buf);
|
||||
buf.extend_from_slice(&checksum.to_le_bytes());
|
||||
|
||||
buf
|
||||
Ok(buf)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -125,15 +146,22 @@ impl BatchObjectHeaderWriter {
|
||||
|
||||
/// Compute the serialized size of each header without actually serializing.
|
||||
/// Returns sizes in the same order as headers were added.
|
||||
pub fn compute_sizes(&self) -> Vec<usize> {
|
||||
self.headers.iter().map(|h| h.serialize().len()).collect()
|
||||
pub fn compute_sizes(&self) -> Result<Vec<usize>, FormatError> {
|
||||
self.headers
|
||||
.iter()
|
||||
.map(|h| h.serialize().map(|b| b.len()))
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Serialize all headers into a single contiguous buffer.
|
||||
/// Returns `(combined_bytes, offsets)` where `offsets[i]` is the byte
|
||||
/// offset of header `i` within the combined buffer.
|
||||
pub fn serialize_all(&self) -> (Vec<u8>, Vec<usize>) {
|
||||
let serialized: Vec<Vec<u8>> = self.headers.iter().map(|h| h.serialize()).collect();
|
||||
pub fn serialize_all(&self) -> Result<(Vec<u8>, Vec<usize>), FormatError> {
|
||||
let serialized: Vec<Vec<u8>> = self
|
||||
.headers
|
||||
.iter()
|
||||
.map(|h| h.serialize())
|
||||
.collect::<Result<_, _>>()?;
|
||||
let total: usize = serialized.iter().map(|s| s.len()).sum();
|
||||
let mut buf = Vec::with_capacity(total);
|
||||
let mut offsets = Vec::with_capacity(serialized.len());
|
||||
@@ -141,7 +169,7 @@ impl BatchObjectHeaderWriter {
|
||||
offsets.push(buf.len());
|
||||
buf.extend_from_slice(s);
|
||||
}
|
||||
(buf, offsets)
|
||||
Ok((buf, offsets))
|
||||
}
|
||||
}
|
||||
|
||||
@@ -159,7 +187,7 @@ mod tests {
|
||||
#[test]
|
||||
fn empty_header_roundtrip() {
|
||||
let writer = ObjectHeaderWriter::new();
|
||||
let bytes = writer.serialize();
|
||||
let bytes = writer.serialize().unwrap();
|
||||
let hdr = ObjectHeader::parse(&bytes, 0, 8, 8).unwrap();
|
||||
assert_eq!(hdr.version, 2);
|
||||
assert_eq!(hdr.messages.len(), 0);
|
||||
@@ -170,7 +198,7 @@ mod tests {
|
||||
let mut writer = ObjectHeaderWriter::new();
|
||||
writer.add_message(MessageType::Dataspace, vec![1, 2, 3, 4]);
|
||||
writer.add_message(MessageType::Datatype, vec![5, 6]);
|
||||
let bytes = writer.serialize();
|
||||
let bytes = writer.serialize().unwrap();
|
||||
let hdr = ObjectHeader::parse(&bytes, 0, 8, 8).unwrap();
|
||||
assert_eq!(hdr.messages.len(), 2);
|
||||
assert_eq!(hdr.messages[0].msg_type, MessageType::Dataspace);
|
||||
@@ -184,12 +212,30 @@ mod tests {
|
||||
let mut writer = ObjectHeaderWriter::new();
|
||||
// Add a message with >255 bytes of payload
|
||||
writer.add_message(MessageType::Datatype, vec![0xAA; 300]);
|
||||
let bytes = writer.serialize();
|
||||
let bytes = writer.serialize().unwrap();
|
||||
let hdr = ObjectHeader::parse(&bytes, 0, 8, 8).unwrap();
|
||||
assert_eq!(hdr.messages.len(), 1);
|
||||
assert_eq!(hdr.messages[0].data.len(), 300);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn oversized_message_is_an_error_not_a_truncated_size() {
|
||||
// 65535 bytes is the largest encodable payload.
|
||||
let mut writer = ObjectHeaderWriter::new();
|
||||
writer.add_message(MessageType::Attribute, vec![0; MAX_MESSAGE_SIZE]);
|
||||
let bytes = writer.serialize().unwrap();
|
||||
let hdr = ObjectHeader::parse(&bytes, 0, 8, 8).unwrap();
|
||||
assert_eq!(hdr.messages[0].data.len(), MAX_MESSAGE_SIZE);
|
||||
|
||||
// One byte more used to be written with its size wrapped to 0.
|
||||
let mut writer = ObjectHeaderWriter::new();
|
||||
writer.add_message(MessageType::Attribute, vec![0; MAX_MESSAGE_SIZE + 1]);
|
||||
assert!(matches!(
|
||||
writer.serialize(),
|
||||
Err(FormatError::SerializationError(_))
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn batch_writer_serialize_all() {
|
||||
let mut batch = BatchObjectHeaderWriter::new();
|
||||
@@ -204,7 +250,7 @@ mod tests {
|
||||
batch.add(w2);
|
||||
assert_eq!(batch.len(), 2);
|
||||
|
||||
let (buf, offsets) = batch.serialize_all();
|
||||
let (buf, offsets) = batch.serialize_all().unwrap();
|
||||
assert_eq!(offsets.len(), 2);
|
||||
assert_eq!(offsets[0], 0);
|
||||
|
||||
@@ -222,7 +268,7 @@ mod tests {
|
||||
fn batch_writer_empty() {
|
||||
let batch = BatchObjectHeaderWriter::new();
|
||||
assert!(batch.is_empty());
|
||||
let (buf, offsets) = batch.serialize_all();
|
||||
let (buf, offsets) = batch.serialize_all().unwrap();
|
||||
assert!(buf.is_empty());
|
||||
assert!(offsets.is_empty());
|
||||
}
|
||||
|
||||
@@ -10,7 +10,7 @@
|
||||
use crate::chunked_read::ChunkInfo;
|
||||
use crate::error::FormatError;
|
||||
use crate::filter_pipeline::FilterPipeline;
|
||||
use crate::filters::decompress_chunk;
|
||||
use crate::filters::decompress_chunk_masked;
|
||||
use crate::lane_partition::{self, LaneStats, PartitionStats};
|
||||
|
||||
/// Threshold: only use parallel decompression when chunk count exceeds this.
|
||||
@@ -84,11 +84,13 @@ pub fn decompress_chunks_lane_partitioned(
|
||||
}
|
||||
let raw_chunk = &file_data[c_addr..c_addr + size];
|
||||
|
||||
let decompressed = if chunk_info.filter_mask == 0 {
|
||||
decompress_chunk(raw_chunk, pipeline, chunk_total_bytes, element_size)?
|
||||
} else {
|
||||
raw_chunk.to_vec()
|
||||
};
|
||||
let decompressed = decompress_chunk_masked(
|
||||
raw_chunk,
|
||||
pipeline,
|
||||
chunk_total_bytes,
|
||||
element_size,
|
||||
chunk_info.filter_mask,
|
||||
)?;
|
||||
|
||||
stats.chunks_processed += 1;
|
||||
stats.compressed_bytes += size as u64;
|
||||
@@ -158,11 +160,13 @@ pub fn decompress_chunks_parallel(
|
||||
}
|
||||
let raw_chunk = &file_data[c_addr..c_addr + size];
|
||||
|
||||
let decompressed = if chunk_info.filter_mask == 0 {
|
||||
decompress_chunk(raw_chunk, pipeline, chunk_total_bytes, element_size)?
|
||||
} else {
|
||||
raw_chunk.to_vec()
|
||||
};
|
||||
let decompressed = decompress_chunk_masked(
|
||||
raw_chunk,
|
||||
pipeline,
|
||||
chunk_total_bytes,
|
||||
element_size,
|
||||
chunk_info.filter_mask,
|
||||
)?;
|
||||
|
||||
Ok(DecompressedChunk {
|
||||
index,
|
||||
@@ -200,11 +204,13 @@ pub fn decompress_chunks_sequential(
|
||||
let raw_chunk = &file_data[c_addr..c_addr + size];
|
||||
|
||||
let decompressed = if let Some(pl) = pipeline {
|
||||
if chunk_info.filter_mask == 0 {
|
||||
decompress_chunk(raw_chunk, pl, chunk_total_bytes, element_size)?
|
||||
} else {
|
||||
raw_chunk.to_vec()
|
||||
}
|
||||
decompress_chunk_masked(
|
||||
raw_chunk,
|
||||
pl,
|
||||
chunk_total_bytes,
|
||||
element_size,
|
||||
chunk_info.filter_mask,
|
||||
)?
|
||||
} else {
|
||||
raw_chunk.to_vec()
|
||||
};
|
||||
|
||||
@@ -22,7 +22,7 @@ use crate::data_read::extract_selection_from_buffer;
|
||||
use crate::dataspace::Dataspace;
|
||||
use crate::error::FormatError;
|
||||
use crate::filter_pipeline::FilterPipeline;
|
||||
use crate::filters::decompress_chunk;
|
||||
use crate::filters::{all_filters_skipped, decompress_chunk_masked};
|
||||
use crate::selection::Selection;
|
||||
|
||||
/// The smallest axis-aligned box containing every selected element, as
|
||||
@@ -325,12 +325,18 @@ pub fn read_selection(
|
||||
expected: at.saturating_add(chunk.chunk_size as usize),
|
||||
available: file_data.len(),
|
||||
})?;
|
||||
// Mirrors the full-read path: a non-zero filter mask means the
|
||||
// chunk was stored unfiltered.
|
||||
// Mirrors the full-read path: filter-mask bit i set means
|
||||
// filter i was not applied to this chunk.
|
||||
let decoded;
|
||||
let data: &[u8] = match pipeline {
|
||||
Some(pl) if chunk.filter_mask == 0 => {
|
||||
decoded = decompress_chunk(raw, pl, chunk_bytes, elem_size as u32)?;
|
||||
Some(pl) if !all_filters_skipped(pl, chunk.filter_mask) => {
|
||||
decoded = decompress_chunk_masked(
|
||||
raw,
|
||||
pl,
|
||||
chunk_bytes,
|
||||
elem_size as u32,
|
||||
chunk.filter_mask,
|
||||
)?;
|
||||
&decoded
|
||||
}
|
||||
_ => raw,
|
||||
|
||||
@@ -43,7 +43,7 @@ impl Default for DatasetCreateProps {
|
||||
fletcher32: false,
|
||||
lz4: false,
|
||||
zstd_level: None,
|
||||
fill_time: FillTime::Alloc,
|
||||
fill_time: FillTime::IfSet,
|
||||
compact: false,
|
||||
alignment: 0,
|
||||
}
|
||||
@@ -335,7 +335,7 @@ mod tests {
|
||||
fn dcpl_defaults() {
|
||||
let dcpl = DatasetCreateProps::new();
|
||||
assert!(dcpl.chunk_dims.is_none());
|
||||
assert_eq!(dcpl.fill_time, FillTime::Alloc);
|
||||
assert_eq!(dcpl.fill_time, FillTime::IfSet);
|
||||
assert!(!dcpl.compact);
|
||||
}
|
||||
|
||||
|
||||
@@ -229,44 +229,47 @@ impl Selection {
|
||||
/// self-describing in length, so the count lets a caller walk a packed list
|
||||
/// of selections — as the Virtual Dataset global-heap block does).
|
||||
///
|
||||
/// Only the forms needed for VDS assembly are decoded: `ALL`, `NONE`, and
|
||||
/// **regular** hyperslabs serialized at **version 3** (the encoding HDF5
|
||||
/// 1.10+/2.0 emit). Point selections, irregular hyperslabs, and older
|
||||
/// hyperslab versions return an error rather than mis-decoding.
|
||||
/// Decodes `ALL`, `NONE`, and hyperslabs at every version libhdf5 writes
|
||||
/// (1: irregular, 4-byte coordinates — the default-format encoding; 2:
|
||||
/// regular, 8-byte; 3: either, variable width). A regular hyperslab maps
|
||||
/// to [`Selection::Hyperslab`]; an *irregular* one (a union of blocks)
|
||||
/// maps to a single-block hyperslab when it has one block, and otherwise to
|
||||
/// [`Selection::Points`] listing the union in row-major order (the order
|
||||
/// libhdf5 iterates it in). Unlimited counts/blocks decode as `u64::MAX`
|
||||
/// (see [`SerializedSelection::decode`] for the raw form). Point
|
||||
/// selections are refused: libhdf5 does not allow them in virtual datasets
|
||||
/// either.
|
||||
pub fn decode_serialized(data: &[u8]) -> Result<(Selection, usize), FormatError> {
|
||||
if data.len() < 8 {
|
||||
return Err(FormatError::UnexpectedEof {
|
||||
expected: 8,
|
||||
available: data.len(),
|
||||
});
|
||||
let (raw, len) = SerializedSelection::decode(data)?;
|
||||
let sel = match raw {
|
||||
SerializedSelection::All => Selection::All,
|
||||
SerializedSelection::None => Selection::None,
|
||||
SerializedSelection::Regular {
|
||||
start,
|
||||
stride,
|
||||
count,
|
||||
block,
|
||||
} => Selection::Hyperslab {
|
||||
start,
|
||||
stride,
|
||||
count,
|
||||
block,
|
||||
},
|
||||
SerializedSelection::Blocks { rank, starts, ends } => {
|
||||
if starts.len() == rank {
|
||||
let block = starts.iter().zip(&ends).map(|(&s, &e)| e - s + 1).collect();
|
||||
Selection::Hyperslab {
|
||||
start: starts,
|
||||
stride: vec![1; rank],
|
||||
count: vec![1; rank],
|
||||
block,
|
||||
}
|
||||
let sel_type = u32::from_le_bytes([data[0], data[1], data[2], data[3]]);
|
||||
let version = u32::from_le_bytes([data[4], data[5], data[6], data[7]]);
|
||||
|
||||
match sel_type {
|
||||
// ALL / NONE: type(4) + version(4) + reserved(4) + length(4) = 16 bytes.
|
||||
3 | 0 => {
|
||||
if data.len() < 16 {
|
||||
return Err(FormatError::UnexpectedEof {
|
||||
expected: 16,
|
||||
available: data.len(),
|
||||
});
|
||||
}
|
||||
let sel = if sel_type == 3 {
|
||||
Selection::All
|
||||
} else {
|
||||
Selection::None
|
||||
Selection::Points(blocks_union_coords(rank, &starts, &ends)?)
|
||||
}
|
||||
}
|
||||
};
|
||||
Ok((sel, 16))
|
||||
}
|
||||
2 => decode_hyperslab_serialized(data, version),
|
||||
1 => Err(FormatError::ChunkedReadError(
|
||||
"VDS point selections are not supported".into(),
|
||||
)),
|
||||
_ => Err(FormatError::ChunkedReadError(
|
||||
"unknown dataspace selection type".into(),
|
||||
)),
|
||||
}
|
||||
Ok((sel, len))
|
||||
}
|
||||
|
||||
/// Enumerate the selected element indices of a **1-D** dataspace of the
|
||||
@@ -314,6 +317,11 @@ impl Selection {
|
||||
"VDS selection rank does not match dataspace rank".into(),
|
||||
));
|
||||
}
|
||||
if count.iter().chain(block.iter()).any(|&v| v == UNLIMITED) {
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"unlimited selection must be clipped before it is enumerated".into(),
|
||||
));
|
||||
}
|
||||
// Selected coordinates along each dimension, in order.
|
||||
let mut per_dim: Vec<Vec<u64>> = Vec::with_capacity(rank);
|
||||
for d in 0..rank {
|
||||
@@ -400,59 +408,174 @@ impl Selection {
|
||||
}
|
||||
}
|
||||
|
||||
/// Decode an `H5S_SEL_HYPER` selection in its serialized form. Only version-3
|
||||
/// **regular** hyperslabs are supported.
|
||||
fn decode_hyperslab_serialized(
|
||||
data: &[u8],
|
||||
version: u32,
|
||||
) -> Result<(Selection, usize), FormatError> {
|
||||
if version != 3 {
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"only version-3 hyperslab selections are supported".into(),
|
||||
));
|
||||
/// Hyperslab count/block value meaning "unlimited" (`H5S_UNLIMITED`).
|
||||
pub const UNLIMITED: u64 = u64::MAX;
|
||||
|
||||
/// Largest number of elements an irregular selection is expanded to when it
|
||||
/// is converted to a point list by [`Selection::decode_serialized`].
|
||||
const MAX_EXPANDED_POINTS: u64 = 1 << 26;
|
||||
|
||||
/// A selection exactly as `H5S_select_serialize` stores it, before it is
|
||||
/// applied to any dataspace.
|
||||
///
|
||||
/// Unlike [`Selection`] this keeps an irregular hyperslab as its list of
|
||||
/// blocks, and a regular hyperslab's count/block may be [`UNLIMITED`] (the
|
||||
/// unlimited selections used by unlimited and "printf" virtual dataset
|
||||
/// mappings).
|
||||
#[derive(Debug, Clone, PartialEq)]
|
||||
pub enum SerializedSelection {
|
||||
/// `H5S_SEL_ALL`.
|
||||
All,
|
||||
/// `H5S_SEL_NONE`.
|
||||
None,
|
||||
/// A regular hyperslab. `count[d]` or `block[d]` may be [`UNLIMITED`].
|
||||
Regular {
|
||||
start: Vec<u64>,
|
||||
stride: Vec<u64>,
|
||||
count: Vec<u64>,
|
||||
block: Vec<u64>,
|
||||
},
|
||||
/// An irregular hyperslab: the union of `starts.len() / rank` blocks, each
|
||||
/// given by its first (`starts`) and last (`ends`, inclusive) coordinate,
|
||||
/// flattened block-major.
|
||||
Blocks {
|
||||
rank: usize,
|
||||
starts: Vec<u64>,
|
||||
ends: Vec<u64>,
|
||||
},
|
||||
}
|
||||
|
||||
fn sel_err(msg: &str) -> FormatError {
|
||||
FormatError::ChunkedReadError(msg.into())
|
||||
}
|
||||
|
||||
/// Bounds-checked little-endian reader over a serialized selection.
|
||||
struct SelReader<'a> {
|
||||
data: &'a [u8],
|
||||
pos: usize,
|
||||
}
|
||||
|
||||
impl SelReader<'_> {
|
||||
fn take(&mut self, n: usize) -> Result<&[u8], FormatError> {
|
||||
let end = self.pos.checked_add(n).filter(|&e| e <= self.data.len());
|
||||
let end = end.ok_or(FormatError::UnexpectedEof {
|
||||
expected: self.pos.saturating_add(n),
|
||||
available: self.data.len(),
|
||||
})?;
|
||||
let s = &self.data[self.pos..end];
|
||||
self.pos = end;
|
||||
Ok(s)
|
||||
}
|
||||
// type(4) ver(4) flags(1) enc_size(1) rank(4) [start,stride,count,block]*rank
|
||||
if data.len() < 14 {
|
||||
return Err(FormatError::UnexpectedEof {
|
||||
expected: 14,
|
||||
available: data.len(),
|
||||
});
|
||||
|
||||
fn uint(&mut self, size: usize) -> Result<u64, FormatError> {
|
||||
let bytes = self.take(size)?;
|
||||
Ok(bytes
|
||||
.iter()
|
||||
.enumerate()
|
||||
.fold(0u64, |v, (i, &b)| v | (b as u64) << (i * 8)))
|
||||
}
|
||||
let flags = data[8];
|
||||
let enc_size = data[9] as usize;
|
||||
// Bit 0 set => regular hyperslab. Irregular hyperslabs list explicit blocks.
|
||||
if flags & 0x01 == 0 {
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"irregular VDS hyperslab selections are not supported".into(),
|
||||
));
|
||||
|
||||
fn remaining(&self) -> usize {
|
||||
self.data.len() - self.pos
|
||||
}
|
||||
if enc_size != 2 && enc_size != 4 && enc_size != 8 {
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"unsupported hyperslab coordinate encoding size".into(),
|
||||
));
|
||||
}
|
||||
let rank = u32::from_le_bytes([data[10], data[11], data[12], data[13]]) as usize;
|
||||
// HDF5 caps dataspace rank at 32 (H5S_MAX_RANK). Reject anything larger so a
|
||||
// corrupt rank can't drive a huge allocation or read loop.
|
||||
if rank > 32 {
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"hyperslab selection rank exceeds maximum (32)".into(),
|
||||
));
|
||||
}
|
||||
let mut pos = 14;
|
||||
let read_coord = |data: &[u8], pos: usize| -> Result<u64, FormatError> {
|
||||
if pos + enc_size > data.len() {
|
||||
return Err(FormatError::UnexpectedEof {
|
||||
expected: pos + enc_size,
|
||||
available: data.len(),
|
||||
});
|
||||
}
|
||||
let mut v = 0u64;
|
||||
for (i, &b) in data[pos..pos + enc_size].iter().enumerate() {
|
||||
v |= (b as u64) << (i * 8);
|
||||
}
|
||||
Ok(v)
|
||||
}
|
||||
|
||||
impl SerializedSelection {
|
||||
/// Decode a serialized selection, returning it and the number of bytes it
|
||||
/// occupies. Mirrors libhdf5's `H5S_select_deserialize`: `ALL`/`NONE` and
|
||||
/// hyperslab versions 1-3 are decoded; point selections (which libhdf5
|
||||
/// refuses in virtual datasets) and malformed input are errors.
|
||||
pub fn decode(data: &[u8]) -> Result<(SerializedSelection, usize), FormatError> {
|
||||
let mut r = SelReader { data, pos: 0 };
|
||||
let sel_type = r.uint(4)?;
|
||||
let version = r.uint(4)?;
|
||||
match sel_type {
|
||||
// ALL / NONE: type(4) + version(4) + reserved(4) + length(4).
|
||||
0 | 3 => {
|
||||
r.take(8)?;
|
||||
let sel = if sel_type == 3 {
|
||||
SerializedSelection::All
|
||||
} else {
|
||||
SerializedSelection::None
|
||||
};
|
||||
Ok((sel, r.pos))
|
||||
}
|
||||
2 => {
|
||||
let sel = decode_hyperslab(&mut r, version)?;
|
||||
Ok((sel, r.pos))
|
||||
}
|
||||
1 => Err(sel_err(
|
||||
"VDS point selections are not supported (libhdf5 rejects them too)",
|
||||
)),
|
||||
_ => Err(sel_err("unknown dataspace selection type")),
|
||||
}
|
||||
}
|
||||
|
||||
/// The single dimension in which this selection is unlimited, if any.
|
||||
pub fn unlimited_dim(&self) -> Option<usize> {
|
||||
match self {
|
||||
SerializedSelection::Regular { count, block, .. } => count
|
||||
.iter()
|
||||
.zip(block)
|
||||
.position(|(&c, &b)| c == UNLIMITED || b == UNLIMITED),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
/// The rank the selection was serialized with (`None` for ALL/NONE, which
|
||||
/// carry no rank).
|
||||
pub fn rank(&self) -> Option<usize> {
|
||||
match self {
|
||||
SerializedSelection::Regular { start, .. } => Some(start.len()),
|
||||
SerializedSelection::Blocks { rank, .. } => Some(*rank),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// `H5S__hyper_deserialize`: after the type and version words.
|
||||
fn decode_hyperslab(r: &mut SelReader, version: u64) -> Result<SerializedSelection, FormatError> {
|
||||
const REGULAR: u8 = 0x01;
|
||||
let (flags, enc_size) = match version {
|
||||
// v1: reserved(4) + length(4), always irregular, 4-byte coordinates.
|
||||
1 => {
|
||||
r.take(8)?;
|
||||
(0u8, 4usize)
|
||||
}
|
||||
// v2: flags(1) + length(4), 8-byte coordinates.
|
||||
2 => {
|
||||
let flags = r.take(1)?[0];
|
||||
r.take(4)?;
|
||||
(flags, 8)
|
||||
}
|
||||
// v3: flags(1) + encoding size(1).
|
||||
3 => {
|
||||
let flags = r.take(1)?[0];
|
||||
let enc = r.take(1)?[0] as usize;
|
||||
(flags, enc)
|
||||
}
|
||||
_ => return Err(sel_err("unsupported hyperslab selection version")),
|
||||
};
|
||||
if flags & !REGULAR != 0 {
|
||||
return Err(sel_err("unknown hyperslab selection flags"));
|
||||
}
|
||||
if !matches!(enc_size, 2 | 4 | 8) {
|
||||
return Err(sel_err("unsupported hyperslab coordinate encoding size"));
|
||||
}
|
||||
let rank = r.uint(4)? as usize;
|
||||
// HDF5 caps dataspace rank at 32 (H5S_MAX_RANK). Reject anything else so a
|
||||
// corrupt rank can't drive a huge allocation or read loop.
|
||||
if rank == 0 || rank > 32 {
|
||||
return Err(sel_err("hyperslab selection rank must be 1..=32"));
|
||||
}
|
||||
// The all-ones value of the encoding width means "unlimited".
|
||||
let unlim_raw = if enc_size == 8 {
|
||||
u64::MAX
|
||||
} else {
|
||||
(1u64 << (enc_size * 8)) - 1
|
||||
};
|
||||
|
||||
if flags & REGULAR != 0 {
|
||||
let (mut start, mut stride, mut count, mut block) = (
|
||||
Vec::with_capacity(rank),
|
||||
Vec::with_capacity(rank),
|
||||
@@ -460,24 +583,104 @@ fn decode_hyperslab_serialized(
|
||||
Vec::with_capacity(rank),
|
||||
);
|
||||
for _ in 0..rank {
|
||||
start.push(read_coord(data, pos)?);
|
||||
pos += enc_size;
|
||||
stride.push(read_coord(data, pos)?);
|
||||
pos += enc_size;
|
||||
count.push(read_coord(data, pos)?);
|
||||
pos += enc_size;
|
||||
block.push(read_coord(data, pos)?);
|
||||
pos += enc_size;
|
||||
start.push(r.uint(enc_size)?);
|
||||
stride.push(r.uint(enc_size)?);
|
||||
let c = r.uint(enc_size)?;
|
||||
count.push(if c == unlim_raw { UNLIMITED } else { c });
|
||||
let b = r.uint(enc_size)?;
|
||||
block.push(if b == unlim_raw { UNLIMITED } else { b });
|
||||
}
|
||||
Ok((
|
||||
Selection::Hyperslab {
|
||||
let unlimited = count
|
||||
.iter()
|
||||
.zip(&block)
|
||||
.filter(|&(&c, &b)| c == UNLIMITED || b == UNLIMITED)
|
||||
.count();
|
||||
if unlimited > 1 {
|
||||
return Err(sel_err(
|
||||
"hyperslab selection is unlimited in more than one dimension",
|
||||
));
|
||||
}
|
||||
for d in 0..rank {
|
||||
// Overlapping blocks are not a valid regular hyperslab.
|
||||
if count[d] > 1 && block[d] != UNLIMITED && block[d] > stride[d] {
|
||||
return Err(sel_err("regular hyperslab blocks overlap"));
|
||||
}
|
||||
}
|
||||
return Ok(SerializedSelection::Regular {
|
||||
start,
|
||||
stride,
|
||||
count,
|
||||
block,
|
||||
},
|
||||
pos,
|
||||
))
|
||||
});
|
||||
}
|
||||
|
||||
// Irregular: number of blocks, then each block's start and end corners.
|
||||
let nblocks = r.uint(enc_size)?;
|
||||
let per_block = (rank * 2 * enc_size) as u64;
|
||||
// Untrusted count: it must fit in what is left of the buffer.
|
||||
if nblocks
|
||||
.checked_mul(per_block)
|
||||
.is_none_or(|need| need > r.remaining() as u64)
|
||||
{
|
||||
return Err(FormatError::UnexpectedEof {
|
||||
expected: r
|
||||
.pos
|
||||
.saturating_add(nblocks.saturating_mul(per_block) as usize),
|
||||
available: r.data.len(),
|
||||
});
|
||||
}
|
||||
let n = nblocks as usize * rank;
|
||||
let (mut starts, mut ends) = (Vec::with_capacity(n), Vec::with_capacity(n));
|
||||
for _ in 0..nblocks {
|
||||
for _ in 0..rank {
|
||||
starts.push(r.uint(enc_size)?);
|
||||
}
|
||||
for _ in 0..rank {
|
||||
ends.push(r.uint(enc_size)?);
|
||||
}
|
||||
}
|
||||
if starts.iter().zip(&ends).any(|(s, e)| e < s) {
|
||||
return Err(sel_err("hyperslab block ends before it starts"));
|
||||
}
|
||||
Ok(SerializedSelection::Blocks { rank, starts, ends })
|
||||
}
|
||||
|
||||
/// The coordinates of the union of the given blocks, in row-major order.
|
||||
fn blocks_union_coords(
|
||||
rank: usize,
|
||||
starts: &[u64],
|
||||
ends: &[u64],
|
||||
) -> Result<Vec<Vec<u64>>, FormatError> {
|
||||
let mut total = 0u64;
|
||||
for (s, e) in starts.chunks_exact(rank).zip(ends.chunks_exact(rank)) {
|
||||
let vol = s
|
||||
.iter()
|
||||
.zip(e)
|
||||
.try_fold(1u64, |acc, (&s, &e)| acc.checked_mul(e - s + 1));
|
||||
total = vol
|
||||
.and_then(|v| total.checked_add(v))
|
||||
.filter(|&t| t <= MAX_EXPANDED_POINTS)
|
||||
.ok_or_else(|| sel_err("irregular hyperslab selection is too large to expand"))?;
|
||||
}
|
||||
let mut out = Vec::with_capacity(total as usize);
|
||||
for (s, e) in starts.chunks_exact(rank).zip(ends.chunks_exact(rank)) {
|
||||
let mut cur = s.to_vec();
|
||||
'block: loop {
|
||||
out.push(cur.clone());
|
||||
for d in (0..rank).rev() {
|
||||
if cur[d] < e[d] {
|
||||
cur[d] += 1;
|
||||
continue 'block;
|
||||
}
|
||||
cur[d] = s[d];
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
// Lexicographic order of coordinates is row-major order.
|
||||
out.sort_unstable();
|
||||
out.dedup();
|
||||
Ok(out)
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -642,11 +845,100 @@ mod tests {
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn decode_irregular_hyperslab_rejected() {
|
||||
fn decode_truncated_irregular_hyperslab_is_error() {
|
||||
// Irregular, rank 1, but the block count is missing.
|
||||
let bytes = [0x02u8, 0, 0, 0, 0x03, 0, 0, 0, 0x00, 0x02, 0x01, 0, 0, 0];
|
||||
assert!(Selection::decode_serialized(&bytes).is_err());
|
||||
}
|
||||
|
||||
/// Version 1 as libhdf5 writes it for the default (earliest) format bounds:
|
||||
/// type, version, reserved(4), length(4), rank(4), nblocks(4), then each
|
||||
/// block's start and inclusive end corner as 4-byte values.
|
||||
fn v1_blocks(rank: u32, blocks: &[(&[u32], &[u32])]) -> Vec<u8> {
|
||||
let mut b = Vec::new();
|
||||
for w in [2u32, 1, 0, 0, rank, blocks.len() as u32] {
|
||||
b.extend_from_slice(&w.to_le_bytes());
|
||||
}
|
||||
for (s, e) in blocks {
|
||||
for v in s.iter().chain(e.iter()) {
|
||||
b.extend_from_slice(&v.to_le_bytes());
|
||||
}
|
||||
}
|
||||
b
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn decode_v1_irregular_single_block() {
|
||||
// Exactly what h5py/HDF5 2.0 writes for `[0:4]` with default libver.
|
||||
let bytes = v1_blocks(1, &[(&[0], &[3])]);
|
||||
let (sel, used) = Selection::decode_serialized(&bytes).unwrap();
|
||||
assert_eq!(used, bytes.len());
|
||||
assert_eq!(sel.iter_linear_1d(8).unwrap(), vec![0, 1, 2, 3]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn decode_v1_irregular_union_is_row_major() {
|
||||
// Blocks given out of order and overlapping still enumerate once each,
|
||||
// in row-major order (libhdf5 iterates the union, not the list).
|
||||
let bytes = v1_blocks(2, &[(&[1, 0], &[1, 1]), (&[0, 2], &[1, 2])]);
|
||||
let (sel, used) = Selection::decode_serialized(&bytes).unwrap();
|
||||
assert_eq!(used, bytes.len());
|
||||
// (0,2) (1,0) (1,1) (1,2) in a 2x3 space.
|
||||
assert_eq!(sel.iter_linear(&[2, 3]).unwrap(), vec![2, 3, 4, 5]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn decode_v2_regular_with_unlimited_count() {
|
||||
// v2: flags(1) + length(4), then 8-byte start/stride/count/block.
|
||||
let mut b = Vec::new();
|
||||
b.extend_from_slice(&2u32.to_le_bytes());
|
||||
b.extend_from_slice(&2u32.to_le_bytes());
|
||||
b.push(0x01);
|
||||
b.extend_from_slice(&36u32.to_le_bytes());
|
||||
b.extend_from_slice(&1u32.to_le_bytes());
|
||||
for v in [0u64, 10, u64::MAX, 10] {
|
||||
b.extend_from_slice(&v.to_le_bytes());
|
||||
}
|
||||
let (raw, used) = SerializedSelection::decode(&b).unwrap();
|
||||
assert_eq!(used, b.len());
|
||||
assert_eq!(raw.unlimited_dim(), Some(0));
|
||||
assert_eq!(
|
||||
raw,
|
||||
SerializedSelection::Regular {
|
||||
start: vec![0],
|
||||
stride: vec![10],
|
||||
count: vec![UNLIMITED],
|
||||
block: vec![10],
|
||||
}
|
||||
);
|
||||
// An unclipped unlimited selection cannot be enumerated.
|
||||
let (sel, _) = Selection::decode_serialized(&b).unwrap();
|
||||
assert!(sel.iter_linear_1d(100).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn decode_v3_two_byte_all_ones_is_unlimited() {
|
||||
let bytes = [
|
||||
0x02, 0, 0, 0, 0x03, 0, 0, 0, 0x01, 0x02, 0x01, 0, 0, 0, //
|
||||
0x00, 0x00, 0x01, 0x00, 0x01, 0x00, 0xFF, 0xFF,
|
||||
];
|
||||
let (raw, _) = SerializedSelection::decode(&bytes).unwrap();
|
||||
assert_eq!(raw.unlimited_dim(), Some(0));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn decode_irregular_block_count_beyond_buffer_is_error() {
|
||||
let mut b = v1_blocks(1, &[(&[0], &[3])]);
|
||||
b[20..24].copy_from_slice(&u32::MAX.to_le_bytes());
|
||||
assert!(Selection::decode_serialized(&b).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn decode_point_selection_is_refused() {
|
||||
let bytes = [1u8, 0, 0, 0, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0];
|
||||
assert!(Selection::decode_serialized(&bytes).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn iter_linear_2d_block_row_major() {
|
||||
// A 2x2 block at the top-left of a 4x4 space => linear 0,1,4,5.
|
||||
|
||||
@@ -154,13 +154,29 @@ pub fn is_shared(msg_flags: u8) -> bool {
|
||||
///
|
||||
/// When the shared flag is set on a message, the data contains a reference
|
||||
/// instead of the actual message content.
|
||||
///
|
||||
/// Assumes the file's length size equals its offset size, which only matters
|
||||
/// for version-1 references; use [`parse_shared_ref_sized`] when the
|
||||
/// superblock's length size is known.
|
||||
pub fn parse_shared_ref(data: &[u8], offset_size: u8) -> Result<SharedMessageRef, FormatError> {
|
||||
parse_shared_ref_sized(data, offset_size, offset_size)
|
||||
}
|
||||
|
||||
/// [`parse_shared_ref`] with the superblock's length size, which locates the
|
||||
/// object header address in a version-1 reference.
|
||||
pub fn parse_shared_ref_sized(
|
||||
data: &[u8],
|
||||
offset_size: u8,
|
||||
length_size: u8,
|
||||
) -> Result<SharedMessageRef, FormatError> {
|
||||
ensure_len(data, 0, 2)?;
|
||||
let version = data[0];
|
||||
let ref_type = data[1];
|
||||
|
||||
// Layouts (HDF5 spec IV.A.2 "Shared Message", and libhdf5's decoder):
|
||||
// v1: version, type, reserved(6), address — always "committed"
|
||||
// v1: version, type, reserved(6), then an old-style symbol table
|
||||
// entry: link-name offset(length_size), object header address,
|
||||
// cache type(4), reserved(4), scratch(16) — always "committed"
|
||||
// v2: version, type, address — always "committed"
|
||||
// v3: version, type, then a fractal-heap ID if type == SOHM, otherwise
|
||||
// an address
|
||||
@@ -177,7 +193,7 @@ pub fn parse_shared_ref(data: &[u8], offset_size: u8) -> Result<SharedMessageRef
|
||||
})
|
||||
};
|
||||
match version {
|
||||
1 => address_at(2 + 6),
|
||||
1 => address_at(2 + 6 + length_size as usize),
|
||||
2 => address_at(2),
|
||||
3 if ref_type == SHARE_TYPE_SOHM => {
|
||||
ensure_len(data, 2, FHEAP_ID_LEN)?;
|
||||
@@ -225,9 +241,12 @@ pub fn parse_sohm_table_message(
|
||||
|
||||
/// Parse the SOHM table structure (signature "SMTB") from the file.
|
||||
///
|
||||
/// Each index entry: index_type(1) + mesg_types(2) + min_mesg_size(4) +
|
||||
/// list_max(2) + btree_min(2) + num_messages(2) + index_addr(offset_size) +
|
||||
/// heap_addr(offset_size)
|
||||
/// Each index entry: version(1) + index_type(1) + mesg_types(2) +
|
||||
/// min_mesg_size(4) + list_max(2) + btree_min(2) + num_messages(2) +
|
||||
/// index_addr(offset_size) + heap_addr(offset_size)
|
||||
///
|
||||
/// The leading per-index version byte (0) was missing here, so every field
|
||||
/// after it was read one byte off — verified against an HDF5 2.0 file.
|
||||
pub fn parse_sohm_table(
|
||||
file_data: &[u8],
|
||||
table_addr: usize,
|
||||
@@ -240,11 +259,16 @@ pub fn parse_sohm_table(
|
||||
}
|
||||
let mut pos = table_addr + 4;
|
||||
let os = offset_size as usize;
|
||||
let entry_size = 1 + 2 + 4 + 2 + 2 + 2 + os + os; // 13 + 2*offset_size
|
||||
let entry_size = 1 + 1 + 2 + 4 + 2 + 2 + 2 + os + os; // 14 + 2*offset_size
|
||||
|
||||
let mut indexes = Vec::with_capacity(nindexes as usize);
|
||||
for _ in 0..nindexes {
|
||||
ensure_len(file_data, pos, entry_size)?;
|
||||
let version = file_data[pos];
|
||||
if version != 0 {
|
||||
return Err(FormatError::InvalidSohmTableVersion(version));
|
||||
}
|
||||
pos += 1;
|
||||
let index_type = file_data[pos];
|
||||
pos += 1;
|
||||
let mesg_types = u16::from_le_bytes([file_data[pos], file_data[pos + 1]]);
|
||||
@@ -381,6 +405,68 @@ pub fn parse_sohm_btree_entries(
|
||||
// ---- SOHM resolution ----
|
||||
|
||||
/// Find the SOHM index that handles the given message type.
|
||||
/// Load a file's SOHM table: superblock → superblock extension → Shared
|
||||
/// Message Table message → SMTB. `Ok(None)` when the file has no superblock
|
||||
/// extension or no shared-message table.
|
||||
pub fn load_sohm_table(
|
||||
file_data: &[u8],
|
||||
offset_size: u8,
|
||||
length_size: u8,
|
||||
) -> Result<Option<SohmTable>, FormatError> {
|
||||
let sig = crate::signature::find_signature(file_data)?;
|
||||
let sb = crate::superblock::Superblock::parse(file_data, sig)?;
|
||||
let Some(ext_addr) = sb
|
||||
.superblock_extension_address
|
||||
.filter(|&a| !is_undefined(a, offset_size))
|
||||
else {
|
||||
return Ok(None);
|
||||
};
|
||||
let ext = ObjectHeader::parse(file_data, ext_addr as usize, offset_size, length_size)?;
|
||||
let Some(msg) = ext
|
||||
.messages
|
||||
.iter()
|
||||
.find(|m| m.msg_type == MessageType::SharedMessageTable)
|
||||
else {
|
||||
return Ok(None);
|
||||
};
|
||||
let table_msg = parse_sohm_table_message(&msg.data, offset_size)?;
|
||||
parse_sohm_table(
|
||||
file_data,
|
||||
table_msg.table_address as usize,
|
||||
table_msg.nindexes,
|
||||
offset_size,
|
||||
)
|
||||
.map(Some)
|
||||
}
|
||||
|
||||
/// Like [`message_data`], but also follows references into the file's SOHM
|
||||
/// heap (shared object header messages), loading the SOHM table on demand.
|
||||
pub fn message_data_with_sohm<'a>(
|
||||
file_data: &[u8],
|
||||
msg: &'a crate::object_header::HeaderMessage,
|
||||
offset_size: u8,
|
||||
length_size: u8,
|
||||
) -> Result<Cow<'a, [u8]>, FormatError> {
|
||||
if !is_shared(msg.flags) {
|
||||
return Ok(Cow::Borrowed(&msg.data));
|
||||
}
|
||||
let shared_ref = parse_shared_ref_sized(&msg.data, offset_size, length_size)?;
|
||||
let table = if shared_ref.heap_id.is_some() {
|
||||
load_sohm_table(file_data, offset_size, length_size)?
|
||||
} else {
|
||||
None
|
||||
};
|
||||
resolve_shared_message_with_sohm(
|
||||
file_data,
|
||||
&shared_ref,
|
||||
msg.msg_type,
|
||||
offset_size,
|
||||
length_size,
|
||||
table.as_ref(),
|
||||
)
|
||||
.map(Cow::Owned)
|
||||
}
|
||||
|
||||
fn find_index_for_msg_type(table: &SohmTable, msg_type: MessageType) -> Option<&SohmIndex> {
|
||||
let type_bit = 1u16 << msg_type.to_u16();
|
||||
table
|
||||
@@ -444,7 +530,7 @@ pub fn message_data<'a>(
|
||||
if !is_shared(msg.flags) {
|
||||
return Ok(Cow::Borrowed(&msg.data));
|
||||
}
|
||||
let shared_ref = parse_shared_ref(&msg.data, offset_size)?;
|
||||
let shared_ref = parse_shared_ref_sized(&msg.data, offset_size, length_size)?;
|
||||
resolve_shared_message(
|
||||
file_data,
|
||||
&shared_ref,
|
||||
@@ -459,7 +545,8 @@ pub fn message_data<'a>(
|
||||
///
|
||||
/// For type 1/3 (shared in another object header), reads the target object header
|
||||
/// and finds the message of the specified type.
|
||||
/// For type 2 (SOHM), uses the fractal heap from the SOHM table.
|
||||
/// For type 2 (SOHM), uses the fractal heap from the file's SOHM table,
|
||||
/// loaded from the superblock extension on demand.
|
||||
pub fn resolve_shared_message(
|
||||
file_data: &[u8],
|
||||
shared_ref: &SharedMessageRef,
|
||||
@@ -467,13 +554,18 @@ pub fn resolve_shared_message(
|
||||
offset_size: u8,
|
||||
length_size: u8,
|
||||
) -> Result<Vec<u8>, FormatError> {
|
||||
let table = if shared_ref.heap_id.is_some() {
|
||||
load_sohm_table(file_data, offset_size, length_size)?
|
||||
} else {
|
||||
None
|
||||
};
|
||||
resolve_shared_message_with_sohm(
|
||||
file_data,
|
||||
shared_ref,
|
||||
target_msg_type,
|
||||
offset_size,
|
||||
length_size,
|
||||
None,
|
||||
table.as_ref(),
|
||||
)
|
||||
}
|
||||
|
||||
@@ -579,15 +671,26 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn parse_v1_ref() {
|
||||
let mut data = Vec::new();
|
||||
data.push(1); // version
|
||||
data.push(0); // type
|
||||
data.extend_from_slice(&[0u8; 6]); // reserved
|
||||
data.extend_from_slice(&0x5678u64.to_le_bytes());
|
||||
// Datatype message of `/group1/dset2` in HDF5's `tcompound.h5`
|
||||
// (written in 2000): version 1, six reserved bytes, then an old-style
|
||||
// symbol table entry — link-name offset 0x10, object header address
|
||||
// 0x590 (the committed datatype `/type1`), cache type, reserved and
|
||||
// scratch.
|
||||
let mut data = vec![1, 0, 0, 0, 0, 0, 0, 0];
|
||||
data.extend_from_slice(&0x10u64.to_le_bytes());
|
||||
data.extend_from_slice(&0x590u64.to_le_bytes());
|
||||
data.extend_from_slice(&[0; 24]);
|
||||
|
||||
let shared = parse_shared_ref(&data, 8).unwrap();
|
||||
let shared = parse_shared_ref_sized(&data, 8, 8).unwrap();
|
||||
assert_eq!(shared.version, 1);
|
||||
assert_eq!(shared.object_header_address, Some(0x5678));
|
||||
assert_eq!(shared.object_header_address, Some(0x590));
|
||||
|
||||
// The name offset is a length: 4 bytes here, then an 8-byte address.
|
||||
let mut data = vec![1, 0, 0, 0, 0, 0, 0, 0];
|
||||
data.extend_from_slice(&0x10u32.to_le_bytes());
|
||||
data.extend_from_slice(&0x590u64.to_le_bytes());
|
||||
let shared = parse_shared_ref_sized(&data, 8, 4).unwrap();
|
||||
assert_eq!(shared.object_header_address, Some(0x590));
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -707,6 +810,7 @@ mod tests {
|
||||
let mut buf = Vec::new();
|
||||
buf.extend_from_slice(b"SMTB");
|
||||
for idx in indexes {
|
||||
buf.push(0); // version
|
||||
buf.push(idx.index_type);
|
||||
buf.extend_from_slice(&idx.mesg_types.to_le_bytes());
|
||||
buf.extend_from_slice(&idx.min_mesg_size.to_le_bytes());
|
||||
|
||||
@@ -11,6 +11,16 @@ pub const HDF5_SIGNATURE: [u8; 8] = [0x89, b'H', b'D', b'F', b'\r', b'\n', 0x1A,
|
||||
/// (powers of two starting at 512, plus offset 0).
|
||||
///
|
||||
/// Returns the byte offset where the signature was found.
|
||||
///
|
||||
/// A non-zero offset means the file starts with a *user block*, and every
|
||||
/// address inside the file is relative to the superblock's position, not to
|
||||
/// byte 0 (libhdf5 uses the signature's position as the base address even
|
||||
/// when the stored base-address field disagrees). The parsers in this crate
|
||||
/// take addresses as indices into `file_data`, so they must be handed the
|
||||
/// bytes from the signature on — use [`split_user_block`]. [`Superblock::parse`]
|
||||
/// refuses a non-zero offset for this reason.
|
||||
///
|
||||
/// [`Superblock::parse`]: crate::superblock::Superblock::parse
|
||||
pub fn find_signature(data: &[u8]) -> Result<usize, FormatError> {
|
||||
// Check offset 0
|
||||
if data.len() >= 8 && data[..8] == HDF5_SIGNATURE {
|
||||
@@ -29,6 +39,17 @@ pub fn find_signature(data: &[u8]) -> Result<usize, FormatError> {
|
||||
Err(FormatError::SignatureNotFound)
|
||||
}
|
||||
|
||||
/// Split a file into its user block and its HDF5 bytes.
|
||||
///
|
||||
/// Returns `(user_block, hdf5)`: `user_block` is everything before the
|
||||
/// superblock signature (empty for most files) and `hdf5` is the rest, in
|
||||
/// which every HDF5 address is a plain index. Pass `hdf5` as `file_data` to
|
||||
/// every parser in this crate, and parse the superblock at offset 0 of it.
|
||||
pub fn split_user_block(data: &[u8]) -> Result<(&[u8], &[u8]), FormatError> {
|
||||
let offset = find_signature(data)?;
|
||||
Ok(data.split_at(offset))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
@@ -88,6 +109,21 @@ mod tests {
|
||||
assert_eq!(find_signature(&data), Err(FormatError::SignatureNotFound));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn split_user_block_rebases_at_the_signature() {
|
||||
let mut data = vec![7u8; 1024];
|
||||
data[512..520].copy_from_slice(&HDF5_SIGNATURE);
|
||||
let (ub, hdf5) = split_user_block(&data).unwrap();
|
||||
assert_eq!(ub.len(), 512);
|
||||
assert_eq!(hdf5.len(), 512);
|
||||
assert_eq!(&hdf5[..8], &HDF5_SIGNATURE);
|
||||
|
||||
data[..8].copy_from_slice(&HDF5_SIGNATURE);
|
||||
let (ub, hdf5) = split_user_block(&data).unwrap();
|
||||
assert!(ub.is_empty());
|
||||
assert_eq!(hdf5.len(), 1024);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn signature_prefers_earliest() {
|
||||
// Signature at both 0 and 512, should return 0
|
||||
|
||||
@@ -39,7 +39,13 @@ pub struct Superblock {
|
||||
pub superblock_extension_address: Option<u64>,
|
||||
/// CRC32C checksum (v2/v3 only).
|
||||
pub checksum: Option<u32>,
|
||||
/// Page size for page-buffer mode (v4 only). `None` for v0–v3.
|
||||
/// Page size of the non-standard "version 4" superblock layout (v4 only).
|
||||
/// `None` for v0–v3.
|
||||
///
|
||||
/// HDF5 has no superblock version 4 — libhdf5 refuses it. A real paged
|
||||
/// file is a v2/v3 superblock whose extension holds a File Space Info
|
||||
/// message (what `FileWriter::with_page_size` writes). This field is kept
|
||||
/// only so such files written by older clawhdf5 versions still parse.
|
||||
pub page_size: Option<u32>,
|
||||
}
|
||||
|
||||
@@ -127,8 +133,9 @@ impl Superblock {
|
||||
|
||||
/// Serialize this superblock to bytes.
|
||||
///
|
||||
/// Writes v2/v3 format, or v4 (with `page_size`) when `self.version == 4`.
|
||||
/// Computes and appends Jenkins lookup3 checksum.
|
||||
/// Writes v2/v3 format, or the non-standard v4 (with `page_size`) when
|
||||
/// `self.version == 4` — which no HDF5 library opens; see
|
||||
/// [`Self::page_size`]. Computes and appends Jenkins lookup3 checksum.
|
||||
pub fn serialize(&self) -> Vec<u8> {
|
||||
let mut buf = Vec::with_capacity(48);
|
||||
buf.extend_from_slice(&HDF5_SIGNATURE);
|
||||
@@ -167,8 +174,18 @@ impl Superblock {
|
||||
|
||||
/// Parse a superblock from `data` starting at `signature_offset`.
|
||||
///
|
||||
/// The signature must be present at the given offset.
|
||||
/// The signature must be present at the given offset, and that offset
|
||||
/// must be 0: every address in an HDF5 file is relative to the
|
||||
/// superblock, so when a file has a user block (signature at 512, 1024,
|
||||
/// …) the caller must pass the bytes from the signature on — see
|
||||
/// [`crate::signature::split_user_block`] — and use that slice as
|
||||
/// `file_data` everywhere. A non-zero offset is refused with
|
||||
/// [`FormatError::UserBlockNotStripped`] because the addresses in the
|
||||
/// returned superblock would otherwise be applied to the wrong bytes.
|
||||
pub fn parse(data: &[u8], signature_offset: usize) -> Result<Superblock, FormatError> {
|
||||
if signature_offset != 0 {
|
||||
return Err(FormatError::UserBlockNotStripped(signature_offset as u64));
|
||||
}
|
||||
let d = data
|
||||
.get(signature_offset..)
|
||||
.ok_or(FormatError::UnexpectedEof {
|
||||
@@ -669,7 +686,16 @@ mod tests {
|
||||
let mut data = vec![0u8; 1024];
|
||||
let v0 = build_v0_bytes(8);
|
||||
data[512..512 + v0.len()].copy_from_slice(&v0);
|
||||
let sb = Superblock::parse(&data, 512).unwrap();
|
||||
// Addresses are relative to the superblock, so parsing in place
|
||||
// (where they would be applied to the whole buffer) is refused...
|
||||
assert_eq!(
|
||||
Superblock::parse(&data, 512),
|
||||
Err(FormatError::UserBlockNotStripped(512))
|
||||
);
|
||||
// ...and the caller parses the bytes from the signature on.
|
||||
let (ub, hdf5) = crate::signature::split_user_block(&data).unwrap();
|
||||
assert_eq!(ub.len(), 512);
|
||||
let sb = Superblock::parse(hdf5, 0).unwrap();
|
||||
assert_eq!(sb.version, 0);
|
||||
assert_eq!(sb.root_group_address, 96);
|
||||
}
|
||||
|
||||
@@ -15,31 +15,83 @@ use crate::datatype::{
|
||||
|
||||
/// Controls when fill values are written to dataset storage.
|
||||
///
|
||||
/// Corresponds to the HDF5 fill value message's "fill time" field.
|
||||
/// Corresponds to the HDF5 fill value message's "fill time" field
|
||||
/// (`H5D_fill_time_t`).
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
|
||||
pub enum FillTime {
|
||||
/// Never write fill values (0x02). Avoids initialization overhead
|
||||
/// for datasets that will be fully written before any read.
|
||||
/// Never write fill values (`H5D_FILL_TIME_NEVER`). Avoids
|
||||
/// initialization overhead for datasets that will be fully written
|
||||
/// before any read.
|
||||
Never,
|
||||
/// Write fill values at allocation time (0x0a). This is the default
|
||||
/// and matches the HDF5 C library's behavior.
|
||||
#[default]
|
||||
/// Write fill values when storage is allocated (`H5D_FILL_TIME_ALLOC`).
|
||||
Alloc,
|
||||
/// Write fill values only when the fill value has been explicitly set (0x06).
|
||||
/// Write fill values at allocation only if one was set explicitly
|
||||
/// (`H5D_FILL_TIME_IFSET`). The default, as in the HDF5 C library.
|
||||
#[default]
|
||||
IfSet,
|
||||
}
|
||||
|
||||
/// Space allocation time written with every fill value message: late
|
||||
/// (`H5D_ALLOC_TIME_LATE`), bits 0-1 of the flags byte.
|
||||
const ALLOC_TIME_LATE: u8 = 2;
|
||||
|
||||
impl FillTime {
|
||||
/// Serialize to the byte used in the fill value message (version 3).
|
||||
/// Serialize to the flags byte of a version 3 fill value message: the
|
||||
/// space allocation time (late) in bits 0-1 and the fill time in bits
|
||||
/// 2-3 (`H5D_FILL_TIME_ALLOC` = 0, `NEVER` = 1, `IFSET` = 2).
|
||||
///
|
||||
/// This used to put `Never` in the ALLOC slot, `Alloc` in IFSET and
|
||||
/// `IfSet` in NEVER, so libhdf5 saw every choice as a different one.
|
||||
pub fn to_byte(self) -> u8 {
|
||||
ALLOC_TIME_LATE | (self.code() << 2)
|
||||
}
|
||||
|
||||
/// Decode the fill time from a version 3 fill value message's flags.
|
||||
pub fn from_byte(flags: u8) -> Option<FillTime> {
|
||||
match (flags >> 2) & 0x03 {
|
||||
0 => Some(FillTime::Alloc),
|
||||
1 => Some(FillTime::Never),
|
||||
2 => Some(FillTime::IfSet),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn code(self) -> u8 {
|
||||
match self {
|
||||
FillTime::Never => 0x02,
|
||||
FillTime::Alloc => 0x0a,
|
||||
FillTime::IfSet => 0x06,
|
||||
FillTime::Alloc => 0,
|
||||
FillTime::Never => 1,
|
||||
FillTime::IfSet => 2,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Serialize a version 3 Fill Value message for a dataset of `dt`: the fill
|
||||
/// time, and the user-defined fill value if there is one (bit 5).
|
||||
pub(crate) fn fill_value_message(
|
||||
fill_time: FillTime,
|
||||
value: Option<&[u8]>,
|
||||
dt: &Datatype,
|
||||
) -> Result<Vec<u8>, crate::error::FormatError> {
|
||||
let mut msg = vec![3, fill_time.to_byte()];
|
||||
if let Some(value) = value {
|
||||
if matches!(dt, Datatype::VariableLength { .. }) {
|
||||
return Err(crate::error::FormatError::SerializationError(
|
||||
"a fill value for a variable-length datatype is not supported".into(),
|
||||
));
|
||||
}
|
||||
if value.len() != dt.type_size() as usize {
|
||||
return Err(crate::error::FormatError::DataSizeMismatch {
|
||||
expected: dt.type_size() as usize,
|
||||
actual: value.len(),
|
||||
});
|
||||
}
|
||||
msg[1] |= 0x20; // fill value defined
|
||||
msg.extend_from_slice(&(value.len() as u32).to_le_bytes());
|
||||
msg.extend_from_slice(value);
|
||||
}
|
||||
Ok(msg)
|
||||
}
|
||||
|
||||
// ---- Datatype constructors ----
|
||||
|
||||
pub fn make_f64_type() -> Datatype {
|
||||
@@ -332,7 +384,11 @@ pub(crate) fn build_attr_message(name: &str, value: &AttrValue) -> AttributeMess
|
||||
raw_data: data.clone(),
|
||||
},
|
||||
AttrValue::String(s) => {
|
||||
let bytes = s.as_bytes();
|
||||
// A fixed-length string type must be at least 1 byte: libhdf5
|
||||
// rejects size 0 ("invalid datatype size") and with it every
|
||||
// attribute on the object. h5py stores "" as one NUL byte.
|
||||
let mut bytes = s.as_bytes().to_vec();
|
||||
bytes.resize(bytes.len().max(1), 0);
|
||||
AttributeMessage {
|
||||
name: name.to_string(),
|
||||
datatype: Datatype::String {
|
||||
@@ -341,11 +397,12 @@ pub(crate) fn build_attr_message(name: &str, value: &AttrValue) -> AttributeMess
|
||||
charset: CharacterSet::Utf8,
|
||||
},
|
||||
dataspace: scalar_ds(),
|
||||
raw_data: bytes.to_vec(),
|
||||
raw_data: bytes,
|
||||
}
|
||||
}
|
||||
AttrValue::StringArray(arr) => {
|
||||
let max_len = arr.iter().map(|s| s.len()).max().unwrap_or(0);
|
||||
// At least 1 byte per element, as for a single string.
|
||||
let max_len = arr.iter().map(|s| s.len()).max().unwrap_or(0).max(1);
|
||||
let mut raw = Vec::new();
|
||||
for s in arr {
|
||||
let mut b = s.as_bytes().to_vec();
|
||||
@@ -431,8 +488,10 @@ pub struct DatasetBuilder {
|
||||
pub(crate) data: Option<Vec<u8>>,
|
||||
pub(crate) attrs: Vec<(String, AttrValue)>,
|
||||
pub(crate) chunk_options: ChunkOptions,
|
||||
/// Controls when fill values are written. Default is `FillTime::Alloc`.
|
||||
/// Controls when fill values are written. Default is `FillTime::IfSet`.
|
||||
pub(crate) fill_time: FillTime,
|
||||
/// User-defined fill value: one element's bytes, as stored.
|
||||
pub(crate) fill_value: Option<Vec<u8>>,
|
||||
/// Use compact (inline) storage: data is stored in the object header.
|
||||
/// Only valid when raw data is <= 65536 bytes and dataset is not chunked.
|
||||
pub(crate) compact: bool,
|
||||
@@ -459,6 +518,7 @@ impl DatasetBuilder {
|
||||
attrs: Vec::new(),
|
||||
chunk_options: ChunkOptions::default(),
|
||||
fill_time: FillTime::default(),
|
||||
fill_value: None,
|
||||
compact: false,
|
||||
alignment: 0,
|
||||
virtual_sources: None,
|
||||
@@ -671,7 +731,12 @@ impl DatasetBuilder {
|
||||
self
|
||||
}
|
||||
|
||||
/// Enable Pcodec lossless numerical compression (clawhdf5 filter ID 32023).
|
||||
/// Enable Pcodec lossless numerical compression (private clawhdf5 filter
|
||||
/// ID 480).
|
||||
///
|
||||
/// **Not interoperable:** pcodec has no registered HDF5 filter ID and no
|
||||
/// libhdf5 plugin, so h5py and other HDF5 readers cannot read the
|
||||
/// dataset — only clawhdf5 built with the `pcodec` feature can.
|
||||
///
|
||||
/// Pcodec achieves 30–94% better compression ratio than Zstd for f32/f64
|
||||
/// columns at 1–5 GiB/s decompression speed (arXiv:2502.06112). Requires
|
||||
@@ -715,10 +780,20 @@ impl DatasetBuilder {
|
||||
self
|
||||
}
|
||||
|
||||
/// Set the dataset's fill value: what readers return for storage that
|
||||
/// was never written (e.g. after the dataset is extended). `value` is one
|
||||
/// element's bytes as stored — the dataset datatype's size and byte order
|
||||
/// (`(-1i32).to_le_bytes()` for an `i32` dataset). A size mismatch, or a
|
||||
/// variable-length datatype, makes `finish` fail.
|
||||
pub fn with_fill_value(&mut self, value: &[u8]) -> &mut Self {
|
||||
self.fill_value = Some(value.to_vec());
|
||||
self
|
||||
}
|
||||
|
||||
/// Use compact (inline) storage for this dataset.
|
||||
///
|
||||
/// The raw data is stored directly in the dataset's object header rather
|
||||
/// than as a separate data blob. Only effective when raw data <= 65536 bytes
|
||||
/// than as a separate data blob. Only effective when raw data <= 65531 bytes
|
||||
/// and the dataset is not chunked.
|
||||
pub fn compact(&mut self) -> &mut Self {
|
||||
self.compact = true;
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -148,7 +148,12 @@ pub fn read_vl_strings(
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
/// Resolve VL byte sequences from raw data.
|
||||
/// Resolve VL sequences from raw data, returning each element's bytes.
|
||||
///
|
||||
/// Each element is the sequence's full encoding — element count × base type
|
||||
/// size bytes, in the base type's byte order — so a sequence of `i32` yields
|
||||
/// four bytes per value. Decode it with the base type (e.g.
|
||||
/// [`crate::data_read::read_as_i64`]).
|
||||
pub fn read_vl_bytes(
|
||||
file_data: &[u8],
|
||||
raw_data: &[u8],
|
||||
@@ -177,8 +182,10 @@ pub fn read_vl_bytes(
|
||||
},
|
||||
)?;
|
||||
|
||||
let len = (vl.length as usize).min(obj.data.len());
|
||||
result.push(obj.data[..len].to_vec());
|
||||
// The heap object holds the whole sequence. `vl.length` counts
|
||||
// elements, not bytes, so it is only the byte length when the base
|
||||
// type is one byte wide.
|
||||
result.push(obj.data.clone());
|
||||
}
|
||||
|
||||
Ok(result)
|
||||
|
||||
@@ -0,0 +1,13 @@
|
||||
# Filter conformance fixtures
|
||||
|
||||
Files written by libhdf5 (and its registered filter plugins), used by the
|
||||
filter regression tests in `src/filters.rs` to compare our decoders against
|
||||
the values h5py/libhdf5 read from the same bytes. Chunk byte ranges quoted in
|
||||
the tests come from h5py's `DatasetID.get_chunk_info`.
|
||||
|
||||
| File | Origin | Licence |
|
||||
|------|--------|---------|
|
||||
| `h5ex_d_lz4.h5` | HDF Group `HDF5Examples/C/H5FLT/tfiles/h5ex_d_lz4.h5` (hdf5 repository) | HDF5 licence (BSD-3-Clause style) |
|
||||
| `noencoder.h5` | HDF Group `test/testfiles/noencoder.h5` (hdf5 repository) | HDF5 licence (BSD-3-Clause style) |
|
||||
| `le_data.h5` | HDF Group `test/testfiles/le_data.h5` (hdf5 repository) | HDF5 licence (BSD-3-Clause style) |
|
||||
| `szip_h5py.h5` | Written for these tests with h5py 3 / libhdf5 2.0.0 (libaec szip): `f8` (8x10, chunks 4x10, `('nn', 8)`), `i8` (8x10, chunks 4x10, `('ec', 4)`), `u2` (70, chunks 35, `('nn', 8)`) | Same as this repository |
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,49 @@
|
||||
"""Generate shared_fill_value.h5: datasets whose Fill Value message is
|
||||
*shared*, in the two ways libhdf5 can share one.
|
||||
|
||||
- /sohm_a, /sohm_b: the file has a shared-object-header-message (SOHM) index
|
||||
for fill values, so libhdf5 stores the fill value (-7, int32) in the SOHM
|
||||
heap and /sohm_b's header holds only a reference to it. Chunked, with only
|
||||
the first chunk written, so the rest reads as the fill value.
|
||||
- /unwritten_a, /unwritten_b: the same, never written: no storage at all,
|
||||
read entirely as the fill value.
|
||||
|
||||
h5py has no API for SOHM indexes, so the file creation property list is
|
||||
configured by calling the libhdf5 bundled in the h5py wheel through ctypes.
|
||||
Written with h5py 3.16.0 / HDF5 2.0.0. Re-run only to regenerate:
|
||||
|
||||
python gen_shared_fill.py shared_fill_value.h5
|
||||
"""
|
||||
import ctypes
|
||||
import glob
|
||||
import os
|
||||
import sys
|
||||
|
||||
import h5py
|
||||
import numpy as np
|
||||
|
||||
libdir = os.path.join(os.path.dirname(os.path.dirname(h5py.__file__)), "h5py.libs")
|
||||
libs = [p for p in glob.glob(os.path.join(libdir, "libhdf5*.so*")) if "_hl" not in os.path.basename(p)]
|
||||
lib = ctypes.CDLL(libs[0])
|
||||
lib.H5open()
|
||||
|
||||
H5O_SHMESG_FILL_FLAG = 1 << 0x0005
|
||||
|
||||
fcpl = h5py.h5p.create(h5py.h5p.FILE_CREATE)
|
||||
lib.H5Pset_shared_mesg_nindexes.argtypes = [ctypes.c_int64, ctypes.c_uint]
|
||||
lib.H5Pset_shared_mesg_index.argtypes = [ctypes.c_int64, ctypes.c_uint, ctypes.c_uint, ctypes.c_uint]
|
||||
assert lib.H5Pset_shared_mesg_nindexes(fcpl.id, 1) >= 0
|
||||
assert lib.H5Pset_shared_mesg_index(fcpl.id, 0, H5O_SHMESG_FILL_FLAG, 0) >= 0
|
||||
|
||||
fapl = h5py.h5p.create(h5py.h5p.FILE_ACCESS)
|
||||
fapl.set_libver_bounds(h5py.h5f.LIBVER_LATEST, h5py.h5f.LIBVER_LATEST)
|
||||
fid = h5py.h5f.create(sys.argv[1].encode(), h5py.h5f.ACC_TRUNC, fcpl=fcpl, fapl=fapl)
|
||||
with h5py.File(fid) as f:
|
||||
# Chunked, with only the first chunk written: the rest reads as fill.
|
||||
# libhdf5 keeps the first copy of a message in its own header; the second
|
||||
# identical one (the `_b` datasets) is the SOHM reference.
|
||||
for name in ("sohm_a", "sohm_b"):
|
||||
d = f.create_dataset(name, shape=(8,), chunks=(4,), dtype="<i4", fillvalue=-7)
|
||||
d[:4] = np.arange(4)
|
||||
for name in ("unwritten_a", "unwritten_b"):
|
||||
f.create_dataset(name, shape=(3,), dtype="<i4", fillvalue=-7)
|
||||
@@ -0,0 +1,13 @@
|
||||
# Legacy (HDF5 1.4/1.6-era) fixtures
|
||||
|
||||
Unmodified copies of the HDF Group's own test files from
|
||||
https://github.com/HDFGroup/hdf5 at a3cf1ea82cc7a66e50029a688121e1b105a7ce88
|
||||
(BSD-style license, see that repository's `LICENSE`). Current libraries cannot
|
||||
write these structures, so they are kept as files.
|
||||
|
||||
| File | Upstream path | Exercises |
|
||||
|---|---|---|
|
||||
| `deflate.h5` | `test/testfiles/deflate.h5` | Data Layout message v1, chunked + deflate (v1 B-tree index) |
|
||||
| `h5ex_g_iterate.h5` | `HDF5Examples/C/H5G/h5ex_g_iterate.h5` | Data Layout message v2, contiguous; an unallocated dataset |
|
||||
| `tarrold.h5` | `test/testfiles/tarrold.h5` | Compound datatype v1 members with legacy array dimensions |
|
||||
| `tcompound.h5` | `tools/test/testfiles/tcompound.h5` | Version-1 shared messages (committed datatypes); compound v1 array members with data |
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
BIN
Binary file not shown.
Binary file not shown.
@@ -83,6 +83,45 @@ fn read_chunked_dataset(file_data: &[u8], dataset_path: &str) -> (Vec<u8>, Datat
|
||||
(raw, datatype, dataspace)
|
||||
}
|
||||
|
||||
/// Helper: read a virtual dataset with `vds::read_virtual_dataset`, giving it
|
||||
/// the dataset's own fill value (same-file sources only).
|
||||
fn read_virtual_fixture(file_data: &[u8], path: &str) -> (Vec<u8>, Datatype) {
|
||||
let sig = find_signature(file_data).unwrap();
|
||||
let sb = Superblock::parse(file_data, sig).unwrap();
|
||||
let addr = resolve_path_any(file_data, &sb, path).unwrap();
|
||||
let hdr =
|
||||
ObjectHeader::parse(file_data, addr as usize, sb.offset_size, sb.length_size).unwrap();
|
||||
let msg = |t: MessageType| hdr.messages.iter().find(|m| m.msg_type == t).unwrap();
|
||||
let ds = Dataspace::parse(&msg(MessageType::Dataspace).data, sb.length_size).unwrap();
|
||||
let (dt, _) = Datatype::parse(&msg(MessageType::Datatype).data).unwrap();
|
||||
let layout = DataLayout::parse(
|
||||
&msg(MessageType::DataLayout).data,
|
||||
sb.offset_size,
|
||||
sb.length_size,
|
||||
)
|
||||
.unwrap();
|
||||
let fill = clawhdf5_format::fill_value::dataset_fill_value_in(
|
||||
file_data,
|
||||
&hdr.messages,
|
||||
sb.offset_size,
|
||||
sb.length_size,
|
||||
)
|
||||
.unwrap();
|
||||
let v = clawhdf5_format::vds::read_virtual_dataset(
|
||||
file_data,
|
||||
&layout,
|
||||
&ds,
|
||||
&dt,
|
||||
fill.as_deref(),
|
||||
sb.offset_size,
|
||||
sb.length_size,
|
||||
None,
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(v.dims, ds.dimensions);
|
||||
(v.data, dt)
|
||||
}
|
||||
|
||||
/// Helper: read any dataset (contiguous or chunked) as f64.
|
||||
fn read_dataset_f64_any(bytes: &[u8], path: &str) -> Vec<f64> {
|
||||
let sig = find_signature(bytes).unwrap();
|
||||
@@ -672,7 +711,7 @@ fn v4_virtual_dataset_same_file_read() {
|
||||
// virt[4:8] <- (unmapped) => fill 0
|
||||
// virt[8:12] <- src_b[0:4] (ALL) => 20,21,22,23
|
||||
let file_data = include_bytes!("fixtures/vds_same_file.h5");
|
||||
let (raw, datatype, _) = read_chunked_dataset(file_data, "virt");
|
||||
let (raw, datatype) = read_virtual_fixture(file_data, "virt");
|
||||
let values = read_as_i32(&raw, &datatype).unwrap();
|
||||
assert_eq!(
|
||||
values,
|
||||
@@ -681,6 +720,38 @@ fn v4_virtual_dataset_same_file_read() {
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn v4_virtual_dataset_raw_api_refuses_to_guess_the_fill_value() {
|
||||
// The raw read API has no fill value message, so a virtual dataset with an
|
||||
// unmapped region is an error there instead of zeros that may be wrong.
|
||||
let file_data = include_bytes!("fixtures/vds_same_file.h5");
|
||||
let sig = find_signature(file_data).unwrap();
|
||||
let sb = Superblock::parse(file_data, sig).unwrap();
|
||||
let addr = resolve_path_any(file_data, &sb, "virt").unwrap();
|
||||
let hdr =
|
||||
ObjectHeader::parse(file_data, addr as usize, sb.offset_size, sb.length_size).unwrap();
|
||||
let msg = |t: MessageType| hdr.messages.iter().find(|m| m.msg_type == t).unwrap();
|
||||
let ds = Dataspace::parse(&msg(MessageType::Dataspace).data, sb.length_size).unwrap();
|
||||
let (dt, _) = Datatype::parse(&msg(MessageType::Datatype).data).unwrap();
|
||||
let layout = DataLayout::parse(
|
||||
&msg(MessageType::DataLayout).data,
|
||||
sb.offset_size,
|
||||
sb.length_size,
|
||||
)
|
||||
.unwrap();
|
||||
let err = read_raw_data_full(
|
||||
file_data,
|
||||
&layout,
|
||||
&ds,
|
||||
&dt,
|
||||
None,
|
||||
sb.offset_size,
|
||||
sb.length_size,
|
||||
)
|
||||
.unwrap_err();
|
||||
assert!(err.to_string().contains("fill value"), "{err}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn v4_virtual_dataset_2d_same_file_read() {
|
||||
// A 4x4 virtual dataset assembled from two 2x2 same-file sources placed as
|
||||
@@ -689,7 +760,7 @@ fn v4_virtual_dataset_2d_same_file_read() {
|
||||
// virt[2:4,2:4] <- src_b = [[5,6],[7,8]]
|
||||
// everything else -> fill 0
|
||||
let file_data = include_bytes!("fixtures/vds_2d_same_file.h5");
|
||||
let (raw, datatype, _) = read_chunked_dataset(file_data, "virt");
|
||||
let (raw, datatype) = read_virtual_fixture(file_data, "virt");
|
||||
let values = read_as_i32(&raw, &datatype).unwrap();
|
||||
assert_eq!(
|
||||
values,
|
||||
|
||||
@@ -312,3 +312,49 @@ fn provenance_mismatch_on_corruption() {
|
||||
"corrupted data should produce hash mismatch"
|
||||
);
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Fuzzer finds, kept as regression tests
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// `fuzz_btree_v2` crash input from 2026-09-20 (82 bytes): a B-tree v2 header
|
||||
/// followed by internal nodes that point back into themselves. It predates the
|
||||
/// depth cap and record budget added to B-tree v2 traversal that day and no
|
||||
/// longer crashes; this replays the fuzz target's exact code path on it so a
|
||||
/// regression fails CI rather than waiting for a fuzz run.
|
||||
#[test]
|
||||
fn fuzz_btree_v2_crash_f98c19dc_is_a_clean_result() {
|
||||
use clawhdf5_format::btree_v2::{BTreeV2Header, collect_btree_v2_records};
|
||||
let data: &[u8] = &[
|
||||
0x42, 0x54, 0x48, 0x44, 0x00, 0x06, 0x00, 0xed, 0xef, 0x00, 0x00, 0x00, 0x00, 0x01, 0x00,
|
||||
0x00, 0x03, 0x40, 0x14, 0x93, 0x42, 0x54, 0x49, 0x4e, 0x42, 0x00, 0x00, 0x00, 0x00, 0x00,
|
||||
0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
|
||||
0x00, 0x00, 0x00, 0x00, 0x42, 0x54, 0x48, 0x44, 0x00, 0x00, 0x00, 0x13, 0x05, 0x00, 0x00,
|
||||
0x00, 0x00, 0x00, 0x00, 0x80, 0x00, 0x00, 0x00, 0x40, 0x14, 0x93, 0x42, 0x54, 0x00, 0x49,
|
||||
0x00, 0x01, 0x4e, 0x42, 0x42, 0x54, 0xbe,
|
||||
];
|
||||
assert_eq!(data.len(), 82);
|
||||
for offset_size in [4u8, 8] {
|
||||
for length_size in [4u8, 8] {
|
||||
if let Ok(header) = BTreeV2Header::parse(data, 0, offset_size, length_size) {
|
||||
let _ = collect_btree_v2_records(data, &header, offset_size, length_size);
|
||||
}
|
||||
}
|
||||
}
|
||||
let (fields, file) = data.split_first_chunk::<20>().unwrap();
|
||||
let header = BTreeV2Header {
|
||||
tree_type: fields[0],
|
||||
node_size: u32::from_le_bytes([fields[1], fields[2], fields[3], fields[4]]),
|
||||
record_size: u16::from_le_bytes([fields[5], fields[6]]),
|
||||
depth: u16::from_le_bytes([fields[7], fields[8]]),
|
||||
root_node_address: u64::from(u32::from_le_bytes([
|
||||
fields[9], fields[10], fields[11], fields[12],
|
||||
])),
|
||||
num_records_in_root: u16::from_le_bytes([fields[13], fields[14]]),
|
||||
total_records: u64::from(u32::from_le_bytes([
|
||||
fields[15], fields[16], fields[17], fields[18],
|
||||
])),
|
||||
};
|
||||
let offset_size = if fields[19] & 1 == 0 { 4 } else { 8 };
|
||||
let _ = collect_btree_v2_records(file, &header, offset_size, 8);
|
||||
}
|
||||
|
||||
@@ -940,3 +940,61 @@ fn provenance_verify_written_file() {
|
||||
.unwrap();
|
||||
assert_eq!(result, clawhdf5_format::provenance::VerifyResult::Ok);
|
||||
}
|
||||
|
||||
// ---- hdf5plugin interop: registered third-party compression filters ----
|
||||
|
||||
/// Write `data` (f64, 1-D, chunked) with `configure` applied, then read it
|
||||
/// back with h5py + hdf5plugin (libhdf5's registered filter plugins) and
|
||||
/// return the values it decodes.
|
||||
#[cfg(any(feature = "lz4", feature = "zstd"))]
|
||||
fn hdf5plugin_roundtrip(
|
||||
tag: &str,
|
||||
data: &[f64],
|
||||
configure: impl FnOnce(&mut clawhdf5_format::type_builders::DatasetBuilder),
|
||||
) -> Vec<f64> {
|
||||
let mut fw = FileWriter::new();
|
||||
let ds = fw.create_dataset("data");
|
||||
ds.with_f64_data(data)
|
||||
.with_shape(&[data.len() as u64])
|
||||
.with_chunks(&[250]);
|
||||
configure(ds);
|
||||
let bytes = fw.finish().unwrap();
|
||||
let path = std::env::temp_dir().join(format!("clawhdf5_hdf5plugin_{tag}.h5"));
|
||||
std::fs::write(&path, &bytes).unwrap();
|
||||
let script = format!(
|
||||
"import h5py,hdf5plugin,json; f=h5py.File('{}','r'); print(json.dumps(f['data'][:].tolist()))",
|
||||
path.display()
|
||||
);
|
||||
let stdout = h5py_read(&path, &script);
|
||||
serde_json::from_str(&stdout).unwrap()
|
||||
}
|
||||
|
||||
/// libhdf5's LZ4 plugin must decode what we write (it could not while we
|
||||
/// wrote a private 4-byte-LE-size framing).
|
||||
#[cfg(feature = "lz4")]
|
||||
#[test]
|
||||
#[ignore = "requires Python h5py + hdf5plugin"]
|
||||
fn hdf5plugin_reads_our_lz4() {
|
||||
let data: Vec<f64> = (0..1000).map(|i| (i % 37) as f64 * 0.5).collect();
|
||||
let got = hdf5plugin_roundtrip("lz4", &data, |ds| {
|
||||
ds.with_lz4();
|
||||
});
|
||||
assert_eq!(got, data);
|
||||
let got = hdf5plugin_roundtrip("lz4_noshuffle", &data, |ds| {
|
||||
ds.with_lz4().without_shuffle();
|
||||
});
|
||||
assert_eq!(got, data);
|
||||
}
|
||||
|
||||
/// libhdf5's Zstandard plugin must decode what we write (it could not while
|
||||
/// our frames lacked the content size).
|
||||
#[cfg(feature = "zstd")]
|
||||
#[test]
|
||||
#[ignore = "requires Python h5py + hdf5plugin"]
|
||||
fn hdf5plugin_reads_our_zstd() {
|
||||
let data: Vec<f64> = (0..1000).map(|i| (i % 37) as f64 * 0.5).collect();
|
||||
let got = hdf5plugin_roundtrip("zstd", &data, |ds| {
|
||||
ds.with_zstd(3);
|
||||
});
|
||||
assert_eq!(got, data);
|
||||
}
|
||||
|
||||
@@ -0,0 +1,646 @@
|
||||
//! Regression tests for writer metadata bugs that produced files libhdf5
|
||||
//! refuses (or reads differently from us), plus the reader-side counterparts.
|
||||
//!
|
||||
//! The plain tests check the bytes we write with our own parser. The
|
||||
//! `#[ignore]`d ones are the interop half: they open what we write in h5py
|
||||
//! (`CLAWHDF5_PYTHON`, as in `writer_h5py_tests.rs`) and run `h5dump` over it.
|
||||
|
||||
use clawhdf5_format::data_layout::DataLayout;
|
||||
use clawhdf5_format::datatype::{Datatype, DatatypeByteOrder, ReferenceType};
|
||||
use clawhdf5_format::file_writer::{AttrValue, FileWriter};
|
||||
use clawhdf5_format::group_v2::resolve_path_any;
|
||||
use clawhdf5_format::message_type::MessageType;
|
||||
use clawhdf5_format::object_header::ObjectHeader;
|
||||
use clawhdf5_format::signature;
|
||||
use clawhdf5_format::superblock::Superblock;
|
||||
use clawhdf5_format::type_builders::{FillTime, make_u8_type};
|
||||
|
||||
// ---- helpers ----
|
||||
|
||||
fn header_at(bytes: &[u8], path: &str) -> (Superblock, ObjectHeader) {
|
||||
let sig = signature::find_signature(bytes).unwrap();
|
||||
let sb = Superblock::parse(bytes, sig).unwrap();
|
||||
let addr = if path == "/" {
|
||||
sb.root_group_address
|
||||
} else {
|
||||
resolve_path_any(bytes, &sb, path).unwrap()
|
||||
};
|
||||
let oh = ObjectHeader::parse(bytes, addr as usize, sb.offset_size, sb.length_size).unwrap();
|
||||
(sb, oh)
|
||||
}
|
||||
|
||||
fn layout_of(bytes: &[u8], path: &str) -> DataLayout {
|
||||
let (sb, oh) = header_at(bytes, path);
|
||||
let msg = oh
|
||||
.messages
|
||||
.iter()
|
||||
.find(|m| m.msg_type == MessageType::DataLayout)
|
||||
.unwrap();
|
||||
DataLayout::parse(&msg.data, sb.offset_size, sb.length_size).unwrap()
|
||||
}
|
||||
|
||||
fn python() -> String {
|
||||
std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string())
|
||||
}
|
||||
|
||||
fn write_tmp(name: &str, bytes: &[u8]) -> std::path::PathBuf {
|
||||
let path = std::env::temp_dir().join(format!("clawhdf5_writer_meta_{name}.h5"));
|
||||
std::fs::write(&path, bytes).unwrap();
|
||||
path
|
||||
}
|
||||
|
||||
/// Run `script` (with `path` bound to the file) under h5py; return stdout.
|
||||
fn h5py(path: &std::path::Path, script: &str) -> String {
|
||||
let full = format!(
|
||||
"import h5py, numpy as np, json\npath = {:?}\n{script}",
|
||||
path.display().to_string()
|
||||
);
|
||||
let o = std::process::Command::new(python())
|
||||
.args(["-c", &full])
|
||||
.output()
|
||||
.expect("python interpreter");
|
||||
assert!(
|
||||
o.status.success(),
|
||||
"h5py failed: {}",
|
||||
String::from_utf8_lossy(&o.stderr)
|
||||
);
|
||||
String::from_utf8(o.stdout).unwrap().trim().to_string()
|
||||
}
|
||||
|
||||
/// `h5dump` must read the whole file without error.
|
||||
fn h5dump_ok(path: &std::path::Path) {
|
||||
let o = std::process::Command::new("h5dump")
|
||||
.arg(path)
|
||||
.output()
|
||||
.expect("h5dump");
|
||||
assert!(
|
||||
o.status.success(),
|
||||
"h5dump failed: {}{}",
|
||||
String::from_utf8_lossy(&o.stdout),
|
||||
String::from_utf8_lossy(&o.stderr)
|
||||
);
|
||||
}
|
||||
|
||||
fn u8_ramp(n: usize) -> Vec<u8> {
|
||||
(0..n).map(|i| (i % 251) as u8).collect()
|
||||
}
|
||||
|
||||
// ---- 1. object header message size limit ----
|
||||
|
||||
#[test]
|
||||
fn attribute_too_big_for_a_header_message_is_an_error() {
|
||||
// Measured: a 70000-byte attribute was written with its message size
|
||||
// wrapped to 16 bits, and libhdf5 refused the whole root group.
|
||||
let mut fw = FileWriter::new();
|
||||
fw.set_root_attr(
|
||||
"a",
|
||||
AttrValue::Raw {
|
||||
datatype: make_u8_type(),
|
||||
shape: vec![70_000],
|
||||
data: u8_ramp(70_000),
|
||||
},
|
||||
);
|
||||
assert!(fw.finish().is_err());
|
||||
|
||||
// 65500 bytes still fits and still works.
|
||||
let mut fw = FileWriter::new();
|
||||
fw.set_root_attr(
|
||||
"a",
|
||||
AttrValue::Raw {
|
||||
datatype: make_u8_type(),
|
||||
shape: vec![65_500],
|
||||
data: u8_ramp(65_500),
|
||||
},
|
||||
);
|
||||
let bytes = fw.finish().unwrap();
|
||||
let (sb, oh) = header_at(&bytes, "/");
|
||||
let attrs = clawhdf5_format::attribute::extract_attributes(&oh, sb.length_size).unwrap();
|
||||
assert_eq!(attrs[0].raw_data, u8_ramp(65_500));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn compact_layout_falls_back_to_contiguous_past_the_message_limit() {
|
||||
// Layout message = 4 bytes + data; data may be at most 65531 bytes.
|
||||
for (n, compact) in [(65_531, true), (65_532, false), (65_534, false)] {
|
||||
let mut fw = FileWriter::new();
|
||||
fw.create_dataset("d").with_u8_data(&u8_ramp(n)).compact();
|
||||
let bytes = fw.finish().unwrap();
|
||||
match layout_of(&bytes, "d") {
|
||||
DataLayout::Compact { data } => {
|
||||
assert!(compact, "{n} bytes must not be compact");
|
||||
assert_eq!(data, u8_ramp(n));
|
||||
}
|
||||
DataLayout::Contiguous { .. } => assert!(!compact, "{n} bytes should be compact"),
|
||||
other => panic!("unexpected layout {other:?}"),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
#[ignore = "requires Python h5py module and h5dump"]
|
||||
fn h5py_reads_compact_datasets_at_the_limit() {
|
||||
for n in [65_531usize, 65_534] {
|
||||
let mut fw = FileWriter::new();
|
||||
fw.create_dataset("d").with_u8_data(&u8_ramp(n)).compact();
|
||||
let path = write_tmp(&format!("compact_{n}"), &fw.finish().unwrap());
|
||||
let out = h5py(
|
||||
&path,
|
||||
"f = h5py.File(path, 'r'); v = f['d'][()]\n\
|
||||
print(bool((v == (np.arange(v.size) % 251).astype(np.uint8)).all()), v.size)",
|
||||
);
|
||||
assert_eq!(out, format!("True {n}"));
|
||||
h5dump_ok(&path);
|
||||
}
|
||||
}
|
||||
|
||||
// ---- 2. Time / BitField / Opaque / Reference datatypes ----
|
||||
|
||||
fn exotic_types() -> Vec<(&'static str, Datatype, Vec<u8>)> {
|
||||
// Four elements each. The object references point at the root group,
|
||||
// which a v3-superblock file without an extension puts at address 48.
|
||||
let refs: Vec<u8> = (0..4).flat_map(|_| 48u64.to_le_bytes()).collect();
|
||||
vec![
|
||||
(
|
||||
"bits",
|
||||
Datatype::BitField {
|
||||
size: 1,
|
||||
byte_order: DatatypeByteOrder::LittleEndian,
|
||||
bit_offset: 0,
|
||||
bit_precision: 8,
|
||||
},
|
||||
vec![1, 2, 4, 8],
|
||||
),
|
||||
(
|
||||
"opaque",
|
||||
Datatype::Opaque {
|
||||
size: 4,
|
||||
tag: b"mytag".to_vec(),
|
||||
},
|
||||
(0..16).collect(),
|
||||
),
|
||||
(
|
||||
"ref",
|
||||
Datatype::Reference {
|
||||
size: 8,
|
||||
ref_type: ReferenceType::Object,
|
||||
},
|
||||
refs,
|
||||
),
|
||||
(
|
||||
"time",
|
||||
Datatype::Time {
|
||||
size: 4,
|
||||
bit_precision: 32,
|
||||
},
|
||||
(0..16).collect(),
|
||||
),
|
||||
]
|
||||
}
|
||||
|
||||
fn exotic_file() -> Vec<u8> {
|
||||
let mut fw = FileWriter::new();
|
||||
for (name, dt, raw) in exotic_types() {
|
||||
fw.create_dataset(name)
|
||||
.with_compound_data(dt.clone(), raw.clone(), 4);
|
||||
fw.set_root_attr(
|
||||
name,
|
||||
AttrValue::Raw {
|
||||
datatype: dt,
|
||||
shape: vec![4],
|
||||
data: raw,
|
||||
},
|
||||
);
|
||||
}
|
||||
fw.finish().unwrap()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn exotic_datatypes_are_written_not_emptied() {
|
||||
let bytes = exotic_file();
|
||||
let (sb, root) = header_at(&bytes, "/");
|
||||
assert_eq!(sb.root_group_address, 48);
|
||||
let attrs = clawhdf5_format::attribute::extract_attributes(&root, sb.length_size).unwrap();
|
||||
for (name, dt, raw) in exotic_types() {
|
||||
let (_, oh) = header_at(&bytes, name);
|
||||
let msg = oh
|
||||
.messages
|
||||
.iter()
|
||||
.find(|m| m.msg_type == MessageType::Datatype)
|
||||
.unwrap();
|
||||
assert_eq!(msg.data, dt.serialize(), "{name}");
|
||||
assert_eq!(Datatype::parse(&msg.data).unwrap().0, dt, "{name}");
|
||||
let attr = attrs.iter().find(|a| a.name == name).unwrap();
|
||||
assert_eq!(attr.datatype, dt, "{name}");
|
||||
assert_eq!(attr.raw_data, raw, "{name}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
#[ignore = "requires Python h5py module and h5dump"]
|
||||
fn h5py_reads_exotic_datatypes() {
|
||||
let path = write_tmp("exotic", &exotic_file());
|
||||
let out = h5py(
|
||||
&path,
|
||||
"from h5py import h5t, h5s\n\
|
||||
f = h5py.File(path, 'r')\n\
|
||||
r = {}\n\
|
||||
buf = np.zeros(4, dtype='V4')\n\
|
||||
f['opaque'].id.read(h5s.ALL, h5s.ALL, buf, mtype=f['opaque'].id.get_type())\n\
|
||||
r['bits'] = f['bits'][()].tolist(), f.attrs['bits'].tolist()\n\
|
||||
r['opaque'] = (f['opaque'].id.get_type().get_tag().decode(),\n\
|
||||
\x20 f.attrs.get_id('opaque').get_type().get_tag().decode(),\n\
|
||||
\x20 buf.tobytes().hex())\n\
|
||||
r['ref'] = [f[x].name for x in f['ref'][()]] + [f[x].name for x in f.attrs['ref']]\n\
|
||||
r['time'] = (f['time'].id.get_type().get_class() == h5t.TIME,\n\
|
||||
\x20 f.attrs.get_id('time').get_type().get_class() == h5t.TIME)\n\
|
||||
print(json.dumps(r))",
|
||||
);
|
||||
let v: serde_json::Value = serde_json::from_str(&out).unwrap();
|
||||
assert_eq!(v["bits"], serde_json::json!([[1, 2, 4, 8], [1, 2, 4, 8]]));
|
||||
assert_eq!(
|
||||
v["opaque"],
|
||||
serde_json::json!(["mytag", "mytag", "000102030405060708090a0b0c0d0e0f"])
|
||||
);
|
||||
assert_eq!(v["ref"], serde_json::json!(vec!["/"; 8]));
|
||||
assert_eq!(v["time"], serde_json::json!([true, true]));
|
||||
h5dump_ok(&path);
|
||||
}
|
||||
|
||||
#[test]
|
||||
#[ignore = "requires Python h5py module and h5dump"]
|
||||
fn raw_attributes_copied_from_h5py_survive_a_rewrite() {
|
||||
// Read Raw attributes of the exotic classes out of an h5py file and write
|
||||
// them back: this used to emit empty datatype messages.
|
||||
let src = std::env::temp_dir().join("clawhdf5_writer_meta_exotic_src.h5");
|
||||
h5py(
|
||||
&src,
|
||||
"from h5py import h5t, h5s, h5a\n\
|
||||
f = h5py.File(path, 'w')\n\
|
||||
f.attrs['ref'] = np.array([f.ref, f.ref], dtype=h5py.ref_dtype)\n\
|
||||
f.attrs.create('opaque', np.frombuffer(b'abcdefgh', dtype='V4'))\n\
|
||||
t = h5t.STD_B16BE.copy()\n\
|
||||
a = h5a.create(f.id, b'bits', t, h5s.create_simple((2,)))\n\
|
||||
a.write(np.array([0x0102, 0x0304], dtype='>u2'), mtype=t)\n\
|
||||
a.close()\n\
|
||||
f.close()",
|
||||
);
|
||||
let src_bytes = std::fs::read(&src).unwrap();
|
||||
let (sb, root) = header_at(&src_bytes, "/");
|
||||
let attrs = clawhdf5_format::attribute::extract_attributes(&root, sb.length_size).unwrap();
|
||||
assert_eq!(attrs.len(), 3);
|
||||
let mut fw = FileWriter::new();
|
||||
for a in &attrs {
|
||||
let data = if a.name == "ref" {
|
||||
// Re-target the references at our root group.
|
||||
48u64.to_le_bytes().repeat(2)
|
||||
} else {
|
||||
a.raw_data.clone()
|
||||
};
|
||||
fw.set_root_attr(
|
||||
&a.name,
|
||||
AttrValue::Raw {
|
||||
datatype: a.datatype.clone(),
|
||||
shape: a.dataspace.dimensions.clone(),
|
||||
data,
|
||||
},
|
||||
);
|
||||
}
|
||||
let path = write_tmp("exotic_copy", &fw.finish().unwrap());
|
||||
let out = h5py(
|
||||
&path,
|
||||
"f = h5py.File(path, 'r')\n\
|
||||
print(json.dumps([[f[x].name for x in f.attrs['ref']],\n\
|
||||
\x20 f.attrs['opaque'].tobytes().decode(),\n\
|
||||
\x20 f.attrs.get_id('bits').get_type().get_order(),\n\
|
||||
\x20 f.attrs['bits'].tolist()]))",
|
||||
);
|
||||
assert_eq!(out, r#"[["/", "/"], "abcdefgh", 1, [258, 772]]"#);
|
||||
h5dump_ok(&path);
|
||||
}
|
||||
|
||||
// ---- 3. paged file-space strategy ----
|
||||
|
||||
fn paged_file(page_size: u32) -> Vec<u8> {
|
||||
let mut fw = FileWriter::new();
|
||||
fw.with_page_size(page_size);
|
||||
fw.create_dataset("d").with_f64_data(&[1.0, 2.0, 3.0]);
|
||||
fw.create_dataset("c")
|
||||
.with_i32_data(&(0..100).collect::<Vec<_>>())
|
||||
.with_chunks(&[10]);
|
||||
fw.set_root_attr("a", AttrValue::I64(7));
|
||||
let mut g = fw.create_group("g");
|
||||
g.create_dataset("e").with_u8_data(&[9; 5000]);
|
||||
fw.add_group(g.finish());
|
||||
fw.finish().unwrap()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn paged_file_has_a_real_superblock() {
|
||||
// Measured: `with_page_size` wrote superblock version 4, which does not
|
||||
// exist ("bad superblock version number" in libhdf5).
|
||||
for ps in [512u32, 4096, 65536] {
|
||||
let bytes = paged_file(ps);
|
||||
let (sb, _) = header_at(&bytes, "/");
|
||||
assert_eq!(sb.version, 3);
|
||||
assert_eq!(bytes.len() % ps as usize, 0);
|
||||
let (_, e) = header_at(&bytes, "g/e");
|
||||
assert!(
|
||||
e.messages
|
||||
.iter()
|
||||
.any(|m| m.msg_type == MessageType::Dataspace)
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
#[ignore = "requires Python h5py module and h5dump"]
|
||||
fn h5py_opens_paged_files() {
|
||||
for ps in [512u32, 4096, 65536] {
|
||||
let path = write_tmp(&format!("paged_{ps}"), &paged_file(ps));
|
||||
let out = h5py(
|
||||
&path,
|
||||
"f = h5py.File(path, 'r')\n\
|
||||
p = f.id.get_create_plist()\n\
|
||||
print(json.dumps([p.get_file_space_strategy()[0], p.get_file_space_page_size(),\n\
|
||||
\x20 f['d'][()].tolist(), int(f['c'][()].sum()), int(f.attrs['a']),\n\
|
||||
\x20 int(f['g/e'][()].sum())]))",
|
||||
);
|
||||
assert_eq!(
|
||||
out,
|
||||
format!("[1, {ps}, [1.0, 2.0, 3.0], 4950, 7, 45000]"),
|
||||
"page size {ps}"
|
||||
);
|
||||
h5dump_ok(&path);
|
||||
}
|
||||
}
|
||||
|
||||
// ---- 4. fill time and fill value ----
|
||||
|
||||
fn fill_message(bytes: &[u8], path: &str) -> clawhdf5_format::object_header::HeaderMessage {
|
||||
let (_, oh) = header_at(bytes, path);
|
||||
oh.messages
|
||||
.into_iter()
|
||||
.find(|m| m.msg_type == MessageType::FillValue)
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
fn fill_file() -> Vec<u8> {
|
||||
let mut fw = FileWriter::new();
|
||||
fw.create_dataset("never")
|
||||
.with_f64_data(&[1.0, 2.0])
|
||||
.fill_time(FillTime::Never);
|
||||
fw.create_dataset("alloc")
|
||||
.with_f64_data(&[1.0, 2.0])
|
||||
.fill_time(FillTime::Alloc);
|
||||
fw.create_dataset("ifset")
|
||||
.with_f64_data(&[1.0, 2.0])
|
||||
.fill_time(FillTime::IfSet);
|
||||
fw.create_dataset("default").with_f64_data(&[1.0, 2.0]);
|
||||
fw.create_dataset("filled")
|
||||
.with_i32_data(&[1, 2, 3, 4])
|
||||
.with_chunks(&[2])
|
||||
.with_maxshape(&[u64::MAX])
|
||||
.with_fill_value(&(-1i32).to_le_bytes());
|
||||
fw.finish().unwrap()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn fill_time_uses_libhdf5_codes() {
|
||||
// H5D_FILL_TIME_ALLOC = 0, NEVER = 1, IFSET = 2, in bits 2-3. Measured:
|
||||
// h5py saw our Never as ALLOC, Alloc as IFSET and IfSet as NEVER.
|
||||
let bytes = fill_file();
|
||||
for (path, code) in [("never", 1), ("alloc", 0), ("ifset", 2), ("default", 2)] {
|
||||
let msg = fill_message(&bytes, path);
|
||||
assert_eq!((msg.data[1] >> 2) & 3, code, "{path}");
|
||||
assert_eq!(msg.data[1] & 3, 2, "{path}: allocation time stays late");
|
||||
}
|
||||
for ft in [FillTime::Never, FillTime::Alloc, FillTime::IfSet] {
|
||||
assert_eq!(FillTime::from_byte(ft.to_byte()), Some(ft));
|
||||
}
|
||||
assert_eq!(FillTime::default(), FillTime::IfSet);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn fill_value_is_written_and_read_back() {
|
||||
let bytes = fill_file();
|
||||
let msg = fill_message(&bytes, "filled");
|
||||
assert_eq!(
|
||||
clawhdf5_format::fill_value::parse_fill_value(&msg).unwrap(),
|
||||
Some((-1i32).to_le_bytes().to_vec())
|
||||
);
|
||||
assert_eq!(
|
||||
clawhdf5_format::fill_value::parse_fill_value(&fill_message(&bytes, "ifset")).unwrap(),
|
||||
None
|
||||
);
|
||||
|
||||
// One element's bytes, no more, no less.
|
||||
let mut fw = FileWriter::new();
|
||||
fw.create_dataset("d")
|
||||
.with_f64_data(&[1.0])
|
||||
.with_fill_value(&[0; 4]);
|
||||
assert!(fw.finish().is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
#[ignore = "requires Python h5py module and h5dump"]
|
||||
fn h5py_sees_our_fill_time_and_fill_value() {
|
||||
let path = write_tmp("fill", &fill_file());
|
||||
let out = h5py(
|
||||
&path,
|
||||
"from h5py import h5d\n\
|
||||
f = h5py.File(path, 'r')\n\
|
||||
names = {h5d.FILL_TIME_NEVER: 'never', h5d.FILL_TIME_ALLOC: 'alloc', h5d.FILL_TIME_IFSET: 'ifset'}\n\
|
||||
t = [names[f[n].id.get_create_plist().get_fill_time()] for n in ('never', 'alloc', 'ifset', 'default')]\n\
|
||||
print(json.dumps([t, int(f['filled'].fillvalue), f['filled'][()].tolist()]))\n\
|
||||
f.close()\n\
|
||||
f = h5py.File(path, 'r+')\n\
|
||||
f['filled'].resize((7,))\n\
|
||||
f.close()\n\
|
||||
print(json.dumps(h5py.File(path, 'r')['filled'][()].tolist()))",
|
||||
);
|
||||
assert_eq!(
|
||||
out,
|
||||
"[[\"never\", \"alloc\", \"ifset\", \"ifset\"], -1, [1, 2, 3, 4]]\n[1, 2, 3, 4, -1, -1, -1]"
|
||||
);
|
||||
h5dump_ok(&path);
|
||||
}
|
||||
|
||||
// ---- 5. empty string attributes ----
|
||||
|
||||
fn empty_string_file() -> Vec<u8> {
|
||||
let mut fw = FileWriter::new();
|
||||
fw.set_root_attr("empty", AttrValue::String(String::new()));
|
||||
fw.set_root_attr("x", AttrValue::String("héllo".into()));
|
||||
fw.set_root_attr(
|
||||
"empties",
|
||||
AttrValue::StringArray(vec![String::new(), String::new()]),
|
||||
);
|
||||
fw.set_root_attr("n", AttrValue::I64(3));
|
||||
fw.finish().unwrap()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn empty_string_attribute_has_a_one_byte_type() {
|
||||
// Measured: "" got a size-0 string type, and libhdf5 then refused every
|
||||
// attribute on the object ("invalid datatype size").
|
||||
let bytes = empty_string_file();
|
||||
let (sb, root) = header_at(&bytes, "/");
|
||||
let attrs = clawhdf5_format::attribute::extract_attributes(&root, sb.length_size).unwrap();
|
||||
for name in ["empty", "empties"] {
|
||||
let a = attrs.iter().find(|a| a.name == name).unwrap();
|
||||
assert_eq!(a.datatype.type_size(), 1, "{name}");
|
||||
let strings = a.read_as_strings().unwrap();
|
||||
assert!(strings.iter().all(String::is_empty), "{name}: {strings:?}");
|
||||
}
|
||||
|
||||
// A size-0 string type handed in directly is refused, not written.
|
||||
let mut fw = FileWriter::new();
|
||||
fw.set_root_attr(
|
||||
"raw",
|
||||
AttrValue::Raw {
|
||||
datatype: Datatype::String {
|
||||
size: 0,
|
||||
padding: clawhdf5_format::datatype::StringPadding::NullPad,
|
||||
charset: clawhdf5_format::datatype::CharacterSet::Ascii,
|
||||
},
|
||||
shape: vec![],
|
||||
data: vec![],
|
||||
},
|
||||
);
|
||||
assert!(fw.finish().is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
#[ignore = "requires Python h5py module and h5dump"]
|
||||
fn h5py_reads_all_attributes_next_to_an_empty_string() {
|
||||
let path = write_tmp("empty_str", &empty_string_file());
|
||||
let out = h5py(
|
||||
&path,
|
||||
"f = h5py.File(path, 'r')\n\
|
||||
d = lambda v: v.decode() if isinstance(v, bytes) else v\n\
|
||||
print(json.dumps([d(f.attrs['empty']), d(f.attrs['x']),\n\
|
||||
\x20 [d(s) for s in f.attrs['empties']], int(f.attrs['n'])], ensure_ascii=False))",
|
||||
);
|
||||
assert_eq!(out, r#"["", "héllo", ["", ""], 3]"#);
|
||||
h5dump_ok(&path);
|
||||
}
|
||||
|
||||
// ---- 6. path-like names ----
|
||||
|
||||
#[test]
|
||||
fn slash_in_a_group_or_dataset_name_is_an_error() {
|
||||
// Measured: create_group("a/b") wrote one link literally named "a/b",
|
||||
// which h5py cannot reach ("component not found"). The writer has no
|
||||
// nested groups, so such names are refused.
|
||||
let mut fw = FileWriter::new();
|
||||
let mut g = fw.create_group("a/b");
|
||||
g.create_dataset("c").with_f64_data(&[1.0]);
|
||||
fw.add_group(g.finish());
|
||||
assert!(fw.finish().is_err());
|
||||
|
||||
let mut fw = FileWriter::new();
|
||||
fw.create_dataset("x/y").with_f64_data(&[1.0]);
|
||||
assert!(fw.finish().is_err());
|
||||
|
||||
let mut fw = FileWriter::new();
|
||||
let mut g = fw.create_group("g");
|
||||
g.create_dataset("x/y").with_f64_data(&[1.0]);
|
||||
fw.add_group(g.finish());
|
||||
assert!(fw.finish().is_err());
|
||||
|
||||
for bad in ["", "."] {
|
||||
let mut fw = FileWriter::new();
|
||||
fw.create_dataset(bad).with_f64_data(&[1.0]);
|
||||
assert!(fw.finish().is_err(), "{bad:?}");
|
||||
}
|
||||
|
||||
// One level of groups still works, and '/' stays legal in attribute names.
|
||||
let mut fw = FileWriter::new();
|
||||
let mut g = fw.create_group("g");
|
||||
g.create_dataset("c").with_f64_data(&[1.0]);
|
||||
g.set_attr("m/s", AttrValue::I64(1));
|
||||
fw.add_group(g.finish());
|
||||
let bytes = fw.finish().unwrap();
|
||||
header_at(&bytes, "g/c");
|
||||
}
|
||||
|
||||
// ---- 7. unknown-message flags on read ----
|
||||
|
||||
#[test]
|
||||
fn unknown_message_flags_follow_libhdf5_on_tbogus() {
|
||||
// libhdf5's own test file (test/testfiles/tbogus.h5): datasets carrying
|
||||
// an unknown message with various flags. libhdf5 (read-only) opens
|
||||
// Dataset1, 2, 4 and 5 and refuses Dataset3 ("unknown message with 'fail
|
||||
// if unknown' flag found"). We used to refuse Dataset2 (bit 3, which only
|
||||
// applies when writing) and open Dataset3 (bit 7, fail always).
|
||||
let bytes = include_bytes!("fixtures/tbogus.h5");
|
||||
let sig = signature::find_signature(bytes).unwrap();
|
||||
let sb = Superblock::parse(bytes, sig).unwrap();
|
||||
for (name, readable) in [
|
||||
("Dataset1", true),
|
||||
("Dataset2", true),
|
||||
("Dataset3", false),
|
||||
("Dataset4", true),
|
||||
("Dataset5", true),
|
||||
] {
|
||||
let addr = resolve_path_any(bytes, &sb, name).unwrap();
|
||||
let parsed = ObjectHeader::parse(bytes, addr as usize, sb.offset_size, sb.length_size);
|
||||
match parsed {
|
||||
Ok(_) => assert!(readable, "{name} must be refused"),
|
||||
Err(e) => {
|
||||
assert!(!readable, "{name} must be readable, got {e:?}");
|
||||
assert!(matches!(
|
||||
e,
|
||||
clawhdf5_format::error::FormatError::UnsupportedMessage(_)
|
||||
));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---- 8. shared fill value messages ----
|
||||
|
||||
#[test]
|
||||
fn shared_fill_value_is_resolved_not_zero() {
|
||||
// gen_shared_fill.py: HDF5 2.0 with a SOHM index for fill values, so each
|
||||
// dataset's fill value message is a reference into the SOHM heap. It
|
||||
// used to be read as "no fill value" (zeros) instead of -7.
|
||||
let bytes = include_bytes!("fixtures/shared_fill_value.h5");
|
||||
for (name, shared) in [
|
||||
("sohm_a", false),
|
||||
("sohm_b", true),
|
||||
("unwritten_a", false),
|
||||
("unwritten_b", true),
|
||||
] {
|
||||
let (sb, oh) = header_at(bytes, name);
|
||||
let msg = oh
|
||||
.messages
|
||||
.iter()
|
||||
.find(|m| m.msg_type == MessageType::FillValue)
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
clawhdf5_format::shared_message::is_shared(msg.flags),
|
||||
shared,
|
||||
"{name}: fixture layout"
|
||||
);
|
||||
if shared {
|
||||
// Without the file the reference cannot be followed: an error,
|
||||
// never a silent default.
|
||||
assert_eq!(
|
||||
clawhdf5_format::fill_value::dataset_fill_value(&oh.messages),
|
||||
Err(clawhdf5_format::error::FormatError::UnresolvedSharedMessage)
|
||||
);
|
||||
}
|
||||
assert_eq!(
|
||||
clawhdf5_format::fill_value::dataset_fill_value_in(
|
||||
bytes,
|
||||
&oh.messages,
|
||||
sb.offset_size,
|
||||
sb.length_size
|
||||
)
|
||||
.unwrap(),
|
||||
Some((-7i32).to_le_bytes().to_vec()),
|
||||
"{name}"
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -268,28 +268,26 @@ impl AsyncHDF5File {
|
||||
///
|
||||
/// Reads the entire file into memory, then parses the superblock.
|
||||
pub async fn open<R: AsyncHDF5Read>(reader: &R) -> Result<Self, AsyncHDF5Error> {
|
||||
let data = reader.read_all().await?;
|
||||
let sig_offset = find_signature(&data)?;
|
||||
let superblock = Superblock::parse(&data, sig_offset)?;
|
||||
Ok(Self { data, superblock })
|
||||
Self::from_bytes(reader.read_all().await?)
|
||||
}
|
||||
|
||||
/// Open an HDF5 file asynchronously from a file path.
|
||||
pub async fn open_path<P: AsRef<Path>>(path: P) -> Result<Self, AsyncHDF5Error> {
|
||||
let data = tokio::fs::read(path).await?;
|
||||
let sig_offset = find_signature(&data)?;
|
||||
let superblock = Superblock::parse(&data, sig_offset)?;
|
||||
Ok(Self { data, superblock })
|
||||
Self::from_bytes(tokio::fs::read(path).await?)
|
||||
}
|
||||
|
||||
/// Open an HDF5 file from bytes already in memory.
|
||||
pub fn from_bytes(data: Vec<u8>) -> Result<Self, AsyncHDF5Error> {
|
||||
let sig_offset = find_signature(&data)?;
|
||||
let superblock = Superblock::parse(&data, sig_offset)?;
|
||||
pub fn from_bytes(mut data: Vec<u8>) -> Result<Self, AsyncHDF5Error> {
|
||||
// HDF5 addresses are relative to the superblock: drop any user block
|
||||
// so they index `data` directly.
|
||||
let user_block = find_signature(&data)?;
|
||||
data.drain(..user_block);
|
||||
let superblock = Superblock::parse(&data, 0)?;
|
||||
Ok(Self { data, superblock })
|
||||
}
|
||||
|
||||
/// Access the raw file bytes.
|
||||
/// Access the file bytes from the superblock on (any user block is
|
||||
/// dropped on open).
|
||||
pub fn as_bytes(&self) -> &[u8] {
|
||||
&self.data
|
||||
}
|
||||
|
||||
@@ -180,7 +180,7 @@ fn mpi_collective_read(vol: &MpiVol, location: &str, path: &str) -> Result<Vec<u
|
||||
use clawhdf5_format::{
|
||||
data_layout::DataLayout, data_read::read_raw_data_full, dataspace::Dataspace,
|
||||
datatype::Datatype, filter_pipeline::FilterPipeline, group_v2::resolve_path_any,
|
||||
message_type::MessageType, object_header::ObjectHeader, signature::find_signature,
|
||||
message_type::MessageType, object_header::ObjectHeader, signature::split_user_block,
|
||||
superblock::Superblock,
|
||||
};
|
||||
use mpi::traits::*;
|
||||
@@ -192,9 +192,10 @@ fn mpi_collective_read(vol: &MpiVol, location: &str, path: &str) -> Result<Vec<u
|
||||
let mut len_buf = [0usize; 1];
|
||||
|
||||
if rank == 0 {
|
||||
let bytes = std::fs::read(location).map_err(VolError::Io)?;
|
||||
let sig = find_signature(&bytes).map_err(|e| VolError::DataError(e.to_string()))?;
|
||||
let sb = Superblock::parse(&bytes, sig).map_err(|e| VolError::DataError(e.to_string()))?;
|
||||
let file = std::fs::read(location).map_err(VolError::Io)?;
|
||||
// Addresses are relative to the superblock: skip any user block.
|
||||
let (_, bytes) = split_user_block(&file).map_err(|e| VolError::DataError(e.to_string()))?;
|
||||
let sb = Superblock::parse(bytes, 0).map_err(|e| VolError::DataError(e.to_string()))?;
|
||||
let addr = resolve_path_any(&bytes, &sb, path)
|
||||
.map_err(|e| VolError::NotFound(format!("{path}: {e}")))?;
|
||||
let oh = ObjectHeader::parse(&bytes, addr as usize, sb.offset_size, sb.length_size)
|
||||
|
||||
@@ -283,12 +283,13 @@ impl VirtualObjectLayer for NativeVol {
|
||||
use clawhdf5_format::{
|
||||
data_layout::DataLayout, data_read::read_raw_data_full, dataspace::Dataspace,
|
||||
datatype::Datatype, filter_pipeline::FilterPipeline, group_v2::resolve_path_any,
|
||||
message_type::MessageType, object_header::ObjectHeader, signature::find_signature,
|
||||
message_type::MessageType, object_header::ObjectHeader, signature::split_user_block,
|
||||
superblock::Superblock,
|
||||
};
|
||||
|
||||
let sig = find_signature(data).map_err(|e| VolError::DataError(e.to_string()))?;
|
||||
let sb = Superblock::parse(data, sig).map_err(|e| VolError::DataError(e.to_string()))?;
|
||||
// Addresses are relative to the superblock: skip any user block.
|
||||
let (_, data) = split_user_block(data).map_err(|e| VolError::DataError(e.to_string()))?;
|
||||
let sb = Superblock::parse(data, 0).map_err(|e| VolError::DataError(e.to_string()))?;
|
||||
let addr = resolve_path_any(data, &sb, path)
|
||||
.map_err(|e| VolError::NotFound(format!("{path}: {e}")))?;
|
||||
|
||||
|
||||
@@ -3,8 +3,11 @@
|
||||
[](https://crates.io/crates/clawhdf5-migrate)
|
||||
[](https://docs.rs/clawhdf5-migrate)
|
||||
|
||||
CLI tool to migrate SQLite agent memory databases (the ZeroClaw layout) to a
|
||||
[clawhdf5-agent](https://crates.io/crates/clawhdf5-agent) store.
|
||||
CLI tool to migrate a SQLite agent-memory database in the `memory_chunks` / `sessions` / `entities` / `relations` layout (table and
|
||||
column names are configurable) to a
|
||||
[clawhdf5-agent](https://crates.io/crates/clawhdf5-agent) store. This is **not**
|
||||
ZeroClaw's schema — ZeroClaw keeps memories in a single `memories` table and
|
||||
does not use clawhdf5.
|
||||
|
||||
The output is written through `clawhdf5-agent`'s own API, so it opens with
|
||||
`HDF5Memory::open` and is searchable immediately: memory records, sessions and
|
||||
|
||||
@@ -12,7 +12,8 @@ use validate::ValidationSummary;
|
||||
|
||||
type BoxErr = Box<dyn std::error::Error>;
|
||||
|
||||
/// Migrate ZeroClaw agent memory from SQLite to a clawhdf5-agent store.
|
||||
/// Migrate agent memory from SQLite (memory_chunks / sessions / entities /
|
||||
/// relations tables) to a clawhdf5-agent store.
|
||||
///
|
||||
/// The output is an ordinary agent store: open it with
|
||||
/// `HDF5Memory::open` (or `clawhdf5-cli --path <store> ...`).
|
||||
@@ -270,7 +271,7 @@ mod tests {
|
||||
use std::path::PathBuf;
|
||||
use tempfile::TempDir;
|
||||
|
||||
/// Create a test SQLite database with the ZeroClaw schema.
|
||||
/// Create a test SQLite database with the default migration layout.
|
||||
fn create_test_db(dir: &TempDir) -> String {
|
||||
let db_path = dir.path().join("test.db");
|
||||
let path_str = db_path.to_str().unwrap().to_string();
|
||||
|
||||
@@ -43,7 +43,7 @@ pub struct Relation {
|
||||
pub timestamp: f64,
|
||||
}
|
||||
|
||||
/// All data read from a ZeroClaw SQLite database.
|
||||
/// All data read from a source SQLite database.
|
||||
#[derive(Debug)]
|
||||
pub struct SqliteData {
|
||||
pub chunks: Vec<MemoryChunk>,
|
||||
@@ -63,7 +63,9 @@ pub struct TableSchema {
|
||||
|
||||
/// Configurable mapping from a SQLite layout to the migration's data model.
|
||||
///
|
||||
/// Defaults to the ZeroClaw schema; the CLI can override the table names so the
|
||||
/// Defaults to the `memory_chunks` / `sessions` / `entities` / `relations`
|
||||
/// layout (not ZeroClaw's schema, despite what earlier docs said); the CLI can
|
||||
/// override the table names so the
|
||||
/// tool can migrate databases whose tables are named differently. Column names
|
||||
/// (and order) are part of the config too, so a library caller can remap them.
|
||||
#[derive(Debug, Clone)]
|
||||
@@ -190,7 +192,7 @@ fn blob_to_f32(blob: &[u8]) -> Vec<f32> {
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Read all data from a ZeroClaw SQLite database.
|
||||
/// Read all data from a source SQLite database.
|
||||
///
|
||||
/// If `skip_deleted` is true, rows with `deleted=1` are excluded from chunks.
|
||||
/// If `embedding_dim` is `None`, auto-detect from the first row (0 when there
|
||||
|
||||
+71
-68
@@ -12,25 +12,23 @@
|
||||
use std::cell::RefCell;
|
||||
use std::collections::HashMap;
|
||||
|
||||
use clawhdf5_format::attribute::extract_attributes_full;
|
||||
use clawhdf5_format::data_layout::DataLayout;
|
||||
use clawhdf5_format::data_read;
|
||||
use clawhdf5_format::dataspace::Dataspace;
|
||||
use clawhdf5_format::datatype::Datatype;
|
||||
use clawhdf5_format::error::FormatError;
|
||||
use clawhdf5_format::filter_pipeline::FilterPipeline;
|
||||
use clawhdf5_format::group_v1::{self, GroupEntry};
|
||||
use clawhdf5_format::group_v1::GroupEntry;
|
||||
use clawhdf5_format::group_v2;
|
||||
use clawhdf5_format::message_type::MessageType;
|
||||
use clawhdf5_format::object_header::ObjectHeader;
|
||||
use clawhdf5_format::signature;
|
||||
use clawhdf5_format::superblock::Superblock;
|
||||
use clawhdf5_format::symbol_table::SymbolTableMessage;
|
||||
|
||||
use clawhdf5_io::HDF5Read;
|
||||
|
||||
use crate::error::Error;
|
||||
use crate::types::{AttrValue, DType, attrs_to_map, classify_datatype};
|
||||
use crate::types::{AttrValue, DType, classify_datatype, read_attrs};
|
||||
|
||||
/// A lazy HDF5 file handle that parses metadata on demand.
|
||||
///
|
||||
@@ -42,6 +40,9 @@ use crate::types::{AttrValue, DType, attrs_to_map, classify_datatype};
|
||||
/// `MemoryReader`, etc.
|
||||
pub struct LazyFile<R: HDF5Read> {
|
||||
reader: R,
|
||||
/// Offset of the superblock in the file (the user-block size); every
|
||||
/// HDF5 address is relative to it.
|
||||
base: usize,
|
||||
superblock: Superblock,
|
||||
root_header: ObjectHeader,
|
||||
/// Cache of parsed object headers, keyed by address.
|
||||
@@ -73,9 +74,9 @@ impl<R: HDF5Read> LazyFile<R> {
|
||||
///
|
||||
/// Parses only the superblock and root group object header.
|
||||
pub fn open(reader: R) -> Result<Self, Error> {
|
||||
let data = reader.as_bytes();
|
||||
let sig_offset = signature::find_signature(data)?;
|
||||
let superblock = Superblock::parse(data, sig_offset)?;
|
||||
let (user_block, data) = signature::split_user_block(reader.as_bytes())?;
|
||||
let base = user_block.len();
|
||||
let superblock = Superblock::parse(data, 0)?;
|
||||
let root_header = ObjectHeader::parse(
|
||||
data,
|
||||
superblock.root_group_address as usize,
|
||||
@@ -84,15 +85,26 @@ impl<R: HDF5Read> LazyFile<R> {
|
||||
)?;
|
||||
Ok(Self {
|
||||
reader,
|
||||
base,
|
||||
superblock,
|
||||
root_header,
|
||||
header_cache: RefCell::new(HashMap::new()),
|
||||
})
|
||||
}
|
||||
|
||||
/// Returns the raw file bytes.
|
||||
/// Returns the file's bytes from the superblock on (after any user
|
||||
/// block), which is the space every HDF5 address in the file indexes.
|
||||
pub fn as_bytes(&self) -> &[u8] {
|
||||
self.reader.as_bytes()
|
||||
self.hdf5_bytes()
|
||||
}
|
||||
|
||||
/// Size of the user block before the superblock (0 for most files).
|
||||
pub fn user_block_size(&self) -> u64 {
|
||||
self.base as u64
|
||||
}
|
||||
|
||||
fn hdf5_bytes(&self) -> &[u8] {
|
||||
&self.reader.as_bytes()[self.base..]
|
||||
}
|
||||
|
||||
/// Returns a reference to the parsed superblock.
|
||||
@@ -110,7 +122,7 @@ impl<R: HDF5Read> LazyFile<R> {
|
||||
|
||||
/// Resolve a path and return a `LazyDataset` handle.
|
||||
pub fn dataset(&self, path: &str) -> Result<LazyDataset<'_, R>, Error> {
|
||||
let data = self.reader.as_bytes();
|
||||
let data = self.hdf5_bytes();
|
||||
let addr = group_v2::resolve_path_any(data, &self.superblock, path)?;
|
||||
let hdr = self.get_or_parse_header(addr)?;
|
||||
if !has_message(&hdr, MessageType::DataLayout) {
|
||||
@@ -124,7 +136,7 @@ impl<R: HDF5Read> LazyFile<R> {
|
||||
|
||||
/// Resolve a path and return a `LazyGroup` handle.
|
||||
pub fn group(&self, path: &str) -> Result<LazyGroup<'_, R>, Error> {
|
||||
let data = self.reader.as_bytes();
|
||||
let data = self.hdf5_bytes();
|
||||
let addr = group_v2::resolve_path_any(data, &self.superblock, path)?;
|
||||
Ok(LazyGroup {
|
||||
file: self,
|
||||
@@ -163,7 +175,7 @@ impl<R: HDF5Read> LazyFile<R> {
|
||||
}
|
||||
|
||||
// Parse and cache
|
||||
let data = self.reader.as_bytes();
|
||||
let data = self.hdf5_bytes();
|
||||
let hdr = ObjectHeader::parse(
|
||||
data,
|
||||
address as usize,
|
||||
@@ -187,7 +199,7 @@ impl<R: HDF5Read> LazyFile<R> {
|
||||
impl<R: HDF5Read> std::fmt::Debug for LazyFile<R> {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.debug_struct("LazyFile")
|
||||
.field("size", &self.reader.as_bytes().len())
|
||||
.field("size", &self.hdf5_bytes().len())
|
||||
.field("superblock_version", &self.superblock.version)
|
||||
.field("cached_headers", &self.header_cache.borrow().len())
|
||||
.finish()
|
||||
@@ -232,17 +244,26 @@ impl<'f, R: HDF5Read> LazyGroup<'f, R> {
|
||||
}
|
||||
|
||||
/// Read all attributes of this group.
|
||||
///
|
||||
/// An attribute that cannot be read — a corrupt or unsupported attribute
|
||||
/// message, or a dense-storage heap object that cannot be located — is
|
||||
/// left out of the map instead of failing every attribute on the object;
|
||||
/// [`attrs_with_errors`](Self::attrs_with_errors) reports which failed.
|
||||
/// Values that are returned are complete (never partially decoded). An
|
||||
/// error in the index of the attributes itself (the attribute info
|
||||
/// message, the dense heap header or B-tree) still fails the call.
|
||||
pub fn attrs(&self) -> Result<HashMap<String, AttrValue>, Error> {
|
||||
self.attrs_with_errors().map(|(attrs, _)| attrs)
|
||||
}
|
||||
|
||||
/// Like [`attrs`](Self::attrs), also returning one error for each
|
||||
/// attribute that could not be read and was left out.
|
||||
pub fn attrs_with_errors(
|
||||
&self,
|
||||
) -> Result<(HashMap<String, AttrValue>, Vec<FormatError>), Error> {
|
||||
let hdr = self.file.get_or_parse_header(self.address)?;
|
||||
let data = self.file.reader.as_bytes();
|
||||
let attr_msgs =
|
||||
extract_attributes_full(data, &hdr, self.file.offset_size(), self.file.length_size())?;
|
||||
Ok(attrs_to_map(
|
||||
&attr_msgs,
|
||||
data,
|
||||
self.file.offset_size(),
|
||||
self.file.length_size(),
|
||||
))
|
||||
let data = self.file.hdf5_bytes();
|
||||
read_attrs(data, &hdr, self.file.offset_size(), self.file.length_size())
|
||||
}
|
||||
|
||||
/// Get a dataset within this group by name.
|
||||
@@ -275,12 +296,14 @@ impl<'f, R: HDF5Read> LazyGroup<'f, R> {
|
||||
})
|
||||
}
|
||||
|
||||
/// This group's links that can be opened: hard links, and soft links
|
||||
/// resolved to their targets (see
|
||||
/// [`group_v2::resolve_group_children`]); dangling, external and
|
||||
/// user-defined links are left out.
|
||||
fn children(&self) -> Result<Vec<GroupEntry>, Error> {
|
||||
let hdr = self.file.get_or_parse_header(self.address)?;
|
||||
let data = self.file.reader.as_bytes();
|
||||
let os = self.file.offset_size();
|
||||
let ls = self.file.length_size();
|
||||
resolve_group_entries(data, &hdr, os, ls).map_err(Error::Format)
|
||||
let data = self.file.hdf5_bytes();
|
||||
group_v2::resolve_group_children(data, &self.file.superblock, self.address)
|
||||
.map_err(Error::Format)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -360,7 +383,7 @@ impl<'f, R: HDF5Read> LazyDataset<'f, R> {
|
||||
let dl = self.data_layout()?;
|
||||
let ds = self.dataspace()?;
|
||||
let dt = self.datatype()?;
|
||||
let slice = data_read::read_raw_data_zerocopy(self.file.reader.as_bytes(), &dl, &ds, &dt)?;
|
||||
let slice = data_read::read_raw_data_zerocopy(self.file.hdf5_bytes(), &dl, &ds, &dt)?;
|
||||
Ok(slice)
|
||||
}
|
||||
|
||||
@@ -400,20 +423,30 @@ impl<'f, R: HDF5Read> LazyDataset<'f, R> {
|
||||
}
|
||||
|
||||
/// Read all attributes of this dataset.
|
||||
///
|
||||
/// An attribute that cannot be read — a corrupt or unsupported attribute
|
||||
/// message, or a dense-storage heap object that cannot be located — is
|
||||
/// left out of the map instead of failing every attribute on the object;
|
||||
/// [`attrs_with_errors`](Self::attrs_with_errors) reports which failed.
|
||||
/// Values that are returned are complete (never partially decoded). An
|
||||
/// error in the index of the attributes itself (the attribute info
|
||||
/// message, the dense heap header or B-tree) still fails the call.
|
||||
pub fn attrs(&self) -> Result<HashMap<String, AttrValue>, Error> {
|
||||
let data = self.file.reader.as_bytes();
|
||||
let attr_msgs = extract_attributes_full(
|
||||
self.attrs_with_errors().map(|(attrs, _)| attrs)
|
||||
}
|
||||
|
||||
/// Like [`attrs`](Self::attrs), also returning one error for each
|
||||
/// attribute that could not be read and was left out.
|
||||
pub fn attrs_with_errors(
|
||||
&self,
|
||||
) -> Result<(HashMap<String, AttrValue>, Vec<FormatError>), Error> {
|
||||
let data = self.file.hdf5_bytes();
|
||||
read_attrs(
|
||||
data,
|
||||
&self.header,
|
||||
self.file.offset_size(),
|
||||
self.file.length_size(),
|
||||
)?;
|
||||
Ok(attrs_to_map(
|
||||
&attr_msgs,
|
||||
data,
|
||||
self.file.offset_size(),
|
||||
self.file.length_size(),
|
||||
))
|
||||
)
|
||||
}
|
||||
|
||||
/// A header message's payload, resolved through the shared-message
|
||||
@@ -479,7 +512,7 @@ impl<'f, R: HDF5Read> LazyDataset<'f, R> {
|
||||
let ds = self.dataspace()?;
|
||||
let dl = self.data_layout()?;
|
||||
let pipeline = self.filter_pipeline()?;
|
||||
let data = self.file.reader.as_bytes();
|
||||
let data = self.file.hdf5_bytes();
|
||||
// Unallocated storage reads as the dataset's fill value.
|
||||
clawhdf5_format::fill_value::read_full_with_fill(
|
||||
&self.header.messages,
|
||||
@@ -530,33 +563,3 @@ fn is_group(header: &ObjectHeader) -> bool {
|
||||
|| m.msg_type == MessageType::SymbolTable
|
||||
})
|
||||
}
|
||||
|
||||
fn resolve_group_entries(
|
||||
file_data: &[u8],
|
||||
object_header: &ObjectHeader,
|
||||
offset_size: u8,
|
||||
length_size: u8,
|
||||
) -> Result<Vec<GroupEntry>, FormatError> {
|
||||
let is_v1 = object_header
|
||||
.messages
|
||||
.iter()
|
||||
.any(|m| m.msg_type == MessageType::SymbolTable);
|
||||
let is_v2 = object_header
|
||||
.messages
|
||||
.iter()
|
||||
.any(|m| m.msg_type == MessageType::LinkInfo || m.msg_type == MessageType::Link);
|
||||
|
||||
if is_v1 {
|
||||
let sym_msg = object_header
|
||||
.messages
|
||||
.iter()
|
||||
.find(|m| m.msg_type == MessageType::SymbolTable)
|
||||
.ok_or_else(|| FormatError::PathNotFound("no symbol table message".into()))?;
|
||||
let stm = SymbolTableMessage::parse(&sym_msg.data, offset_size)?;
|
||||
group_v1::resolve_v1_group_entries(file_data, &stm, offset_size, length_size)
|
||||
} else if is_v2 {
|
||||
group_v2::resolve_v2_group_entries(file_data, object_header, offset_size, length_size)
|
||||
} else {
|
||||
Ok(Vec::new())
|
||||
}
|
||||
}
|
||||
|
||||
@@ -7,25 +7,23 @@
|
||||
|
||||
use std::collections::HashMap;
|
||||
|
||||
use clawhdf5_format::attribute::extract_attributes_full;
|
||||
use clawhdf5_format::data_layout::DataLayout;
|
||||
use clawhdf5_format::data_read;
|
||||
use clawhdf5_format::dataspace::Dataspace;
|
||||
use clawhdf5_format::datatype::Datatype;
|
||||
use clawhdf5_format::error::FormatError;
|
||||
use clawhdf5_format::filter_pipeline::FilterPipeline;
|
||||
use clawhdf5_format::group_v1::{self, GroupEntry};
|
||||
use clawhdf5_format::group_v1::GroupEntry;
|
||||
use clawhdf5_format::group_v2;
|
||||
use clawhdf5_format::message_type::MessageType;
|
||||
use clawhdf5_format::object_header::ObjectHeader;
|
||||
use clawhdf5_format::signature;
|
||||
use clawhdf5_format::superblock::Superblock;
|
||||
use clawhdf5_format::symbol_table::SymbolTableMessage;
|
||||
|
||||
use clawhdf5_io::MmapReader;
|
||||
|
||||
use crate::error::Error;
|
||||
use crate::types::{AttrValue, DType, attrs_to_map, classify_datatype};
|
||||
use crate::types::{AttrValue, DType, classify_datatype, read_attrs};
|
||||
|
||||
/// An HDF5 file opened via memory mapping.
|
||||
///
|
||||
@@ -34,6 +32,9 @@ use crate::types::{AttrValue, DType, attrs_to_map, classify_datatype};
|
||||
/// `&[u8]` slice via [`MmapDataset::read_raw_slice`].
|
||||
pub struct MmapFile {
|
||||
reader: MmapReader,
|
||||
/// Offset of the superblock in the mapped file (the user-block size);
|
||||
/// every HDF5 address is relative to it.
|
||||
base: usize,
|
||||
superblock: Superblock,
|
||||
}
|
||||
|
||||
@@ -41,10 +42,25 @@ impl MmapFile {
|
||||
/// Open an HDF5 file using memory-mapped I/O.
|
||||
pub fn open<P: AsRef<std::path::Path>>(path: P) -> Result<Self, Error> {
|
||||
let reader = MmapReader::open(path).map_err(Error::Io)?;
|
||||
let data = reader.as_bytes();
|
||||
let sig_offset = signature::find_signature(data)?;
|
||||
let superblock = Superblock::parse(data, sig_offset)?;
|
||||
Ok(Self { reader, superblock })
|
||||
let (user_block, data) = signature::split_user_block(reader.as_bytes())?;
|
||||
let base = user_block.len();
|
||||
let superblock = Superblock::parse(data, 0)?;
|
||||
Ok(Self {
|
||||
reader,
|
||||
base,
|
||||
superblock,
|
||||
})
|
||||
}
|
||||
|
||||
/// The file's bytes from the superblock on — the space HDF5 addresses
|
||||
/// index into.
|
||||
fn hdf5_bytes(&self) -> &[u8] {
|
||||
&self.reader.as_bytes()[self.base..]
|
||||
}
|
||||
|
||||
/// Size of the user block before the superblock (0 for most files).
|
||||
pub fn user_block_size(&self) -> u64 {
|
||||
self.base as u64
|
||||
}
|
||||
|
||||
/// Returns a handle to the root group.
|
||||
@@ -57,7 +73,7 @@ impl MmapFile {
|
||||
|
||||
/// Resolve a path and return a `MmapDataset` handle.
|
||||
pub fn dataset(&self, path: &str) -> Result<MmapDataset<'_>, Error> {
|
||||
let data = self.reader.as_bytes();
|
||||
let data = self.hdf5_bytes();
|
||||
let addr = group_v2::resolve_path_any(data, &self.superblock, path)?;
|
||||
let hdr = self.parse_header(addr)?;
|
||||
if !has_message(&hdr, MessageType::DataLayout) {
|
||||
@@ -71,7 +87,7 @@ impl MmapFile {
|
||||
|
||||
/// Resolve a path and return a `MmapGroup` handle.
|
||||
pub fn group(&self, path: &str) -> Result<MmapGroup<'_>, Error> {
|
||||
let data = self.reader.as_bytes();
|
||||
let data = self.hdf5_bytes();
|
||||
let addr = group_v2::resolve_path_any(data, &self.superblock, path)?;
|
||||
Ok(MmapGroup {
|
||||
file: self,
|
||||
@@ -79,9 +95,11 @@ impl MmapFile {
|
||||
})
|
||||
}
|
||||
|
||||
/// Returns the raw file bytes (zero-copy from mmap).
|
||||
/// Returns the file's bytes from the superblock on (after any user
|
||||
/// block), zero-copy from the mmap. Every HDF5 address in the file
|
||||
/// indexes this slice.
|
||||
pub fn as_bytes(&self) -> &[u8] {
|
||||
self.reader.as_bytes()
|
||||
self.hdf5_bytes()
|
||||
}
|
||||
|
||||
/// Returns a reference to the parsed superblock.
|
||||
@@ -91,7 +109,7 @@ impl MmapFile {
|
||||
|
||||
fn parse_header(&self, address: u64) -> Result<ObjectHeader, FormatError> {
|
||||
ObjectHeader::parse(
|
||||
self.reader.as_bytes(),
|
||||
self.hdf5_bytes(),
|
||||
address as usize,
|
||||
self.superblock.offset_size,
|
||||
self.superblock.length_size,
|
||||
@@ -154,17 +172,26 @@ impl<'f> MmapGroup<'f> {
|
||||
}
|
||||
|
||||
/// Read all attributes of this group.
|
||||
///
|
||||
/// An attribute that cannot be read — a corrupt or unsupported attribute
|
||||
/// message, or a dense-storage heap object that cannot be located — is
|
||||
/// left out of the map instead of failing every attribute on the object;
|
||||
/// [`attrs_with_errors`](Self::attrs_with_errors) reports which failed.
|
||||
/// Values that are returned are complete (never partially decoded). An
|
||||
/// error in the index of the attributes itself (the attribute info
|
||||
/// message, the dense heap header or B-tree) still fails the call.
|
||||
pub fn attrs(&self) -> Result<HashMap<String, AttrValue>, Error> {
|
||||
let data = self.file.reader.as_bytes();
|
||||
self.attrs_with_errors().map(|(attrs, _)| attrs)
|
||||
}
|
||||
|
||||
/// Like [`attrs`](Self::attrs), also returning one error for each
|
||||
/// attribute that could not be read and was left out.
|
||||
pub fn attrs_with_errors(
|
||||
&self,
|
||||
) -> Result<(HashMap<String, AttrValue>, Vec<FormatError>), Error> {
|
||||
let hdr = self.file.parse_header(self.address)?;
|
||||
let attr_msgs =
|
||||
extract_attributes_full(data, &hdr, self.file.offset_size(), self.file.length_size())?;
|
||||
Ok(attrs_to_map(
|
||||
&attr_msgs,
|
||||
data,
|
||||
self.file.offset_size(),
|
||||
self.file.length_size(),
|
||||
))
|
||||
let data = self.file.hdf5_bytes();
|
||||
read_attrs(data, &hdr, self.file.offset_size(), self.file.length_size())
|
||||
}
|
||||
|
||||
/// Get a dataset within this group by name.
|
||||
@@ -197,12 +224,14 @@ impl<'f> MmapGroup<'f> {
|
||||
})
|
||||
}
|
||||
|
||||
/// This group's links that can be opened: hard links, and soft links
|
||||
/// resolved to their targets (see
|
||||
/// [`group_v2::resolve_group_children`]); dangling, external and
|
||||
/// user-defined links are left out.
|
||||
fn children(&self) -> Result<Vec<GroupEntry>, Error> {
|
||||
let data = self.file.reader.as_bytes();
|
||||
let hdr = self.file.parse_header(self.address)?;
|
||||
let os = self.file.offset_size();
|
||||
let ls = self.file.length_size();
|
||||
resolve_group_entries(data, &hdr, os, ls).map_err(Error::Format)
|
||||
let data = self.file.hdf5_bytes();
|
||||
group_v2::resolve_group_children(data, &self.file.superblock, self.address)
|
||||
.map_err(Error::Format)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -326,7 +355,7 @@ impl<'f> MmapDataset<'f> {
|
||||
actual: sz,
|
||||
}));
|
||||
}
|
||||
let data = self.file.reader.as_bytes();
|
||||
let data = self.file.hdf5_bytes();
|
||||
let a = addr as usize;
|
||||
if a + sz > data.len() {
|
||||
return Err(Error::Format(FormatError::UnexpectedEof {
|
||||
@@ -341,20 +370,30 @@ impl<'f> MmapDataset<'f> {
|
||||
}
|
||||
|
||||
/// Read all attributes of this dataset.
|
||||
///
|
||||
/// An attribute that cannot be read — a corrupt or unsupported attribute
|
||||
/// message, or a dense-storage heap object that cannot be located — is
|
||||
/// left out of the map instead of failing every attribute on the object;
|
||||
/// [`attrs_with_errors`](Self::attrs_with_errors) reports which failed.
|
||||
/// Values that are returned are complete (never partially decoded). An
|
||||
/// error in the index of the attributes itself (the attribute info
|
||||
/// message, the dense heap header or B-tree) still fails the call.
|
||||
pub fn attrs(&self) -> Result<HashMap<String, AttrValue>, Error> {
|
||||
let data = self.file.reader.as_bytes();
|
||||
let attr_msgs = extract_attributes_full(
|
||||
self.attrs_with_errors().map(|(attrs, _)| attrs)
|
||||
}
|
||||
|
||||
/// Like [`attrs`](Self::attrs), also returning one error for each
|
||||
/// attribute that could not be read and was left out.
|
||||
pub fn attrs_with_errors(
|
||||
&self,
|
||||
) -> Result<(HashMap<String, AttrValue>, Vec<FormatError>), Error> {
|
||||
let data = self.file.hdf5_bytes();
|
||||
read_attrs(
|
||||
data,
|
||||
&self.header,
|
||||
self.file.offset_size(),
|
||||
self.file.length_size(),
|
||||
)?;
|
||||
Ok(attrs_to_map(
|
||||
&attr_msgs,
|
||||
data,
|
||||
self.file.offset_size(),
|
||||
self.file.length_size(),
|
||||
))
|
||||
)
|
||||
}
|
||||
|
||||
/// A header message's payload, resolved through the shared-message
|
||||
@@ -423,7 +462,7 @@ impl<'f> MmapDataset<'f> {
|
||||
// Unallocated storage reads as the dataset's fill value.
|
||||
clawhdf5_format::fill_value::read_full_with_fill(
|
||||
&self.header.messages,
|
||||
self.file.reader.as_bytes(),
|
||||
self.file.hdf5_bytes(),
|
||||
&dl,
|
||||
&ds,
|
||||
dt.type_size() as usize,
|
||||
@@ -431,7 +470,7 @@ impl<'f> MmapDataset<'f> {
|
||||
self.file.length_size(),
|
||||
|| {
|
||||
Ok(data_read::read_raw_data_full(
|
||||
self.file.reader.as_bytes(),
|
||||
self.file.hdf5_bytes(),
|
||||
&dl,
|
||||
&ds,
|
||||
&dt,
|
||||
@@ -470,33 +509,3 @@ fn is_group(header: &ObjectHeader) -> bool {
|
||||
|| m.msg_type == MessageType::SymbolTable
|
||||
})
|
||||
}
|
||||
|
||||
fn resolve_group_entries(
|
||||
file_data: &[u8],
|
||||
object_header: &ObjectHeader,
|
||||
offset_size: u8,
|
||||
length_size: u8,
|
||||
) -> Result<Vec<GroupEntry>, FormatError> {
|
||||
let is_v1 = object_header
|
||||
.messages
|
||||
.iter()
|
||||
.any(|m| m.msg_type == MessageType::SymbolTable);
|
||||
let is_v2 = object_header
|
||||
.messages
|
||||
.iter()
|
||||
.any(|m| m.msg_type == MessageType::LinkInfo || m.msg_type == MessageType::Link);
|
||||
|
||||
if is_v1 {
|
||||
let sym_msg = object_header
|
||||
.messages
|
||||
.iter()
|
||||
.find(|m| m.msg_type == MessageType::SymbolTable)
|
||||
.ok_or_else(|| FormatError::PathNotFound("no symbol table message".into()))?;
|
||||
let stm = SymbolTableMessage::parse(&sym_msg.data, offset_size)?;
|
||||
group_v1::resolve_v1_group_entries(file_data, &stm, offset_size, length_size)
|
||||
} else if is_v2 {
|
||||
group_v2::resolve_v2_group_entries(file_data, object_header, offset_size, length_size)
|
||||
} else {
|
||||
Ok(Vec::new())
|
||||
}
|
||||
}
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user