Compare commits
227
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
dda28d6c72 | ||
|
|
408f69ec1d | ||
|
|
73a01f1256 | ||
|
|
956e55c76a | ||
|
|
846c35455d | ||
|
|
ca779b2864 | ||
|
|
20bd381c87 | ||
|
|
5a202f3791 | ||
|
|
8dcce084ca | ||
|
|
45d617c39e | ||
|
|
d345ffbf80 | ||
|
|
17edfe2cf0 | ||
|
|
546fdb84fa | ||
|
|
05b0192a60 | ||
|
|
8bcae3c78e | ||
|
|
f0ecae38b6 | ||
|
|
400e3a9fec | ||
|
|
bd1d8f1a59 | ||
|
|
751edeb7e6 | ||
|
|
b43bd2e67f | ||
|
|
81a0e8685d | ||
|
|
41b7837d0a | ||
|
|
8c51b05b9c | ||
|
|
24412a0e59 | ||
|
|
3bcd443e63 | ||
|
|
37770f594a | ||
|
|
a5e41c1a53 | ||
|
|
b0a1e4f9a6 | ||
|
|
bd36fe883b | ||
|
|
d102c06306 | ||
|
|
6e8421a81e | ||
|
|
8ce6eca34d | ||
|
|
c3850a0b66 | ||
|
|
2bc4cb46a6 | ||
|
|
f7d88bb4fb | ||
|
|
f99587c27d | ||
|
|
2d4b211523 | ||
|
|
10da8f0d09 | ||
|
|
78c769f179 | ||
|
|
006bf3b131 | ||
|
|
8cbbef3fae | ||
|
|
63648c7000 | ||
|
|
91644d8aaf | ||
|
|
72306c6013 | ||
|
|
e60bde3579 | ||
|
|
c85a8222cc | ||
|
|
2b68791f6a | ||
|
|
f7c362cef5 | ||
|
|
13c095a3da | ||
|
|
591aa71d12 | ||
|
|
b9a2ce3077 | ||
|
|
743c32b512 | ||
|
|
993214723e | ||
|
|
3938f7f8a2 | ||
|
|
dd40bea467 | ||
|
|
afae86f3ea | ||
|
|
f713847e65 | ||
|
|
17fc8b1964 | ||
|
|
9238605661 | ||
|
|
738b9491b2 | ||
|
|
e10df68ed8 | ||
|
|
c4d96c1390 | ||
|
|
b8492bd28d | ||
|
|
a5bd70216c | ||
|
|
17f09375ad | ||
|
|
386bd1d41e | ||
|
|
b5e43bacd7 | ||
|
|
a14ccc36bf | ||
|
|
7f52a6f3ba | ||
|
|
f325d111f3 | ||
|
|
699ee9c447 | ||
|
|
9416c58723 | ||
|
|
6a8ee3ec7f | ||
|
|
a59d83d47d | ||
|
|
bb39be7f24 | ||
|
|
845a9d0125 | ||
|
|
7d7a7e75d4 | ||
|
|
0685037593 | ||
|
|
e92faa23a6 | ||
|
|
40968b3578 | ||
|
|
310448bfcb | ||
|
|
e73ac2af09 | ||
|
|
e01160299a | ||
|
|
5461a13984 | ||
|
|
056092b082 | ||
|
|
a5ca970015 | ||
|
|
e7a7951f1e | ||
|
|
1f71f3bcbc | ||
|
|
3cf8cd86f2 | ||
|
|
34987ec194 | ||
|
|
b58d61cfb7 | ||
|
|
1abd93e0f8 | ||
|
|
6dfd239011 | ||
|
|
07094e34a9 | ||
|
|
f4dee1cd08 | ||
|
|
e38f9123db | ||
|
|
a42b646689 | ||
|
|
74f9f50086 | ||
|
|
d16544b928 | ||
|
|
3b24e6753b | ||
|
|
e815eb922f | ||
|
|
bb78d70b99 | ||
|
|
a7de15534c | ||
|
|
10d1029ead | ||
|
|
883980f2bd | ||
|
|
d6e426e6d5 | ||
|
|
256e7b89e4 | ||
|
|
f2e704abf3 | ||
|
|
61f36516d7 | ||
|
|
45720fe5a6 | ||
|
|
adf961c883 | ||
|
|
b4a44a2e66 | ||
|
|
17fa783dce | ||
|
|
90e050944f | ||
|
|
a6e90f3ee3 | ||
|
|
0555794850 | ||
|
|
efc2dc53c9 | ||
|
|
5c2f656fe7 | ||
|
|
945b13a1f1 | ||
|
|
9179aa356e | ||
|
|
e94a52a88b | ||
|
|
d54a0f4737 | ||
|
|
aadfd18d4c | ||
|
|
2c6c6c176e | ||
|
|
190918a478 | ||
|
|
38d0d4de02 | ||
|
|
1c85986079 | ||
|
|
c7092722aa | ||
|
|
36356ba8a1 | ||
|
|
85eb7f5ce2 | ||
|
|
8196fab72a | ||
|
|
8ebd488d9e | ||
|
|
42b81d9f1c | ||
|
|
72b9cfb1e1 | ||
|
|
650f355219 | ||
|
|
7f5cfee281 | ||
|
|
c5302e587e | ||
|
|
e1115bc92a | ||
|
|
36d7a6f234 | ||
|
|
4b23ad697c | ||
|
|
e7f2d8575d | ||
|
|
7c1968a34a | ||
|
|
6db13c60b8 | ||
|
|
d99426be94 | ||
|
|
f5505fb03d | ||
|
|
95dcb04454 | ||
|
|
1dba7b465a | ||
|
|
57e938c438 | ||
|
|
3000b40cf3 | ||
|
|
bc820fbd8c | ||
|
|
9066d34eaa | ||
|
|
540fa08907 | ||
|
|
14876b8ae5 | ||
|
|
5935e13866 | ||
|
|
8c3ef996ea | ||
|
|
74fdf0582b | ||
|
|
44f5f8b5c5 | ||
|
|
2f252df084 | ||
|
|
c8c2930fc0 | ||
|
|
417c9516ca | ||
|
|
53dbddb07b | ||
|
|
081341b433 | ||
|
|
d074385944 | ||
|
|
4a1876faf2 | ||
|
|
be88e3fec7 | ||
|
|
e162c013fd | ||
|
|
06dda26d85 | ||
|
|
585e14d5e2 | ||
|
|
183d96ee26 | ||
|
|
bba1560416 | ||
|
|
b36998ef01 | ||
|
|
aef8e766ae | ||
|
|
9ea44d473d | ||
|
|
46203ea761 | ||
|
|
75bdb53342 | ||
|
|
dd5b3f6633 | ||
|
|
79dfa78e8f | ||
|
|
87d64588e5 | ||
|
|
bdadf3447c | ||
|
|
0c65a27b00 | ||
|
|
db9af7972c | ||
|
|
7706697feb | ||
|
|
4ecac65f22 | ||
|
|
c0f704c381 | ||
|
|
00b0cb0035 | ||
|
|
a7920bd4b3 | ||
|
|
7e43b5366c | ||
|
|
dce5559ff2 | ||
|
|
1b3bbb054a | ||
|
|
a8fb758489 | ||
|
|
73bb068264 | ||
|
|
5c8323cb1e | ||
|
|
dbaf3f505d | ||
|
|
c470244a6f | ||
|
|
d0db83812b | ||
|
|
5e4aa1c6bf | ||
|
|
1cceb930b2 | ||
|
|
fc7ae6549a | ||
|
|
735db117a7 | ||
|
|
e9b37a9602 | ||
|
|
4bed8b3765 | ||
|
|
36d689bc2c | ||
|
|
e7c08e06b4 | ||
|
|
c5049eb734 | ||
|
|
6f6bc97850 | ||
|
|
0cb72e8a60 | ||
|
|
e338d58ad5 | ||
|
|
8b85d9364b | ||
|
|
6598a7d02f | ||
|
|
114a2dfcba | ||
|
|
56a8c2f3d0 | ||
|
|
7b16dc90d6 | ||
|
|
4a5544da1d | ||
|
|
f4c6d43a3f | ||
|
|
e5e087f9ab | ||
|
|
16c9ee0554 | ||
|
|
1d767e3b93 | ||
|
|
a91df3f1c3 | ||
|
|
b41272487a | ||
|
|
0bc7a293ae | ||
|
|
dea02f5214 | ||
|
|
97e65f2adf | ||
|
|
fb58300b3f | ||
|
|
eb196e824f | ||
|
|
367faad7f7 | ||
|
|
0901fb1499 | ||
|
|
e9aeb110b7 |
+63
-12
@@ -9,31 +9,45 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
container: rust:latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- name: Cache cargo registry/target
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: |
|
||||
~/.cargo/registry
|
||||
~/.cargo/git
|
||||
target
|
||||
key: ${{ runner.os }}-cargo-${{ hashFiles('**/Cargo.lock') }}
|
||||
# Plain git rather than actions/checkout: that is a JavaScript action,
|
||||
# and rust:latest has no `node`, so it failed with exit 127 before any
|
||||
# code was built — on every push. actions/cache went for the same reason.
|
||||
- name: Check out
|
||||
run: |
|
||||
git init -q .
|
||||
git remote add origin "${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}.git"
|
||||
for i in 1 2 3; do git fetch -q --depth 1 origin "${GITHUB_SHA}" && break; sleep 5; done
|
||||
git checkout -q FETCH_HEAD
|
||||
- name: Install rustfmt & clippy components
|
||||
run: rustup component add rustfmt clippy
|
||||
- name: Install thumbv7em-none-eabihf target
|
||||
run: rustup target add thumbv7em-none-eabihf
|
||||
- name: Install wasm32-unknown-unknown target
|
||||
# ci-test.sh builds the reader and clawhdf5-wasm for the browser.
|
||||
run: rustup target add wasm32-unknown-unknown
|
||||
- name: Install Python interop dependencies
|
||||
# The interop suites used to skip silently when python3/h5py were
|
||||
# missing, so they never ran in CI. Install them and make a missing
|
||||
# dependency a failure (CLAWHDF5_REQUIRE_INTEROP below).
|
||||
run: |
|
||||
apt-get update
|
||||
apt-get install -y --no-install-recommends python3 python3-venv
|
||||
# cmake builds libz-ng-sys for the opt-in `fast-deflate` (zlib-ng)
|
||||
# steps in ci-test.sh; rust:latest does not ship it. The default
|
||||
# build (pure-Rust zlib-rs) does not need it.
|
||||
# hdf5-tools: h5ls/h5stat/h5dump/h5diff, which the h5rs
|
||||
# (clawhdf5-tools) interop tests compare against.
|
||||
apt-get install -y --no-install-recommends python3 python3-venv cmake hdf5-tools
|
||||
python3 -m venv /opt/interop
|
||||
/opt/interop/bin/pip install --no-cache-dir h5py numpy netCDF4 xarray
|
||||
# maturin + pytest: ci-test.sh builds the Python package
|
||||
# (crates/clawhdf5-py) and runs its tests against h5py.
|
||||
/opt/interop/bin/pip install --no-cache-dir h5py numpy netCDF4 xarray hdf5plugin maturin pytest
|
||||
echo "/opt/interop/bin" >> "$GITHUB_PATH"
|
||||
- name: Show interop library versions
|
||||
run: /opt/interop/bin/python -c "import h5py, netCDF4; print('h5py', h5py.__version__, 'HDF5', h5py.version.hdf5_version, 'netCDF4', netCDF4.__version__)"
|
||||
# h5dump's version too: the h5rs dump test requires its exact output
|
||||
# (checked against Debian's 1.14.5 in rust:latest and 1.14.6).
|
||||
run: |
|
||||
/opt/interop/bin/python -c "import h5py, netCDF4, hdf5plugin; print('h5py', h5py.__version__, 'HDF5', h5py.version.hdf5_version, 'netCDF4', netCDF4.__version__, 'hdf5plugin', hdf5plugin.version)"
|
||||
h5dump --version
|
||||
- name: Run CI script
|
||||
env:
|
||||
# Name the interpreter outright rather than relying on $GITHUB_PATH
|
||||
@@ -44,3 +58,40 @@ jobs:
|
||||
CLAWHDF5_PYTHON: /opt/interop/bin/python
|
||||
CLAWHDF5_REQUIRE_INTEROP: "1"
|
||||
run: bash scripts/ci-test.sh
|
||||
|
||||
test-arm64:
|
||||
# The aarch64 kernels in clawhdf5-accel — NEON `dot_i8`, including the
|
||||
# SDOT path, and the f32 NEON kernels — are cfg'd out on x86, so the job
|
||||
# above never compiles, lints or tests them.
|
||||
#
|
||||
# `linux_arm64` is served by two runners that execute differently:
|
||||
# vision-01 runs steps on the host (Rust already installed) and vision-02
|
||||
# runs them in docker.gitea.com/runner-images. So the steps work in both:
|
||||
# no `container:`, no JavaScript actions (they are fetched from GitHub,
|
||||
# which not every runner reliably reaches), and an explicit `+stable`
|
||||
# toolchain rather than whatever a host happens to default to.
|
||||
runs-on: linux_arm64
|
||||
env:
|
||||
CARGO_NET_RETRY: "10"
|
||||
CARGO_TERM_COLOR: always
|
||||
steps:
|
||||
- name: Check out
|
||||
run: |
|
||||
git init -q .
|
||||
git remote add origin "${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}.git"
|
||||
for i in 1 2 3; do git fetch -q --depth 1 origin "${GITHUB_SHA}" && break; sleep 5; done
|
||||
git checkout -q FETCH_HEAD
|
||||
- name: Rust stable
|
||||
run: |
|
||||
export PATH="$HOME/.cargo/bin:$PATH"
|
||||
command -v rustup >/dev/null || curl -sSf --retry 5 https://sh.rustup.rs | sh -s -- -y --profile minimal --default-toolchain none
|
||||
rustup toolchain install stable --profile minimal --component clippy
|
||||
echo "$HOME/.cargo/bin" >> "$GITHUB_PATH"
|
||||
- name: Confirm aarch64
|
||||
run: |
|
||||
test "$(uname -m)" = aarch64
|
||||
if grep -q asimddp /proc/cpuinfo; then echo "dot-product extension present: SDOT kernel runs"; else echo "no dot-product extension: plain NEON kernel runs"; fi
|
||||
- name: Clippy (aarch64 kernels)
|
||||
run: cargo +stable clippy -p clawhdf5-accel --all-targets -- -D warnings
|
||||
- name: Test
|
||||
run: cargo +stable test -p clawhdf5-accel -p clawhdf5-ann -p clawhdf5-format
|
||||
|
||||
@@ -0,0 +1,56 @@
|
||||
name: Conformance
|
||||
# Nightly: read every file of the pinned public HDF5 corpora with clawhdf5 and
|
||||
# with h5py/libhdf5 and compare (conformance/run.sh; CONFORMANCE.md explains
|
||||
# the method). Fails on any panic, hang, crash or out-of-memory in clawhdf5,
|
||||
# and when the ok count drops below conformance/baseline.json or a file the
|
||||
# baseline lists as ok stops being ok. The report is printed into the job log;
|
||||
# nothing is uploaded (artifact actions are JavaScript, which rust:latest
|
||||
# cannot run — see CLAUDE.md).
|
||||
on:
|
||||
schedule:
|
||||
- cron: "17 3 * * *"
|
||||
workflow_dispatch:
|
||||
jobs:
|
||||
conformance:
|
||||
runs-on: ubuntu-latest
|
||||
container: rust:latest
|
||||
timeout-minutes: 60
|
||||
env:
|
||||
CARGO_NET_RETRY: "10"
|
||||
steps:
|
||||
# Plain git, not actions/checkout (a JavaScript action; see ci.yml).
|
||||
- name: Check out
|
||||
run: |
|
||||
git init -q .
|
||||
git remote add origin "${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}.git"
|
||||
for i in 1 2 3; do git fetch -q --depth 1 origin "${GITHUB_SHA}" && break; sleep 5; done
|
||||
git checkout -q FETCH_HEAD
|
||||
- name: Install h5py, h5dump and the probe's codec libraries
|
||||
# hdf5-tools: h5dump for the CVE-corpus comparison. libaec-dev and
|
||||
# pkg-config: the probe builds clawhdf5-format with `szip` (the core
|
||||
# crates' default build needs neither).
|
||||
run: |
|
||||
apt-get update
|
||||
apt-get install -y --no-install-recommends python3 python3-venv hdf5-tools libaec-dev pkg-config
|
||||
python3 -m venv /opt/conformance
|
||||
/opt/conformance/bin/pip install --no-cache-dir -r conformance/requirements.txt
|
||||
/opt/conformance/bin/python -c "import h5py, hdf5plugin; print('h5py', h5py.__version__, 'HDF5', h5py.version.hdf5_version, 'hdf5plugin', hdf5plugin.version)"
|
||||
h5dump --version
|
||||
- name: Probe unit tests
|
||||
run: cargo test --release --manifest-path conformance/probe/Cargo.toml
|
||||
env:
|
||||
CARGO_TARGET_DIR: conformance/.cache/target
|
||||
- name: Sweep
|
||||
# The corpora come from GitHub (pinned commits, conformance/corpus.txt),
|
||||
# so this job needs a runner that reaches github.com.
|
||||
env:
|
||||
CLAWHDF5_PYTHON: /opt/conformance/bin/python
|
||||
run: bash conformance/run.sh
|
||||
- name: Report
|
||||
if: always()
|
||||
run: |
|
||||
if [ -f CONFORMANCE.md ]; then cat CONFORMANCE.md; else echo "no report was generated"; fi
|
||||
if [ -f conformance/.cache/results/summary.md ]; then
|
||||
echo; echo "---- per-file detail (conformance/.cache/results/summary.md) ----"
|
||||
cat conformance/.cache/results/summary.md
|
||||
fi
|
||||
@@ -5,3 +5,5 @@ benchmarks/longmemeval/*.json
|
||||
# Local model weights (MiniLM etc.) — large, not committed
|
||||
weights/
|
||||
.venv
|
||||
__pycache__/
|
||||
.pytest_cache/
|
||||
|
||||
+1083
-144
File diff suppressed because it is too large
Load Diff
+1250
File diff suppressed because it is too large
Load Diff
@@ -1,33 +1,40 @@
|
||||
# clawhdf5
|
||||
|
||||
## Purpose
|
||||
Pure-Rust HDF5 format implementation with HNSW vector search, WAL-backed persistence, agent memory storage, and GPU-accelerated I/O. Used by ZeroClaw as its persistent memory and knowledge graph backend.
|
||||
Pure-Rust HDF5 format implementation with HNSW vector search, WAL-backed persistence, agent memory storage, and GPU-accelerated vector search. A standalone library. Its one verified consumer is ClawBrainHub (`.brain` files); no agent framework integrates it (OpenClaw and ZeroClaw claims were withdrawn on 2026-09-25 — neither was ever true).
|
||||
|
||||
## Architecture
|
||||
|
||||
Cargo workspace with 16 crates under `crates/` (plus `libaec-sys`, an internal FFI bindings crate for the optional `szip` feature):
|
||||
Cargo workspace with 18 crates under `crates/` (plus `libaec-sys`, an internal FFI bindings crate for the optional `szip` feature):
|
||||
|
||||
| Crate | Role |
|
||||
|-------|------|
|
||||
| `clawhdf5-format` | HDF5 binary spec parser (superblock, B-tree, heap) — also holds shared type definitions and physical constants |
|
||||
| `clawhdf5-io` | Read/write implementation |
|
||||
| `clawhdf5-filters` | Compression filters (gzip, LZ4, Zstd, Blosc) |
|
||||
| `clawhdf5-filters` | Deflate backends (zlib-rs, zlib-ng, Apple Compression); the HDF5 filter pipeline, the filter registry (`clawhdf5_format::filter_registry`) and the other codecs (LZ4, Zstd, SZIP, N-Bit, scale-offset, pcodec, and the pure-Rust plugin filters LZF, bitshuffle, bzip2, Blosc 1) live in `clawhdf5-format`. No Blosc2 or ZFP. |
|
||||
| `clawhdf5-derive` | Proc-macro derive for HDF5-serializable structs |
|
||||
| `clawhdf5` | Main facade crate |
|
||||
| `clawhdf5-netcdf4` | NetCDF-4 compatibility layer |
|
||||
| `clawhdf5-ann` | HNSW approximate nearest-neighbor vector index |
|
||||
| `clawhdf5-agent` | Agent memory, session history, knowledge graph storage |
|
||||
| `clawhdf5-gpu` | GPU-accelerated I/O via wgpu (hand-written WGSL compute shaders) |
|
||||
| `clawhdf5-gpu` | GPU vector distance computation via wgpu (hand-written WGSL compute shaders) — not dataset I/O |
|
||||
| `clawhdf5-accel` | CPU SIMD acceleration path |
|
||||
| `clawhdf5-migrate` | Schema migration engine |
|
||||
| `clawhdf5-migrate` | SQLite → HDF5 agent-memory migration |
|
||||
| `clawhdf5-android` | Android JNI bindings |
|
||||
| `clawhdf5-cli` | Command-line interface |
|
||||
| `clawhdf5-cli` | Command-line interface (agent memory) |
|
||||
| `clawhdf5-tools` | `h5rs`: pure-Rust HDF5 tools — `ls`, `dump` (DDL / hdf5-json), `stat`, `diff`, `check` (structural + checksum validator) |
|
||||
| `clawhdf5-napi` | Node.js native addon bindings |
|
||||
| `clawhdf5-py` | PyO3 Python bindings |
|
||||
| `clawhdf5-wasm` | WebAssembly (wasm-bindgen) reader for the browser; demo in `examples/wasm-viewer/` |
|
||||
| `clawhdf5-bench` | Benchmark suite |
|
||||
|
||||
## Key Features
|
||||
- Zero-dependency HDF5 read/write (no libhdf5 C library required)
|
||||
- Zero-C-dependency HDF5 read/write: no libhdf5, and deflate defaults to
|
||||
pure-Rust zlib-rs (`fast-deflate` opts into zlib-ng, which needs cmake).
|
||||
`ci-test.sh` fails if a C-building crate enters the core crates' default
|
||||
tree. flate2 must keep `runtime_detection` with zlib-rs — without it zlib-rs
|
||||
loses SIMD and inflates 3.5x slower. MSRV is 1.92 (`rust-version`, checked
|
||||
in CI).
|
||||
- HNSW vector index for semantic similarity search over agent memories — the
|
||||
`clawhdf5-agent` `hnsw` feature is **on by default**, so `hybrid_search` uses
|
||||
the approximate `clawhdf5-ann` index for the vector stage (the index mirrors
|
||||
@@ -39,13 +46,20 @@ Cargo workspace with 16 crates under `crates/` (plus `libaec-sys`, an internal F
|
||||
(plain closest-M capped recall on clustered data: 0.31 recall@10 at 100K). Its
|
||||
graph is saved to `<store>.h5.ann` at each checkpoint and reloaded by `open()`
|
||||
(tied to the checkpoint by a generation id; stale/damaged sidecars are
|
||||
ignored and the index rebuilt). `MemoryConfig::quantized_index` (off by
|
||||
default, persisted) stores the index's own copy of the embeddings as `i8`,
|
||||
ignored and the index rebuilt). `MemoryConfig::quantized_index` (**on by
|
||||
default** for new stores, persisted; stores predating the setting load as
|
||||
`false` and keep their f32 index — guarded by
|
||||
`tests/fixtures/store_v2_5_0.h5`; CLI opt-out is `create --f32-index`)
|
||||
stores the index's own copy of the embeddings as `i8`,
|
||||
which roughly halves a loaded store's memory (2.72x -> 1.74x the raw vectors
|
||||
at 100K); because quantised distances are approximate and `ef` cannot
|
||||
compensate, the query path then re-scores the candidate pool against the
|
||||
exact embeddings, which holds recall at the f32 index's level and costs
|
||||
~13% of QPS. `hybrid_search` keeps one incremental BM25
|
||||
exact embeddings, which holds recall at the f32 index's level. It is also
|
||||
faster at equal recall: 1.63x the QPS on x86-64 (AVX2) and 1.18x on a
|
||||
Raspberry Pi 5 (`clawhdf5_accel::dot_i8`, NEON `SDOT` via inline asm since
|
||||
the intrinsic is unstable; plain NEON on pre-dotprod cores). The aarch64
|
||||
code is `cfg`'d out on x86, so x86 CI never compiles or lints it — test it
|
||||
on real ARM (`rpivision02`, 10.0.2.3, is a Pi 5). `hybrid_search` keeps one incremental BM25
|
||||
index for the life of the store and never writes the store: Hebbian
|
||||
activation boosts are persisted by the next checkpoint (or on drop), not per
|
||||
query. Measure any search-path change with
|
||||
@@ -75,8 +89,51 @@ Cargo workspace with 16 crates under `crates/` (plus `libaec-sys`, an internal F
|
||||
`export` do). An unreadable WAL (torn header, bad magic) is quarantined to
|
||||
`<store>.h5.wal.corrupt-<ts>` rather than blocking `open()`; a WAL with an
|
||||
unknown *newer* version still fails and is left untouched.
|
||||
- `MemoryConfig::compression` uses deflate by default; enable the agent's
|
||||
`zstd` feature to compress embeddings with Zstd instead (links libzstd).
|
||||
- `MemoryConfig::float16` (**on by default** for new stores, persisted;
|
||||
existing stores keep their recorded `false` — guarded by the v2.5.0
|
||||
fixture in `tests/float16_store.rs`; CLI opt-out is `create --f32`) writes
|
||||
`/memory/embeddings` as IEEE half precision (48% smaller file at 100K;
|
||||
LongMemEval with real MiniLM embeddings identical to f32).
|
||||
`MemoryCache::half_precision` rounds each embedding as it enters the cache (push, update, WAL replay, and on load of a store still
|
||||
`f32` on disk), so memory and file agree bit for bit; the conversions live
|
||||
in `clawhdf5_format::float16` and must stay the single implementation.
|
||||
Values beyond ±65504 are `MemoryError::InvalidEntry`. Interop: every file
|
||||
must open in h5py — `f32` datasets and empty datasets did not until
|
||||
2026-09-23 (see `docs/known-issues.md`); the agent's `h5py_interop` test
|
||||
guards a whole store.
|
||||
- `HDF5Memory::search(query_emb, text, &SearchOptions)` is the full search
|
||||
path: optional source-channel filter (applied before ranking; exact scan of
|
||||
the allowed records whenever cheaper than `pool × M` index distance
|
||||
evaluations, and as the fallback when the pool comes back short), fusion,
|
||||
activation scaling, optional re-ranking and confidence rejection.
|
||||
`hybrid_search`/`hybrid_search_with` are thin wrappers; `ClawhdfBackend`
|
||||
(the `openclaw` module) is `search` with re-rank + confidence on.
|
||||
- **OpenClaw is not supported** (decided 2026-09-25): clawhdf5 is not an
|
||||
OpenClaw memory plugin and never was — the old `memory.backend = "clawhdf5"`
|
||||
config was never valid. Don't reintroduce OpenClaw claims; `docs/openclaw.md`
|
||||
records what a real plugin would need.
|
||||
- **ZeroClaw does not use clawhdf5** (checked 2026-09-25 against upstream
|
||||
v0.8.5 and the `osobh/zeroclaw` fork, and their full history): no
|
||||
`clawhdf5` feature or backend exists; ZeroClaw's memory backends are
|
||||
sqlite/lucid/postgres/qdrant/markdown/none behind its own `Memory` trait.
|
||||
`clawhdf5-migrate`'s default SQLite layout (`memory_chunks`, `sessions`,
|
||||
`entities`, `relations`) is not ZeroClaw's schema either (ZeroClaw's is a
|
||||
`memories` table). Don't reintroduce integration claims without an
|
||||
integration and a test against the real consumer. Measure changes with
|
||||
`search_harness --options-study`.
|
||||
- `MemoryConfig::compression` is off by default; when on, embeddings are
|
||||
deflate-compressed, or Zstd with the agent's `zstd` feature (links libzstd).
|
||||
- Signed checkpoints (`clawhdf5-agent` `signing` module): with
|
||||
`HDF5Memory::set_signing_key` every checkpoint stores an Ed25519-signed
|
||||
manifest (SHA-256 per record in a Merkle tree + settings/sessions/graph
|
||||
hashes; per-record hashes in `/integrity/record_hashes`);
|
||||
`HDF5Memory::verify(path, &pk)` locates edits. The hashes must cover exactly
|
||||
what the file persists in the form the loader returns it (strings lose
|
||||
trailing NULs; an empty WAL mark is not written) or untouched stores stop
|
||||
verifying — `tests/signed_store.rs` round-trips awkward strings. The key is
|
||||
never persisted; a signed store refuses to checkpoint without it
|
||||
(`MemoryError::SigningKeyRequired`, and `MemoryError` is `#[non_exhaustive]`).
|
||||
WAL entries after the checkpoint are not covered.
|
||||
- `Dataset::verify_provenance()` (clawhdf5 facade, `provenance` feature, on by
|
||||
default) recomputes a dataset's SHA-256 and compares it against the
|
||||
`_provenance_sha256` attribute written automatically on save when
|
||||
@@ -93,7 +150,15 @@ Cargo workspace with 16 crates under `crates/` (plus `libaec-sys`, an internal F
|
||||
Alerts never block a save — drain them with `HDF5Memory::take_anomaly_alerts`.
|
||||
`MemorySource` for this bookkeeping is inferred from the caller-supplied
|
||||
`source_channel` string (a heuristic, not an authenticated trust boundary).
|
||||
- GPU-accelerated batch I/O for large dataset processing
|
||||
- GPU-accelerated vector distance computation (`clawhdf5-gpu`, wgpu); HDF5 I/O itself is CPU-only
|
||||
- Browser: `clawhdf5-wasm` (wasm-bindgen, read-only, file held in memory;
|
||||
no Zstd/SZIP since they link C) and the `examples/wasm-viewer/` page.
|
||||
`examples/wasm-viewer/test/run.sh` builds the package (needs the
|
||||
`wasm-bindgen` CLI at the crate's exact version) and tests it under Node
|
||||
and headless Chromium (a Playwright download in `~/.cache/ms-playwright`
|
||||
on tank); the CI container has neither, so CI runs the native
|
||||
`clawhdf5-wasm` `h5py_interop` test on the same fixture. Size numbers are
|
||||
in the example's README.
|
||||
- Python and Node.js bindings for cross-language use
|
||||
- NetCDF-4 compatibility for scientific data interop
|
||||
|
||||
@@ -109,12 +174,40 @@ cargo build --release
|
||||
cargo test --workspace
|
||||
```
|
||||
|
||||
### CI
|
||||
`.gitea/workflows/ci.yml` has two jobs, both green as of 2026-09-22:
|
||||
- **`test`** (`ubuntu-latest`, in `rust:latest`) runs `scripts/ci-test.sh` with
|
||||
the h5py/netCDF4 interop suites required (`CLAWHDF5_REQUIRE_INTEROP=1`).
|
||||
Served by the `tank` and `architect` runners.
|
||||
- **`test-arm64`** (`linux_arm64`) lints and tests the aarch64 code — the NEON
|
||||
kernels are `cfg`'d out on x86, so this is the only place they are built.
|
||||
Served by `vision-01` (host mode) and `vision-02` (Docker), so steps must
|
||||
work in both.
|
||||
|
||||
Keep workflows free of JavaScript actions (`actions/checkout`, `actions/cache`,
|
||||
…): `rust:latest` has no `node`, and not every runner reaches GitHub, where
|
||||
they are fetched from. Check out with plain `git` instead. The `test` job
|
||||
installs `cmake` for the opt-in `fast-deflate` (zlib-ng) steps; the default
|
||||
build needs no C toolchain, so `test-arm64` does not.
|
||||
All runners are on `gitea-runner` 3.5.0, from `docker.gitea.com/act_runner`
|
||||
— `gitea/act_runner:latest` on Docker Hub is frozen at 0.6.1.
|
||||
|
||||
### CLI
|
||||
```bash
|
||||
cargo run -p clawhdf5-cli -- --help
|
||||
# create, save, search, recall, stats, flush-wal, agents-md, export, snapshot subcommands
|
||||
```
|
||||
|
||||
### HDF5 tools (`h5rs`, crate `clawhdf5-tools`)
|
||||
```bash
|
||||
cargo run -p clawhdf5-tools -- ls -r file.h5 # also dump [--json], stat, diff, check
|
||||
bash scripts/h5rs-fuzz.sh # every subcommand over the CVE corpus: no panic/crash/hang
|
||||
bash scripts/h5rs-check-ok-files.sh --data # check passes every fully-read conformance file
|
||||
```
|
||||
Its interop tests compare against h5ls/h5stat/h5dump/h5diff (Debian
|
||||
`hdf5-tools`, installed in CI); `dump` must stay byte-identical to h5dump on
|
||||
the test files.
|
||||
|
||||
### Python bindings
|
||||
```bash
|
||||
cd crates/clawhdf5-py
|
||||
@@ -123,4 +216,12 @@ python -c "import clawhdf5; print(clawhdf5.__version__)"
|
||||
```
|
||||
|
||||
## Integration
|
||||
ZeroClaw imports this as a Cargo feature (`clawhdf5` feature flag) to persist agent memory with HNSW vector search for context retrieval.
|
||||
- **ClawBrainHub** (`clawverse/clawbrainhub` on git.redclaw.dev) is the one
|
||||
verified consumer: `cbh-core` reads and writes `.brain` files through the
|
||||
facade (`File`, `FileBuilder`, `AttrValue`, `Selection`), `cbh-scanner`
|
||||
uses the facade, and `cbh-cli` uses `clawhdf5_agent::bm25::BM25Index`. It
|
||||
depends on this repo by path (`../clawhdf5`), so it builds against whatever
|
||||
is checked out — changes to those APIs reach it directly. Verified
|
||||
2026-09-25 against main: builds, and its 204 tests pass.
|
||||
- OpenClaw and ZeroClaw were both described as consumers; neither integrates
|
||||
clawhdf5 (see Key Features and `docs/openclaw.md`).
|
||||
|
||||
+298
@@ -0,0 +1,298 @@
|
||||
# clawhdf5 conformance report
|
||||
|
||||
Every HDF5 file of eight public corpora (pinned by commit) is read twice — by
|
||||
clawhdf5 (`conformance/probe`, the same `clawhdf5-format` calls the facade
|
||||
makes) and by h5py/libhdf5 (`conformance/ref.py`) — and the two readings are
|
||||
compared object by object: the set of hard-linked objects, each dataset's and
|
||||
attribute's shape, and a SHA-256 of its values in a canonical encoding. The
|
||||
CVE corpus is also run through `h5dump`. Each side runs under a timeout and an
|
||||
address-space limit, so a hang, crash or runaway allocation is recorded, not
|
||||
fatal. This file is generated by `conformance/run.sh`; do not edit it by hand.
|
||||
|
||||
## Run
|
||||
|
||||
| | |
|
||||
|---|---|
|
||||
| date | 2026-09-26 14:18 UTC |
|
||||
| clawhdf5 commit | `73a01f1256fb9bf1b1e7601f755af9e8273cec4e` |
|
||||
| machine | `tank`: AMD Ryzen 7 7800X3D 8-Core Processor, 16 CPUs, 61 GiB, Linux 7.0.0-34-generic x86_64 |
|
||||
| command | `conformance/run.sh --no-fetch --update-baseline` |
|
||||
| rustc | rustc 1.98.1 (48a229cea 2026-09-01) |
|
||||
| reference | h5py 3.16.0, HDF5 2.0.0, numpy 2.5.3, hdf5plugin 7.1.0, Python 3.14.4 |
|
||||
| h5dump | Version 1.14.6 (CVE corpus only) |
|
||||
| limits | 20 s timeout (SIGKILL), 4096 MiB address space, per process; 16 files in parallel |
|
||||
| runtime | 23 s probing + comparing (0 s fetch/build before it) |
|
||||
|
||||
## Results
|
||||
|
||||
A file's class is the first that applies:
|
||||
|
||||
- **panic / hang / crash / oom** — clawhdf5 panicked (caught per object or not), hit the timeout, died on a signal, or failed an allocation. The CI gate fails on any of these.
|
||||
- **h5py-cannot-read** — libhdf5 could not open the file (or itself crashed or hung). Nothing to compare against; most are the deliberately malformed CVE reproducers.
|
||||
- **our-error** — clawhdf5 returned an error for something h5py reads.
|
||||
- **mismatch** — both read it, but the shapes, values, object set or attribute set differ.
|
||||
- **ok** — every object h5py reads, clawhdf5 reads identically.
|
||||
|
||||
| corpus | files | ok | our-error | mismatch | h5py-cannot-read | panic | hang | crash | oom |
|
||||
|---|---|---|---|---|---|---|---|---|---|
|
||||
| NCAS-CMS_pyfive | 33 | 32 | 0 | 1 | 0 | 0 | 0 | 0 | 0 |
|
||||
| cve_hdf5 | 147 | 100 | 6 | 9 | 32 | 0 | 0 | 0 | 0 |
|
||||
| h5py_data | 4 | 4 | 0 | 0 | 0 | 0 | 0 | 0 | 0 |
|
||||
| hdf5 | 466 | 392 | 4 | 10 | 60 | 0 | 0 | 0 | 0 |
|
||||
| netcdf-c | 20 | 20 | 0 | 0 | 0 | 0 | 0 | 0 | 0 |
|
||||
| netcdf4-python | 18 | 18 | 0 | 0 | 0 | 0 | 0 | 0 | 0 |
|
||||
| usnistgov_h5wasm | 5 | 5 | 0 | 0 | 0 | 0 | 0 | 0 | 0 |
|
||||
| xarray-data | 4 | 4 | 0 | 0 | 0 | 0 | 0 | 0 | 0 |
|
||||
| **all** | **697** | **575** | **10** | **20** | **92** | **0** | **0** | **0** | **0** |
|
||||
|
||||
2 of the 20 mismatches are a known h5py bug, not ours (see *Known not-our-bug*).
|
||||
|
||||
Corpora (fetched by `conformance/fetch-corpus.sh` into the gitignored `conformance/.cache/`):
|
||||
|
||||
| corpus | source | commit |
|
||||
|---|---|---|
|
||||
| hdf5 | https://github.com/HDFGroup/hdf5 | `a3cf1ea82cc7` |
|
||||
| cve_hdf5 | https://github.com/HDFGroup/cve_hdf5 | `3fd1f5ae3869` |
|
||||
| netcdf-c | https://github.com/Unidata/netcdf-c | `beb7b9585273` |
|
||||
| NCAS-CMS_pyfive | https://github.com/NCAS-CMS/pyfive | `8cf07b874913` |
|
||||
| usnistgov_h5wasm | https://github.com/usnistgov/h5wasm | `02f6336527d2` |
|
||||
| netcdf4-python | https://github.com/Unidata/netcdf4-python | `6e67576d39ae` |
|
||||
| xarray-data | https://github.com/pydata/xarray-data | `a35297e9da2c` |
|
||||
| h5py_data | https://github.com/h5py/h5py (`h5py/tests/data_files`) | `b2f0347c4200` |
|
||||
|
||||
## Panics, hangs, crashes, out-of-memory
|
||||
|
||||
None.
|
||||
|
||||
## Our-error root causes
|
||||
|
||||
Grouped by normalised error message. *files* counts files whose class this cause affects.
|
||||
|
||||
| files | objects | error | examples |
|
||||
|---:|---:|---|---|
|
||||
| 3 | 3 | `DataSizeMismatch { expected: N, actual: N }` | `cve_hdf5/cvefiles/cve-2020-18494.h5`, `cve_hdf5/cvefiles/cve-2024-32623.h5`, `cve_hdf5/cvefiles/cve-2025-2309.h5` |
|
||||
| 2 | 2 | `ChunkedReadError("…")` | `cve_hdf5/cvefiles/cve-2025-2308.h5`, `hdf5/test/testfiles/bad_nbit_parms_walk.h5` |
|
||||
| 2 | 2 | `UnsupportedFilter(N)` | `hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_blosc2.h5`, `hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_zfp.h5` |
|
||||
| 1 | 1 | `UnexpectedEof { expected: N, available: N }` | `cve_hdf5/cvefiles/cve-2019-9151.h5` |
|
||||
| 1 | 1 | `MissingMessage(Dataspace)` | `cve_hdf5/cvefiles/cve-2024-33874.h5` |
|
||||
| 1 | 1 | `InvalidObjectHeaderVersion(N)` | `hdf5/tools/test/testfiles/h5clear_mdc_image.h5` |
|
||||
|
||||
## Mismatch root causes
|
||||
|
||||
| files | objects | cause | examples |
|
||||
|---:|---:|---|---|
|
||||
| 13 | 14 | `missing-object` | `cve_hdf5/cvefiles/cve-2019-8397.h5`, `cve_hdf5/cvefiles/cve-2019-8398.h5`, `cve_hdf5/cvefiles/cve-2021-46243.h5` (+10 more) |
|
||||
| 2 | 6 | `extra-attr` | `cve_hdf5/cvefiles/cve-2018-17438`, `cve_hdf5/cvefiles/cve-2018-17439` |
|
||||
| 1 | 1 | `attr-values: ours=vlen(>u8) h5py=object layout=- filters=-` | `NCAS-CMS_pyfive/tests/data/attr_datatypes.hdf5` |
|
||||
| 1 | 4 | `extra-object` | `cve_hdf5/cvefiles/cve-2021-46244.h5` |
|
||||
| 1 | 1 | `values: ours=<f4 h5py=float32 layout=chunked filters=-` | `cve_hdf5/cvefiles/cve-2025-44904.h5` |
|
||||
| 1 | 1 | `values: ours=>i2 h5py=>i2 layout=chunked filters=[6]` | `cve_hdf5/cvefiles/cve-2025-44905.h5` |
|
||||
| 1 | 1 | `values: ours=>f4 h5py=>f4 layout=chunked filters=[2]` | `cve_hdf5/cvefiles/cve-2025-44905.h5` |
|
||||
| 1 | 1 | `values: ours=<f4 h5py=float32 layout=chunked filters=[2]` | `cve_hdf5/cvefiles/cve-2025-44905.h5` |
|
||||
| 1 | 1 | `values: ours=((<i4)[6, 3])[4] h5py=(('<i4', (6, 3)), (4,)) layout=contiguous filters=-` | `hdf5/tools/test/testfiles/tarray3.h5` |
|
||||
| 1 | 1 | `values: ours=vlen({r:>f4,i:>f4}8) h5py=object layout=contiguous filters=-` | `hdf5/tools/test/testfiles/tcomplex_be.h5` |
|
||||
|
||||
## CVE corpus: clawhdf5 vs h5dump vs h5py
|
||||
|
||||
The 147 files of [HDFGroup/cve_hdf5](https://github.com/HDFGroup/cve_hdf5) — reproducers for
|
||||
published libhdf5 CVEs and fuzzer finds. *read* = produced output (possibly with per-object
|
||||
errors), *error* = refused cleanly. h5dump exits non-zero on any error anywhere in a file, so
|
||||
its read/error split is not comparable with the other two rows; the panic, crash, hang and oom
|
||||
columns are.
|
||||
|
||||
| tool | read | error | panic | crash | hang | oom |
|
||||
|---|---:|---:|---:|---:|---:|---:|
|
||||
| clawhdf5 | 140 | 7 | 0 | 0 | 0 | 0 |
|
||||
| h5dump 1.14.6 | 16 | 129 | 0 | 2 | 0 | 0 |
|
||||
| h5py 3.16.0 / HDF5 2.0.0 | 115 | 31 | 0 | 1 | 0 | 0 |
|
||||
|
||||
<details><summary>Per-file outcomes</summary>
|
||||
|
||||
| file | h5dump | h5py | clawhdf5 | class |
|
||||
|---|---|---|---|---|
|
||||
| cvefiles/cve-2016-4330.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2016-4331.h5 | error exit | read 25 obj, 1 errors | read 25 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2016-4332-mtime-new.h5 | error exit | read 25 obj, 1 errors | read 25 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2016-4332-mtime.h5 | error exit | read 4 obj, 3 errors | read 4 obj, 3 errors | ok |
|
||||
| cvefiles/cve-2016-4332-stab.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2016-4333.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2017-17505.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2017-17506.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2017-17507.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2017-17508.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2017-17509.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-11202.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-11203.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-11204.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-11205.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-11206-new.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-11206-old.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-11207.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-13866.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||
| cvefiles/cve-2018-13867.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-13868.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-13869.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-13870.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-13871.h5 | error exit | read 2 obj | read 2 obj | ok |
|
||||
| cvefiles/cve-2018-13872.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-13873.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-13874.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||
| cvefiles/cve-2018-13875.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-13876.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||
| cvefiles/cve-2018-14031.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-14033.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-14034.h5 | error exit | read 1 obj, 2 errors | read 1 obj | ok |
|
||||
| cvefiles/cve-2018-14035.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-14460.h5 | error exit | read 3 obj, 2 errors | read 3 obj, 2 errors | ok |
|
||||
| cvefiles/cve-2018-15671.h5 | ok | read 1 obj | read 1 obj | ok |
|
||||
| cvefiles/cve-2018-15672.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-16438.h5 | error exit | read 1 obj, 1 errors | read 1 obj | ok |
|
||||
| cvefiles/cve-2018-17233.h5 | error exit | read 6 obj, 1 errors | read 6 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-17234.h5 | error exit | read 6 obj, 1 errors | read 6 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-17237.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-17432.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-17433 | error exit | open error | open error | h5py-cannot-read |
|
||||
| cvefiles/cve-2018-17434.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-17435.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-17436 | error exit | open error | open error | h5py-cannot-read |
|
||||
| cvefiles/cve-2018-17437.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2018-17438 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | mismatch |
|
||||
| cvefiles/cve-2018-17439 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | mismatch |
|
||||
| cvefiles/cve-2019-8396.h5 | error exit | read 3 obj, 2 errors | read 3 obj, 2 errors | ok |
|
||||
| cvefiles/cve-2019-8397.h5 | error exit | read 3 obj, 2 errors | read 2 obj, 1 errors | mismatch |
|
||||
| cvefiles/cve-2019-8398.h5 | error exit | read 3 obj, 2 errors | read 2 obj, 1 errors | mismatch |
|
||||
| cvefiles/cve-2019-9151.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 2 errors | our-error |
|
||||
| cvefiles/cve-2019-9152.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2020-10809 | error exit | open error | open error | h5py-cannot-read |
|
||||
| cvefiles/cve-2020-10810.h5 | error exit | open error | read 2 obj | h5py-cannot-read |
|
||||
| cvefiles/cve-2020-10811.h5 | error exit | read 25 obj, 1 errors | read 25 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2020-10812.h5 | error exit | open error | read 2 obj | h5py-cannot-read |
|
||||
| cvefiles/cve-2020-18232.h5 | error exit | read 3 obj, 2 errors | read 3 obj, 2 errors | ok |
|
||||
| cvefiles/cve-2020-18494.h5 | ok | read 2 obj | read 2 obj, 1 errors | our-error |
|
||||
| cvefiles/cve-2021-36977.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2021-37501.h5 | error exit | read 18 obj, 1 errors | read 18 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2021-45829.h5 | error exit | read 1 obj, 2 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2021-45830.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2021-45833.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2021-46242.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2021-46243.h5 | error exit | read 3 obj, 2 errors | read 2 obj, 1 errors | mismatch |
|
||||
| cvefiles/cve-2021-46244.h5 | error exit | read 2 obj, 1 errors | read 6 obj, 4 errors | mismatch |
|
||||
| cvefiles/cve-2024-29157.h5 | error exit | read 4 obj, 7 errors | read 4 obj, 7 errors | ok |
|
||||
| cvefiles/cve-2024-29158.h5 | ok | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-29159.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-29160.h5 | error exit | read 4 obj, 1 errors | read 4 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-29161.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-29162.h5 | error exit | read 17 obj, 4 errors | read 17 obj, 4 errors | ok |
|
||||
| cvefiles/cve-2024-29163.h5 | error exit | read 7 obj, 1 errors | read 7 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-29164.h5 | ok | read 3 obj | read 3 obj | ok |
|
||||
| cvefiles/cve-2024-29165.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-29166.h5 | error exit | read 17 obj, 2 errors | read 17 obj | ok |
|
||||
| cvefiles/cve-2024-32605.h5 | ok | read 6 obj, 1 errors | read 6 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-32606.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-32607-1.h5 | ok | read 10 obj | read 10 obj | ok |
|
||||
| cvefiles/cve-2024-32607-2.h5 | error exit | read 9 obj, 1 errors | read 9 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-32608.h5 | error exit | read 6 obj, 1 errors | read 6 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-32609.h5 | error exit | SIGSEGV | read 3 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2024-32610.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-32611.h5 | ok | read 6 obj | read 6 obj | ok |
|
||||
| cvefiles/cve-2024-32612.h5 | ok | read 3 obj | read 3 obj | ok |
|
||||
| cvefiles/cve-2024-32613.h5 | error exit | read 7 obj, 1 errors | read 7 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-32614.h5 | error exit | read 25 obj, 2 errors | read 25 obj, 2 errors | ok |
|
||||
| cvefiles/cve-2024-32615.h5 | error exit | read 4 obj, 1 errors | read 4 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-32616.h5 | error exit | read 10 obj, 7 errors | read 10 obj, 6 errors | ok |
|
||||
| cvefiles/cve-2024-32617.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-32618.h5 | error exit | read 4 obj, 2 errors | read 3 obj, 1 errors | mismatch |
|
||||
| cvefiles/cve-2024-32619.h5 | error exit | read 3 obj, 2 errors | read 3 obj, 2 errors | ok |
|
||||
| cvefiles/cve-2024-32620.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-32621.h5 | ok | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-32622.h5 | ok | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-32623.h5 | ok | read 6 obj | read 6 obj, 1 errors | our-error |
|
||||
| cvefiles/cve-2024-32624.h5 | error exit | read 6 obj, 1 errors | read 6 obj | ok |
|
||||
| cvefiles/cve-2024-33873.h5 | error exit | read 4 obj, 1 errors | read 4 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-33874.h5 | ok | read 6 obj, 1 errors | read 6 obj, 2 errors | our-error |
|
||||
| cvefiles/cve-2024-33875.h5 | ok | read 2 obj | read 2 obj | ok |
|
||||
| cvefiles/cve-2024-33876.h5 | ok | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2024-33877.h5 | error exit | read 8 obj, 1 errors | read 8 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2025-2153.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2025-2308.h5 | error exit | read 25 obj, 1 errors | read 25 obj, 2 errors | our-error |
|
||||
| cvefiles/cve-2025-2309.h5 | ok | read 6 obj, 1 errors | read 6 obj, 1 errors | our-error |
|
||||
| cvefiles/cve-2025-2310.h5 | error exit | read 24 obj, 8 errors | read 24 obj, 8 errors | ok |
|
||||
| cvefiles/cve-2025-2912.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2025-2913.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2025-2914.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2025-2915.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2025-2923.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2025-2924.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2025-2925.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2025-2926.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2025-44904.h5 | error exit | read 25 obj, 1 errors | read 25 obj, 1 errors | mismatch |
|
||||
| cvefiles/cve-2025-44905.h5 | error exit | read 25 obj, 3 errors | read 25 obj, 3 errors | mismatch |
|
||||
| cvefiles/cve-2025-6269-1.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2025-6269-2.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2025-6269-3.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2025-6269-4.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2025-6270-1.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2025-6270-2.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2025-6270-3.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2025-6516.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2025-6750.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2025-6816.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2025-6817.h5 | error exit | open error | read 1 obj | h5py-cannot-read |
|
||||
| cvefiles/cve-2025-6818.h5 | error exit | open error | read 1 obj | h5py-cannot-read |
|
||||
| cvefiles/cve-2025-6856.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2025-6857.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2025-6858.h5 | SIGSEGV | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2025-7067.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2025-7068.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2025-7069.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| cvefiles/cve-2026-26200.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2026-34734.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||
| cvefiles/cve-2026-92627.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||
| cvefiles/unknown-1.h5 | error exit | read 11 obj, 1 errors | read 11 obj, 1 errors | ok |
|
||||
| fuzzerfiles/gh-4431-poc-03.h5 | error exit | read 1 obj | read 1 obj | ok |
|
||||
| fuzzerfiles/gh-4432-poc-05.h5 | SIGSEGV | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| fuzzerfiles/gh-4433-poc-08.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| fuzzerfiles/gh-4434-poc-09.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||
| fuzzerfiles/gh-4435-poc-10.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||
| fuzzerfiles/gh-4585.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||
| fuzzerfiles/gh_2649_flawed.h5 | error exit | read 9 obj, 1 errors | read 9 obj, 1 errors | ok |
|
||||
| fuzzerfiles/gh_2649_plain_model.h5 | ok | read 10 obj | read 10 obj | ok |
|
||||
|
||||
</details>
|
||||
|
||||
## Known not-our-bug
|
||||
|
||||
- **h5py big-endian variable-length sequences.** h5py returns the elements of a VL sequence
|
||||
whose base type is big-endian with the file's big-endian bytes but a native (little-endian)
|
||||
numpy dtype, so the values it reports are byte-swapped garbage; `h5dump` prints the values
|
||||
clawhdf5 reads. Reproducer: `h5py.vlen_dtype(np.dtype('>f4'))` dataset holding `[1.0, 2.0]`
|
||||
reads back in h5py as `[4.6e-41, 9.0e-44]`. Affected here: `NCAS-CMS_pyfive/tests/data/attr_datatypes.hdf5`, `hdf5/tools/test/testfiles/tcomplex_be.h5`.
|
||||
- **Non-IEEE floats and partial-precision integers (N-Bit).** libhdf5 converts a float whose
|
||||
bit layout is not IEEE (e.g. `H5Tset_precision` for the N-Bit filter) or an integer with a
|
||||
bit offset / reduced precision into the plain numpy type of the same size. The probe
|
||||
compares such values as converted numbers, not raw file bytes (before 2026-09-25 it compared
|
||||
raw bytes, which reported every N-Bit float dataset as a mismatch).
|
||||
- **Types h5py widens.** Where h5py reads a type into a numpy type of a different size
|
||||
(FP8 -> float16, bfloat16 -> float32, x87 long double -> float128) the values are not
|
||||
compared (shape and presence still are): dataset file type size 1 -> numpy float16 (2) (15x), attr file type size 1 -> numpy float16 (2) (15x), dataset file type size 2 -> numpy float32 (4) (2x), dataset file type size 8 -> numpy float128 (16) (1x), dataset file type size 12 -> numpy float128 (16) (1x), attr file type size 2 -> numpy float32 (4) (1x), dataset file type size 2 -> numpy >f4 (4) (1x), attr file type size 2 -> numpy >f4 (4) (1x).
|
||||
- **References** are compared by presence only (`R`), not by target.
|
||||
|
||||
## Objects h5py fails on but clawhdf5 reads
|
||||
|
||||
- 19 x `OSError: Can't synchronously read data (no appropriate function for conversion path)`
|
||||
- 1 x `TypeError: unhandled dtype kind M (dtype('…'))`
|
||||
- 1 x `TypeError: No NumPy equivalent for TypeTimeID exists`
|
||||
- 1 x `KeyError: "…"`
|
||||
- 1 x `ValueError: Insufficient precision in available types to represent (N, N, N, N, N)`
|
||||
|
||||
## Reproduce
|
||||
|
||||
```sh
|
||||
# needs: Rust, python3 with h5py numpy hdf5plugin (conformance/requirements.txt), h5dump (hdf5-tools), git
|
||||
CLAWHDF5_PYTHON=/path/to/venv/bin/python conformance/run.sh
|
||||
```
|
||||
|
||||
The corpus (about 450 MB of sparse checkouts) is cached in `conformance/.cache/`; results for
|
||||
every file, both sides' raw JSON and stderr, are in `conformance/.cache/results/`.
|
||||
`conformance/baseline.json` holds the ok files the nightly CI job (`.gitea/workflows/conformance.yml`)
|
||||
must keep; `conformance/run.sh --update-baseline` rewrites it.
|
||||
+15
-1
@@ -16,13 +16,18 @@ members = [
|
||||
"crates/clawhdf5-cli",
|
||||
"crates/clawhdf5-napi",
|
||||
"crates/clawhdf5-bench",
|
||||
"crates/clawhdf5-tools",
|
||||
"crates/clawhdf5-wasm",
|
||||
"crates/libaec-sys",
|
||||
]
|
||||
resolver = "2"
|
||||
|
||||
[workspace.package]
|
||||
version = "2.6.0"
|
||||
version = "2.7.0"
|
||||
edition = "2024"
|
||||
# Oldest toolchain that builds the whole workspace; CI checks it. wgpu (in
|
||||
# clawhdf5-gpu) requires 1.92.
|
||||
rust-version = "1.92"
|
||||
license = "MIT"
|
||||
repository = "https://git.redclaw.dev/quantumclaw/clawhdf5"
|
||||
|
||||
@@ -31,3 +36,12 @@ tempfile = "3"
|
||||
criterion = { version = "0.5", features = ["html_reports"] }
|
||||
half = "2.7"
|
||||
serde = { version = "1", features = ["derive"] }
|
||||
|
||||
# The browser build of clawhdf5-wasm (examples/wasm-viewer/build.sh): size
|
||||
# over speed, whole-program optimisation. Native profiles are unaffected.
|
||||
[profile.wasm-release]
|
||||
inherits = "release"
|
||||
opt-level = "s"
|
||||
lto = true
|
||||
codegen-units = 1
|
||||
panic = "abort"
|
||||
|
||||
@@ -3,24 +3,103 @@
|
||||
**The memory layer AI agents deserve. One file. Pure Rust. Zero C dependencies.**
|
||||
|
||||
[](LICENSE)
|
||||
[](https://www.rust-lang.org)
|
||||
[](#performance)
|
||||
[](BENCHMARKS.md#longmemeval-results)
|
||||
[](BENCHMARKS.md#memory-footprint)
|
||||
[](https://www.rust-lang.org)
|
||||
[](#building)
|
||||
[](BENCHMARKS.md#longmemeval-results)
|
||||
[](BENCHMARKS.md#memory-footprint-1)
|
||||
|
||||
ClawHDF5 is a pure-Rust HDF5 implementation combined with a research-grade agent memory engine. It gives AI agents persistent, searchable, cryptographically verifiable memory — all stored in a single portable file.
|
||||
ClawHDF5 is a pure-Rust HDF5 implementation combined with a research-grade agent memory engine. It gives AI agents persistent, searchable, cryptographically verifiable memory (Ed25519-signed checkpoints) — all stored in a single portable file.
|
||||
|
||||
> **Two things live here:**
|
||||
> - **A general-purpose, pure-Rust HDF5 library** — zero C dependencies, NetCDF-4 support, SIMD/GPU acceleration. See the **[Crate Map](#crate-map)** and **[BENCHMARKS.md](BENCHMARKS.md)** for the libhdf5 head-to-head numbers.
|
||||
> - **An agent memory layer built on top of it** — vector search, knowledge graph, hippocampal-style consolidation, in `clawhdf5-agent`.
|
||||
|
||||
```
|
||||
cargo add clawhdf5 # core HDF5 read/write, no agent layer
|
||||
cargo add clawhdf5-agent --features agent # + agent memory layer
|
||||
The crates are not on crates.io yet, so depend on them from git:
|
||||
|
||||
```toml
|
||||
[dependencies]
|
||||
clawhdf5 = { git = "https://git.redclaw.dev/quantumclaw/clawhdf5" } # core HDF5 read/write
|
||||
clawhdf5-agent = { git = "https://git.redclaw.dev/quantumclaw/clawhdf5" } # + agent memory layer
|
||||
```
|
||||
|
||||
> **C dependencies, precisely:** the core crates (`clawhdf5`, `clawhdf5-agent`,
|
||||
> `-format`, `-io`, `-filters`, `-ann`, `-accel`, `-netcdf4`, `-cli`) build no C
|
||||
> code by default — no libhdf5, and deflate is the pure-Rust
|
||||
> [zlib-rs](https://github.com/trifectatechfoundation/zlib-rs), which matches
|
||||
> zlib-ng on HDF5 reads and writes and produces byte-identical output
|
||||
> ([BENCHMARKS.md § Deflate backend](BENCHMARKS.md#deflate-backend-zlib-rs-vs-zlib-ng)).
|
||||
> CI fails if a C-building crate enters their default dependency tree. C comes
|
||||
> in only when you ask for it: `fast-deflate` (zlib-ng, needs cmake), `zstd`,
|
||||
> `szip`, the BLAS backends, `clawhdf5-migrate` (bundled SQLite) and the
|
||||
> Node.js bindings.
|
||||
|
||||
> **New here?** Start with the **[Quickstart Guide](docs/QUICKSTART.md)** · See **[Use Cases](docs/USE_CASES.md)** · Read **[Benchmarks](BENCHMARKS.md)**
|
||||
|
||||
## What's new (v2.2 → v2.7, and unreleased)
|
||||
|
||||
Five releases in September 2026. Details, including upgrade notes and every
|
||||
breaking change, are in [CHANGELOG.md](CHANGELOG.md).
|
||||
|
||||
**HDF5 correctness (read these if you read files with an earlier release)**
|
||||
- **Extensible Array chunk indexes returned wrong data** past the 36th chunk —
|
||||
any dataset with one unlimited dimension. Silent: plausible numbers from the
|
||||
wrong chunks. Fixed in v2.7.0; re-read affected data.
|
||||
- Fixed and Extensible Array checksums are now verified, so a corrupt chunk
|
||||
index is `ChecksumMismatch` instead of wrong data (v2.7.0).
|
||||
- Compound datatypes written with default libver bounds (plain
|
||||
`h5py.File(path, 'w')`) were mis-parsed; HDF5 2.0 compound v5 and native
|
||||
complex (class 11) types now parse (v2.2.0–v2.3.0).
|
||||
- Committed datatypes, fill values, soft links and `H5T_STD_REF` references now
|
||||
read correctly; external links and external raw data are explicit errors;
|
||||
`attrs()` no longer silently drops attributes (v2.3.0–v2.5.0).
|
||||
- Datasets indexed by a version-2 B-tree now read (v2.5.0).
|
||||
|
||||
**Security and robustness**
|
||||
- A crafted file could abort any reader via B-tree v2 recursion or explode it
|
||||
via shared children; both are now fast errors (v2.7.0).
|
||||
- Virtual-dataset source paths are confined to the file's directory; chunked
|
||||
reads use overflow-checked sizes and fallible allocation, and the facade
|
||||
writes files atomically (v2.3.0).
|
||||
- Agent store: single-writer lock plus `open_read_only`; a crash between
|
||||
checkpoint and WAL truncate no longer duplicates entries; unreadable WALs are
|
||||
quarantined instead of blocking `open()` (v2.3.0).
|
||||
|
||||
**Search quality and speed**
|
||||
- HNSW neighbour selection now uses the paper's diversity heuristic: recall@10
|
||||
at 100K went from 0.31 to 0.98 (v2.4.0).
|
||||
- `hybrid_search` is 79–190× faster than v2.3.0 (p50 0.07 ms at 1K, 4.65 ms at
|
||||
100K). It no longer rebuilds BM25 or rewrites the store per query, and the
|
||||
HNSW graph is persisted (v2.4.0).
|
||||
- Default fusion weights are now the measured 0.4 / 0.6 (v2.5.0). Re-ranking had
|
||||
been discarding the retrieval score, costing the Markdown backend 40.6pp of
|
||||
Hit@1; fixed in v2.6.0.
|
||||
- Selection reads whose bounding box covers at most half the dataset decode
|
||||
only the chunks they touch (a 64×64 window: 105 ms to 0.39 ms), and full
|
||||
reads are 1.2–1.9× faster (v2.5.0).
|
||||
|
||||
**Memory**
|
||||
- A loaded store holds ~30% less (embeddings stored once, v2.6.0), and the
|
||||
int8 HNSW index, **on by default for new stores** (unreleased), brings a
|
||||
100K × 384 store to 1.74× the raw vectors. At equal recall it is also faster
|
||||
than `f32`: 1.63× QPS on AVX2, 1.18× on a Raspberry Pi 5 (NEON `SDOT`).
|
||||
|
||||
**Interop and search (unreleased)**
|
||||
- **Files we write now open in h5py and libhdf5.** Every `f32` dataset —
|
||||
including every agent store's embeddings — and every empty dataset was
|
||||
refused by libhdf5. Both were write-side bugs in every release; agent stores
|
||||
fix themselves at their next checkpoint. See
|
||||
[docs/known-issues.md](docs/known-issues.md).
|
||||
- `MemoryConfig::float16` now stores half-precision embeddings (it was
|
||||
ignored), and is on by default for new stores: 48% smaller files, and
|
||||
identical LongMemEval retrieval on real embeddings.
|
||||
- `HDF5Memory::search` with `SearchOptions`: filter by source channel (exact
|
||||
filtered top-k, never slower than unfiltered), and opt-in re-ranking and
|
||||
confidence rejection, which used to be reachable only through `ClawhdfBackend`.
|
||||
|
||||
**Tooling**
|
||||
- CI now runs the h5py/netCDF4 interop suites for real (they had been skipping
|
||||
silently) and runs an aarch64 job for the NEON kernels.
|
||||
|
||||
---
|
||||
|
||||
## Why ClawhDF5?
|
||||
@@ -33,16 +112,16 @@ Every AI agent needs memory. Today that means scattered Markdown files, SQLite d
|
||||
| Keyword search | Separate FTS engine | Integrated BM25 |
|
||||
| Knowledge graph | Neo4j or none | In-file graph with spreading activation |
|
||||
| Memory consolidation | Manual pruning | Hippocampal-inspired automatic tiers |
|
||||
| Temporal queries | Custom code | Native temporal index (716ns) |
|
||||
| Multi-modal | Multiple stores | Unified cross-modal search |
|
||||
| Security | Hope for the best | Provenance tracking + anomaly detection |
|
||||
| Temporal queries | Custom code | Native temporal index (622 ns range query over 10K) |
|
||||
| Multi-modal | Multiple stores | Unified cross-modal search (exact scan: 842 µs over 1K records) |
|
||||
| Integrity | Hope for the best | Ed25519-signed checkpoints that pinpoint any edited record, chained-CRC WAL, checksummed chunk indexes, write-anomaly alerts |
|
||||
| Portability | Config + DB + files | **One `.h5` file. Copy it anywhere.** |
|
||||
|
||||
---
|
||||
|
||||
## Performance
|
||||
|
||||
Vector search and agent-memory operations below are benchmarked on Intel i7-12650H (10C/16T), 384-dim embeddings, Criterion.rs. The HDF5 Core I/O table immediately below is from a separate, independently reproduced run (see its own hardware note).
|
||||
The brute-force/IVF vector search, agent-memory, on-disk footprint and consolidation figures below were measured 2026-09-24 on tank (AMD Ryzen 7 7800X3D, 8C/16T), commit 5c8323c, 384-dim embeddings; the commands are in [BENCHMARKS.md](BENCHMARKS.md). Exceptions are marked where they appear: the HDF5 Core I/O table immediately below is from a separate, independently reproduced run (see its own hardware note), and the HNSW `f32`/`i8` table and the in-memory `i8` column were not re-measured on 2026-09-24.
|
||||
|
||||
### HDF5 Core I/O (vs libhdf5 1.14.6)
|
||||
|
||||
@@ -58,31 +137,63 @@ Figures below are from an independent reproduction run on a second machine (AMD
|
||||
| Sequential read (100K f32) | 23.3 µs | 63.6 µs | **2.7×** |
|
||||
| Sequential write (100K f32) | 210 µs | 189 µs | **≈ tie** |
|
||||
|
||||
The chunked-write row was re-measured on the same machine on 2026-09-23, after
|
||||
the default deflate backend became pure-Rust zlib-rs: 1.46 ms against
|
||||
libhdf5's 51.4 ms (**35×**), and 1.48 ms with zlib-ng. libhdf5's own time on
|
||||
that machine moved from 65.0 to 51.4 ms between the two dates, which is most
|
||||
of the difference from 45×; compare same-day numbers only.
|
||||
|
||||
### Vector Search
|
||||
|
||||
| Scale | Flat | IVF (nprobe=10) | IVF-PQ | vs MemX¹ |
|
||||
|-------|------|-----------------|--------|----------|
|
||||
| 1K | **54 µs** | — | — | — |
|
||||
| 10K | 753 µs | **27 µs** | — | — |
|
||||
| 100K | 11.4 ms | 1.32 ms | **1.19 ms** | ~8–76× (see caveat) |
|
||||
**HNSW (the default backend for `hybrid_search`)** — `search_harness`, clustered
|
||||
384-dim data, M = 16, ef_construction = 64, recall measured against an exact scan.
|
||||
See [BENCHMARKS.md § Search harness](BENCHMARKS.md#search-harness-baseline-v230)
|
||||
and [§ Quantising the index copy](BENCHMARKS.md#quantising-the-index-copy-quantized_index):
|
||||
|
||||
> Reproduced on the same second machine (Ryzen 7 7800X3D) with a corrected,
|
||||
> apples-to-apples SIMD/scalar/parallel comparison methodology — see
|
||||
> [BENCHMARKS.md § Independent Validation: tank — LongMemEval & Vector
|
||||
> Search](BENCHMARKS.md#independent-validation-tank--longmemeval--vector-search-ryzen-7-7800x3d-2026-08-05).
|
||||
| N = 100K, ef = 64 | recall@10 | QPS | build |
|
||||
|---|---:|---:|---:|
|
||||
| `f32` index | 0.9945 | 13 399 | 3.2 s |
|
||||
| `i8` index + exact re-score (**default for new stores**) | 0.9940 | **21 848** | **1.8 s** |
|
||||
|
||||
Before the v2.4.0 neighbour-selection fix, recall@10 at 100K was 0.31. These
|
||||
two rows are a paired comparison (medians of alternating runs, same binary).
|
||||
A single `f32` run on 2026-09-24 measured recall 0.9945, 19 001 QPS and a
|
||||
2.7 s build; the int8 row was not re-run, so the pair has not been re-checked
|
||||
([§ Quantising the index copy](BENCHMARKS.md#quantising-the-index-copy-quantized_index)).
|
||||
|
||||
**Brute-force and IVF paths** (Criterion, tank, 2026-09-24):
|
||||
|
||||
| Scale | Flat | IVF (nprobe=10) | IVF-PQ | MemX¹ (claimed, end-to-end) |
|
||||
|-------|------|-----------------|--------|----------|
|
||||
| 1K | **47.4 µs** | — | — | — |
|
||||
| 10K | 500.5 µs | **24.8 µs** | — | — |
|
||||
| 100K | 6.58 ms | 592 µs | **869 µs** | <90 ms |
|
||||
|
||||
> These replace figures from the original i7-12650H run (flat 54 µs / 753 µs /
|
||||
> 11.4 ms); a 2026-08-05 run on tank had already matched the new ones — see
|
||||
> [BENCHMARKS.md § Vector Search Latency](BENCHMARKS.md#vector-search-latency).
|
||||
|
||||
### Agent Memory Operations
|
||||
|
||||
| Operation | Latency | Scale |
|
||||
|-----------|---------|-------|
|
||||
| Hybrid search (RRF) | **222 µs** | 1K records |
|
||||
| BM25 keyword search | **67 µs** | 1K records |
|
||||
| Knowledge graph BFS | **24 µs** | 1K entities |
|
||||
| Spreading activation | **17 µs** | 100 entities |
|
||||
| Temporal range query | **716 ns** | 10K timestamps |
|
||||
| Consolidation cycle | **164 µs** | 1K records |
|
||||
| Memory write (WAL) | **18 µs** | per record (group-commit append; HDF5 batched at flush) |
|
||||
| Importance gate | **61 ns** | per record |
|
||||
| Hybrid search (`HDF5Memory::hybrid_search`, p50) | **0.07 ms** / 0.49 ms / 4.69 ms | 1K / 10K / 100K records |
|
||||
| BM25 keyword search | **20.4 µs** | 1K records |
|
||||
| Knowledge graph BFS | **23.1 µs** | 1K entities |
|
||||
| Spreading activation | **10.1 µs** | 100 entities |
|
||||
| Temporal range query | **622 ns** | 10K timestamps |
|
||||
| Consolidation cycle | **115.2 µs** | 1K records |
|
||||
| Cross-modal search (exact scan, 2 embeddings per record) | **842.0 µs** / 8.44 ms | 1K / 10K records |
|
||||
| Memory write (WAL) | **26.1 µs** | per record (group-commit append; HDF5 batched at flush) |
|
||||
| Importance gate | **57.6 ns** | per record (trivial skip) |
|
||||
|
||||
The old 18 µs WAL write was undated, from another machine: v2.3.0 measures
|
||||
24.3 µs on the same hardware as this table, the same as an `f32` store today.
|
||||
`float16` stores (the new default) add ~2 µs for rounding; the int8 index adds
|
||||
nothing. See [BENCHMARKS.md § Write Path](BENCHMARKS.md#write-path).
|
||||
Knowledge-graph traversal was briefly 6.5x slower (155 µs) until this re-run
|
||||
found and fixed an adjacency index rebuilt on every traversal; see
|
||||
[§ Knowledge Graph](BENCHMARKS.md#knowledge-graph).
|
||||
|
||||
### Chunked Write Throughput (codec comparison)
|
||||
|
||||
@@ -97,7 +208,7 @@ by default (AoS→SoA byte transpose, +157–204% throughput for float data):
|
||||
|
||||
Use `.with_zstd(3)` or `.with_deflate(6)` for write-heavy workloads — both now perform at ~720–750 MiB/s on large matrices. Use `.with_pcodec()` for write-once/read-many workloads where compression ratio matters more than encode speed. Disable auto-shuffle with `.without_shuffle()` for byte arrays that don't benefit from AoS→SoA transposition.
|
||||
|
||||
> ¹ MemX ([arxiv:2603.16171](https://arxiv.org/abs/2603.16171), March 2026): Rust + libSQL, claims <90ms at 100K records. **Not like-for-like:** MemX's figure is *end-to-end* (embeddings + FTS5 + four-factor re-ranking); ours is a *single component* (raw vector search). The ratio overstates the real advantage by an unquantified margin — order-of-magnitude indication only. See [BENCHMARKS.md](BENCHMARKS.md#comparison-to-memx-arxiv260316171).
|
||||
> ¹ MemX ([arxiv:2603.16171](https://arxiv.org/abs/2603.16171), March 2026): Rust + libSQL, claims <90ms at 100K records. **Not like-for-like:** MemX's figure is *end-to-end* (embeddings + FTS5 + four-factor re-ranking); ours is a *single component* (raw vector search), so the two columns are not comparable and no ratio is given. See [BENCHMARKS.md](BENCHMARKS.md#comparison-to-memx-arxiv260316171).
|
||||
|
||||
### LongMemEval Retrieval Recall
|
||||
|
||||
@@ -115,13 +226,17 @@ declaration:
|
||||
|
||||
Hybrid is the strongest configuration, which is what running two retrieval stages
|
||||
is for. The weights matter more than the stages: a sweep of `vector_weight` from
|
||||
0.0 to 1.0 found the long-standing `0.7/0.3` default is **strictly dominated** by
|
||||
`0.4/0.6` — better on Hit@1, Hit@5, Hit@10 and MRR at both granularities. Use
|
||||
`0.4/0.6`, or `0.3/0.7` if rank-1 precision matters most. See
|
||||
[BENCHMARKS.md § Weight sweep](BENCHMARKS.md#longmemeval-results).
|
||||
0.0 to 1.0 found the old `0.7/0.3` default is **strictly dominated** by
|
||||
`0.4/0.6` — better on Hit@1, Hit@5, Hit@10 and MRR at both granularities. Since
|
||||
v2.5.0 `0.4/0.6` is the default (`hybrid::DEFAULT_FUSION`, used by
|
||||
`unified_search`, `hybrid_search_with` and `ClawhdfBackend`); callers that
|
||||
pass weights to `hybrid_search` explicitly choose their own. Use `0.3/0.7` if
|
||||
rank-1 precision matters most. Reciprocal rank fusion is selectable
|
||||
(`hybrid::Fusion::Rrf`) but measured worse than the weighted sum. See
|
||||
[BENCHMARKS.md § Weight sweep](BENCHMARKS.md#weight-sweep--full-haystack-n500).
|
||||
|
||||
Vector embeddings require `--features embeddings`; without it the vector stage is
|
||||
inert and only the BM25 row is produced, which is what every previously published
|
||||
The benchmark's vector stage requires `clawhdf5-bench`'s `embeddings` feature
|
||||
(real MiniLM embeddings); without it the vector stage is inert and only the BM25 row is produced, which is what every previously published
|
||||
number here measured.
|
||||
|
||||
On the easier `longmemeval_oracle` variant (evidence sessions only) the same
|
||||
@@ -146,19 +261,53 @@ retrieval recall reported as QA accuracy typically overstates by 20–30 points.
|
||||
|
||||
### Memory Footprint
|
||||
|
||||
| Records | File Size | Bytes/Record | With Compression |
|
||||
|---------|-----------|--------------|------------------|
|
||||
| 1K | ~6.5 MB | ~6.5 KB | ~2.1 MB (3.1x) |
|
||||
| 10K | ~65 MB | ~6.5 KB | ~21 MB (3.1x) |
|
||||
| 100K | ~645 MB | ~6.5 KB | ~208 MB (3.1x) |
|
||||
**On disk** — 384-dim `float16` embeddings (the default for new stores),
|
||||
200-char text, `footprint_bench`
|
||||
([BENCHMARKS.md § Memory Footprint](BENCHMARKS.md#memory-footprint-1)):
|
||||
|
||||
| Records | File Size | Bytes/Record | Gzip-6 compressed |
|
||||
|---------|-----------|--------------|-------------------|
|
||||
| 1K | 810.4 KB | 829 B | 56.4 KB |
|
||||
| 10K | 7.8 MB | 820 B | 471.3 KB |
|
||||
| 100K | 76.7 MB | 803 B | 4.5 MB |
|
||||
|
||||
The benchmark's synthetic embeddings and text are far more repetitive than
|
||||
real data (only 40 distinct texts), so no column here is an expectation for
|
||||
real data. The compressed column is an upper bound, and the Bytes/Record
|
||||
column is optimistic too: it is not an uncompressed figure, because the store
|
||||
always deflates its text (any string dataset of 4 KiB or more) whatever
|
||||
`MemoryConfig::compression` says. The `float16` embeddings alone are 768 B per
|
||||
record, so 200 characters of real text would take a record above 820 B.
|
||||
This table used to show `f32` stores (1.7 KB per record, 169.8 MB at 100K);
|
||||
those were not re-measured. The float16 study compares the two on the same
|
||||
data: 100K × 384 records take 80.8 MiB as `float16` and 154.0 MiB as `f32`.
|
||||
|
||||
**In memory** — a store reopened from disk, 384-dim `f32`, measured with a
|
||||
counting allocator ([BENCHMARKS.md § Memory footprint](BENCHMARKS.md#memory-footprint)):
|
||||
|
||||
| Records | Raw vectors | Reopened, `f32` index | Reopened, `i8` index (default) |
|
||||
|---------|-------------|-----------------------|--------------------------------|
|
||||
| 1K | 1 MiB | 4 MiB (2.40x) | 2 MiB (1.64x) |
|
||||
| 10K | 15 MiB | 44 MiB (3.03x) | 27 MiB (1.81x) |
|
||||
| 100K | 146 MiB | 399 MiB (2.72x) | **256 MiB (1.74x)** |
|
||||
|
||||
Down from 505 MiB (3.44x) at 100K before v2.6.0, when the cache held every
|
||||
embedding twice. The `f32` column was re-measured on 2026-09-24 and reproduced
|
||||
exactly; the `i8` column was not re-run.
|
||||
|
||||
### Consolidation Efficiency
|
||||
|
||||
1,000 records (10 signal + 990 noise), `working_capacity = 100`
|
||||
([BENCHMARKS.md § Consolidation Efficiency](BENCHMARKS.md#consolidation-efficiency)):
|
||||
|
||||
| Metric | Before | After | Delta |
|
||||
|--------|--------|-------|-------|
|
||||
| Records in store | 1,000 | ~110 | −89% |
|
||||
| Hit@1 recall | ~60% | ~90% | +30% |
|
||||
| Search latency | ~2.8 ms | ~0.3 ms | **9x faster** |
|
||||
| Records in store | 1,000 | 100 | −90% |
|
||||
| Hit@1 recall (signal records) | 100% | 100% | no loss |
|
||||
| Search latency (avg) | 2.22 ms | 0.24 ms | **9.3x faster** |
|
||||
|
||||
The consolidation cycle that does this took 0.13 ms; a cycle over 10K records
|
||||
takes 2.81 ms and over 100K 46.7 ms.
|
||||
|
||||
**Full benchmark details: [BENCHMARKS.md](BENCHMARKS.md)**
|
||||
|
||||
@@ -166,74 +315,74 @@ retrieval recall reported as QA accuracy typically overstates by 20–30 points.
|
||||
|
||||
## Agent Memory Architecture
|
||||
|
||||
ClawhDF5's agent memory engine implements research from 15+ recent papers on agentic memory systems. It's not a toy — it's the real thing.
|
||||
ClawhDF5's agent memory engine draws on 15+ recent papers on agentic memory systems (see [Research Foundation](#research-foundation)).
|
||||
|
||||
```
|
||||
┌─────────────────┐
|
||||
│ Agent Query │
|
||||
└────────┬────────┘
|
||||
│
|
||||
┌────────────▼────────────┐
|
||||
│ Hybrid Retrieval │
|
||||
│ Vector + BM25 + RRF │
|
||||
└────────────┬────────────┘
|
||||
│
|
||||
┌──────────────────▼──────────────────┐
|
||||
│ Multi-Factor Re-Ranking │
|
||||
│ temporal · authority · activation │
|
||||
└──────────────────┬──────────────────┘
|
||||
│
|
||||
┌────────────▼────────────┐
|
||||
│ Confidence Rejection │
|
||||
┌─────────────────▼──────────────────┐
|
||||
│ HDF5Memory::search │
|
||||
│ optional source-channel filter │
|
||||
│ HNSW vector + BM25 keyword │
|
||||
│ weighted fusion (0.4 / 0.6) │
|
||||
│ × √(Hebbian activation) │
|
||||
└─────────────────┬──────────────────┘
|
||||
│ opt-in (SearchOptions);
|
||||
│ ClawhdfBackend turns both on
|
||||
┌─────────────────▼──────────────────┐
|
||||
│ Multi-factor re-ranking │
|
||||
│ relevance · recency · authority · │
|
||||
│ activation │
|
||||
├────────────────────────────────────┤
|
||||
│ Confidence rejection │
|
||||
│ (suppress bad matches) │
|
||||
└────────────┬────────────┘
|
||||
└─────────────────┬──────────────────┘
|
||||
│
|
||||
┌────────────────────────▼────────────────────────┐
|
||||
│ Memory Store (HDF5) │
|
||||
│ │
|
||||
│ ┌───────────┐ ┌───────────┐ ┌───────────────┐ │
|
||||
│ │ Working │→│ Episodic │→│ Semantic │ │
|
||||
│ │ (bounded) │ │ (bounded) │ │ (long-term) │ │
|
||||
│ └───────────┘ └───────────┘ └───────────────┘ │
|
||||
│ │
|
||||
│ ┌──────────┐ ┌──────────┐ ┌────────────────┐ │
|
||||
│ │Knowledge │ │Temporal │ │ Multi-Modal │ │
|
||||
│ │ Graph │ │ Index │ │ Embeddings │ │
|
||||
│ └──────────┘ └──────────┘ └────────────────┘ │
|
||||
│ │
|
||||
│ ┌──────────┐ ┌──────────┐ ┌────────────────┐ │
|
||||
│ │Provenance│ │ Anomaly │ │ Source │ │
|
||||
│ │ Tracking │ │Detection │ │ Isolation │ │
|
||||
│ └──────────┘ └──────────┘ └────────────────┘ │
|
||||
└─────────────────────────────────────────────────┘
|
||||
│
|
||||
┌────────┴────────┐
|
||||
│ agent_memory.h5 │
|
||||
│ single file │
|
||||
└─────────────────┘
|
||||
┌────────────────────────────▼────────────────────────────┐
|
||||
│ In memory │
|
||||
│ cache (flat f32 embeddings) · BM25 index · HNSW index │
|
||||
│ provenance ledger + anomaly alerts (session-scoped) │
|
||||
└────────────────────────────┬────────────────────────────┘
|
||||
│ WAL append; checkpoint
|
||||
┌────────────────────────────▼────────────────────────────┐
|
||||
│ agent_memory.h5 /meta · /memory · /sessions · │
|
||||
│ /knowledge_graph │
|
||||
│ agent_memory.h5.wal chained-CRC write-ahead log │
|
||||
│ agent_memory.h5.ann HNSW graph (derived, rebuildable) │
|
||||
│ agent_memory.h5.lock single-writer lock │
|
||||
└─────────────────────────────────────────────────────────┘
|
||||
```
|
||||
|
||||
Consolidation tiers (Working → Episodic → Semantic), the knowledge-graph
|
||||
algorithms, temporal and multi-modal indexes are library components you drive
|
||||
directly; the store persists the records, sessions and graph they work over.
|
||||
|
||||
### Module Overview
|
||||
|
||||
| Module | What It Does |
|
||||
|--------|-------------|
|
||||
| **`knowledge`** | Entity/relation graph with BFS traversal, spreading activation, fuzzy entity resolution |
|
||||
| **`consolidation`** | Three-tier memory (Working → Episodic → Semantic) with importance scoring and time-decay |
|
||||
| **`hybrid`** | Vector + BM25 fusion with Reciprocal Rank Fusion (RRF, k=60). The vector stage uses the HNSW index by default (`hnsw` feature, on by default); disable with `--no-default-features --features float16` for an exact linear scan |
|
||||
| **`reranker`** | Multi-factor re-ranking: temporal recency, source authority, activation weight |
|
||||
| **`confidence`** | Low-confidence rejection — suppresses spurious recalls when nothing matches |
|
||||
| **`knowledge`** | Entity/relation graph with BFS traversal, spreading activation, fuzzy (Levenshtein) entity resolution |
|
||||
| **`consolidation`** | Three-tier memory (Working → Episodic → Semantic) with importance scoring, novelty, and time-decay |
|
||||
| **`hybrid`** | Vector + BM25 fusion. Default is a min-max-normalised weighted sum, vector 0.4 / keyword 0.6 (`hybrid::DEFAULT_FUSION`, tuned on LongMemEval); RRF is available via `Fusion::Rrf` / `hybrid_search_with`. The vector stage uses the HNSW index by default (`hnsw` feature); disable with `--no-default-features --features float16` for an exact linear scan |
|
||||
| **`reranker`** | Multi-factor re-ranking: retrieval relevance (leads, weight 1.0), temporal recency, source authority, activation weight. Opt-in via `SearchOptions::with_rerank`; on in `ClawhdfBackend` |
|
||||
| **`confidence`** | Low-confidence rejection — suppresses spurious recalls when nothing matches. Opt-in via `SearchOptions::with_confidence`; on in `ClawhdfBackend` |
|
||||
| **`temporal`** | Sorted timestamp index, session DAG, entity timeline, temporal query hints |
|
||||
| **`multimodal`** | Cross-modal search across text/image/audio/video embeddings |
|
||||
| **`provenance`** | Source attribution, FNV-1a content hashing, integrity verification |
|
||||
| **`anomaly`** | Write rate limiting, 15 injection pattern detectors, source distribution analysis |
|
||||
| **`openclaw`** | OpenClaw integration: MemoryBackend trait, Markdown ↔ HDF5 conversion |
|
||||
| **`signing`** | Ed25519-signed checkpoints: SHA-256 per record in a Merkle tree, plus hashes of settings, sessions and the knowledge graph; `HDF5Memory::verify` names any edited record |
|
||||
| **`provenance`** | Source attribution and an unkeyed FNV-1a content hash per record, held in memory for the session, for detecting accidental corruption (not tamper-proof) |
|
||||
| **`anomaly`** | Write rate limiting, 15 injection-pattern detectors, source-distribution analysis. Alerts never block a save; drain them with `take_anomaly_alerts` |
|
||||
| **`openclaw`** | `ClawhdfBackend`: a Markdown-oriented backend (ingest by section, search, read back by path, export). Named for OpenClaw, but **not an OpenClaw plugin** — see [docs/openclaw.md](docs/openclaw.md) |
|
||||
| **`vector_search`** | Flat cosine, pre-normed, SIMD, BLAS, GPU, parallel search paths |
|
||||
| **`ivf` / `pq`** | IVF-PQ approximate nearest neighbor for billion-scale search |
|
||||
| **`bm25`** | BM25 keyword index with TF-IDF scoring |
|
||||
| **`ivf` / `pq`** | Standalone IVF and IVF-PQ indexes (benchmarked to 100K vectors); not used by `HDF5Memory`, whose ANN index is HNSW |
|
||||
| **`bm25`** | Incremental Okapi BM25 inverted index, kept for the life of the store; optional stemming |
|
||||
| **`query_expand`** | Synonym / acronym / temporal query expansion |
|
||||
| **`entity_extract`** | Rule-based entity extraction from text chunks into the knowledge graph |
|
||||
| **`wal`** | Write-ahead log for crash-safe persistence; each entry is CRC32-checked on replay, so a corrupted entry stops replay there instead of loading bad data |
|
||||
| **`wal`** | Write-ahead log (v4) with a chained CRC32 per entry, so a corrupted, reordered, duplicated or spliced entry stops replay; checkpoints record a WAL mark so nothing is applied twice. Appends are not fsynced |
|
||||
| **`memory_strategy`** | Pluggable strategies: save-every, semantic-shift, user-correction detection |
|
||||
| **`decision_gate`** | Sub-microsecond trivial/substantive classification |
|
||||
| **`ephemeral`** | In-memory TTL/LFU working tier |
|
||||
| **`async_memory`** | Tokio-based async wrapper over the memory store (`async` feature) |
|
||||
|
||||
---
|
||||
@@ -259,13 +408,86 @@ let values = ds.read_f64()?;
|
||||
assert_eq!(values, vec![22.5, 23.1, 21.8]);
|
||||
```
|
||||
|
||||
### Groups and links
|
||||
|
||||
```rust
|
||||
use clawhdf5::{AttrValue, FileBuilder};
|
||||
|
||||
let mut b = FileBuilder::new();
|
||||
// A path creates its missing intermediate groups, as in h5py.
|
||||
b.create_dataset("run/2026/temps").with_f64_data(&[22.5, 23.1]);
|
||||
// Builders nest; a group added at an existing path is merged into it.
|
||||
let mut run = b.create_group("run");
|
||||
run.set_attr("operator", AttrValue::String("ana".into()));
|
||||
let mut cal = run.create_group("calibration");
|
||||
cal.track_order(true); // h5py lists members in insertion order
|
||||
cal.create_dataset("offset").with_f64_data(&[0.1]);
|
||||
run.add_group(cal.finish());
|
||||
b.add_group(run.finish());
|
||||
b.add_soft_link("latest", "/run/2026"); // h5py.SoftLink
|
||||
b.add_hard_link("temps", "/run/2026/temps"); // f["temps"] = f["run/2026/temps"]
|
||||
b.add_external_link("raw", "raw.h5", "/data");
|
||||
b.write("groups.h5")?;
|
||||
```
|
||||
|
||||
A group holds at most 65 535 links; more is an error, as is a link over
|
||||
65 515 bytes (a very long soft-link target) in a group of more than 8 links.
|
||||
|
||||
### Python
|
||||
|
||||
`crates/clawhdf5-py` is a Python package (PyO3 + numpy) that reads HDF5 with
|
||||
an h5py-shaped API and no libhdf5. It is not on PyPI; build it with
|
||||
[maturin](https://www.maturin.rs) into a virtualenv:
|
||||
|
||||
```bash
|
||||
python -m venv .venv && . .venv/bin/activate
|
||||
pip install maturin numpy
|
||||
maturin develop --release -m crates/clawhdf5-py/Cargo.toml
|
||||
python -c "import clawhdf5; print(clawhdf5.__version__)"
|
||||
```
|
||||
|
||||
```python
|
||||
import numpy as np
|
||||
import clawhdf5
|
||||
|
||||
with clawhdf5.File("data.h5", "r") as f:
|
||||
print(list(f.keys())) # sorted member names, like h5py
|
||||
ds = f["group/temperatures"] # relative or absolute ("/group/...") paths
|
||||
print(ds.shape, ds.dtype) # dtype is the numpy dtype h5py reports
|
||||
block = ds[100:200, ::4] # a small selection reads only its chunks
|
||||
row = ds[-1] # integers drop the axis
|
||||
picked = ds[[1, 5, 9], :] # one increasing index list per key
|
||||
units = ds.attrs["units"] # attributes come back as h5py returns them
|
||||
everything = np.asarray(ds)
|
||||
|
||||
records = f["table"] # compound -> numpy structured array
|
||||
ids = records["id"] # one field
|
||||
```
|
||||
|
||||
Reads cover integers and IEEE floats of every width in either byte order,
|
||||
`bool`, enums, complex, fixed and variable-length strings, variable-length
|
||||
sequences, opaque, HDF5 array types and compounds; other types (references,
|
||||
bitfields, ...) raise `TypeError` instead of returning guessed data. Keys
|
||||
follow h5py (negative steps, `None` and boolean masks are refused). The
|
||||
read itself runs with the GIL released, so Python threads read in parallel.
|
||||
A selection whose bounding box covers at most half the dataset decodes only
|
||||
the chunks (or contiguous rows) that box overlaps; a larger one — including
|
||||
a strided slice across the whole dataset — decodes the whole dataset, as
|
||||
do datasets that are compact, virtual, unwritten, or chunked with a
|
||||
non-default fill value (`docs/known-issues.md`). An index list is read one
|
||||
group of neighbouring chunks at a time.
|
||||
Writing (`File(path, "w")`, `create_dataset`, `create_group`, `attrs[...] =`)
|
||||
covers `float64`, `float32`, `int64`, `int32` and `uint8` arrays. The tests
|
||||
in `crates/clawhdf5-py/tests` compare every read with h5py; run them with
|
||||
`pip install pytest h5py && pytest crates/clawhdf5-py/tests`.
|
||||
|
||||
### Agent Memory
|
||||
|
||||
```rust
|
||||
use clawhdf5_agent::{HDF5Memory, MemoryConfig, MemoryEntry, AgentMemory};
|
||||
|
||||
// Create memory store
|
||||
let config = MemoryConfig::new("agent.h5", "my-agent", 384);
|
||||
let config = MemoryConfig::new("agent.h5".into(), "my-agent", 384);
|
||||
let mut memory = HDF5Memory::create(config)?;
|
||||
|
||||
// Save a memory
|
||||
@@ -278,13 +500,67 @@ memory.save(MemoryEntry {
|
||||
tags: "preference".into(),
|
||||
})?;
|
||||
|
||||
// Search
|
||||
let results = memory.search(&query_embedding, 5)?;
|
||||
// Hybrid search: vector + BM25, weighted 0.4 / 0.6 (the measured default)
|
||||
let results = memory.hybrid_search(&query_embedding, "user preferences", 0.4, 0.6, 5);
|
||||
for result in results {
|
||||
println!("[{:.3}] {}", result.score, result.chunk);
|
||||
}
|
||||
```
|
||||
|
||||
### Search Options
|
||||
|
||||
```rust
|
||||
use clawhdf5_agent::SearchOptions;
|
||||
use clawhdf5_agent::confidence::ConfidenceConfig;
|
||||
use clawhdf5_agent::reranker::ReRankConfig;
|
||||
|
||||
// Only memories from these source channels; still a full page of k results.
|
||||
let work = memory.search(
|
||||
&query_embedding,
|
||||
"deadline",
|
||||
&SearchOptions::new(5).with_sources(["slack", "email"]),
|
||||
);
|
||||
|
||||
// Re-rank by relevance, recency, source authority and activation, then drop
|
||||
// low-confidence results — the pipeline ClawhdfBackend runs.
|
||||
let careful = memory.search(
|
||||
&query_embedding,
|
||||
"user preferences",
|
||||
&SearchOptions::new(5)
|
||||
.with_rerank(ReRankConfig::default())
|
||||
.with_confidence(ConfidenceConfig::default()),
|
||||
);
|
||||
```
|
||||
|
||||
### Signed Checkpoints
|
||||
|
||||
```rust
|
||||
use clawhdf5_agent::signing;
|
||||
|
||||
// Once, somewhere safe: keep the secret key, publish the public key.
|
||||
let key = signing::generate_key();
|
||||
let public = key.verifying_key();
|
||||
|
||||
// Every checkpoint is signed from now on. The key is never written to disk;
|
||||
// a signed store refuses to checkpoint without it.
|
||||
memory.set_signing_key(key);
|
||||
memory.flush_wal()?;
|
||||
|
||||
// Anyone holding the public key can check the file, e.g. after copying it.
|
||||
let report = HDF5Memory::verify(std::path::Path::new("agent.h5"), &public)?;
|
||||
assert!(report.is_valid());
|
||||
// On a tampered file: report.changed_records lists the records that differ.
|
||||
```
|
||||
|
||||
The signature covers every record (text, embedding as stored, channel,
|
||||
timestamp, session, tags, deleted flag, activation), the store's settings,
|
||||
its sessions and its knowledge graph — a change made with any tool is caught.
|
||||
It covers checkpoints, not saves still in the WAL
|
||||
(`report.wal_entries_unsigned` counts those). CLI: `clawhdf5-cli keygen`,
|
||||
`--signing-key <file>` on writing commands, and `verify --public-key`.
|
||||
Signing adds about 20% to a checkpoint and 32 bytes per record to the file
|
||||
([BENCHMARKS.md § Signed checkpoints](BENCHMARKS.md#signed-checkpoints)).
|
||||
|
||||
### Knowledge Graph
|
||||
|
||||
```rust
|
||||
@@ -309,8 +585,8 @@ let neighbors = kg.bfs_neighbors(alice, 2); // 2-hop neighborhood
|
||||
let activated = kg.spreading_activation(&[alice], 0.5, 0.01, 5);
|
||||
|
||||
// Entity resolution — fuzzy matching
|
||||
let resolved = kg.resolve_or_create("alice", "person", -1, 2);
|
||||
// Returns existing Alice entity (Levenshtein distance ≤ 2)
|
||||
let (id, created) = kg.resolve_or_create("alice", "person", -1, 2);
|
||||
// id == alice, created == false: matched the existing entity (Levenshtein distance ≤ 2)
|
||||
```
|
||||
|
||||
### Memory Consolidation
|
||||
@@ -321,15 +597,19 @@ use clawhdf5_agent::consolidation::*;
|
||||
let config = ConsolidationConfig::default();
|
||||
let mut engine = ConsolidationEngine::new(config);
|
||||
|
||||
// Add memories — automatically scored for importance
|
||||
engine.add_memory("User prefers dark mode", vec![0.1, 0.2, ...], MemorySource::User);
|
||||
engine.add_memory("ok", vec![0.0, 0.0, ...], MemorySource::System);
|
||||
let now = 1_700_000_000.0; // seconds since the epoch
|
||||
|
||||
// Add memories — automatically scored for importance.
|
||||
// Elevated sources (System, …) go through a separate, explicit API.
|
||||
let id = engine.add_memory("User prefers dark mode".into(), vec![0.1, 0.2, ...], UntrustedSource::User, now);
|
||||
engine.add_trusted_memory("ok".into(), vec![0.0, 0.0, ...], TrustedSource::System, now);
|
||||
|
||||
// Access a memory (reactivates it)
|
||||
engine.access_memory(0);
|
||||
engine.access_memory(id, now);
|
||||
|
||||
// Run consolidation cycle
|
||||
let stats = engine.consolidate();
|
||||
engine.consolidate(now);
|
||||
let stats = engine.get_stats();
|
||||
// Working memories promote to Episodic (if important enough)
|
||||
// Episodic memories promote to Semantic (if accessed enough)
|
||||
// Low-decay memories get evicted when tiers are full
|
||||
@@ -351,19 +631,25 @@ let ids = index.range_query(1700000000.0, 1700010800.0);
|
||||
let recent = index.latest(10);
|
||||
```
|
||||
|
||||
### OpenClaw Integration
|
||||
### Markdown Backend
|
||||
|
||||
`ClawhdfBackend` ingests Markdown by section and searches it with the full
|
||||
pipeline. It is a library API — clawhdf5 is **not** an OpenClaw memory plugin
|
||||
([docs/openclaw.md](docs/openclaw.md)). Sections stored this way carry no
|
||||
embedding, so their search is keyword-only unless you save records with
|
||||
vectors through `save_entry`.
|
||||
|
||||
```rust
|
||||
use clawhdf5_agent::openclaw::*;
|
||||
|
||||
// Create backend
|
||||
let mut backend = ClawhdfBackend::create("memory.h5", "agent-1", 384)?;
|
||||
let mut backend = ClawhdfBackend::create(std::path::Path::new("memory.h5"), 384)?;
|
||||
|
||||
// Ingest existing Markdown memory files
|
||||
let md = std::fs::read_to_string("MEMORY.md")?;
|
||||
let count = backend.ingest_markdown("MEMORY.md", &md)?;
|
||||
|
||||
// Search (uses full pipeline: RRF → re-rank → confidence filter)
|
||||
// Search (full pipeline: weighted vector + BM25 fusion → re-rank → confidence filter)
|
||||
let results = backend.search("user preferences", &query_embedding, 5);
|
||||
|
||||
// Export back to Markdown
|
||||
@@ -375,29 +661,31 @@ let exported = backend.export_markdown("MEMORY.md")?;
|
||||
## Crate Map
|
||||
|
||||
```
|
||||
clawhdf5 workspace (16 crates, ~92K lines of Rust; plus libaec-sys, an
|
||||
internal FFI bindings crate for the optional szip feature)
|
||||
clawhdf5 workspace (17 crates, ~86K lines of Rust in src/, ~104K with tests
|
||||
and benches; plus libaec-sys, an internal FFI bindings
|
||||
crate for the optional szip feature)
|
||||
│
|
||||
├── Core HDF5
|
||||
│ ├── clawhdf5-format — Binary parser/writer (no_std), shared type definitions
|
||||
│ ├── clawhdf5-io — I/O abstraction (buffered, mmap, async)
|
||||
│ ├── clawhdf5-filters — Fast deflate path (zlib-ng); lz4/zstd/pcodec/szip filters live in clawhdf5-format
|
||||
│ ├── clawhdf5-format — Binary parser/writer (no_std-capable), shared type definitions
|
||||
│ ├── clawhdf5-io — I/O abstraction (file/memory readers; optional mmap, async, HSDS, MPI)
|
||||
│ ├── clawhdf5-filters — Fast deflate path (zlib-ng); the filter registry and the lz4/zstd/pcodec/szip/LZF/bitshuffle/bzip2/Blosc filters live in clawhdf5-format
|
||||
│ ├── clawhdf5-derive — Proc macros
|
||||
│ ├── clawhdf5 — High-level API
|
||||
│ ├── clawhdf5-netcdf4 — NetCDF-4 support
|
||||
│ ├── clawhdf5-accel — SIMD (NEON, AVX2, AVX-512)
|
||||
│ ├── clawhdf5-accel — SIMD (AVX2, NEON incl. SDOT int8; AVX-512 behind `avx512`)
|
||||
│ └── clawhdf5-gpu — GPU compute (wgpu, hand-written WGSL compute shaders)
|
||||
│
|
||||
├── Agent Memory
|
||||
│ ├── clawhdf5-agent — Memory engine (20.9K lines, 32 modules; WAL is CRC32-checked per entry)
|
||||
│ ├── clawhdf5-ann — HNSW approximate nearest neighbor (default backend; optional `parallel` feature)
|
||||
│ ├── clawhdf5-agent — Memory engine (24.7K lines, 32 modules; chained-CRC WAL)
|
||||
│ ├── clawhdf5-ann — HNSW approximate nearest neighbor (default backend; f32 or int8 storage; `parallel` build)
|
||||
│ ├── clawhdf5-migrate — SQLite → HDF5 migration
|
||||
│ ├── clawhdf5-android — Android JNI bridge
|
||||
│ └── clawhdf5-cli — CLI tool
|
||||
│
|
||||
├── Bindings
|
||||
│ ├── clawhdf5-py — Python (PyO3)
|
||||
│ └── clawhdf5-napi — Node.js (napi-rs)
|
||||
│ ├── clawhdf5-napi — Node.js (napi-rs)
|
||||
│ └── clawhdf5-wasm — Browser (WebAssembly, wasm-bindgen; read-only)
|
||||
│
|
||||
└── Tooling
|
||||
└── clawhdf5-bench — Benchmark suite
|
||||
@@ -411,10 +699,10 @@ ClawhDF5's agent memory design draws from 15+ recent papers:
|
||||
|
||||
| Paper | Key Insight | ClawhDF5 Module |
|
||||
|-------|-------------|-----------------|
|
||||
| **MemX** (2026) | RRF + multi-factor re-ranking | `hybrid`, `reranker` |
|
||||
| **Graph-Native Cognitive Memory** (2026) | Graph-structured belief revision | `knowledge` |
|
||||
| **MemX** (2026) | Hybrid fusion + multi-factor re-ranking | `hybrid`, `reranker` |
|
||||
| **Graph-Native Cognitive Memory** (2026) | Graph-structured memory (weighted, timestamped relations; entity timelines) | `knowledge`, `temporal` |
|
||||
| **CraniMem** (2026) | Bounded hippocampal memory | `consolidation` |
|
||||
| **D-MEM** (2026) | Reward prediction error gating | `consolidation` |
|
||||
| **D-MEM** (2026) | Surprise-gated storage (implemented as a novelty score) | `consolidation` |
|
||||
| **SYNAPSE** (2025) | Spreading activation for recall | `knowledge` |
|
||||
| **RAGdb** (2025) | Zero-dependency edge RAG | Architecture |
|
||||
| **MemoryGraft** (2025) | Memory poisoning attacks | `anomaly`, `provenance` |
|
||||
@@ -429,23 +717,45 @@ ClawhDF5's agent memory design draws from 15+ recent papers:
|
||||
|
||||
| Flag | Default | Description |
|
||||
|------|---------|-------------|
|
||||
| `agent` | no | Full agent memory layer |
|
||||
| `float16` | **yes** | Half-precision embedding storage (2× compression) |
|
||||
| `float16` | **yes** | Half-precision cosine kernel (`cosine_similarity_f16`). Half-precision *storage* is the `MemoryConfig::float16` setting below, and needs no feature |
|
||||
| `hnsw` | **yes** | HNSW approximate vector index for `hybrid_search` (via `clawhdf5-ann`); disable for an exact linear scan |
|
||||
|
||||
`MemoryConfig::quantized_index` (off by default) stores the HNSW index's own
|
||||
copy of the embeddings as `i8`, roughly halving a loaded store's memory
|
||||
(2.72x -> 1.74x the raw vectors at 100k x 384). Quantised distances are
|
||||
approximate, so the query path re-scores the candidate pool against the exact
|
||||
embeddings the store already holds — recall matches the `f32` index, at about
|
||||
13% fewer queries per second. See `BENCHMARKS.md`, "Quantising the index copy".
|
||||
| `parallel` | no | Rayon parallel search |
|
||||
| `parallel` | **yes** | Parallel HNSW bulk build (same graph, ~3× faster on 16 cores) and Rayon brute-force search strategies |
|
||||
| `zstd` | no | Compress embeddings with Zstd instead of deflate when `MemoryConfig::compression` is on (links libzstd) |
|
||||
| `fast-math` | no | BLAS matrix-vector multiply |
|
||||
| `accelerate` | no | Apple Accelerate / AMX (macOS) |
|
||||
| `openblas` | no | OpenBLAS (Linux) |
|
||||
| `gpu` | no | GPU search via wgpu |
|
||||
| `async` | no | Tokio async with background flush |
|
||||
|
||||
To opt out of the parallel build: `--no-default-features --features float16,hnsw`.
|
||||
For an exact linear cosine scan instead of HNSW: `--no-default-features --features float16`.
|
||||
|
||||
`MemoryConfig::hnsw_m`, `hnsw_ef_construction` and `hnsw_ef_search` tune the
|
||||
vector index (16 / 64 / scale-with-`k` by default) and are stored with the
|
||||
file.
|
||||
|
||||
`MemoryConfig::quantized_index` (**on by default** for new stores) holds the
|
||||
HNSW index's own copy of the embeddings as `i8`, roughly halving a loaded
|
||||
store's memory (2.72x -> 1.74x the raw vectors at 100k x 384). Quantised
|
||||
distances are approximate, so the query path re-scores the candidate pool
|
||||
against the exact embeddings the store already holds, which keeps recall at the
|
||||
`f32` index's level. It is also **faster**: 1.63x the queries per second at
|
||||
equal recall on x86-64 (AVX2) and 1.18x on a Raspberry Pi 5 (NEON `SDOT`), with
|
||||
index builds 1.8x and 2.3x faster respectively. Stores created before the
|
||||
setting existed keep their `f32` index; opt out for new stores with
|
||||
`quantized_index = false` or `clawhdf5-cli create --f32-index`. See
|
||||
[BENCHMARKS.md § Quantising the index copy](BENCHMARKS.md#quantising-the-index-copy-quantized_index).
|
||||
|
||||
`MemoryConfig::float16` (**on by default** for new stores) stores the
|
||||
embeddings on disk as IEEE half precision (numpy `float16`): at 100K × 384 the
|
||||
file drops from 154 to 81 MiB, checkpoints and opens get faster, and on the
|
||||
full LongMemEval haystack with real MiniLM embeddings every retrieval metric
|
||||
matches `f32`. Embeddings are rounded as they are saved, so the store searches
|
||||
the same before and after a reopen; values must lie within ±65504. Existing
|
||||
stores keep their setting. Opt out with `float16 = false` or
|
||||
`clawhdf5-cli create --f32` — e.g. for unnormalised vectors. See
|
||||
[BENCHMARKS.md § float16 embedding storage](BENCHMARKS.md#float16-embedding-storage-memoryconfigfloat16).
|
||||
|
||||
### `clawhdf5-format`
|
||||
|
||||
| Flag | Default | Description |
|
||||
@@ -454,26 +764,47 @@ embeddings the store already holds — recall matches the `f32` index, at about
|
||||
| `deflate` | yes | Deflate compression |
|
||||
| `checksum` | yes | Jenkins lookup3 verification |
|
||||
| `provenance` | yes | SHA-256 provenance attributes |
|
||||
| `fast-deflate` | **yes** | zlib-ng backend for faster deflate |
|
||||
| `system-zlib-decompress` | **yes** | Use the system zlib for decompression where available |
|
||||
| `zlib-rs` | **yes** | Pure-Rust deflate backend ([zlib-rs](https://github.com/trifectatechfoundation/zlib-rs)) |
|
||||
| `fast-deflate` | no | zlib-ng deflate backend instead (C; needs `cmake`). Overrides `zlib-rs` when both are on |
|
||||
| `system-zlib-decompress` | **yes** | Use Apple's system libz for decompression (macOS only; no effect elsewhere) |
|
||||
| `parallel` | no | Parallel chunk encoding + compression (rayon) |
|
||||
| `fast-checksum` | no | crc32fast-accelerated checksums |
|
||||
| `lz4` | no | LZ4 block compression filter (id 32004) |
|
||||
| `zstd` | no | Zstandard compression filter (id 32015) |
|
||||
| `pcodec` | no | Pcodec lossless numerical codec (id 32023, via `pco` crate) |
|
||||
| `system-zlib` / `zlib-rs` | no | Alternative zlib backends for deflate |
|
||||
| `pcodec` | no | Pcodec lossless numerical codec (via `pco` crate). Private, unregistered filter id 480: **only clawhdf5 can read these datasets** (h5py/libhdf5 cannot). Files from clawhdf5 <= 2.7.0 used id 32023, which is registered to Granular BitRound; they still read. |
|
||||
| `system-zlib` | no | System zlib backend for deflate (C) |
|
||||
| `blake3_hash` | no | BLAKE3 content hashing for provenance |
|
||||
| `szip` | no | SZIP filter (id 4) via libaec (C, through the internal `libaec-sys` crate) |
|
||||
| `lzf` | **yes** | LZF filter (id 32000), h5py's built-in `compression="lzf"`: read and write. No dependencies |
|
||||
| `bitshuffle` | no | Bitshuffle filter (id 32008) with its LZ4 and Zstandard modes: read and write. Pure Rust (lz4_flex, ruzstd) |
|
||||
| `bzip2` | no | bzip2 filter (id 307): read and write. Pure Rust (the `bzip2` crate's libbz2-rs-sys backend compiles no C) |
|
||||
| `blosc` | no | Blosc 1 filter (id 32001): reads BloscLZ, LZ4/LZ4HC, Snappy, Zlib and Zstandard frames with byte or bit shuffle; writes LZ4, Snappy, Zlib or Zstandard (not BloscLZ). Pure Rust |
|
||||
| `plugin-filters` | no | All four above |
|
||||
|
||||
Blosc2 (32026) and ZFP (32013) are not implemented: reading them fails with
|
||||
`UnsupportedFilter`, whose message names the filter. Any other filter can be
|
||||
supplied at run time with `filter_registry::register_filter` (a decoder
|
||||
closure, or a `FilterCodec` that also encodes). The facade (`clawhdf5`)
|
||||
forwards `lzf`, `bitshuffle`, `bzip2`, `blosc` and `plugin-filters`. Write
|
||||
with `DatasetBuilder::with_lzf()`, `with_bitshuffle(..)`, `with_bzip2(..)`
|
||||
and `with_blosc(..)`; h5py + hdf5plugin read the result (tested both ways in
|
||||
`crates/clawhdf5/tests/plugin_filters_interop.rs`). The pure-Rust Zstandard
|
||||
encoder has one level (about zstd's level 1); no speed or ratio claims are
|
||||
made for these codecs.
|
||||
|
||||
### `clawhdf5-ann`
|
||||
|
||||
| Flag | Default | Description |
|
||||
|------|---------|-------------|
|
||||
| `parallel` | no | Rayon-parallel neighbor-distance computation during HNSW graph pruning |
|
||||
| `parallel` | no | Batched bulk build runs neighbour planning and back-link pruning on a Rayon pool; the graph is identical with or without it (enabled by `clawhdf5-agent`'s default `parallel`) |
|
||||
|
||||
### `clawhdf5-io`
|
||||
|
||||
| Flag | Default | Description |
|
||||
|------|---------|-------------|
|
||||
| `mmap` | no | Memory-mapped reads (`memmap2`) |
|
||||
| `async` | no | Tokio-based async I/O |
|
||||
| `hsds` | no | HSDS (HDF REST service) client |
|
||||
| `mpi-io` | no | MPI-backed I/O via the `mpi` crate |
|
||||
|
||||
> **Parallel I/O (MPI) limitation:** `mpi-io`'s read path is a root-rank read
|
||||
@@ -487,17 +818,17 @@ embeddings the store already holds — recall matches the `f32` index, at about
|
||||
## Building
|
||||
|
||||
```bash
|
||||
# Default
|
||||
# Default (pure Rust: no cmake or C compiler needed)
|
||||
cargo build --workspace
|
||||
|
||||
# Agent memory with all accelerations (Linux)
|
||||
cargo build -p clawhdf5-agent --features "agent,float16,parallel,fast-math"
|
||||
cargo build -p clawhdf5-agent --features fast-math
|
||||
|
||||
# Agent memory with Apple Accelerate (macOS)
|
||||
cargo build -p clawhdf5-agent --features "agent,float16,accelerate,parallel,gpu"
|
||||
cargo build -p clawhdf5-agent --features "accelerate,gpu"
|
||||
|
||||
# Tests
|
||||
cargo test --workspace # all 1,650+ tests
|
||||
cargo test --workspace # all 1,850+ tests
|
||||
cargo test -p clawhdf5-agent # agent memory tests
|
||||
scripts/ci-test.sh # what CI runs: fmt, clippy matrix, tests,
|
||||
# h5py/netCDF4 interop, no_std
|
||||
@@ -519,25 +850,42 @@ cargo bench -p clawhdf5-bench # h5bench-equivalent I/O suite
|
||||
|
||||
```
|
||||
agent_memory.h5
|
||||
├── /meta
|
||||
│ ├── schema_version: "1.0"
|
||||
│ ├── agent_id, embedder, embedding_dim
|
||||
│ └── created_at
|
||||
├── /meta (attributes)
|
||||
│ ├── schema_version: "1.0", edgehdf5_version
|
||||
│ ├── agent_id, embedder, embedding_dim, chunk_size, overlap, created_at
|
||||
│ ├── float16, compression, compression_level, compact_threshold,
|
||||
│ │ hebbian_boost, decay_factor, wal_enabled, wal_max_entries
|
||||
│ ├── quantized_index, hnsw_m, hnsw_ef_construction, hnsw_ef_search
|
||||
│ ├── wal_applied_len, wal_applied_crc (WAL mark of the last checkpoint)
|
||||
│ └── ann_generation (ties the .ann sidecar to this checkpoint)
|
||||
├── /memory
|
||||
│ ├── chunks: string[N]
|
||||
│ ├── embeddings: f32[N × D] (or f16 with float16 flag)
|
||||
│ ├── embeddings: f32[N × D], or f16 for a `float16` store
|
||||
│ │ (chunked; deflate, or Zstd with the `zstd`
|
||||
│ │ feature, when compression is on)
|
||||
│ ├── source_channel: string[N]
|
||||
│ ├── timestamps: f64[N]
|
||||
│ ├── session_ids: string[N]
|
||||
│ ├── tags: string[N]
|
||||
│ ├── tombstones: u8[N]
|
||||
│ └── norms: f32[N] (pre-computed L2)
|
||||
│ ├── norms: f32[N] (pre-computed L2)
|
||||
│ └── activation_weights: f32[N] (Hebbian)
|
||||
├── /sessions
|
||||
│ ├── ids: string[S]
|
||||
│ └── summaries: string[S]
|
||||
│ ├── ids, channels, summaries: string[S]
|
||||
│ ├── start_idxs, end_idxs: i64[S]
|
||||
│ └── timestamps: f64[S]
|
||||
└── /knowledge_graph
|
||||
├── entity_names: string[E]
|
||||
├── relation_srcs: i64[R]
|
||||
├── relation_tgts: i64[R]
|
||||
└── relation_types: string[R]
|
||||
├── entity_ids, entity_emb_idxs: i64[E]; entity_names, entity_types: string[E]
|
||||
├── relation_srcs, relation_tgts: i64[R]; relation_types: string[R]
|
||||
├── relation_weights: f32[R]; relation_ts: f64[R]
|
||||
└── alias_strings: string[A]; alias_entity_ids: i64[A] (when aliases exist)
|
||||
```
|
||||
|
||||
Alongside the store: `<store>.h5.wal` (write-ahead log), `<store>.h5.ann`
|
||||
(HNSW graph; derived, safe to delete) and `<store>.h5.lock` (single-writer
|
||||
lock). A second writer gets `MemoryError::Locked`; use
|
||||
`HDF5Memory::open_read_only` for a lock-free point-in-time view.
|
||||
|
||||
---
|
||||
|
||||
## Migration
|
||||
@@ -556,9 +904,39 @@ Replace in `Cargo.toml` and source:
|
||||
|
||||
```bash
|
||||
cargo install --path crates/clawhdf5-migrate
|
||||
clawhdf5-migrate --sqlite old.db --hdf5 memory.h5 --agent-id my-agent --embedding-dim 384
|
||||
clawhdf5-migrate --sqlite old.db --hdf5 memory.h5 --agent-id my-agent --embedder minilm
|
||||
```
|
||||
|
||||
The output is an ordinary `clawhdf5-agent` store, written through the agent's
|
||||
own API: open it with `HDF5Memory::open` (or `clawhdf5-cli --path memory.h5 …`)
|
||||
and search it straight away. The source must use the `memory_chunks` / `sessions` / `entities` / `relations` layout (names are
|
||||
configurable with `--*-table`); note that this is not ZeroClaw's schema, and
|
||||
ZeroClaw does not use clawhdf5. What carries over:
|
||||
|
||||
| SQLite | Agent store |
|
||||
|--------|-------------|
|
||||
| `memory_chunks` | memory records (text, embedding, source channel, timestamp, session id, tags); rows with `deleted = 1` become deleted records, or are left out with `--skip-deleted` |
|
||||
| `sessions` | sessions (id, start/end index, channel, summary, timestamp) |
|
||||
| `entities`, `relations` | knowledge graph entities and relations; entities get new ids and relations are re-pointed at them |
|
||||
|
||||
The chunk `id` column has no counterpart in the agent store, so records are
|
||||
written in `id` order and numbered from 0. Embeddings are stored as float16
|
||||
like any new store; `--f32` keeps full precision (and is required for values
|
||||
beyond ±65504). The embedding dimension is detected from the first row unless
|
||||
`--embedding-dim` is given, and every row must have it: a row of another length
|
||||
is an error, never truncated or padded. A source with no memory records (only
|
||||
sessions or the graph) needs `--embedding-dim`, since a store's dimension is
|
||||
fixed when it is created. Every row is checked before the output is created,
|
||||
so a source that cannot be migrated leaves an existing store at `--hdf5` as it
|
||||
was. `--incremental` adds to an existing store only the rows it does not
|
||||
already hold; the source must have the store's dimension, and records already
|
||||
in the store take the source's deleted flag (a row deleted in SQLite since the
|
||||
last run is deleted in the store; one un-deleted there is written again, as
|
||||
the agent has no un-delete). The tool reads the result back with
|
||||
`HDF5Memory::open_read_only`, compares it with the source (every row with
|
||||
`--validate-full`) and checks that a migrated record is found by search;
|
||||
`--dry-run` only counts the rows.
|
||||
|
||||
---
|
||||
|
||||
## Roadmap
|
||||
@@ -572,10 +950,10 @@ See [ROADMAP.md](ROADMAP.md) for the full implementation tracker.
|
||||
- ✅ Temporal reasoning with sub-µs queries
|
||||
- ✅ Memory security + anomaly detection
|
||||
- ✅ Multi-modal memory (text/image/audio/video)
|
||||
- ✅ OpenClaw integration layer
|
||||
- ✅ Markdown ingest/export backend (`ClawhdfBackend`); an OpenClaw plugin was never built — see [docs/openclaw.md](docs/openclaw.md)
|
||||
- ✅ Comprehensive Criterion benchmarks
|
||||
|
||||
**Phase 2** — MemoryArena and LongMemEval academic benchmarks are done (see [BENCHMARKS.md](BENCHMARKS.md), reproduced on a second machine); remaining: publish the OpenClaw TypeScript bridge to npm, crates.io/PyPI publishing.
|
||||
**Phase 2** — MemoryArena and LongMemEval academic benchmarks are done (see [BENCHMARKS.md](BENCHMARKS.md), reproduced on a second machine); remaining: crates.io/PyPI publishing. The Node bindings are unpublished and known to be broken ([known issues](docs/known-issues.md)).
|
||||
|
||||
---
|
||||
|
||||
@@ -592,6 +970,6 @@ MIT
|
||||
---
|
||||
|
||||
<p align="center">
|
||||
<em>Built by <a href="https://github.com/redclawsystems">RedClaw Systems</a></em><br>
|
||||
<em>~92,000 lines of Rust. Zero C dependencies. One file to remember everything.</em>
|
||||
<em>Built by <a href="https://git.redclaw.dev/quantumclaw">RedClaw Systems</a></em><br>
|
||||
<em>~86,000 lines of Rust. Zero C dependencies. One file to remember everything.</em>
|
||||
</p>
|
||||
|
||||
+14
-8
@@ -105,24 +105,30 @@
|
||||
|
||||
---
|
||||
|
||||
## Track 7: OpenClaw Integration
|
||||
**Status:** 🟢 Complete
|
||||
## Track 7: OpenClaw Integration — withdrawn (2026-09-25)
|
||||
**Status:** ⚪ Withdrawn (the items below were library work; no OpenClaw integration shipped)
|
||||
**Priority:** Critical (for adoption)
|
||||
**Crates:** `clawhdf5-agent`, `clawhdf5-napi`
|
||||
|
||||
- [x] **7.1** Memory backend trait — MemoryBackend with search/get/write/ingest/export/stats
|
||||
- [x] **7.2** Hybrid retrieval pipeline — ClawhdfBackend wires RRF → reranker → confidence rejection
|
||||
- [x] **7.3** Markdown import/export — MarkdownParser + MarkdownExporter with line tracking + metadata
|
||||
- [x] **7.4** memory_search tool — backed by full hybrid retrieval pipeline
|
||||
- [x] **7.5** memory_get tool — get() with path + line range support
|
||||
- [x] **7.4** `search()` — backed by the full hybrid retrieval pipeline (a Rust method; no OpenClaw tool was ever registered)
|
||||
- [x] **7.5** `get()` — read back by path, with a line slice (not an OpenClaw tool either)
|
||||
- [x] **7.6** Compaction integration — run_compaction() (decay + compact + WAL flush), run_consolidation() (hippocampal engine), tick_session(), flush_wal()
|
||||
- [x] **7.7** Config surface — `memory.backend = "clawhdf5"` schema documented in docs/openclaw-config.md
|
||||
- [x] **7.8** Documentation + migration guide — docs/migration-guide.md, docs/openclaw-integration.md (architecture, full API reference, code patterns)
|
||||
- [ ] **7.7** ~~Config surface — `memory.backend = "clawhdf5"`~~ — never valid OpenClaw config; docs removed
|
||||
- [ ] **7.8** ~~Documentation + migration guide~~ — removed: they described an integration that never worked
|
||||
|
||||
**Node.js bridge:** `clawhdf5-napi` (napi-rs) → `@redclaw/clawhdf5` npm package with full TypeScript types.
|
||||
**Node.js bridge:** `clawhdf5-napi` (napi-rs) and a TypeScript wrapper in `packages/clawhdf5-node` exist but are unpublished, untested in CI and known to be broken (docs/known-issues.md).
|
||||
|
||||
---
|
||||
|
||||
> **Withdrawn.** None of this track produced a working OpenClaw integration: no
|
||||
> plugin was built, the documented `memory.backend = "clawhdf5"` config was never
|
||||
> valid in any OpenClaw release, and the Node package was never published. The
|
||||
> Rust `ClawhdfBackend` remains as a library API. Not pursued for now; see
|
||||
> [docs/openclaw.md](docs/openclaw.md) for what a plugin would need today.
|
||||
|
||||
## Track 8: Benchmarking & Validation
|
||||
**Status:** 🟢 Complete
|
||||
**Priority:** High
|
||||
@@ -142,7 +148,7 @@
|
||||
|
||||
**Phase 1:** ~~Tracks 1, 2, 3 — core memory intelligence~~ 🟢 Complete
|
||||
**Phase 2:** ~~Track 4 (temporal) + Track 5 (security)~~ 🟢 Complete
|
||||
**Phase 3:** ~~Track 6 (multi-modal) + Track 7 (OpenClaw integration)~~ 🟢 Complete
|
||||
**Phase 3:** ~~Track 6 (multi-modal)~~ 🟢 Complete; Track 7 (OpenClaw integration) withdrawn
|
||||
**Phase 4:** ~~Track 8 (benchmarking + validation)~~ 🟢 Complete
|
||||
|
||||
All 8 tracks delivered. 1,650+ tests passing, zero clippy warnings.
|
||||
|
||||
@@ -0,0 +1,3 @@
|
||||
/.cache/
|
||||
# pin the probe's dependencies (the workspace lock is not committed)
|
||||
!/probe/Cargo.lock
|
||||
@@ -0,0 +1,39 @@
|
||||
# Conformance sweep
|
||||
|
||||
Reads every HDF5 file of eight public corpora with clawhdf5 and with
|
||||
h5py/libhdf5, compares the two readings object by object, and writes
|
||||
[`CONFORMANCE.md`](../CONFORMANCE.md).
|
||||
|
||||
```sh
|
||||
CLAWHDF5_PYTHON=/path/to/venv/bin/python conformance/run.sh # ~30 s once the corpus is cached
|
||||
conformance/run.sh --update-baseline # after an intended change in results
|
||||
```
|
||||
|
||||
Needs Rust, `git`, `h5dump` (Debian/Ubuntu `hdf5-tools`), `libaec` (for the
|
||||
probe's `szip` feature; `libaec-dev`), and a Python with the packages in
|
||||
`requirements.txt`. The first run downloads about 450 MB of sparse checkouts.
|
||||
|
||||
| file | role |
|
||||
|---|---|
|
||||
| `corpus.txt` | the corpora: git URL, pinned commit, swept root, sparse-checkout patterns |
|
||||
| `fetch-corpus.sh` | shallow, sparse, blob-filtered checkout of each pinned commit into `.cache/src/` (gitignored); no-op when already there |
|
||||
| `list_files.py` | which files are probed (HDF5/netCDF-4 extensions minus netCDF classic, plus the CVE reproducers) |
|
||||
| `probe/` | the clawhdf5 side: a standalone crate (outside the workspace, so `cargo test --workspace` never builds it) that walks a file with `clawhdf5-format` and prints canonical JSON |
|
||||
| `ref.py` | the h5py side: the same JSON from h5py |
|
||||
| `run_one.sh` | runs both sides on one file (and `h5dump` on the CVE corpus) under a timeout and an address-space limit |
|
||||
| `compare.py` | classifies each file (ok / our-error / mismatch / h5py-cannot-read / panic / hang / crash / oom) and groups root causes |
|
||||
| `report.py` | writes `CONFORMANCE.md` |
|
||||
| `check.py` | the gate: fails on any panic/hang/crash/oom, on an ok count below `baseline.json`, or on a baseline-ok file that is no longer ok |
|
||||
| `baseline.json` | the ok files the gate holds the line on |
|
||||
| `requirements.txt` | pinned h5py / numpy / hdf5plugin / netCDF4 |
|
||||
|
||||
Results for every file (both sides' JSON and stderr, `results.csv`,
|
||||
`results.json`, `summary.md`) are left in `.cache/results/`.
|
||||
|
||||
The nightly job is `.gitea/workflows/conformance.yml`; it prints the report
|
||||
into the job log.
|
||||
|
||||
The canonical value encoding both sides hash is documented at the top of
|
||||
`probe/src/main.rs`. Values are compared as libhdf5 presents them: a float
|
||||
with a non-IEEE bit layout (N-Bit) or an integer with a bit offset is compared
|
||||
as the converted number, not as raw file bytes.
|
||||
@@ -0,0 +1,624 @@
|
||||
{
|
||||
"comment": "conformance/run.sh fails if the ok count drops below `ok` or a file in `ok_files` stops being ok. Regenerate with `conformance/run.sh --update-baseline` after an intended change.",
|
||||
"commit": "73a01f1256fb9bf1b1e7601f755af9e8273cec4e",
|
||||
"date": "2026-09-26 14:18 UTC",
|
||||
"reference": "h5py 3.16.0 / HDF5 2.0.0",
|
||||
"files": 697,
|
||||
"ok": 575,
|
||||
"counts": {
|
||||
"h5py-cannot-read": 92,
|
||||
"mismatch": 20,
|
||||
"ok": 575,
|
||||
"our-error": 10
|
||||
},
|
||||
"per_corpus": {
|
||||
"NCAS-CMS_pyfive": {
|
||||
"mismatch": 1,
|
||||
"ok": 32
|
||||
},
|
||||
"cve_hdf5": {
|
||||
"h5py-cannot-read": 32,
|
||||
"mismatch": 9,
|
||||
"ok": 100,
|
||||
"our-error": 6
|
||||
},
|
||||
"h5py_data": {
|
||||
"ok": 4
|
||||
},
|
||||
"hdf5": {
|
||||
"h5py-cannot-read": 60,
|
||||
"mismatch": 10,
|
||||
"ok": 392,
|
||||
"our-error": 4
|
||||
},
|
||||
"netcdf-c": {
|
||||
"ok": 20
|
||||
},
|
||||
"netcdf4-python": {
|
||||
"ok": 18
|
||||
},
|
||||
"usnistgov_h5wasm": {
|
||||
"ok": 5
|
||||
},
|
||||
"xarray-data": {
|
||||
"ok": 4
|
||||
}
|
||||
},
|
||||
"ok_files": [
|
||||
"NCAS-CMS_pyfive/tests/compact.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/btreev2.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/chunked.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/cmip_bad_eg.nc",
|
||||
"NCAS-CMS_pyfive/tests/data/compressed.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/compressed_v1.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/dataset_datatypes.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/dataset_multidim.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/dim_scales.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/earliest.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/enum_h5variable.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/enum_variable.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/enum_variable.nc",
|
||||
"NCAS-CMS_pyfive/tests/data/enums_from_netcdf.nc",
|
||||
"NCAS-CMS_pyfive/tests/data/fillvalue_earliest.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/fillvalue_latest.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/filter_pipeline_v2.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/fletcher32.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/fractal_heap_no_mci_rlat.nc",
|
||||
"NCAS-CMS_pyfive/tests/data/groups.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/h5netcdf_test.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/issue23_A.nc",
|
||||
"NCAS-CMS_pyfive/tests/data/issue23_A_contiguous.nc",
|
||||
"NCAS-CMS_pyfive/tests/data/issue23_B.nc",
|
||||
"NCAS-CMS_pyfive/tests/data/latest.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/netcdf4_classic.nc",
|
||||
"NCAS-CMS_pyfive/tests/data/new_style_groups.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/noy_AERmonZ_UKESM1-0-LL_piControl_r1i1p1f2_gnz_200001-200012.nc",
|
||||
"NCAS-CMS_pyfive/tests/data/references.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/data/resizable.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/opaque_datetime.hdf5",
|
||||
"NCAS-CMS_pyfive/tests/opaque_fixed.hdf5",
|
||||
"cve_hdf5/cvefiles/cve-2016-4330.h5",
|
||||
"cve_hdf5/cvefiles/cve-2016-4331.h5",
|
||||
"cve_hdf5/cvefiles/cve-2016-4332-mtime-new.h5",
|
||||
"cve_hdf5/cvefiles/cve-2016-4332-mtime.h5",
|
||||
"cve_hdf5/cvefiles/cve-2016-4333.h5",
|
||||
"cve_hdf5/cvefiles/cve-2017-17505.h5",
|
||||
"cve_hdf5/cvefiles/cve-2017-17506.h5",
|
||||
"cve_hdf5/cvefiles/cve-2017-17507.h5",
|
||||
"cve_hdf5/cvefiles/cve-2017-17508.h5",
|
||||
"cve_hdf5/cvefiles/cve-2017-17509.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-11202.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-11203.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-11204.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-11205.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-11206-new.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-11206-old.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-11207.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-13867.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-13868.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-13869.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-13870.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-13871.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-13872.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-13873.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-13875.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-14031.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-14033.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-14034.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-14035.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-14460.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-15671.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-15672.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-16438.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-17233.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-17234.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-17237.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-17432.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-17434.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-17435.h5",
|
||||
"cve_hdf5/cvefiles/cve-2018-17437.h5",
|
||||
"cve_hdf5/cvefiles/cve-2019-8396.h5",
|
||||
"cve_hdf5/cvefiles/cve-2019-9152.h5",
|
||||
"cve_hdf5/cvefiles/cve-2020-10811.h5",
|
||||
"cve_hdf5/cvefiles/cve-2020-18232.h5",
|
||||
"cve_hdf5/cvefiles/cve-2021-36977.h5",
|
||||
"cve_hdf5/cvefiles/cve-2021-37501.h5",
|
||||
"cve_hdf5/cvefiles/cve-2021-45829.h5",
|
||||
"cve_hdf5/cvefiles/cve-2021-45833.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-29157.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-29158.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-29159.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-29160.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-29161.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-29162.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-29163.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-29164.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-29165.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-29166.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32605.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32606.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32607-1.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32607-2.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32608.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32610.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32611.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32612.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32613.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32614.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32615.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32616.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32617.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32619.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32620.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32621.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32622.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-32624.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-33873.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-33875.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-33876.h5",
|
||||
"cve_hdf5/cvefiles/cve-2024-33877.h5",
|
||||
"cve_hdf5/cvefiles/cve-2025-2310.h5",
|
||||
"cve_hdf5/cvefiles/cve-2025-2924.h5",
|
||||
"cve_hdf5/cvefiles/cve-2025-2925.h5",
|
||||
"cve_hdf5/cvefiles/cve-2025-6269-1.h5",
|
||||
"cve_hdf5/cvefiles/cve-2025-6269-2.h5",
|
||||
"cve_hdf5/cvefiles/cve-2025-6269-3.h5",
|
||||
"cve_hdf5/cvefiles/cve-2025-6269-4.h5",
|
||||
"cve_hdf5/cvefiles/cve-2025-6516.h5",
|
||||
"cve_hdf5/cvefiles/cve-2025-6857.h5",
|
||||
"cve_hdf5/cvefiles/cve-2025-7067.h5",
|
||||
"cve_hdf5/cvefiles/cve-2026-26200.h5",
|
||||
"cve_hdf5/cvefiles/cve-2026-34734.h5",
|
||||
"cve_hdf5/cvefiles/cve-2026-92627.h5",
|
||||
"cve_hdf5/cvefiles/unknown-1.h5",
|
||||
"cve_hdf5/fuzzerfiles/gh-4431-poc-03.h5",
|
||||
"cve_hdf5/fuzzerfiles/gh-4432-poc-05.h5",
|
||||
"cve_hdf5/fuzzerfiles/gh-4433-poc-08.h5",
|
||||
"cve_hdf5/fuzzerfiles/gh-4435-poc-10.h5",
|
||||
"cve_hdf5/fuzzerfiles/gh_2649_flawed.h5",
|
||||
"cve_hdf5/fuzzerfiles/gh_2649_plain_model.h5",
|
||||
"h5py_data/compound-dtype-complex.h5",
|
||||
"h5py_data/vlen_string_dset.h5",
|
||||
"h5py_data/vlen_string_dset_utc.h5",
|
||||
"h5py_data/vlen_string_s390x.h5",
|
||||
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_bitgroom.h5",
|
||||
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_blosc.h5",
|
||||
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_bshuf.h5",
|
||||
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_bzip2.h5",
|
||||
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_granularbr.h5",
|
||||
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_jpeg.h5",
|
||||
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_lz4.h5",
|
||||
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_lzf.h5",
|
||||
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_zstd.h5",
|
||||
"hdf5/HDF5Examples/C/H5G/16/h5ex_g_iterate.h5",
|
||||
"hdf5/HDF5Examples/C/H5G/16/h5ex_g_traverse.h5",
|
||||
"hdf5/HDF5Examples/C/H5G/h5ex_g_iterate.h5",
|
||||
"hdf5/HDF5Examples/C/H5G/h5ex_g_traverse.h5",
|
||||
"hdf5/HDF5Examples/C/H5G/h5ex_g_visit.h5",
|
||||
"hdf5/HDF5Examples/FORTRAN/H5G/h5ex_g_iterate.h5",
|
||||
"hdf5/HDF5Examples/FORTRAN/H5G/h5ex_g_traverse.h5",
|
||||
"hdf5/HDF5Examples/FORTRAN/H5G/h5ex_g_visit.h5",
|
||||
"hdf5/HDF5Examples/JAVA/H5G/h5ex_g_iterate.h5",
|
||||
"hdf5/HDF5Examples/JAVA/H5G/h5ex_g_visit.h5",
|
||||
"hdf5/HDF5Examples/JAVA/compat/H5G/110/h5ex_g_iterate.h5",
|
||||
"hdf5/HDF5Examples/JAVA/compat/H5G/110/h5ex_g_visit.h5",
|
||||
"hdf5/HDF5Examples/JAVA/compat/H5G/h5ex_g_iterate.h5",
|
||||
"hdf5/HDF5Examples/JAVA/compat/H5G/h5ex_g_visit.h5",
|
||||
"hdf5/c++/test/th5s.h5",
|
||||
"hdf5/hl/test/testfiles/test_ds_be.h5",
|
||||
"hdf5/hl/test/testfiles/test_ds_be_new_ref-32bit.h5",
|
||||
"hdf5/hl/test/testfiles/test_ds_be_new_ref.h5",
|
||||
"hdf5/hl/test/testfiles/test_ds_le.h5",
|
||||
"hdf5/hl/test/testfiles/test_ds_le_new_ref.h5",
|
||||
"hdf5/hl/test/testfiles/test_ld.h5",
|
||||
"hdf5/hl/test/testfiles/test_table_be.h5",
|
||||
"hdf5/hl/test/testfiles/test_table_cray.h5",
|
||||
"hdf5/hl/test/testfiles/test_table_le.h5",
|
||||
"hdf5/test/testfiles/aggr.h5",
|
||||
"hdf5/test/testfiles/bad_chunk_ndims.h5",
|
||||
"hdf5/test/testfiles/bad_compound.h5",
|
||||
"hdf5/test/testfiles/bad_offset.h5",
|
||||
"hdf5/test/testfiles/be_data.h5",
|
||||
"hdf5/test/testfiles/be_extlink1.h5",
|
||||
"hdf5/test/testfiles/be_extlink2.h5",
|
||||
"hdf5/test/testfiles/btree_idx_1_6.h5",
|
||||
"hdf5/test/testfiles/btree_idx_1_8.h5",
|
||||
"hdf5/test/testfiles/charsets.h5",
|
||||
"hdf5/test/testfiles/corrupt_stab_msg.h5",
|
||||
"hdf5/test/testfiles/deflate.h5",
|
||||
"hdf5/test/testfiles/file_image_core_test.h5",
|
||||
"hdf5/test/testfiles/filespace_1_6.h5",
|
||||
"hdf5/test/testfiles/filespace_1_8.h5",
|
||||
"hdf5/test/testfiles/fill18.h5",
|
||||
"hdf5/test/testfiles/fill_old.h5",
|
||||
"hdf5/test/testfiles/filter_error.h5",
|
||||
"hdf5/test/testfiles/fsm_aggr_nopersist.h5",
|
||||
"hdf5/test/testfiles/fsm_aggr_persist.h5",
|
||||
"hdf5/test/testfiles/group_old.h5",
|
||||
"hdf5/test/testfiles/h5fc_ext1_f.h5",
|
||||
"hdf5/test/testfiles/h5fc_ext1_i.h5",
|
||||
"hdf5/test/testfiles/h5fc_ext2_if.h5",
|
||||
"hdf5/test/testfiles/h5fc_ext2_sf.h5",
|
||||
"hdf5/test/testfiles/h5fc_ext3_isf.h5",
|
||||
"hdf5/test/testfiles/h5fc_ext_none.h5",
|
||||
"hdf5/test/testfiles/le_data.h5",
|
||||
"hdf5/test/testfiles/le_extlink1.h5",
|
||||
"hdf5/test/testfiles/le_extlink2.h5",
|
||||
"hdf5/test/testfiles/memleak_H5O_dtype_decode_helper_H5Odtype.h5",
|
||||
"hdf5/test/testfiles/mergemsg.h5",
|
||||
"hdf5/test/testfiles/noencoder.h5",
|
||||
"hdf5/test/testfiles/none.h5",
|
||||
"hdf5/test/testfiles/paged_nopersist.h5",
|
||||
"hdf5/test/testfiles/paged_persist.h5",
|
||||
"hdf5/test/testfiles/specmetaread.h5",
|
||||
"hdf5/test/testfiles/tarrold.h5",
|
||||
"hdf5/test/testfiles/tbad_msg_count.h5",
|
||||
"hdf5/test/testfiles/tbogus.h5",
|
||||
"hdf5/test/testfiles/test_filters_be.h5",
|
||||
"hdf5/test/testfiles/test_filters_le.h5",
|
||||
"hdf5/test/testfiles/th5s.h5",
|
||||
"hdf5/test/testfiles/tlayouto.h5",
|
||||
"hdf5/test/testfiles/tmisc38a.h5",
|
||||
"hdf5/test/testfiles/tmisc38b.h5",
|
||||
"hdf5/test/testfiles/tmtimen.h5",
|
||||
"hdf5/test/testfiles/tmtimeo.h5",
|
||||
"hdf5/test/testfiles/tnullspace.h5",
|
||||
"hdf5/test/testfiles/tsizeslheap.h5",
|
||||
"hdf5/tools/test/testfiles/bigendian/tdset2.h5",
|
||||
"hdf5/tools/test/testfiles/binfp64.h5",
|
||||
"hdf5/tools/test/testfiles/binin16.h5",
|
||||
"hdf5/tools/test/testfiles/binin32.h5",
|
||||
"hdf5/tools/test/testfiles/binin8.h5",
|
||||
"hdf5/tools/test/testfiles/binin8w.h5",
|
||||
"hdf5/tools/test/testfiles/binuin16.h5",
|
||||
"hdf5/tools/test/testfiles/binuin32.h5",
|
||||
"hdf5/tools/test/testfiles/bounds_latest_latest.h5",
|
||||
"hdf5/tools/test/testfiles/charsets.h5",
|
||||
"hdf5/tools/test/testfiles/compounds_array_vlen1.h5",
|
||||
"hdf5/tools/test/testfiles/compounds_array_vlen2.h5",
|
||||
"hdf5/tools/test/testfiles/err_attr_dspace.h5",
|
||||
"hdf5/tools/test/testfiles/file_space.h5",
|
||||
"hdf5/tools/test/testfiles/filter_fail.h5",
|
||||
"hdf5/tools/test/testfiles/h5clear_fsm_persist_equal.h5",
|
||||
"hdf5/tools/test/testfiles/h5clear_fsm_persist_less.h5",
|
||||
"hdf5/tools/test/testfiles/h5clear_fsm_persist_noclose.h5",
|
||||
"hdf5/tools/test/testfiles/h5clear_fsm_persist_user_equal.h5",
|
||||
"hdf5/tools/test/testfiles/h5clear_fsm_persist_user_less.h5",
|
||||
"hdf5/tools/test/testfiles/h5clear_sec2_v0.h5",
|
||||
"hdf5/tools/test/testfiles/h5clear_sec2_v2.h5",
|
||||
"hdf5/tools/test/testfiles/h5copy_extlinks_src.h5",
|
||||
"hdf5/tools/test/testfiles/h5copy_extlinks_trg.h5",
|
||||
"hdf5/tools/test/testfiles/h5copy_ref.h5",
|
||||
"hdf5/tools/test/testfiles/h5copytst.h5",
|
||||
"hdf5/tools/test/testfiles/h5copytst_new.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_attr1.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_attr2.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_attr3.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_attr_v_level1.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_attr_v_level2.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_basic1.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_basic2.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_comp_vl_strs.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_danglelinks1.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_danglelinks2.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_dset1.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_dset2.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_dset3.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_dset_zero_dim_size1.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_dset_zero_dim_size2.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_dtypes.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_empty.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_enum_invalid_values.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_eps1.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_eps2.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_exclude1-1.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_exclude1-2.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_exclude2-1.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_exclude2-2.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_exclude3-1.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_exclude3-2.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_ext2softlink_src.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_ext2softlink_trg.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_extlink_src.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_extlink_trg.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_grp_recurse1.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_grp_recurse2.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_grp_recurse_ext1.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_grp_recurse_ext2-1.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_grp_recurse_ext2-2.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_grp_recurse_ext2-3.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_hyper1.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_hyper2.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_linked_softlink.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_links.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_onion_dset_1d.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_onion_dset_ext.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_onion_objs.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_softlinks.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_strings1.h5",
|
||||
"hdf5/tools/test/testfiles/h5diff_strings2.h5",
|
||||
"hdf5/tools/test/testfiles/h5fc_edge_v3.h5",
|
||||
"hdf5/tools/test/testfiles/h5fc_err_level.h5",
|
||||
"hdf5/tools/test/testfiles/h5fc_ext1_f.h5",
|
||||
"hdf5/tools/test/testfiles/h5fc_ext1_i.h5",
|
||||
"hdf5/tools/test/testfiles/h5fc_ext1_s.h5",
|
||||
"hdf5/tools/test/testfiles/h5fc_ext2_if.h5",
|
||||
"hdf5/tools/test/testfiles/h5fc_ext2_is.h5",
|
||||
"hdf5/tools/test/testfiles/h5fc_ext2_sf.h5",
|
||||
"hdf5/tools/test/testfiles/h5fc_ext3_isf.h5",
|
||||
"hdf5/tools/test/testfiles/h5fc_ext_none.h5",
|
||||
"hdf5/tools/test/testfiles/h5fc_non_v3.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_CVE-2018-14460.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_CVE-2018-17432.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_aggr.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_attr.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_attr_refs.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_deflate.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_early.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_ext.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_f32le.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_f32le_ex.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_fill.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_filters.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_fletcher.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_fsm_aggr_nopersist.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_fsm_aggr_persist.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_hlink.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_int32le_1d.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_int32le_1d_ex.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_int32le_2d.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_int32le_2d_ex.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_int32le_3d.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_int32le_3d_ex.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_layout.UD.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_layout.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_layout2.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_layout3.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_layouto.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_named_dtypes.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_nbit.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_nested_8bit_enum.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_nested_8bit_enum_deflated.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_none.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_objs.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_paged_nopersist.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_paged_persist.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_refs.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_shuffle.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_soffset.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_szip.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_uint8be.h5",
|
||||
"hdf5/tools/test/testfiles/h5repack_uint8be_ex.h5",
|
||||
"hdf5/tools/test/testfiles/h5stat_err_old_fill.h5",
|
||||
"hdf5/tools/test/testfiles/h5stat_err_old_layout.h5",
|
||||
"hdf5/tools/test/testfiles/h5stat_err_refcount.h5",
|
||||
"hdf5/tools/test/testfiles/h5stat_filters.h5",
|
||||
"hdf5/tools/test/testfiles/h5stat_idx.h5",
|
||||
"hdf5/tools/test/testfiles/h5stat_newgrat.h5",
|
||||
"hdf5/tools/test/testfiles/h5stat_threshold.h5",
|
||||
"hdf5/tools/test/testfiles/h5stat_tsohm.h5",
|
||||
"hdf5/tools/test/testfiles/mod_h5clear_mdc_image.h5",
|
||||
"hdf5/tools/test/testfiles/non_comparables1.h5",
|
||||
"hdf5/tools/test/testfiles/non_comparables2.h5",
|
||||
"hdf5/tools/test/testfiles/old_h5fc_ext1_f.h5",
|
||||
"hdf5/tools/test/testfiles/old_h5fc_ext1_i.h5",
|
||||
"hdf5/tools/test/testfiles/old_h5fc_ext1_s.h5",
|
||||
"hdf5/tools/test/testfiles/old_h5fc_ext2_if.h5",
|
||||
"hdf5/tools/test/testfiles/old_h5fc_ext2_is.h5",
|
||||
"hdf5/tools/test/testfiles/old_h5fc_ext2_sf.h5",
|
||||
"hdf5/tools/test/testfiles/old_h5fc_ext3_isf.h5",
|
||||
"hdf5/tools/test/testfiles/old_h5fc_ext_none.h5",
|
||||
"hdf5/tools/test/testfiles/packedbits.h5",
|
||||
"hdf5/tools/test/testfiles/t128bit_float.h5",
|
||||
"hdf5/tools/test/testfiles/tCVE-2021-37501_attr_decode.h5",
|
||||
"hdf5/tools/test/testfiles/tCVE_2018_11206_fill_new.h5",
|
||||
"hdf5/tools/test/testfiles/tCVE_2018_11206_fill_old.h5",
|
||||
"hdf5/tools/test/testfiles/taindices.h5",
|
||||
"hdf5/tools/test/testfiles/tarray1.h5",
|
||||
"hdf5/tools/test/testfiles/tarray1_big.h5",
|
||||
"hdf5/tools/test/testfiles/tarray2.h5",
|
||||
"hdf5/tools/test/testfiles/tarray4.h5",
|
||||
"hdf5/tools/test/testfiles/tarray5.h5",
|
||||
"hdf5/tools/test/testfiles/tarray8.h5",
|
||||
"hdf5/tools/test/testfiles/tattr.h5",
|
||||
"hdf5/tools/test/testfiles/tattr2.h5",
|
||||
"hdf5/tools/test/testfiles/tattr4_be.h5",
|
||||
"hdf5/tools/test/testfiles/tattrintsize.h5",
|
||||
"hdf5/tools/test/testfiles/tattrreg.h5",
|
||||
"hdf5/tools/test/testfiles/tbfloat16.h5",
|
||||
"hdf5/tools/test/testfiles/tbfloat16_be.h5",
|
||||
"hdf5/tools/test/testfiles/tbigdims.h5",
|
||||
"hdf5/tools/test/testfiles/tbinary.h5",
|
||||
"hdf5/tools/test/testfiles/tbitnopaque.h5",
|
||||
"hdf5/tools/test/testfiles/tchar.h5",
|
||||
"hdf5/tools/test/testfiles/tcmpdattrintsize.h5",
|
||||
"hdf5/tools/test/testfiles/tcmpdintarray.h5",
|
||||
"hdf5/tools/test/testfiles/tcmpdints.h5",
|
||||
"hdf5/tools/test/testfiles/tcmpdintsize.h5",
|
||||
"hdf5/tools/test/testfiles/tcomplex.h5",
|
||||
"hdf5/tools/test/testfiles/tcompound.h5",
|
||||
"hdf5/tools/test/testfiles/tcompound_complex.h5",
|
||||
"hdf5/tools/test/testfiles/tcompound_complex2.h5",
|
||||
"hdf5/tools/test/testfiles/tdatareg.h5",
|
||||
"hdf5/tools/test/testfiles/tdset.h5",
|
||||
"hdf5/tools/test/testfiles/tdset2.h5",
|
||||
"hdf5/tools/test/testfiles/tdset_idx.h5",
|
||||
"hdf5/tools/test/testfiles/tempty.h5",
|
||||
"hdf5/tools/test/testfiles/textlink.h5",
|
||||
"hdf5/tools/test/testfiles/textlinkfar.h5",
|
||||
"hdf5/tools/test/testfiles/textlinksrc.h5",
|
||||
"hdf5/tools/test/testfiles/textlinktar.h5",
|
||||
"hdf5/tools/test/testfiles/textpfe.h5",
|
||||
"hdf5/tools/test/testfiles/tfcontents2.h5",
|
||||
"hdf5/tools/test/testfiles/tfilters.h5",
|
||||
"hdf5/tools/test/testfiles/tfloat16.h5",
|
||||
"hdf5/tools/test/testfiles/tfloat16_be.h5",
|
||||
"hdf5/tools/test/testfiles/tfloat4.h5",
|
||||
"hdf5/tools/test/testfiles/tfloat6.h5",
|
||||
"hdf5/tools/test/testfiles/tfloat8.h5",
|
||||
"hdf5/tools/test/testfiles/tfloatsattrs.h5",
|
||||
"hdf5/tools/test/testfiles/tfpformat.h5",
|
||||
"hdf5/tools/test/testfiles/tfvalues.h5",
|
||||
"hdf5/tools/test/testfiles/tgroup.h5",
|
||||
"hdf5/tools/test/testfiles/tgrp_comments.h5",
|
||||
"hdf5/tools/test/testfiles/tgrpnullspace.h5",
|
||||
"hdf5/tools/test/testfiles/thlink.h5",
|
||||
"hdf5/tools/test/testfiles/thyperslab.h5",
|
||||
"hdf5/tools/test/testfiles/tintascii.h5",
|
||||
"hdf5/tools/test/testfiles/tints4dims.h5",
|
||||
"hdf5/tools/test/testfiles/tintsattrs.h5",
|
||||
"hdf5/tools/test/testfiles/tintsnodata.h5",
|
||||
"hdf5/tools/test/testfiles/tlarge_objname.h5",
|
||||
"hdf5/tools/test/testfiles/tldouble.h5",
|
||||
"hdf5/tools/test/testfiles/tldouble_scalar.h5",
|
||||
"hdf5/tools/test/testfiles/tlonglinks.h5",
|
||||
"hdf5/tools/test/testfiles/tloop.h5",
|
||||
"hdf5/tools/test/testfiles/tnamed_dtype_attr.h5",
|
||||
"hdf5/tools/test/testfiles/tnestedcmpddt.h5",
|
||||
"hdf5/tools/test/testfiles/tnestedcomp.h5",
|
||||
"hdf5/tools/test/testfiles/tno-subset.h5",
|
||||
"hdf5/tools/test/testfiles/tnullspace.h5",
|
||||
"hdf5/tools/test/testfiles/torderattr.h5",
|
||||
"hdf5/tools/test/testfiles/tordergr.h5",
|
||||
"hdf5/tools/test/testfiles/trefer_attr.h5",
|
||||
"hdf5/tools/test/testfiles/trefer_compat.h5",
|
||||
"hdf5/tools/test/testfiles/trefer_ext1.h5",
|
||||
"hdf5/tools/test/testfiles/trefer_ext2.h5",
|
||||
"hdf5/tools/test/testfiles/trefer_grp.h5",
|
||||
"hdf5/tools/test/testfiles/trefer_obj.h5",
|
||||
"hdf5/tools/test/testfiles/trefer_obj_del.h5",
|
||||
"hdf5/tools/test/testfiles/trefer_param.h5",
|
||||
"hdf5/tools/test/testfiles/trefer_reg.h5",
|
||||
"hdf5/tools/test/testfiles/trefer_reg_1d.h5",
|
||||
"hdf5/tools/test/testfiles/tsaf.h5",
|
||||
"hdf5/tools/test/testfiles/tscalarattrintsize.h5",
|
||||
"hdf5/tools/test/testfiles/tscalarintattrsize.h5",
|
||||
"hdf5/tools/test/testfiles/tscalarintsize.h5",
|
||||
"hdf5/tools/test/testfiles/tscalarstring.h5",
|
||||
"hdf5/tools/test/testfiles/tslink.h5",
|
||||
"hdf5/tools/test/testfiles/tsoftlinks.h5",
|
||||
"hdf5/tools/test/testfiles/tst_onion_dset_1d.h5",
|
||||
"hdf5/tools/test/testfiles/tst_onion_dset_ext.h5",
|
||||
"hdf5/tools/test/testfiles/tst_onion_objs.h5",
|
||||
"hdf5/tools/test/testfiles/tstr.h5",
|
||||
"hdf5/tools/test/testfiles/tstr2.h5",
|
||||
"hdf5/tools/test/testfiles/tstr3.h5",
|
||||
"hdf5/tools/test/testfiles/tudfilter.h5",
|
||||
"hdf5/tools/test/testfiles/tudfilter2.h5",
|
||||
"hdf5/tools/test/testfiles/tvldtypes1.h5",
|
||||
"hdf5/tools/test/testfiles/tvldtypes2.h5",
|
||||
"hdf5/tools/test/testfiles/tvldtypes3.h5",
|
||||
"hdf5/tools/test/testfiles/tvldtypes4.h5",
|
||||
"hdf5/tools/test/testfiles/tvldtypes5.h5",
|
||||
"hdf5/tools/test/testfiles/tvlenstr_array.h5",
|
||||
"hdf5/tools/test/testfiles/tvlstr.h5",
|
||||
"hdf5/tools/test/testfiles/tvms.h5",
|
||||
"hdf5/tools/test/testfiles/txtfp32.h5",
|
||||
"hdf5/tools/test/testfiles/txtfp64.h5",
|
||||
"hdf5/tools/test/testfiles/txtin16.h5",
|
||||
"hdf5/tools/test/testfiles/txtin32.h5",
|
||||
"hdf5/tools/test/testfiles/txtin8.h5",
|
||||
"hdf5/tools/test/testfiles/txtstr.h5",
|
||||
"hdf5/tools/test/testfiles/txtuin16.h5",
|
||||
"hdf5/tools/test/testfiles/txtuin32.h5",
|
||||
"hdf5/tools/test/testfiles/vds/1_a.h5",
|
||||
"hdf5/tools/test/testfiles/vds/1_b.h5",
|
||||
"hdf5/tools/test/testfiles/vds/1_c.h5",
|
||||
"hdf5/tools/test/testfiles/vds/1_d.h5",
|
||||
"hdf5/tools/test/testfiles/vds/1_e.h5",
|
||||
"hdf5/tools/test/testfiles/vds/1_f.h5",
|
||||
"hdf5/tools/test/testfiles/vds/1_vds.h5",
|
||||
"hdf5/tools/test/testfiles/vds/2_a.h5",
|
||||
"hdf5/tools/test/testfiles/vds/2_b.h5",
|
||||
"hdf5/tools/test/testfiles/vds/2_c.h5",
|
||||
"hdf5/tools/test/testfiles/vds/2_d.h5",
|
||||
"hdf5/tools/test/testfiles/vds/2_e.h5",
|
||||
"hdf5/tools/test/testfiles/vds/2_vds.h5",
|
||||
"hdf5/tools/test/testfiles/vds/3_1_vds.h5",
|
||||
"hdf5/tools/test/testfiles/vds/3_2_vds.h5",
|
||||
"hdf5/tools/test/testfiles/vds/4_0.h5",
|
||||
"hdf5/tools/test/testfiles/vds/4_1.h5",
|
||||
"hdf5/tools/test/testfiles/vds/4_2.h5",
|
||||
"hdf5/tools/test/testfiles/vds/4_vds.h5",
|
||||
"hdf5/tools/test/testfiles/vds/5_a.h5",
|
||||
"hdf5/tools/test/testfiles/vds/5_b.h5",
|
||||
"hdf5/tools/test/testfiles/vds/5_c.h5",
|
||||
"hdf5/tools/test/testfiles/vds/5_vds.h5",
|
||||
"hdf5/tools/test/testfiles/vds/a.h5",
|
||||
"hdf5/tools/test/testfiles/vds/b.h5",
|
||||
"hdf5/tools/test/testfiles/vds/c.h5",
|
||||
"hdf5/tools/test/testfiles/vds/d.h5",
|
||||
"hdf5/tools/test/testfiles/vds/f-0.h5",
|
||||
"hdf5/tools/test/testfiles/vds/f-3.h5",
|
||||
"hdf5/tools/test/testfiles/vds/vds-eiger.h5",
|
||||
"hdf5/tools/test/testfiles/vds/vds-percival-unlim-maxmin.h5",
|
||||
"hdf5/tools/test/testfiles/xml/tbitfields.h5",
|
||||
"hdf5/tools/test/testfiles/xml/tcompound2.h5",
|
||||
"hdf5/tools/test/testfiles/xml/tdset2.h5",
|
||||
"hdf5/tools/test/testfiles/xml/tenum.h5",
|
||||
"hdf5/tools/test/testfiles/xml/test35.nc",
|
||||
"hdf5/tools/test/testfiles/xml/tloop2.h5",
|
||||
"hdf5/tools/test/testfiles/xml/tname-amp.h5",
|
||||
"hdf5/tools/test/testfiles/xml/tname-apos.h5",
|
||||
"hdf5/tools/test/testfiles/xml/tname-gt.h5",
|
||||
"hdf5/tools/test/testfiles/xml/tname-lt.h5",
|
||||
"hdf5/tools/test/testfiles/xml/tname-quot.h5",
|
||||
"hdf5/tools/test/testfiles/xml/tname-sp.h5",
|
||||
"hdf5/tools/test/testfiles/xml/tnodata.h5",
|
||||
"hdf5/tools/test/testfiles/xml/tobjref.h5",
|
||||
"hdf5/tools/test/testfiles/xml/topaque.h5",
|
||||
"hdf5/tools/test/testfiles/xml/tref-escapes-at.h5",
|
||||
"hdf5/tools/test/testfiles/xml/tref-escapes.h5",
|
||||
"hdf5/tools/test/testfiles/xml/tref.h5",
|
||||
"hdf5/tools/test/testfiles/xml/tstring-at.h5",
|
||||
"hdf5/tools/test/testfiles/xml/tstring.h5",
|
||||
"hdf5/tools/test/testfiles/zerodim.h5",
|
||||
"netcdf-c/h5_test/ref_tst_h_compounds.h5",
|
||||
"netcdf-c/h5_test/ref_tst_h_compounds2.h5",
|
||||
"netcdf-c/nc_test4/ref_hdf5_compat1.nc",
|
||||
"netcdf-c/nc_test4/ref_hdf5_compat2.nc",
|
||||
"netcdf-c/nc_test4/ref_hdf5_compat3.nc",
|
||||
"netcdf-c/nc_test4/ref_szip.h5",
|
||||
"netcdf-c/nc_test4/ref_tst_compounds.nc",
|
||||
"netcdf-c/nc_test4/ref_tst_dims.nc",
|
||||
"netcdf-c/nc_test4/ref_tst_interops4.nc",
|
||||
"netcdf-c/nc_test4/ref_tst_xplatform2_1.nc",
|
||||
"netcdf-c/nc_test4/ref_tst_xplatform2_2.nc",
|
||||
"netcdf-c/nc_test4/tdset.h5",
|
||||
"netcdf-c/ncdump/ref_nc_test_netcdf4_4_0.nc",
|
||||
"netcdf-c/ncdump/ref_no_ncproperty.nc",
|
||||
"netcdf-c/ncdump/ref_provenance_v1.nc",
|
||||
"netcdf-c/ncdump/ref_test_corrupt_magic.nc",
|
||||
"netcdf-c/ncdump/ref_tst_compounds2.nc",
|
||||
"netcdf-c/ncdump/ref_tst_compounds3.nc",
|
||||
"netcdf-c/ncdump/ref_tst_compounds4.nc",
|
||||
"netcdf-c/ncdump/ref_tst_irish_rover.nc",
|
||||
"netcdf4-python/examples/data/prmsl.2000.nc",
|
||||
"netcdf4-python/examples/data/prmsl.2001.nc",
|
||||
"netcdf4-python/examples/data/prmsl.2002.nc",
|
||||
"netcdf4-python/examples/data/prmsl.2003.nc",
|
||||
"netcdf4-python/examples/data/prmsl.2004.nc",
|
||||
"netcdf4-python/examples/data/prmsl.2005.nc",
|
||||
"netcdf4-python/examples/data/prmsl.2006.nc",
|
||||
"netcdf4-python/examples/data/prmsl.2007.nc",
|
||||
"netcdf4-python/examples/data/prmsl.2008.nc",
|
||||
"netcdf4-python/examples/data/prmsl.2009.nc",
|
||||
"netcdf4-python/examples/data/prmsl.2010.nc",
|
||||
"netcdf4-python/examples/data/prmsl.2011.nc",
|
||||
"netcdf4-python/examples/data/rtofs_glo_3dz_f006_6hrly_reg3.nc",
|
||||
"netcdf4-python/test/20171025_2056.Cloud_Top_Height.nc",
|
||||
"netcdf4-python/test/issue1152.nc",
|
||||
"netcdf4-python/test/issue671.nc",
|
||||
"netcdf4-python/test/issue672.nc",
|
||||
"netcdf4-python/test/test_gold.nc",
|
||||
"usnistgov_h5wasm/test/array.h5",
|
||||
"usnistgov_h5wasm/test/compressed.h5",
|
||||
"usnistgov_h5wasm/test/empty.h5",
|
||||
"usnistgov_h5wasm/test/float16.h5",
|
||||
"usnistgov_h5wasm/test/vlen.h5",
|
||||
"xarray-data/ROMS_example.nc",
|
||||
"xarray-data/basin_mask.nc",
|
||||
"xarray-data/imerghh_730.hdf5",
|
||||
"xarray-data/precipitation.nc4"
|
||||
]
|
||||
}
|
||||
Executable
+88
@@ -0,0 +1,88 @@
|
||||
#!/usr/bin/env python3
|
||||
"""check.py <results_dir> <baseline.json> [--update]
|
||||
|
||||
The conformance gate. Fails (exit 1) when
|
||||
* clawhdf5 panicked, hung, crashed or ran out of memory on any file, or
|
||||
* the ok count fell below the baseline's, or
|
||||
* a file the baseline lists as ok is no longer ok (even if another file
|
||||
became ok and the total held).
|
||||
New ok files are reported so the baseline can be raised (--update rewrites it
|
||||
from the results).
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
|
||||
FATAL = ("panic", "hang", "crash", "oom")
|
||||
|
||||
|
||||
def main():
|
||||
args = [a for a in sys.argv[1:] if not a.startswith("--")]
|
||||
update = "--update" in sys.argv
|
||||
res_dir, base_path = args
|
||||
res = json.load(open(os.path.join(res_dir, "results.json")))
|
||||
rows = res["rows"]
|
||||
counts = {}
|
||||
per_corpus = {}
|
||||
for r in rows:
|
||||
counts[r["class"]] = counts.get(r["class"], 0) + 1
|
||||
pc = per_corpus.setdefault(r["corpus"], {})
|
||||
pc[r["class"]] = pc.get(r["class"], 0) + 1
|
||||
ok_files = sorted(r["file"] for r in rows if r["class"] == "ok")
|
||||
|
||||
if update:
|
||||
meta = {}
|
||||
mp = os.path.join(res_dir, "report-meta.json")
|
||||
if os.path.exists(mp):
|
||||
meta = json.load(open(mp))
|
||||
base = {
|
||||
"comment": "conformance/run.sh fails if the ok count drops below `ok` or a file in `ok_files` stops being ok. "
|
||||
"Regenerate with `conformance/run.sh --update-baseline` after an intended change.",
|
||||
"commit": meta.get("commit", ""),
|
||||
"date": meta.get("date", ""),
|
||||
"reference": meta.get("reference", ""),
|
||||
"files": len(rows),
|
||||
"ok": len(ok_files),
|
||||
"counts": dict(sorted(counts.items())),
|
||||
"per_corpus": {k: dict(sorted(v.items())) for k, v in sorted(per_corpus.items())},
|
||||
"ok_files": ok_files,
|
||||
}
|
||||
with open(base_path, "w") as fh:
|
||||
json.dump(base, fh, indent=1)
|
||||
fh.write("\n")
|
||||
print(f"baseline updated: {len(ok_files)} ok of {len(rows)} files -> {base_path}")
|
||||
return 0
|
||||
|
||||
base = json.load(open(base_path))
|
||||
failures = []
|
||||
fatal = [r for r in rows if r["class"] in FATAL]
|
||||
for r in fatal:
|
||||
failures.append(f"{r['class']}: {r['file']}: {r['ours_detail'][:200]}")
|
||||
if len(ok_files) < base["ok"]:
|
||||
failures.append(f"ok count dropped: {len(ok_files)} < baseline {base['ok']}")
|
||||
now_ok = set(ok_files)
|
||||
by_file = {r["file"]: r for r in rows}
|
||||
for f in base["ok_files"]:
|
||||
if f not in now_ok:
|
||||
r = by_file.get(f)
|
||||
why = f"now {r['class']}: {(r['ours_detail'] or r['first_issue'])[:200]}" if r else "no longer in the corpus"
|
||||
failures.append(f"regressed: {f}: {why}")
|
||||
gained = sorted(now_ok - set(base["ok_files"]))
|
||||
|
||||
print(f"conformance: {len(ok_files)} ok of {len(rows)} files (baseline {base['ok']} of {base['files']}); "
|
||||
+ ", ".join(f"{k} {v}" for k, v in sorted(counts.items())))
|
||||
if gained:
|
||||
print(f"{len(gained)} file(s) newly ok — raise the baseline with `conformance/run.sh --update-baseline`:")
|
||||
for f in gained:
|
||||
print(f" + {f}")
|
||||
if failures:
|
||||
print(f"CONFORMANCE GATE FAILED ({len(failures)}):")
|
||||
for f in failures:
|
||||
print(f" - {f}")
|
||||
return 1
|
||||
print("conformance gate passed")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
Executable
+289
@@ -0,0 +1,289 @@
|
||||
#!/usr/bin/env python3
|
||||
"""compare.py <results_dir>: classify each file and group failures by root cause.
|
||||
|
||||
Writes <results_dir>/results.csv, results.json and summary.md.
|
||||
File classes (first match wins):
|
||||
hang, oom, crash, panic ours: timeout / allocation failure / signal / any panic (caught or not)
|
||||
h5py-cannot-read libhdf5/h5py failed to open the file (or crashed/hung)
|
||||
our-error we fail to open, list, or read something h5py reads
|
||||
mismatch we read something with different shape/values, or a different object set
|
||||
ok
|
||||
"""
|
||||
import collections
|
||||
import csv
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
|
||||
R = sys.argv[1]
|
||||
RUNS = os.path.join(R, "runs")
|
||||
|
||||
|
||||
def load(d, name):
|
||||
rc_p = os.path.join(d, name + ".rc")
|
||||
if not os.path.exists(rc_p):
|
||||
return None
|
||||
rc = int(open(rc_p).read().strip() or -1)
|
||||
err = open(os.path.join(d, name + ".err"), errors="replace").read()
|
||||
js = None
|
||||
try:
|
||||
js = json.load(open(os.path.join(d, name + ".json")))
|
||||
except Exception: # noqa: BLE001
|
||||
pass
|
||||
return {"rc": rc, "err": err, "json": js}
|
||||
|
||||
|
||||
def proc_status(p):
|
||||
"""-> (status, detail)"""
|
||||
if p is None:
|
||||
return "missing", ""
|
||||
rc, err = p["rc"], p["err"]
|
||||
first_panic = next((ln for ln in err.splitlines() if ln.startswith("PANIC:") or "panicked at" in ln), "")
|
||||
if rc == 0 and p["json"] is not None:
|
||||
return "ok", ""
|
||||
if rc == 137 or rc == 124:
|
||||
return "hang", f"timeout ({os.environ.get('TMO', '20')} s)"
|
||||
if "memory allocation of" in err or "MemoryError" in err or "std::bad_alloc" in err:
|
||||
m = re.search(r"memory allocation of \d+ bytes failed", err)
|
||||
return "oom", m.group(0) if m else "allocation failure"
|
||||
if "overflowed its stack" in err:
|
||||
return "crash", "stack overflow"
|
||||
if rc == 101:
|
||||
return "panic", first_panic or (err.strip().splitlines() or [""])[-1]
|
||||
if rc in (134, 139, 136, 135, 132) or rc > 128:
|
||||
sig = {134: "SIGABRT", 139: "SIGSEGV", 136: "SIGFPE", 135: "SIGBUS", 132: "SIGILL"}.get(rc, f"signal {rc - 128}")
|
||||
tail = [ln for ln in err.strip().splitlines() if ln.strip()][-1:]
|
||||
return "crash", f"{sig}: {tail[0][:200] if tail else ''}"
|
||||
tail = [ln for ln in err.strip().splitlines() if ln.strip()][-1:]
|
||||
return "crash", f"rc={rc}: {tail[0][:200] if tail else ''}"
|
||||
|
||||
|
||||
def norm(msg):
|
||||
m = msg.split("\n")[0]
|
||||
m = re.sub(r"0x[0-9a-fA-F]+", "X", m)
|
||||
m = re.sub(r'"[^"]*"', '"…"', m)
|
||||
m = re.sub(r"'[^']*'", "'…'", m)
|
||||
m = re.sub(r"\d+", "N", m)
|
||||
return m[:160]
|
||||
|
||||
|
||||
def panic_head(msg):
|
||||
"""First line + first clawhdf5 frame of a PANIC record."""
|
||||
lines = msg.split("\n")
|
||||
frame = next((ln.strip() for ln in lines[1:] if "clawhdf5_format" in ln), "")
|
||||
return lines[0][:300], frame[:300]
|
||||
|
||||
|
||||
def eq_shape(a, b):
|
||||
return a == b
|
||||
|
||||
|
||||
rows = []
|
||||
issues_by_file = {}
|
||||
root_causes = collections.defaultdict(lambda: {"files": set(), "count": 0, "examples": []})
|
||||
mismatch_causes = collections.defaultdict(lambda: {"files": set(), "count": 0, "examples": []})
|
||||
panics = []
|
||||
ref_only_errors = collections.Counter()
|
||||
incomparable = collections.Counter()
|
||||
|
||||
|
||||
def add(bucket, key, file, example):
|
||||
b = bucket[key]
|
||||
b["count"] += 1
|
||||
if file not in b["files"] and len(b["examples"]) < 6:
|
||||
b["examples"].append(example)
|
||||
b["files"].add(file)
|
||||
|
||||
|
||||
files = [ln.strip() for ln in open(os.path.join(R, "files.txt")) if ln.strip()]
|
||||
for rel in files:
|
||||
d = os.path.join(RUNS, rel.replace("/", "__"))
|
||||
corpus = rel.split("/")[0]
|
||||
ours, ref = load(d, "ours"), load(d, "ref")
|
||||
h5dump = load(d, "h5dump")
|
||||
os_, od = proc_status(ours)
|
||||
rs, rd = proc_status(ref)
|
||||
oj = ours["json"] if ours else None
|
||||
rj = ref["json"] if ref else None
|
||||
issues = [] # (kind, detail)
|
||||
caught_panics = []
|
||||
|
||||
def scan_err(path, what, msg):
|
||||
if msg.startswith("PANIC:"):
|
||||
caught_panics.append((path, what, msg))
|
||||
|
||||
if oj:
|
||||
for o in oj.get("objects", []):
|
||||
for k in ("error", "attrs_error", "list_error"):
|
||||
if k in o:
|
||||
scan_err(o["path"], k, o[k])
|
||||
for an, av in (o.get("attrs") or {}).items():
|
||||
if "error" in av:
|
||||
scan_err(o["path"], f"attr {an}", av["error"])
|
||||
if oj.get("open_error", "").startswith("PANIC:"):
|
||||
caught_panics.append(("<open>", "open", oj["open_error"]))
|
||||
|
||||
ref_open_fail = rs != "ok" or (rj is not None and "open_error" in rj)
|
||||
ours_open_err = oj.get("open_error") if oj else None
|
||||
n_obj = n_ok = 0
|
||||
if os_ == "ok" and rj and not ref_open_fail and not ours_open_err:
|
||||
ro = {x["path"]: x for x in rj.get("objects", [])}
|
||||
oo = {x["path"]: x for x in oj.get("objects", [])}
|
||||
our_list_errors = [x for x in oo.values() if "list_error" in x]
|
||||
for p in sorted(set(ro) | set(oo)):
|
||||
a, b = ro.get(p), oo.get(p)
|
||||
n_obj += 1
|
||||
if a is None:
|
||||
issues.append(("mismatch", f"extra object {p} (kind={b.get('kind')})", "extra-object", b))
|
||||
continue
|
||||
if b is None:
|
||||
if our_list_errors:
|
||||
continue # accounted for by the list_error
|
||||
issues.append(("mismatch", f"missing object {p} (kind={a.get('kind')})", "missing-object", a))
|
||||
continue
|
||||
ok = True
|
||||
if a.get("kind") != b.get("kind") and "error" not in b and "error" not in a:
|
||||
issues.append(("mismatch", f"{p}: kind {a.get('kind')} vs ours {b.get('kind')}", "kind", b))
|
||||
ok = False
|
||||
for k in ("error", "list_error", "attrs_error"):
|
||||
if k in b and k not in a:
|
||||
issues.append(("our-error", f"{p}: {k}: {b[k]}", b[k], b))
|
||||
ok = False
|
||||
elif k in a and k not in b and k == "error":
|
||||
ref_only_errors[norm(a[k])] += 1
|
||||
if a.get("kind") == "dataset" and "error" not in a and "error" not in b:
|
||||
if "skipped" in a or "skipped" in b:
|
||||
pass
|
||||
elif a.get("converted"):
|
||||
incomparable[f"dataset {a['converted']}"] += 1
|
||||
elif a.get("shape") != b.get("shape"):
|
||||
issues.append(("mismatch", f"{p}: shape {a.get('shape')} vs ours {b.get('shape')}", "shape", b))
|
||||
ok = False
|
||||
elif a.get("hash") != b.get("hash"):
|
||||
issues.append(("mismatch", f"{p}: values differ (h5py {a.get('dtype')} vs ours {b.get('dtype')})", "values", b | {"ref_head": a.get("head"), "ref_dtype": a.get("dtype")}))
|
||||
ok = False
|
||||
ra, oa = a.get("attrs") or {}, b.get("attrs") or {}
|
||||
if "attrs_error" not in b and "attrs_error" not in a:
|
||||
for an in sorted(set(ra) | set(oa)):
|
||||
x, y = ra.get(an), oa.get(an)
|
||||
if x is None:
|
||||
issues.append(("mismatch", f"{p}@{an}: extra attribute", "extra-attr", y or {}))
|
||||
elif y is None:
|
||||
issues.append(("mismatch", f"{p}@{an}: missing attribute", "missing-attr", x))
|
||||
elif "error" in y and "error" not in x:
|
||||
issues.append(("our-error", f"{p}@{an}: {y['error']}", y["error"], y))
|
||||
elif "error" in x:
|
||||
continue
|
||||
elif x.get("converted"):
|
||||
incomparable[f"attr {x['converted']}"] += 1
|
||||
elif x.get("shape") != y.get("shape"):
|
||||
issues.append(("mismatch", f"{p}@{an}: attr shape {x.get('shape')} vs ours {y.get('shape')}", "attr-shape", y | {"ref_dtype": x.get("dtype")}))
|
||||
elif x.get("hash") != y.get("hash"):
|
||||
issues.append(("mismatch", f"{p}@{an}: attr values differ (h5py {x.get('dtype')} vs ours {y.get('dtype')})", "attr-values", y | {"ref_head": x.get("head"), "ref_dtype": x.get("dtype")}))
|
||||
if ok:
|
||||
n_ok += 1
|
||||
|
||||
# classify
|
||||
if os_ in ("hang", "oom", "crash", "panic"):
|
||||
cls = os_
|
||||
elif caught_panics:
|
||||
cls = "panic"
|
||||
elif ref_open_fail:
|
||||
cls = "h5py-cannot-read"
|
||||
elif ours_open_err:
|
||||
cls = "our-error"
|
||||
issues.append(("our-error", f"open: {ours_open_err}", ours_open_err, {}))
|
||||
elif any(i[0] == "our-error" for i in issues):
|
||||
cls = "our-error"
|
||||
elif issues:
|
||||
cls = "mismatch"
|
||||
else:
|
||||
cls = "ok"
|
||||
|
||||
if os_ in ("hang", "oom", "crash", "panic") or caught_panics:
|
||||
panics.append({
|
||||
"file": rel, "class": cls, "detail": od,
|
||||
"stderr": (ours["err"] if ours else "")[:3000],
|
||||
"caught": [(p, w, m[:2500]) for p, w, m in caught_panics[:3]],
|
||||
"n_caught": len(caught_panics),
|
||||
})
|
||||
for kind, detail, key, rec in issues:
|
||||
if kind == "our-error":
|
||||
add(root_causes, norm(key), rel, detail[:300])
|
||||
else:
|
||||
if key in ("values", "attr-values", "shape", "attr-shape"):
|
||||
mk = f"{key}: ours={rec.get('dtype')} h5py={rec.get('ref_dtype')} layout={rec.get('layout','-')} filters={rec.get('filters','-')}"
|
||||
else:
|
||||
mk = key
|
||||
add(mismatch_causes, mk, rel, detail[:300] + (f" | ref_head={rec.get('ref_head')} our_head={rec.get('head')}" if rec.get("ref_head") else ""))
|
||||
ref_detail = rd if rs != "ok" else ((rj or {}).get("open_error") or "")
|
||||
h5d = ""
|
||||
if h5dump:
|
||||
rc = h5dump["rc"]
|
||||
h5d = {0: "ok", 1: "error", 137: "hang", 124: "hang", 134: "SIGABRT", 139: "SIGSEGV", 136: "SIGFPE", 135: "SIGBUS"}.get(rc, f"rc={rc}")
|
||||
if "memory allocation" in h5dump["err"] or "Cannot allocate" in h5dump["err"]:
|
||||
h5d += "(oom)"
|
||||
rows.append({
|
||||
"file": rel, "corpus": corpus, "class": cls,
|
||||
"ours": os_ if os_ != "ok" else ("open-error" if ours_open_err else ("panic" if caught_panics else "ok")),
|
||||
"ours_detail": (od or ours_open_err or (caught_panics[0][2].split("\n")[0] if caught_panics else ""))[:300],
|
||||
"ref": rs if rs != "ok" else ("open-error" if (rj or {}).get("open_error") else "ok"),
|
||||
"ref_detail": ref_detail[:300],
|
||||
"h5dump_1_14_6": h5d,
|
||||
"h5dump_detail": ([ln for ln in h5dump["err"].splitlines() if ln.strip()][-1:] or [""])[0][:200] if h5dump else "",
|
||||
"objects": n_obj, "objects_ok": n_ok,
|
||||
"issues": len(issues), "first_issue": issues[0][1][:300] if issues else "",
|
||||
"superblock": (oj or {}).get("superblock_version", ""),
|
||||
})
|
||||
# the first issues of each file, for report.py's known-cause matching
|
||||
issues_by_file[rel] = [
|
||||
{"kind": k, "key": key, "detail": det[:300], "ours_dtype": rec.get("dtype"), "ref_dtype": rec.get("ref_dtype")}
|
||||
for k, det, key, rec in issues[:50]
|
||||
]
|
||||
|
||||
with open(os.path.join(R, "results.csv"), "w", newline="") as fh:
|
||||
w = csv.DictWriter(fh, fieldnames=list(rows[0].keys()))
|
||||
w.writeheader()
|
||||
w.writerows(rows)
|
||||
|
||||
|
||||
def ser(b):
|
||||
return {k: {"files": len(v["files"]), "count": v["count"], "examples": v["examples"], "file_list": sorted(v["files"])} for k, v in sorted(b.items(), key=lambda kv: -len(kv[1]["files"]))}
|
||||
|
||||
|
||||
json.dump({"rows": rows, "issues": issues_by_file, "root_causes": ser(root_causes), "mismatch_causes": ser(mismatch_causes),
|
||||
"panics": panics, "incomparable": incomparable.most_common(), "ref_only_errors": ref_only_errors.most_common()},
|
||||
open(os.path.join(R, "results.json"), "w"), indent=1)
|
||||
|
||||
classes = ["ok", "our-error", "mismatch", "h5py-cannot-read", "hang", "panic", "crash", "oom"]
|
||||
by_corpus = collections.defaultdict(collections.Counter)
|
||||
for r in rows:
|
||||
by_corpus[r["corpus"]][r["class"]] += 1
|
||||
by_corpus["ALL"][r["class"]] += 1
|
||||
lines = ["# Conformance sweep summary", "", "| corpus | files | " + " | ".join(classes) + " |", "|---" * (len(classes) + 2) + "|"]
|
||||
for c in sorted(by_corpus, key=lambda k: (k == "ALL", k)):
|
||||
cnt = by_corpus[c]
|
||||
lines.append(f"| {c} | {sum(cnt.values())} | " + " | ".join(str(cnt.get(k, 0)) for k in classes) + " |")
|
||||
lines += ["", "## Panics / hangs / crashes / OOM", ""]
|
||||
for p in panics:
|
||||
lines.append(f"- **{p['file']}** [{p['class']}] {p['detail']}")
|
||||
for path, what, m in p["caught"][:1]:
|
||||
lines.append(" ```\n " + f"{path} ({what}): " + m.replace("\n", "\n ")[:1500] + "\n ```")
|
||||
if not p["caught"] and p["stderr"]:
|
||||
lines.append(" ```\n " + p["stderr"].strip()[:1500].replace("\n", "\n ") + "\n ```")
|
||||
lines += ["", "## Our-error root causes (files affected)", ""]
|
||||
for k, v in ser(root_causes).items():
|
||||
lines.append(f"- [{v['files']} files, {v['count']} objs] `{k}`")
|
||||
for ex in v["examples"][:3]:
|
||||
lines.append(f" - {ex}")
|
||||
lines += ["", "## Mismatch root causes", ""]
|
||||
for k, v in ser(mismatch_causes).items():
|
||||
lines.append(f"- [{v['files']} files, {v['count']} objs] `{k}`")
|
||||
for ex in v["examples"][:3]:
|
||||
lines.append(f" - {ex}")
|
||||
lines += ["", "## Objects h5py fails on but we read (top)", ""]
|
||||
for k, n in ref_only_errors.most_common(15):
|
||||
lines.append(f"- {n} x `{k}`")
|
||||
open(os.path.join(R, "summary.md"), "w").write("\n".join(lines) + "\n")
|
||||
print("\n".join(lines[:4 + len(by_corpus)]))
|
||||
@@ -0,0 +1,19 @@
|
||||
# Conformance corpora, pinned by commit. fetch-corpus.sh reads this file.
|
||||
#
|
||||
# name git-url commit root [sparse-checkout patterns...]
|
||||
#
|
||||
# `root` is the directory inside the checkout that is swept ("." = all of it).
|
||||
# Patterns are git non-cone sparse-checkout patterns; none = whole repository.
|
||||
# Every file under <root> with an HDF5/netCDF-4 extension is probed; for
|
||||
# cve_hdf5 the extension-less files in cvefiles/ and fuzzerfiles/ are too.
|
||||
# Licences: each corpus keeps its upstream licence; nothing here is committed
|
||||
# to this repository — the files are downloaded into the gitignored cache.
|
||||
hdf5 https://github.com/HDFGroup/hdf5.git a3cf1ea82cc7a66e50029a688121e1b105a7ce88 . *.h5 *.he5 *.nc *.hdf5 *.h5f
|
||||
cve_hdf5 https://github.com/HDFGroup/cve_hdf5.git 3fd1f5ae3869e01b8ae02b41d7108de7ffb1a374 .
|
||||
netcdf-c https://github.com/Unidata/netcdf-c.git beb7b9585273c1548386231a59b809d906359033 . /nc_test4/*.nc /ncdump/*.nc /nc_test4/*.h5 /ncdump/*.h5 /h5_test/*.h5 /hdf5_test/*.h5
|
||||
NCAS-CMS_pyfive https://github.com/NCAS-CMS/pyfive.git 8cf07b8749133f41c5e30b8a4c604486f687fe74 . *.h5 *.hdf5 *.hdf *.nc *.he5
|
||||
usnistgov_h5wasm https://github.com/usnistgov/h5wasm.git 02f6336527d2812783fcedabfbf42127ec8d06d2 . *.h5 *.hdf5 *.hdf *.nc *.he5
|
||||
netcdf4-python https://github.com/Unidata/netcdf4-python.git 6e67576d39aef8091fb20bd767b4f1a52ddc1bec . *.nc *.h5
|
||||
xarray-data https://github.com/pydata/xarray-data.git a35297e9da2cc99c811014f0c8a4297345a5c28d . /basin_mask.nc /precipitation.nc4 /imerghh_730.hdf5 /eraint_uvz.nc /ROMS_example.nc /tiny.nc
|
||||
# h5py 3.16.0 (tag 3.16.0), its test data files.
|
||||
h5py_data https://github.com/h5py/h5py.git b2f0347c4200333acd89b43733f1caa0c115162f h5py/tests/data_files /h5py/tests/data_files/*
|
||||
Executable
+39
@@ -0,0 +1,39 @@
|
||||
#!/usr/bin/env bash
|
||||
# fetch-corpus.sh [cache_dir]
|
||||
#
|
||||
# Download the corpora pinned in conformance/corpus.txt into the (gitignored)
|
||||
# cache: <cache>/src/<name> is a shallow, sparse, blob-filtered checkout of the
|
||||
# pinned commit and <cache>/corpus/<name> links to the swept root inside it.
|
||||
# A corpus already checked out at its pinned commit is left alone, so a second
|
||||
# run costs nothing and needs no network.
|
||||
set -euo pipefail
|
||||
HERE="$(cd "$(dirname "$0")" && pwd)"
|
||||
CACHE="${1:-${CONFORMANCE_CACHE:-$HERE/.cache}}"
|
||||
mkdir -p "$CACHE/src" "$CACHE/corpus"
|
||||
CACHE="$(cd "$CACHE" && pwd)"
|
||||
|
||||
retry() { local i; for i in 1 2 3 4; do "$@" && return 0; sleep $((i * 5)); done; return 1; }
|
||||
|
||||
grep -v '^[[:space:]]*\(#\|$\)' "$HERE/corpus.txt" | while read -r name url commit root patterns; do
|
||||
src="$CACHE/src/$name"
|
||||
if [ -d "$src/.git" ] && [ "$(git -C "$src" rev-parse HEAD 2>/dev/null)" = "$commit" ]; then
|
||||
echo "cached $name @ ${commit:0:12}"
|
||||
else
|
||||
echo "fetching $name @ ${commit:0:12} from $url"
|
||||
rm -rf "$src"
|
||||
git init -q "$src"
|
||||
git -C "$src" remote add origin "$url"
|
||||
git -C "$src" config advice.detachedHead false
|
||||
if [ -n "$patterns" ]; then
|
||||
git -C "$src" config core.sparseCheckout true
|
||||
# no-cone patterns (globs); `set -f` keeps the shell from expanding them
|
||||
(set -f; printf '%s\n' $patterns) > "$src/.git/info/sparse-checkout"
|
||||
fi
|
||||
retry git -C "$src" fetch -q --depth 1 --filter=blob:none origin "$commit"
|
||||
retry git -C "$src" checkout -q FETCH_HEAD
|
||||
got="$(git -C "$src" rev-parse HEAD)"
|
||||
[ "$got" = "$commit" ] || { echo "error: $name checked out $got, expected $commit" >&2; exit 1; }
|
||||
fi
|
||||
ln -sfn "$src/$root" "$CACHE/corpus/$name"
|
||||
done
|
||||
echo "corpus ready in $CACHE/corpus"
|
||||
@@ -0,0 +1,48 @@
|
||||
#!/usr/bin/env python3
|
||||
"""list_files.py <corpus_dir>: print the files the sweep probes, one per line,
|
||||
as <corpus>/<path> in byte order.
|
||||
|
||||
* every file named *.h5 *.hdf5 *.he5 *.nc *.nc4 *.hdf *.h5f in each corpus,
|
||||
except netCDF classic / 64-bit-offset / CDF5 files (magic "CDF"): they are
|
||||
not HDF5, so neither side can read them and they say nothing;
|
||||
* plus, for cve_hdf5, every file in cvefiles/ and fuzzerfiles/ except
|
||||
.md/.c sources — the reproducers are mostly extension-less, and they are
|
||||
kept whatever their bytes look like (that is their point).
|
||||
"""
|
||||
import os
|
||||
import sys
|
||||
|
||||
EXTS = (".h5", ".hdf5", ".he5", ".nc", ".nc4", ".hdf", ".h5f")
|
||||
|
||||
|
||||
def walk(top):
|
||||
for dirpath, dirnames, filenames in os.walk(top):
|
||||
dirnames[:] = [d for d in dirnames if d != ".git"]
|
||||
for fn in filenames:
|
||||
p = os.path.join(dirpath, fn)
|
||||
if os.path.isfile(p) and not os.path.islink(p):
|
||||
yield os.path.relpath(p, top)
|
||||
|
||||
|
||||
def main(root):
|
||||
out = set()
|
||||
for corpus in sorted(os.listdir(root)):
|
||||
top = os.path.join(root, corpus)
|
||||
if not os.path.isdir(top):
|
||||
continue
|
||||
for rel in walk(top):
|
||||
path = os.path.join(top, rel)
|
||||
if rel.lower().endswith(EXTS):
|
||||
with open(path, "rb") as fh:
|
||||
if fh.read(3) == b"CDF":
|
||||
continue
|
||||
out.add(f"{corpus}/{rel}")
|
||||
elif corpus == "cve_hdf5" and rel.split(os.sep)[0] in ("cvefiles", "fuzzerfiles") \
|
||||
and not rel.endswith((".md", ".c")):
|
||||
out.add(f"{corpus}/{rel}")
|
||||
for f in sorted(out, key=lambda s: s.encode()):
|
||||
print(f)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main(sys.argv[1])
|
||||
Generated
+492
@@ -0,0 +1,492 @@
|
||||
# This file is automatically @generated by Cargo.
|
||||
# It is not intended for manual editing.
|
||||
version = 4
|
||||
|
||||
[[package]]
|
||||
name = "adler2"
|
||||
version = "2.0.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "320119579fcad9c21884f5c4861d16174d0e06250625266f50fe6898340abefa"
|
||||
|
||||
[[package]]
|
||||
name = "better_io"
|
||||
version = "0.2.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ef0a3155e943e341e557863e69a708999c94ede624e37865c8e2a91b94efa78f"
|
||||
|
||||
[[package]]
|
||||
name = "block-buffer"
|
||||
version = "0.10.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "3078c7629b62d3f0439517fa394996acacc5cbc91c5a20d8c658e77abd503a71"
|
||||
dependencies = [
|
||||
"generic-array",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "byteorder"
|
||||
version = "1.5.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "1fd0f2584146f6f2ef48085050886acf353beff7305ebd1ae69500e27c67f64b"
|
||||
|
||||
[[package]]
|
||||
name = "bzip2"
|
||||
version = "0.6.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f3a53fac24f34a81bc9954b5d6cfce0c21e18ec6959f44f56e8e90e4bb7c346c"
|
||||
dependencies = [
|
||||
"libbz2-rs-sys",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "cc"
|
||||
version = "1.5.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f360145194ee8e21db5ee7f3fcd4fe52210864c75c985dae33218202c8bbe040"
|
||||
dependencies = [
|
||||
"find-msvc-tools",
|
||||
"jobserver",
|
||||
"libc",
|
||||
"shlex",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "cfg-if"
|
||||
version = "1.0.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "4e7648175b45a9a48536d676f68d918270699102aa8dab5496df06904c914600"
|
||||
|
||||
[[package]]
|
||||
name = "clawhdf5-format"
|
||||
version = "2.7.0"
|
||||
dependencies = [
|
||||
"byteorder",
|
||||
"bzip2",
|
||||
"flate2",
|
||||
"libaec-sys",
|
||||
"libc",
|
||||
"lz4_flex",
|
||||
"pco",
|
||||
"portable-atomic",
|
||||
"ruzstd",
|
||||
"sha2",
|
||||
"snap",
|
||||
"zstd",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "conformance-probe"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"clawhdf5-format",
|
||||
"serde_json",
|
||||
"sha2",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "cpufeatures"
|
||||
version = "0.2.17"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "59ed5838eebb26a2bb2e58f6d5b5316989ae9d08bab10e0e6d103e656d1b0280"
|
||||
dependencies = [
|
||||
"libc",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "crc32fast"
|
||||
version = "1.5.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "01a7799fd6b852db0e61728dde9a204c423b44d689dbd432522543614b490e78"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "crunchy"
|
||||
version = "0.2.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "460fbee9c2c2f33933d720630a6a0bac33ba7053db5344fac858d4b8952d77d5"
|
||||
|
||||
[[package]]
|
||||
name = "crypto-common"
|
||||
version = "0.1.7"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "78c8292055d1c1df0cce5d180393dc8cce0abec0a7102adb6c7b1eef6016d60a"
|
||||
dependencies = [
|
||||
"generic-array",
|
||||
"typenum",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "digest"
|
||||
version = "0.10.7"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292"
|
||||
dependencies = [
|
||||
"block-buffer",
|
||||
"crypto-common",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "dtype_dispatch"
|
||||
version = "0.2.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ab23e69df104e2fd85ee63a533a22d2132ef5975dc6b36f9f3e5a7305e4a8ed7"
|
||||
|
||||
[[package]]
|
||||
name = "find-msvc-tools"
|
||||
version = "0.1.14"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "aedcfb3409746eddb02b9e19ebda1c3394f759a152e48ee875a0844d1b955484"
|
||||
|
||||
[[package]]
|
||||
name = "flate2"
|
||||
version = "1.1.10"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "6e634e2e0ebac1ee034020da1ca582e17ffe4e0f5e985823721e168928136dcb"
|
||||
dependencies = [
|
||||
"crc32fast",
|
||||
"miniz_oxide",
|
||||
"zlib-rs",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "generic-array"
|
||||
version = "0.14.7"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "85649ca51fd72272d7821adaf274ad91c288277713d9c18820d8499a7ff69e9a"
|
||||
dependencies = [
|
||||
"typenum",
|
||||
"version_check",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "getrandom"
|
||||
version = "0.4.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "300e883d756b2e4ec94e02791f39b04b522276138852cfc41d9fb7e904106099"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"libc",
|
||||
"r-efi",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "half"
|
||||
version = "2.7.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "6ea2d84b969582b4b1864a92dc5d27cd2b77b622a8d79306834f1be5ba20d84b"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"crunchy",
|
||||
"zerocopy",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "itoa"
|
||||
version = "1.0.18"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682"
|
||||
|
||||
[[package]]
|
||||
name = "jobserver"
|
||||
version = "0.1.35"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "1c00acbd29eabad4a2392fa0e921c874934dbbf4194312ad20f04a0ed67a3cb3"
|
||||
dependencies = [
|
||||
"getrandom",
|
||||
"libc",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "libaec-sys"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"pkg-config",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "libbz2-rs-sys"
|
||||
version = "0.2.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "34b357333733e8260735ba5894eb928c02ecc69c78715f01a8019e7fa7f2db4c"
|
||||
|
||||
[[package]]
|
||||
name = "libc"
|
||||
version = "0.2.189"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2"
|
||||
|
||||
[[package]]
|
||||
name = "lz4_flex"
|
||||
version = "0.11.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "373f5eceeeab7925e0c1098212f2fbc4d416adec9d35051a6ab251e824c1854a"
|
||||
dependencies = [
|
||||
"twox-hash",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "memchr"
|
||||
version = "2.8.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98"
|
||||
|
||||
[[package]]
|
||||
name = "miniz_oxide"
|
||||
version = "0.9.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b63fbc4a50860e98e7b2aa7804ded1db5cbc3aff9193adaff57a6931bf7c4b4c"
|
||||
dependencies = [
|
||||
"adler2",
|
||||
"simd-adler32",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "pco"
|
||||
version = "1.0.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "386342cad4c6e97f081568e5d910ea7d871314c843aa8fc564f2a6b64cab9456"
|
||||
dependencies = [
|
||||
"better_io",
|
||||
"dtype_dispatch",
|
||||
"half",
|
||||
"rand_xoshiro",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "pkg-config"
|
||||
version = "0.3.34"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f6b464fbc74e149a392436b17d523f769e057cb6877f6a5c4618bc6f11800548"
|
||||
|
||||
[[package]]
|
||||
name = "portable-atomic"
|
||||
version = "1.15.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "05c8b63e8d9609db387f0324918f81d68fe27748f084ef092fb35954d0539a85"
|
||||
|
||||
[[package]]
|
||||
name = "proc-macro2"
|
||||
version = "1.0.107"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9"
|
||||
dependencies = [
|
||||
"unicode-ident",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "quote"
|
||||
version = "1.0.47"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "r-efi"
|
||||
version = "6.0.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f8dcc9c7d52a811697d2151c701e0d08956f92b0e24136cf4cf27b57a6a0d9bf"
|
||||
|
||||
[[package]]
|
||||
name = "rand_core"
|
||||
version = "0.6.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ec0be4795e2f6a28069bec0b5ff3e2ac9bafc99e6a9a7dc3547996c5c816922c"
|
||||
|
||||
[[package]]
|
||||
name = "rand_xoshiro"
|
||||
version = "0.6.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "6f97cdb2a36ed4183de61b2f824cc45c9f1037f28afe0a322e9fff4c108b5aaa"
|
||||
dependencies = [
|
||||
"rand_core",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "ruzstd"
|
||||
version = "0.9.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "a252f5e20f038fe7b4ea53e073e65398d652c864cc162fc77c56c2f13717b888"
|
||||
dependencies = [
|
||||
"twox-hash",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "serde"
|
||||
version = "1.0.229"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba"
|
||||
dependencies = [
|
||||
"serde_core",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "serde_core"
|
||||
version = "1.0.229"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48"
|
||||
dependencies = [
|
||||
"serde_derive",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "serde_derive"
|
||||
version = "1.0.229"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"syn 3.0.6",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "serde_json"
|
||||
version = "1.0.151"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14"
|
||||
dependencies = [
|
||||
"itoa",
|
||||
"memchr",
|
||||
"serde",
|
||||
"serde_core",
|
||||
"zmij",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "sha2"
|
||||
version = "0.10.9"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "a7507d819769d01a365ab707794a4084392c824f54a7a6a7862f8c3d0892b283"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"cpufeatures",
|
||||
"digest",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "shlex"
|
||||
version = "2.0.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba"
|
||||
|
||||
[[package]]
|
||||
name = "simd-adler32"
|
||||
version = "0.3.10"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "3a219298ac11a56ea9a6d2120044824d6f01aeb034955e7af7bc16858527deea"
|
||||
|
||||
[[package]]
|
||||
name = "snap"
|
||||
version = "1.1.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "199905e6153d6405f9728fe44daace35f8f837bbf830bb6e85fbd5828709a886"
|
||||
|
||||
[[package]]
|
||||
name = "syn"
|
||||
version = "2.0.119"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "872831b642d1a07999a962a351ed35b955ea2cfc8f3862091e2a240a84f17297"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"unicode-ident",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "syn"
|
||||
version = "3.0.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "8593e8e72159ed2257d083c7a454a85cbf854f37a0966d8d483aff8c8a3ebcee"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"unicode-ident",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "twox-hash"
|
||||
version = "2.1.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "5283634e518fe9e82c7b20520bb4bc209009fd16c82077c802f8111ecbb0117a"
|
||||
|
||||
[[package]]
|
||||
name = "typenum"
|
||||
version = "1.20.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b6f5e870be6c3b371b77fe0ee0bafb859fa4964b4404c27de1d380043c4dda20"
|
||||
|
||||
[[package]]
|
||||
name = "unicode-ident"
|
||||
version = "1.0.26"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d245f478577f809a851594d02313b640fb437e0bb33866753cff937863096954"
|
||||
|
||||
[[package]]
|
||||
name = "version_check"
|
||||
version = "0.9.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a"
|
||||
|
||||
[[package]]
|
||||
name = "zerocopy"
|
||||
version = "0.8.59"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "6df92bf3d9227be3d53173901ddbffac2babc27ae50f397776ffd6dc33f800cb"
|
||||
dependencies = [
|
||||
"zerocopy-derive",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "zerocopy-derive"
|
||||
version = "0.8.59"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ac4f328cf2f05d084e496c3e9c3f33ed0a183656a16e1fcec4d464d8373aec82"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"syn 2.0.119",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "zlib-rs"
|
||||
version = "0.6.8"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b268e58e7c693d7c271f93ffc4ba3b380412554231c85bf61ca7af91042a4112"
|
||||
|
||||
[[package]]
|
||||
name = "zmij"
|
||||
version = "1.0.23"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b"
|
||||
|
||||
[[package]]
|
||||
name = "zstd"
|
||||
version = "0.13.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e91ee311a569c327171651566e07972200e76fcfe2242a4fa446149a3881c08a"
|
||||
dependencies = [
|
||||
"zstd-safe",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "zstd-safe"
|
||||
version = "7.3.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "64d80649ab6db9d9f6f9c80a40becd948eda4714a0a5ac8c4d157a32231c7882"
|
||||
dependencies = [
|
||||
"zstd-sys",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "zstd-sys"
|
||||
version = "2.1.0+zstd.1.5.7"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0ef0a8027ec3ee71300ab3bcbcd0393f434aa72b91ca6d635a39941deae8eea0"
|
||||
dependencies = [
|
||||
"cc",
|
||||
"pkg-config",
|
||||
]
|
||||
@@ -0,0 +1,25 @@
|
||||
[package]
|
||||
name = "conformance-probe"
|
||||
version = "0.1.0"
|
||||
edition = "2024"
|
||||
rust-version = "1.92"
|
||||
publish = false
|
||||
description = "Walks an HDF5 file with clawhdf5-format and prints a canonical JSON description (see conformance/README.md)"
|
||||
|
||||
# Deliberately outside the main workspace: `cargo test --workspace` never
|
||||
# builds it, and it links the optional C codecs (zstd, libaec) that the core
|
||||
# crates' default build must not.
|
||||
[workspace]
|
||||
|
||||
[dependencies]
|
||||
clawhdf5-format = { path = "../../crates/clawhdf5-format", features = ["lz4", "zstd", "szip", "pcodec", "plugin-filters"] }
|
||||
serde_json = "1"
|
||||
sha2 = "0.10"
|
||||
|
||||
[profile.release]
|
||||
# Keep panics catchable (the probe records them per object) and turn integer
|
||||
# overflow into a reported panic instead of silent wraparound.
|
||||
debug = 1
|
||||
overflow-checks = true
|
||||
debug-assertions = true
|
||||
panic = "unwind"
|
||||
@@ -0,0 +1,885 @@
|
||||
//! Conformance probe: walks an HDF5 file with clawhdf5-format (the same calls
|
||||
//! the `clawhdf5` facade makes) and prints a canonical JSON description:
|
||||
//! every hard-linked object (sorted-name DFS, deduplicated by header address),
|
||||
//! and for each dataset / attribute its shape plus the SHA-256 of its values
|
||||
//! in a canonical encoding shared with `ref.py`.
|
||||
//!
|
||||
//! Canonical value encoding (per element, concatenated, row-major):
|
||||
//! int / float / bitfield / enum / time : element bytes, little-endian
|
||||
//! non-IEEE-layout float (e.g. N-Bit) : the IEEE float of the same size it converts to
|
||||
//! int with bit offset / short precision: the full-width integer it converts to
|
||||
//! opaque : raw bytes
|
||||
//! compound : members in declaration order (padding dropped)
|
||||
//! array : base elements row-major
|
||||
//! string (fixed or VL) : b'S' + u32le len + bytes (cut at first NUL, trailing spaces stripped)
|
||||
//! VL sequence : b'V' + u32le count + base elements
|
||||
//! reference : b'R' (payload not compared)
|
||||
//!
|
||||
//! Every object is processed inside catch_unwind; a caught panic is recorded
|
||||
//! with its message, location and the clawhdf5 frames of its backtrace.
|
||||
|
||||
use std::cell::RefCell;
|
||||
use std::collections::HashSet;
|
||||
use std::panic::{self, AssertUnwindSafe};
|
||||
|
||||
use clawhdf5_format::attribute::extract_attributes_full;
|
||||
use clawhdf5_format::data_layout::DataLayout;
|
||||
use clawhdf5_format::data_read;
|
||||
use clawhdf5_format::dataspace::{Dataspace, DataspaceType};
|
||||
use clawhdf5_format::datatype::{Datatype, DatatypeByteOrder};
|
||||
use clawhdf5_format::filter_pipeline::FilterPipeline;
|
||||
use clawhdf5_format::group_v1::{self, GroupEntry};
|
||||
use clawhdf5_format::group_v2;
|
||||
use clawhdf5_format::message_type::MessageType;
|
||||
use clawhdf5_format::object_header::ObjectHeader;
|
||||
use clawhdf5_format::signature;
|
||||
use clawhdf5_format::superblock::Superblock;
|
||||
use clawhdf5_format::symbol_table::SymbolTableMessage;
|
||||
use clawhdf5_format::vl_data::{VlResolver, check_element_size};
|
||||
use serde_json::{Map, Value, json};
|
||||
use sha2::{Digest, Sha256};
|
||||
|
||||
const MAX_BYTES: u64 = 200 * 1024 * 1024;
|
||||
const MAX_OBJECTS: usize = 200_000;
|
||||
|
||||
thread_local! {
|
||||
static LAST_PANIC: RefCell<Option<String>> = const { RefCell::new(None) };
|
||||
}
|
||||
|
||||
fn install_hook() {
|
||||
panic::set_hook(Box::new(|info| {
|
||||
let msg = if let Some(s) = info.payload().downcast_ref::<&str>() {
|
||||
s.to_string()
|
||||
} else if let Some(s) = info.payload().downcast_ref::<String>() {
|
||||
s.clone()
|
||||
} else {
|
||||
"<non-string panic>".into()
|
||||
};
|
||||
let loc = info
|
||||
.location()
|
||||
.map(|l| format!("{}:{}", l.file(), l.line()))
|
||||
.unwrap_or_default();
|
||||
let bt = std::backtrace::Backtrace::force_capture().to_string();
|
||||
// keep only frames from clawhdf5 code
|
||||
let mut frames = Vec::new();
|
||||
let lines: Vec<&str> = bt.lines().collect();
|
||||
for (i, l) in lines.iter().enumerate() {
|
||||
let t = l.trim();
|
||||
if t.contains("clawhdf5_format::") || t.contains("conformance_probe::") {
|
||||
let at = lines
|
||||
.get(i + 1)
|
||||
.map(|n| n.trim())
|
||||
.filter(|n| n.starts_with("at "))
|
||||
.map(|n| {
|
||||
let n = n.trim_start_matches("at ");
|
||||
match n.find("/crates/") {
|
||||
Some(p) => n[p + 1..].to_string(),
|
||||
None => n.to_string(),
|
||||
}
|
||||
})
|
||||
.unwrap_or_default();
|
||||
let name = t.split_once(": ").map(|x| x.1).unwrap_or(t);
|
||||
frames.push(format!("{name} ({at})"));
|
||||
if frames.len() >= 12 {
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
let full = format!("PANIC: {msg} @ {loc}\n {}", frames.join("\n "));
|
||||
eprintln!("{full}");
|
||||
LAST_PANIC.with(|p| *p.borrow_mut() = Some(full));
|
||||
}));
|
||||
}
|
||||
|
||||
/// Run `f`, turning a panic into Err("PANIC: ...").
|
||||
fn guarded<T>(f: impl FnOnce() -> Result<T, String>) -> Result<T, String> {
|
||||
match panic::catch_unwind(AssertUnwindSafe(f)) {
|
||||
Ok(r) => r,
|
||||
Err(_) => Err(LAST_PANIC
|
||||
.with(|p| p.borrow_mut().take())
|
||||
.unwrap_or_else(|| "PANIC: <unknown>".into())),
|
||||
}
|
||||
}
|
||||
|
||||
fn e<E: std::fmt::Debug>(x: E) -> String {
|
||||
format!("{x:?}")
|
||||
}
|
||||
|
||||
struct Ctx<'a> {
|
||||
data: &'a [u8],
|
||||
os: u8,
|
||||
ls: u8,
|
||||
base_dir: std::path::PathBuf,
|
||||
/// Resolves variable-length elements as the library does (null
|
||||
/// elements, strings cut at a NUL, heap objects of the wrong size
|
||||
/// refused), caching each heap collection.
|
||||
vl: RefCell<VlResolver<'a>>,
|
||||
}
|
||||
|
||||
impl<'a> Ctx<'a> {
|
||||
fn header(&self, addr: u64) -> Result<ObjectHeader, String> {
|
||||
ObjectHeader::parse(self.data, addr as usize, self.os, self.ls).map_err(e)
|
||||
}
|
||||
|
||||
fn payload(&self, h: &ObjectHeader, t: MessageType) -> Result<Option<Vec<u8>>, String> {
|
||||
match h.messages.iter().find(|m| m.msg_type == t) {
|
||||
None => Ok(None),
|
||||
Some(m) => {
|
||||
clawhdf5_format::shared_message::message_data(self.data, m, self.os, self.ls)
|
||||
.map(|c| Some(c.into_owned()))
|
||||
.map_err(e)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn canon(&self, dt: &Datatype, b: &[u8], out: &mut Vec<u8>) -> Result<(), String> {
|
||||
let size = dt.type_size() as usize;
|
||||
if b.len() < size {
|
||||
return Err(format!(
|
||||
"canon: element slice {} < type size {size}",
|
||||
b.len()
|
||||
));
|
||||
}
|
||||
match dt {
|
||||
Datatype::FloatingPoint { .. } if !ieee_layout(dt) => {
|
||||
canon_custom_float(dt, &b[..size], out)?
|
||||
}
|
||||
Datatype::FixedPoint { .. } if partial_int(dt) => {
|
||||
canon_partial_int(dt, &b[..size], out)?
|
||||
}
|
||||
Datatype::FixedPoint { byte_order, .. }
|
||||
| Datatype::BitField { byte_order, .. }
|
||||
| Datatype::FloatingPoint { byte_order, .. } => match byte_order {
|
||||
DatatypeByteOrder::LittleEndian => out.extend_from_slice(&b[..size]),
|
||||
DatatypeByteOrder::BigEndian => out.extend(b[..size].iter().rev()),
|
||||
DatatypeByteOrder::Vax => return Err("canon: VAX byte order".into()),
|
||||
},
|
||||
Datatype::Time { .. } | Datatype::Opaque { .. } => out.extend_from_slice(&b[..size]),
|
||||
Datatype::String { .. } => canon_str(&b[..size], out),
|
||||
Datatype::Compound { members, .. } => {
|
||||
for m in members {
|
||||
let off = m.byte_offset as usize;
|
||||
let ms = m.datatype.type_size() as usize;
|
||||
if off.checked_add(ms).is_none_or(|end| end > size) {
|
||||
return Err(format!("canon: member {} out of bounds", m.name));
|
||||
}
|
||||
self.canon(&m.datatype, &b[off..off + ms], out)?;
|
||||
}
|
||||
}
|
||||
Datatype::Reference { .. } => out.push(b'R'),
|
||||
Datatype::Enumeration { base_type, .. } => self.canon(base_type, b, out)?,
|
||||
Datatype::Array {
|
||||
base_type,
|
||||
dimensions,
|
||||
} => {
|
||||
let n: usize = dimensions.iter().map(|d| *d as usize).product();
|
||||
let bs = base_type.type_size() as usize;
|
||||
for i in 0..n {
|
||||
self.canon(base_type, &b[i * bs..], out)?;
|
||||
}
|
||||
}
|
||||
Datatype::VariableLength {
|
||||
size: vl_size,
|
||||
is_string,
|
||||
base_type,
|
||||
..
|
||||
} => {
|
||||
check_element_size(*vl_size, self.os).map_err(e)?;
|
||||
let el = &b[..size];
|
||||
if *is_string {
|
||||
let s = self.vl.borrow_mut().string_bytes(el).map_err(e)?;
|
||||
canon_str(&s[0], out);
|
||||
} else {
|
||||
let bs = base_type.type_size() as usize;
|
||||
// The borrow ends here: the base type may itself be
|
||||
// variable-length.
|
||||
let seq = self.vl.borrow_mut().sequences(el, bs).map_err(e)?;
|
||||
let seq = &seq[0];
|
||||
let len = seq.len() / bs;
|
||||
out.push(b'V');
|
||||
out.extend_from_slice(&(len as u32).to_le_bytes());
|
||||
for i in 0..len {
|
||||
self.canon(base_type, &seq[i * bs..], out)?;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Returns (shape json, n_elements)
|
||||
fn shape(ds: &Dataspace) -> (Value, u64) {
|
||||
match ds.space_type {
|
||||
DataspaceType::Null => (Value::String("null".into()), 0),
|
||||
DataspaceType::Scalar => (json!([]), 1),
|
||||
DataspaceType::Simple => {
|
||||
let n = ds.dimensions.iter().fold(1u64, |a, d| a.saturating_mul(*d));
|
||||
(json!(ds.dimensions), n)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn hash_values(
|
||||
&self,
|
||||
dt: &Datatype,
|
||||
raw: &[u8],
|
||||
n: u64,
|
||||
rec: &mut Map<String, Value>,
|
||||
) -> Result<(), String> {
|
||||
let size = dt.type_size() as usize;
|
||||
let need = (n as usize).checked_mul(size).ok_or("n*size overflow")?;
|
||||
if raw.len() != need {
|
||||
return Err(format!(
|
||||
"raw length {} != n_elements {n} * type_size {size}",
|
||||
raw.len()
|
||||
));
|
||||
}
|
||||
let mut canon = Vec::with_capacity(need);
|
||||
for i in 0..n as usize {
|
||||
self.canon(dt, &raw[i * size..(i + 1) * size], &mut canon)?;
|
||||
}
|
||||
let h = Sha256::digest(&canon);
|
||||
rec.insert("hash".into(), Value::String(hex(&h)));
|
||||
rec.insert(
|
||||
"head".into(),
|
||||
Value::String(hex(&canon[..canon.len().min(48)])),
|
||||
);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// VDS source files resolve next to the virtual file; like the library,
|
||||
/// refuse absolute paths and `..`.
|
||||
fn vds_resolver(
|
||||
&self,
|
||||
) -> impl Fn(&str) -> Result<Option<Vec<u8>>, clawhdf5_format::error::FormatError> + use<> {
|
||||
let base = self.base_dir.clone();
|
||||
move |name: &str| {
|
||||
use clawhdf5_format::error::FormatError;
|
||||
let p = std::path::Path::new(name);
|
||||
if p.is_absolute()
|
||||
|| p.components()
|
||||
.any(|c| matches!(c, std::path::Component::ParentDir))
|
||||
{
|
||||
return Err(FormatError::ChunkedReadError(format!("refused {name}")));
|
||||
}
|
||||
match std::fs::read(base.join(p)) {
|
||||
Ok(b) => Ok(Some(b)),
|
||||
Err(err) if err.kind() == std::io::ErrorKind::NotFound => Ok(None),
|
||||
Err(err) => Err(FormatError::ChunkedReadError(err.to_string())),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn read_named_datatype(&self, h: &ObjectHeader) -> Result<(), String> {
|
||||
let dtb = self
|
||||
.payload(h, MessageType::Datatype)?
|
||||
.ok_or("MissingMessage(Datatype)")?;
|
||||
Datatype::parse_in_header(&dtb, h.version).map_err(e)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn read_dataset(&self, h: &ObjectHeader, rec: &mut Map<String, Value>) -> Result<(), String> {
|
||||
let dtb = self
|
||||
.payload(h, MessageType::Datatype)?
|
||||
.ok_or("MissingMessage(Datatype)")?;
|
||||
let (dt, _) = Datatype::parse_in_header(&dtb, h.version).map_err(e)?;
|
||||
rec.insert("dtype".into(), Value::String(dtype_str(&dt)));
|
||||
let dsb = self
|
||||
.payload(h, MessageType::Dataspace)?
|
||||
.ok_or("MissingMessage(Dataspace)")?;
|
||||
let mut ds = Dataspace::parse(&dsb, self.ls).map_err(e)?;
|
||||
// A virtual dataset's extent can come from its sources (unlimited /
|
||||
// printf mappings), as h5py reports it, rather than the stored one.
|
||||
if let Some(lm) = h
|
||||
.messages
|
||||
.iter()
|
||||
.find(|m| m.msg_type == MessageType::DataLayout)
|
||||
&& let Ok(dl @ DataLayout::Virtual { .. }) =
|
||||
DataLayout::parse(&lm.data, self.os, self.ls)
|
||||
{
|
||||
let resolver = self.vds_resolver();
|
||||
ds.dimensions = clawhdf5_format::vds::virtual_dataset_extent(
|
||||
self.data,
|
||||
&dl,
|
||||
&ds,
|
||||
self.os,
|
||||
self.ls,
|
||||
Some(&resolver),
|
||||
)
|
||||
.map_err(e)?;
|
||||
}
|
||||
let (shape, n) = Self::shape(&ds);
|
||||
rec.insert("shape".into(), shape);
|
||||
if n.saturating_mul(dt.type_size() as u64) > MAX_BYTES {
|
||||
rec.insert("skipped".into(), Value::String("too large".into()));
|
||||
return Ok(());
|
||||
}
|
||||
let lm = h
|
||||
.messages
|
||||
.iter()
|
||||
.find(|m| m.msg_type == MessageType::DataLayout)
|
||||
.ok_or("MissingMessage(DataLayout)")?;
|
||||
let dl = DataLayout::parse(&lm.data, self.os, self.ls).map_err(e)?;
|
||||
rec.insert(
|
||||
"layout".into(),
|
||||
Value::String(
|
||||
match &dl {
|
||||
DataLayout::Compact { .. } => "compact",
|
||||
DataLayout::Contiguous { .. } => "contiguous",
|
||||
DataLayout::Chunked { .. } => "chunked",
|
||||
DataLayout::Virtual { .. } => "virtual",
|
||||
}
|
||||
.into(),
|
||||
),
|
||||
);
|
||||
let pipeline = match self.payload(h, MessageType::FilterPipeline)? {
|
||||
Some(p) => Some(FilterPipeline::parse(&p).map_err(e)?),
|
||||
None => None,
|
||||
};
|
||||
if let Some(p) = &pipeline {
|
||||
rec.insert(
|
||||
"filters".into(),
|
||||
json!(p.filters.iter().map(|f| f.filter_id).collect::<Vec<_>>()),
|
||||
);
|
||||
}
|
||||
let raw = if matches!(dl, DataLayout::Virtual { .. }) {
|
||||
let resolver = self.vds_resolver();
|
||||
let fill = clawhdf5_format::fill_value::dataset_fill_value_in(
|
||||
self.data,
|
||||
&h.messages,
|
||||
self.os,
|
||||
self.ls,
|
||||
)
|
||||
.map_err(e)?;
|
||||
clawhdf5_format::vds::read_virtual_dataset(
|
||||
self.data,
|
||||
&dl,
|
||||
&ds,
|
||||
&dt,
|
||||
fill.as_deref(),
|
||||
self.os,
|
||||
self.ls,
|
||||
Some(&resolver),
|
||||
)
|
||||
.map_err(e)?
|
||||
.data
|
||||
} else {
|
||||
let cache = clawhdf5_format::chunk_cache::ChunkCache::new();
|
||||
clawhdf5_format::fill_value::read_full_with_fill::<clawhdf5_format::error::FormatError>(
|
||||
&h.messages,
|
||||
self.data,
|
||||
&dl,
|
||||
&ds,
|
||||
dt.type_size() as usize,
|
||||
self.os,
|
||||
self.ls,
|
||||
|| {
|
||||
data_read::read_raw_data_cached(
|
||||
self.data,
|
||||
&dl,
|
||||
&ds,
|
||||
&dt,
|
||||
pipeline.as_ref(),
|
||||
self.os,
|
||||
self.ls,
|
||||
&cache,
|
||||
)
|
||||
},
|
||||
)
|
||||
.map_err(e)?
|
||||
};
|
||||
self.hash_values(&dt, &raw, n, rec)
|
||||
}
|
||||
|
||||
fn attrs(&self, h: &ObjectHeader) -> Result<Map<String, Value>, String> {
|
||||
let msgs = extract_attributes_full(self.data, h, self.os, self.ls).map_err(e)?;
|
||||
let mut out = Map::new();
|
||||
for a in &msgs {
|
||||
let r = guarded(|| {
|
||||
let mut rec = Map::new();
|
||||
rec.insert("dtype".into(), Value::String(dtype_str(&a.datatype)));
|
||||
let (shape, n) = Self::shape(&a.dataspace);
|
||||
rec.insert("shape".into(), shape);
|
||||
self.hash_values(&a.datatype, &a.raw_data, n, &mut rec)?;
|
||||
Ok(rec)
|
||||
});
|
||||
let v = match r {
|
||||
Ok(rec) => Value::Object(rec),
|
||||
Err(msg) => json!({ "error": msg }),
|
||||
};
|
||||
out.insert(a.name.clone(), v);
|
||||
}
|
||||
Ok(out)
|
||||
}
|
||||
|
||||
fn entries(&self, h: &ObjectHeader) -> Result<Vec<GroupEntry>, String> {
|
||||
let v1 = h
|
||||
.messages
|
||||
.iter()
|
||||
.find(|m| m.msg_type == MessageType::SymbolTable);
|
||||
if let Some(m) = v1 {
|
||||
let stm = SymbolTableMessage::parse(&m.data, self.os).map_err(e)?;
|
||||
group_v1::resolve_v1_group_entries(self.data, &stm, self.os, self.ls).map_err(e)
|
||||
} else if h
|
||||
.messages
|
||||
.iter()
|
||||
.any(|m| m.msg_type == MessageType::LinkInfo || m.msg_type == MessageType::Link)
|
||||
{
|
||||
group_v2::resolve_v2_group_entries(self.data, h, self.os, self.ls).map_err(e)
|
||||
} else {
|
||||
Ok(Vec::new())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Element bytes as an unsigned integer (at most 16 bytes), honouring byte order.
|
||||
fn element_bits(b: &[u8], byte_order: &DatatypeByteOrder) -> Result<u128, String> {
|
||||
if b.len() > 16 {
|
||||
return Err(format!("canon: {}-byte numeric element", b.len()));
|
||||
}
|
||||
let mut v = 0u128;
|
||||
match byte_order {
|
||||
DatatypeByteOrder::LittleEndian => {
|
||||
for (i, x) in b.iter().enumerate() {
|
||||
v |= u128::from(*x) << (8 * i);
|
||||
}
|
||||
}
|
||||
DatatypeByteOrder::BigEndian => {
|
||||
for x in b {
|
||||
v = (v << 8) | u128::from(*x);
|
||||
}
|
||||
}
|
||||
DatatypeByteOrder::Vax => return Err("canon: VAX byte order".into()),
|
||||
}
|
||||
Ok(v)
|
||||
}
|
||||
|
||||
fn field(v: u128, pos: u32, len: u32) -> u128 {
|
||||
if len == 0 || pos >= 128 {
|
||||
return 0;
|
||||
}
|
||||
let v = v >> pos;
|
||||
if len >= 128 {
|
||||
v
|
||||
} else {
|
||||
v & ((1u128 << len) - 1)
|
||||
}
|
||||
}
|
||||
|
||||
/// True when a float's bit fields are exactly IEEE 754 binary16/32/64 for its
|
||||
/// size. h5py hands back such a type's bytes untouched; any other layout (an
|
||||
/// N-Bit `H5Tset_precision` float, say) is *converted* by libhdf5 into the
|
||||
/// numpy float of the same size, so comparing raw bytes would be meaningless.
|
||||
fn ieee_layout(dt: &Datatype) -> bool {
|
||||
let Datatype::FloatingPoint {
|
||||
size,
|
||||
bit_offset,
|
||||
bit_precision,
|
||||
exponent_location,
|
||||
exponent_size,
|
||||
mantissa_location,
|
||||
mantissa_size,
|
||||
exponent_bias,
|
||||
..
|
||||
} = dt
|
||||
else {
|
||||
return true;
|
||||
};
|
||||
let std = match size {
|
||||
2 => (16, 10, 5, 10, 15),
|
||||
4 => (32, 23, 8, 23, 127),
|
||||
8 => (64, 52, 11, 52, 1023),
|
||||
_ => return true, // no same-size numpy float to convert to: compare raw
|
||||
};
|
||||
*bit_offset == 0
|
||||
&& (
|
||||
*bit_precision,
|
||||
*exponent_location,
|
||||
*exponent_size,
|
||||
*mantissa_size,
|
||||
*exponent_bias,
|
||||
) == (std.0, std.1, std.2, std.3, std.4)
|
||||
&& *mantissa_location == 0
|
||||
}
|
||||
|
||||
/// Canonicalise a non-IEEE-layout float the way libhdf5's float->float
|
||||
/// conversion presents it to h5py: as the IEEE float of the same size.
|
||||
/// Assumes the implied-leading-one normalisation and the sign bit at the top
|
||||
/// of the precision (what `H5Tset_precision` produces; the parser does not
|
||||
/// keep either field).
|
||||
fn canon_custom_float(dt: &Datatype, b: &[u8], out: &mut Vec<u8>) -> Result<(), String> {
|
||||
let Datatype::FloatingPoint {
|
||||
size,
|
||||
byte_order,
|
||||
bit_offset,
|
||||
bit_precision,
|
||||
exponent_location,
|
||||
exponent_size,
|
||||
mantissa_location,
|
||||
mantissa_size,
|
||||
exponent_bias,
|
||||
} = dt
|
||||
else {
|
||||
unreachable!()
|
||||
};
|
||||
let (esize, msize) = (u32::from(*exponent_size), u32::from(*mantissa_size));
|
||||
if esize == 0 || esize > 30 || msize > 64 {
|
||||
return Err(format!("canon: unsupported float layout e{esize} m{msize}"));
|
||||
}
|
||||
let v = element_bits(b, byte_order)?;
|
||||
let sign_pos = (u32::from(*bit_offset) + u32::from(*bit_precision)).saturating_sub(1);
|
||||
let neg = field(v, sign_pos, 1) == 1;
|
||||
let e = field(v, u32::from(*exponent_location), esize) as i64;
|
||||
let m = field(v, u32::from(*mantissa_location), msize);
|
||||
let emax = (1i64 << esize) - 1;
|
||||
let bias = i64::from(*exponent_bias);
|
||||
let mag = if e == emax {
|
||||
if m == 0 { f64::INFINITY } else { f64::NAN }
|
||||
} else if e == 0 {
|
||||
(m as f64) * 2f64.powi((1 - bias - msize as i64) as i32)
|
||||
} else {
|
||||
((1u128 << msize) as f64 + m as f64) * 2f64.powi((e - bias - msize as i64) as i32)
|
||||
};
|
||||
let x = if neg { -mag } else { mag };
|
||||
match size {
|
||||
2 => out
|
||||
.extend_from_slice(&clawhdf5_format::float16::f32_to_f16_bits(x as f32).to_le_bytes()),
|
||||
4 => out.extend_from_slice(&(x as f32).to_le_bytes()),
|
||||
8 => out.extend_from_slice(&x.to_le_bytes()),
|
||||
_ => unreachable!("ieee_layout keeps other sizes raw"),
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Integers stored with a bit offset or reduced precision (N-Bit): libhdf5
|
||||
/// converts them to the full-width integer of the same size, shifting the
|
||||
/// value down and sign-extending from the top precision bit.
|
||||
fn canon_partial_int(dt: &Datatype, b: &[u8], out: &mut Vec<u8>) -> Result<(), String> {
|
||||
let Datatype::FixedPoint {
|
||||
size,
|
||||
byte_order,
|
||||
signed,
|
||||
bit_offset,
|
||||
bit_precision,
|
||||
} = dt
|
||||
else {
|
||||
unreachable!()
|
||||
};
|
||||
let prec = u32::from(*bit_precision);
|
||||
let v = element_bits(b, byte_order)?;
|
||||
let mut x = field(v, u32::from(*bit_offset), prec);
|
||||
if *signed && prec > 0 && prec < 128 && field(x, prec - 1, 1) == 1 {
|
||||
x |= !0u128 << prec;
|
||||
}
|
||||
out.extend_from_slice(&x.to_le_bytes()[..*size as usize]);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn partial_int(dt: &Datatype) -> bool {
|
||||
matches!(dt, Datatype::FixedPoint { size, bit_offset, bit_precision, .. }
|
||||
if *bit_offset != 0 || u32::from(*bit_precision) != size * 8)
|
||||
}
|
||||
|
||||
fn canon_str(b: &[u8], out: &mut Vec<u8>) {
|
||||
let cut = b.iter().position(|&c| c == 0).unwrap_or(b.len());
|
||||
let mut s = &b[..cut];
|
||||
while let [rest @ .., b' '] = s {
|
||||
s = rest;
|
||||
}
|
||||
out.push(b'S');
|
||||
out.extend_from_slice(&(s.len() as u32).to_le_bytes());
|
||||
out.extend_from_slice(s);
|
||||
}
|
||||
|
||||
fn hex(b: &[u8]) -> String {
|
||||
b.iter().map(|x| format!("{x:02x}")).collect()
|
||||
}
|
||||
|
||||
fn dtype_str(dt: &Datatype) -> String {
|
||||
match dt {
|
||||
Datatype::FixedPoint {
|
||||
size,
|
||||
signed,
|
||||
byte_order,
|
||||
..
|
||||
} => {
|
||||
format!(
|
||||
"{}{}{}",
|
||||
bo(byte_order),
|
||||
if *signed { "i" } else { "u" },
|
||||
size
|
||||
)
|
||||
}
|
||||
Datatype::FloatingPoint {
|
||||
size, byte_order, ..
|
||||
} => format!("{}f{}", bo(byte_order), size),
|
||||
Datatype::BitField {
|
||||
size, byte_order, ..
|
||||
} => format!("{}b{}", bo(byte_order), size),
|
||||
Datatype::Time { size, .. } => format!("time{size}"),
|
||||
Datatype::String { size, .. } => format!("S{size}"),
|
||||
Datatype::Opaque { size, .. } => format!("V{size}"),
|
||||
Datatype::Compound { size, members } => format!(
|
||||
"{{{}}}{size}",
|
||||
members
|
||||
.iter()
|
||||
.map(|m| format!("{}:{}", m.name, dtype_str(&m.datatype)))
|
||||
.collect::<Vec<_>>()
|
||||
.join(",")
|
||||
),
|
||||
Datatype::Reference { ref_type, .. } => format!("ref({ref_type:?})"),
|
||||
Datatype::Enumeration { base_type, .. } => format!("enum({})", dtype_str(base_type)),
|
||||
Datatype::VariableLength {
|
||||
is_string: true, ..
|
||||
} => "vlstr".into(),
|
||||
Datatype::VariableLength { base_type, .. } => format!("vlen({})", dtype_str(base_type)),
|
||||
Datatype::Array {
|
||||
base_type,
|
||||
dimensions,
|
||||
} => format!("({}){dimensions:?}", dtype_str(base_type)),
|
||||
}
|
||||
}
|
||||
|
||||
fn bo(b: &DatatypeByteOrder) -> &'static str {
|
||||
match b {
|
||||
DatatypeByteOrder::LittleEndian => "<",
|
||||
DatatypeByteOrder::BigEndian => ">",
|
||||
DatatypeByteOrder::Vax => "vax",
|
||||
}
|
||||
}
|
||||
|
||||
fn is_group(h: &ObjectHeader) -> bool {
|
||||
h.messages.iter().any(|m| {
|
||||
matches!(
|
||||
m.msg_type,
|
||||
MessageType::LinkInfo | MessageType::Link | MessageType::SymbolTable
|
||||
)
|
||||
})
|
||||
}
|
||||
|
||||
fn main() {
|
||||
install_hook();
|
||||
let path = std::env::args().nth(1).expect("usage: probe <file>");
|
||||
let mut top = Map::new();
|
||||
top.insert("file".into(), Value::String(path.clone()));
|
||||
let data = match std::fs::read(&path) {
|
||||
Ok(d) => d,
|
||||
Err(err) => {
|
||||
top.insert("open_error".into(), Value::String(format!("Io({err})")));
|
||||
println!("{}", Value::Object(top));
|
||||
return;
|
||||
}
|
||||
};
|
||||
// Every address is relative to the superblock: look at the file from
|
||||
// there on (past any user block), as libhdf5 does.
|
||||
let hdf5: &[u8] = match signature::find_signature(&data) {
|
||||
Ok(off) => &data[off..],
|
||||
Err(_) => &data,
|
||||
};
|
||||
let sb = guarded(|| Superblock::parse(hdf5, 0).map_err(e));
|
||||
let sb = match sb {
|
||||
Ok(sb) => sb,
|
||||
Err(msg) => {
|
||||
top.insert("open_error".into(), Value::String(msg));
|
||||
println!("{}", Value::Object(top));
|
||||
return;
|
||||
}
|
||||
};
|
||||
// libhdf5 refuses a truncated file and reads nothing past the recorded
|
||||
// end of file.
|
||||
let base = (data.len() - hdf5.len()) as u64;
|
||||
let hdf5 = match sb.data_end(base, data.len() as u64) {
|
||||
Ok(end) => &hdf5[..end as usize],
|
||||
Err(err) => {
|
||||
top.insert("open_error".into(), Value::String(e(err)));
|
||||
println!("{}", Value::Object(top));
|
||||
return;
|
||||
}
|
||||
};
|
||||
top.insert("superblock_version".into(), json!(sb.version));
|
||||
let ctx = Ctx {
|
||||
data: hdf5,
|
||||
os: sb.offset_size,
|
||||
ls: sb.length_size,
|
||||
base_dir: std::path::Path::new(&path)
|
||||
.parent()
|
||||
.map(|p| p.to_path_buf())
|
||||
.unwrap_or_default(),
|
||||
vl: RefCell::new(VlResolver::new(hdf5, sb.offset_size, sb.length_size)),
|
||||
};
|
||||
let mut objects: Vec<Value> = Vec::new();
|
||||
let mut visited = HashSet::new();
|
||||
let mut soft_v1 = 0u64;
|
||||
// explicit DFS stack: (address, path)
|
||||
let mut stack: Vec<(u64, String)> = vec![(sb.root_group_address, "/".to_string())];
|
||||
while let Some((addr, p)) = stack.pop() {
|
||||
if objects.len() >= MAX_OBJECTS {
|
||||
top.insert("truncated".into(), json!(true));
|
||||
break;
|
||||
}
|
||||
if !visited.insert(addr) {
|
||||
continue;
|
||||
}
|
||||
let mut rec = Map::new();
|
||||
rec.insert("path".into(), Value::String(p.clone()));
|
||||
let r = guarded(|| {
|
||||
let h = ctx.header(addr)?;
|
||||
Ok(h)
|
||||
});
|
||||
let h = match r {
|
||||
Ok(h) => h,
|
||||
Err(msg) => {
|
||||
rec.insert("kind".into(), Value::String("unknown".into()));
|
||||
rec.insert("error".into(), Value::String(msg));
|
||||
objects.push(Value::Object(rec));
|
||||
continue;
|
||||
}
|
||||
};
|
||||
let is_ds = h
|
||||
.messages
|
||||
.iter()
|
||||
.any(|m| m.msg_type == MessageType::DataLayout);
|
||||
let kind = if is_ds {
|
||||
"dataset"
|
||||
} else if is_group(&h) || addr == sb.root_group_address {
|
||||
"group"
|
||||
} else if h
|
||||
.messages
|
||||
.iter()
|
||||
.any(|m| m.msg_type == MessageType::Datatype)
|
||||
{
|
||||
"datatype"
|
||||
} else {
|
||||
"unknown"
|
||||
};
|
||||
rec.insert("kind".into(), Value::String(kind.into()));
|
||||
if kind == "dataset"
|
||||
&& let Err(msg) = guarded(|| ctx.read_dataset(&h, &mut rec))
|
||||
{
|
||||
rec.insert("error".into(), Value::String(msg));
|
||||
}
|
||||
// Opening a committed datatype decodes it (h5py's `f[name]` fails on
|
||||
// one libhdf5 cannot decode), so decode it here too.
|
||||
if kind == "datatype"
|
||||
&& let Err(msg) = guarded(|| ctx.read_named_datatype(&h))
|
||||
{
|
||||
rec.insert("error".into(), Value::String(msg));
|
||||
}
|
||||
if kind != "datatype" {
|
||||
match guarded(|| ctx.attrs(&h)) {
|
||||
Ok(m) => {
|
||||
rec.insert("attrs".into(), Value::Object(m));
|
||||
}
|
||||
Err(msg) => {
|
||||
rec.insert("attrs_error".into(), Value::String(msg));
|
||||
}
|
||||
}
|
||||
}
|
||||
if kind == "group" {
|
||||
match guarded(|| ctx.entries(&h)) {
|
||||
Ok(mut ents) => {
|
||||
ents.retain(|en| {
|
||||
if en.cache_type == 2 {
|
||||
soft_v1 += 1;
|
||||
false
|
||||
} else {
|
||||
true
|
||||
}
|
||||
});
|
||||
ents.sort_by(|a, b| a.name.cmp(&b.name));
|
||||
let base = if p == "/" { String::new() } else { p.clone() };
|
||||
for en in ents.into_iter().rev() {
|
||||
stack.push((en.object_header_address, format!("{base}/{}", en.name)));
|
||||
}
|
||||
}
|
||||
Err(msg) => {
|
||||
rec.insert("list_error".into(), Value::String(msg));
|
||||
}
|
||||
}
|
||||
}
|
||||
objects.push(Value::Object(rec));
|
||||
}
|
||||
if soft_v1 > 0 {
|
||||
top.insert("v1_soft_link_entries".into(), json!(soft_v1));
|
||||
}
|
||||
top.insert("objects".into(), Value::Array(objects));
|
||||
println!("{}", Value::Object(top));
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// The N-Bit float of libhdf5's `test/testfiles/le_data.h5`
|
||||
/// (`Nbit_float_data_le`): offset 7, precision 20, sign bit 26, exponent
|
||||
/// 20+6 (bias 31), mantissa 7+13.
|
||||
fn nbit_f32(byte_order: DatatypeByteOrder) -> Datatype {
|
||||
Datatype::FloatingPoint {
|
||||
size: 4,
|
||||
byte_order,
|
||||
bit_offset: 7,
|
||||
bit_precision: 20,
|
||||
exponent_location: 20,
|
||||
exponent_size: 6,
|
||||
mantissa_location: 7,
|
||||
mantissa_size: 13,
|
||||
exponent_bias: 31,
|
||||
}
|
||||
}
|
||||
|
||||
fn canon_one(dt: &Datatype, bytes: &[u8]) -> Vec<u8> {
|
||||
let mut out = Vec::new();
|
||||
canon_custom_float(dt, bytes, &mut out).unwrap();
|
||||
out
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn nbit_float_canonicalises_to_the_value_libhdf5_returns() {
|
||||
let le = nbit_f32(DatatypeByteOrder::LittleEndian);
|
||||
let be = nbit_f32(DatatypeByteOrder::BigEndian);
|
||||
assert!(!ieee_layout(&le));
|
||||
// 1.0: exponent = bias, mantissa 0
|
||||
let one: u32 = 31 << 20;
|
||||
assert_eq!(canon_one(&le, &one.to_le_bytes()), 1.0f32.to_le_bytes());
|
||||
assert_eq!(canon_one(&be, &one.to_be_bytes()), 1.0f32.to_le_bytes());
|
||||
// -2.1999512 (h5py's reading of the file's -2.2): sign, e = 32, m = 819
|
||||
let v: u32 = (1 << 26) | (32 << 20) | (819 << 7);
|
||||
assert_eq!(
|
||||
canon_one(&le, &v.to_le_bytes()),
|
||||
(-2.199_951_2f32).to_le_bytes()
|
||||
);
|
||||
assert_eq!(canon_one(&le, &[0; 4]), 0.0f32.to_le_bytes());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ieee_floats_keep_their_raw_bytes() {
|
||||
let f32le = Datatype::FloatingPoint {
|
||||
size: 4,
|
||||
byte_order: DatatypeByteOrder::LittleEndian,
|
||||
bit_offset: 0,
|
||||
bit_precision: 32,
|
||||
exponent_location: 23,
|
||||
exponent_size: 8,
|
||||
mantissa_location: 0,
|
||||
mantissa_size: 23,
|
||||
exponent_bias: 127,
|
||||
};
|
||||
assert!(ieee_layout(&f32le));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn partial_precision_int_is_shifted_and_sign_extended() {
|
||||
let dt = Datatype::FixedPoint {
|
||||
size: 4,
|
||||
byte_order: DatatypeByteOrder::BigEndian,
|
||||
signed: true,
|
||||
bit_offset: 4,
|
||||
bit_precision: 17,
|
||||
};
|
||||
assert!(partial_int(&dt));
|
||||
let stored = (((-5i32) as u32) & 0x1_FFFF) << 4;
|
||||
let mut out = Vec::new();
|
||||
canon_partial_int(&dt, &stored.to_be_bytes(), &mut out).unwrap();
|
||||
assert_eq!(out, (-5i32).to_le_bytes());
|
||||
}
|
||||
}
|
||||
Executable
+259
@@ -0,0 +1,259 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Reference probe: same JSON as the Rust `conformance-probe`, produced with h5py.
|
||||
|
||||
Walk: iterative DFS from '/', children in sorted (UTF-8 byte) name order, hard
|
||||
links only, each object once (first path wins, deduplicated by object identity).
|
||||
Canonical value encoding: see harness/src/main.rs.
|
||||
"""
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
import struct
|
||||
import sys
|
||||
|
||||
import numpy as np
|
||||
import h5py
|
||||
|
||||
try:
|
||||
import hdf5plugin # noqa: F401 registers blosc/lz4/zstd/bzip2/... filters
|
||||
except Exception: # pragma: no cover
|
||||
pass
|
||||
|
||||
MAX_BYTES = 200 * 1024 * 1024
|
||||
MAX_OBJECTS = 200_000
|
||||
|
||||
|
||||
def canon_str(b, out):
|
||||
if isinstance(b, str):
|
||||
b = b.encode("utf-8", "surrogateescape")
|
||||
b = bytes(b)
|
||||
cut = b.find(b"\x00")
|
||||
if cut >= 0:
|
||||
b = b[:cut]
|
||||
b = b.rstrip(b" ")
|
||||
out += b"S" + struct.pack("<I", len(b)) + b
|
||||
|
||||
|
||||
def simple(dt):
|
||||
if dt.fields:
|
||||
return all(simple(dt.fields[n][0]) for n in dt.names)
|
||||
if dt.subdtype:
|
||||
return simple(dt.subdtype[0])
|
||||
return dt.kind in "iufcbV"
|
||||
|
||||
|
||||
def packed(dt):
|
||||
if dt.fields:
|
||||
return np.dtype([(n, packed(dt.fields[n][0])) for n in dt.names])
|
||||
if dt.subdtype:
|
||||
base, shape = dt.subdtype
|
||||
return np.dtype((packed(base), shape))
|
||||
if dt.kind in "iufcb":
|
||||
return dt.newbyteorder("<")
|
||||
return dt
|
||||
|
||||
|
||||
def canon_el(dt, val, out):
|
||||
if dt.fields:
|
||||
for n in dt.names:
|
||||
canon_el(dt.fields[n][0], val[n], out)
|
||||
return
|
||||
if dt.subdtype:
|
||||
base, _ = dt.subdtype
|
||||
for x in np.asarray(val).reshape(-1):
|
||||
canon_el(base, x, out)
|
||||
return
|
||||
k = dt.kind
|
||||
if k in "iufcb":
|
||||
out += np.asarray(val, dtype=dt).astype(dt.newbyteorder("<")).tobytes()
|
||||
elif k == "V":
|
||||
out += np.asarray(val, dtype=dt).tobytes()
|
||||
elif k == "S":
|
||||
canon_str(val, out)
|
||||
elif k == "O":
|
||||
if h5py.check_string_dtype(dt) is not None:
|
||||
canon_str(val if val is not None else b"", out)
|
||||
elif h5py.check_ref_dtype(dt) is not None:
|
||||
out += b"R"
|
||||
else:
|
||||
base = h5py.check_vlen_dtype(dt)
|
||||
if base is None:
|
||||
raise TypeError(f"unhandled object dtype {dt!r}")
|
||||
arr = np.asarray(val if val is not None else [], dtype=base).reshape(-1)
|
||||
out += b"V" + struct.pack("<I", arr.shape[0])
|
||||
if simple(base):
|
||||
out += arr.astype(packed(base)).tobytes()
|
||||
else:
|
||||
for x in arr:
|
||||
canon_el(base, x, out)
|
||||
elif k == "U":
|
||||
canon_str(str(val), out)
|
||||
else:
|
||||
raise TypeError(f"unhandled dtype kind {k} ({dt!r})")
|
||||
|
||||
|
||||
def has_obj(dt):
|
||||
if dt.fields:
|
||||
return any(has_obj(dt.fields[n][0]) for n in dt.names)
|
||||
if dt.subdtype:
|
||||
return has_obj(dt.subdtype[0])
|
||||
return dt.kind == "O"
|
||||
|
||||
|
||||
def note_conversion(tid, dt, rec):
|
||||
"""h5py converts some file types (FP8, bfloat16, x87 long double, ...) to a
|
||||
different-sized numpy type; then value bytes are not comparable."""
|
||||
try:
|
||||
if not has_obj(dt) and tid.get_size() != dt.itemsize:
|
||||
rec["converted"] = f"file type size {tid.get_size()} -> numpy {dt} ({dt.itemsize})"
|
||||
except Exception: # noqa: BLE001
|
||||
pass
|
||||
|
||||
|
||||
def hash_values(arr, dt, rec):
|
||||
if dt.subdtype is not None:
|
||||
# h5py expands an HDF5 array element type into trailing array dims
|
||||
dt = dt.subdtype[0]
|
||||
arr = np.asarray(arr, dtype=dt)
|
||||
if simple(dt):
|
||||
c = np.ascontiguousarray(arr).astype(packed(dt)).tobytes()
|
||||
else:
|
||||
out = bytearray()
|
||||
for x in arr.reshape(-1):
|
||||
canon_el(dt, x, out)
|
||||
c = bytes(out)
|
||||
rec["hash"] = hashlib.sha256(c).hexdigest()
|
||||
rec["head"] = c[:48].hex()
|
||||
|
||||
|
||||
def err(e):
|
||||
s = f"{type(e).__name__}: {e}"
|
||||
return s.splitlines()[0][:400] if s else type(e).__name__
|
||||
|
||||
|
||||
def shape_of(s):
|
||||
return "null" if s is None else list(s)
|
||||
|
||||
|
||||
def n_bytes(shape, tid):
|
||||
n = 1
|
||||
for d in shape or ():
|
||||
n *= d
|
||||
return n * tid.get_size()
|
||||
|
||||
|
||||
def read_attrs(obj):
|
||||
out = {}
|
||||
names = sorted(obj.attrs.keys(), key=lambda s: s.encode("utf-8", "surrogateescape"))
|
||||
for name in names:
|
||||
rec = {}
|
||||
try:
|
||||
aid = obj.attrs.get_id(name)
|
||||
rec["dtype"] = str(aid.dtype)
|
||||
rec["shape"] = shape_of(aid.shape)
|
||||
note_conversion(aid.get_type(), aid.dtype, rec)
|
||||
if aid.shape is None:
|
||||
hash_values(np.empty((0,), dtype=aid.dtype), aid.dtype, rec)
|
||||
else:
|
||||
val = obj.attrs[name]
|
||||
hash_values(val, aid.dtype, rec)
|
||||
except Exception as e: # noqa: BLE001
|
||||
rec = {"error": err(e)}
|
||||
out[name] = rec
|
||||
return out
|
||||
|
||||
|
||||
def main(path):
|
||||
top = {"file": path}
|
||||
try:
|
||||
f = h5py.File(path, "r")
|
||||
except Exception as e: # noqa: BLE001
|
||||
top["open_error"] = err(e)
|
||||
print(json.dumps(top))
|
||||
return
|
||||
objects = []
|
||||
seen = set()
|
||||
stack = [("/", None)]
|
||||
while stack:
|
||||
p, obj = stack.pop()
|
||||
if len(objects) >= MAX_OBJECTS:
|
||||
top["truncated"] = True
|
||||
break
|
||||
rec = {"path": p}
|
||||
try:
|
||||
if obj is None:
|
||||
obj = f[p]
|
||||
key = hash(obj.id) # h5py ObjectID hash = (fileno, object address/token)
|
||||
except Exception as e: # noqa: BLE001
|
||||
rec["kind"] = "unknown"
|
||||
rec["error"] = err(e)
|
||||
objects.append(rec)
|
||||
continue
|
||||
if key in seen:
|
||||
continue
|
||||
seen.add(key)
|
||||
if isinstance(obj, h5py.Dataset):
|
||||
kind = "dataset"
|
||||
elif isinstance(obj, h5py.Group):
|
||||
kind = "group"
|
||||
elif isinstance(obj, h5py.Datatype):
|
||||
kind = "datatype"
|
||||
else:
|
||||
kind = "unknown"
|
||||
rec["kind"] = kind
|
||||
if kind == "dataset":
|
||||
try:
|
||||
dt = obj.dtype
|
||||
rec["dtype"] = str(dt)
|
||||
rec["shape"] = shape_of(obj.shape)
|
||||
note_conversion(obj.id.get_type(), dt, rec)
|
||||
if obj.shape is None:
|
||||
hash_values(np.empty((0,), dtype=dt), dt, rec)
|
||||
elif n_bytes(obj.shape, obj.id.get_type()) > MAX_BYTES:
|
||||
rec["skipped"] = "too large"
|
||||
else:
|
||||
arr = np.empty(obj.shape, dtype=dt)
|
||||
if arr.size:
|
||||
try:
|
||||
obj.read_direct(arr)
|
||||
except Exception: # noqa: BLE001
|
||||
arr = obj[()]
|
||||
hash_values(arr, dt, rec)
|
||||
except Exception as e: # noqa: BLE001
|
||||
rec["error"] = err(e)
|
||||
if kind != "datatype":
|
||||
try:
|
||||
rec["attrs"] = read_attrs(obj)
|
||||
except Exception as e: # noqa: BLE001
|
||||
rec["attrs_error"] = err(e)
|
||||
if kind == "group":
|
||||
try:
|
||||
names = sorted(obj.keys(), key=lambda s: s.encode("utf-8", "surrogateescape"))
|
||||
base = "" if p == "/" else p
|
||||
kids = []
|
||||
for n in names:
|
||||
try:
|
||||
link = obj.get(n, getlink=True)
|
||||
except Exception: # noqa: BLE001
|
||||
link = None
|
||||
if link is not None and not isinstance(link, h5py.HardLink):
|
||||
continue
|
||||
kids.append(f"{base}/{n}")
|
||||
for k in reversed(kids):
|
||||
stack.append((k, None))
|
||||
except Exception as e: # noqa: BLE001
|
||||
rec["list_error"] = err(e)
|
||||
objects.append(rec)
|
||||
top["objects"] = objects
|
||||
print(json.dumps(top), flush=True)
|
||||
# Exit without tearing down the h5py objects: freeing them for some files
|
||||
# that hold references (hdf5's h5repack_attr_refs.h5, cve-2024-32623.h5)
|
||||
# makes libhdf5 2.0 abort with "free(): chunks in smallbin corrupted"
|
||||
# about half the time. That happens after the reading is done, so it says
|
||||
# nothing about what h5py read, but it flipped those files between ok and
|
||||
# h5py-cannot-read from one run to the next.
|
||||
os._exit(0)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main(sys.argv[1])
|
||||
@@ -0,0 +1,335 @@
|
||||
#!/usr/bin/env python3
|
||||
"""report.py <results_dir> <CONFORMANCE.md> <corpus_dir>
|
||||
|
||||
Render the sweep's results (compare.py's results.json plus the raw per-side
|
||||
runs) as CONFORMANCE.md, and write <results_dir>/report-meta.json (commit,
|
||||
date, versions) for check.py --update.
|
||||
"""
|
||||
import collections
|
||||
import datetime
|
||||
import json
|
||||
import os
|
||||
import platform
|
||||
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
import h5py
|
||||
import numpy
|
||||
|
||||
try:
|
||||
import hdf5plugin
|
||||
HDF5PLUGIN = hdf5plugin.version
|
||||
except Exception: # noqa: BLE001
|
||||
HDF5PLUGIN = "not installed"
|
||||
|
||||
R, OUT_MD, CORPUS = sys.argv[1], sys.argv[2], sys.argv[3]
|
||||
HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
ROOT = os.path.dirname(HERE)
|
||||
CLASSES = ["ok", "our-error", "mismatch", "h5py-cannot-read", "panic", "hang", "crash", "oom"]
|
||||
|
||||
|
||||
def sh(*cmd, cwd=ROOT):
|
||||
try:
|
||||
return subprocess.run(cmd, cwd=cwd, capture_output=True, text=True, timeout=30).stdout.strip()
|
||||
except Exception: # noqa: BLE001
|
||||
return ""
|
||||
|
||||
|
||||
def cpu_model():
|
||||
try:
|
||||
for ln in open("/proc/cpuinfo"):
|
||||
if ln.startswith(("model name", "Model")):
|
||||
return ln.split(":", 1)[1].strip()
|
||||
except OSError:
|
||||
pass
|
||||
return platform.processor() or "unknown"
|
||||
|
||||
|
||||
def mem_gib():
|
||||
try:
|
||||
for ln in open("/proc/meminfo"):
|
||||
if ln.startswith("MemTotal:"):
|
||||
return f"{int(ln.split()[1]) / 1048576:.0f} GiB"
|
||||
except OSError:
|
||||
pass
|
||||
return "?"
|
||||
|
||||
|
||||
res = json.load(open(os.path.join(R, "results.json")))
|
||||
meta_run = json.load(open(os.path.join(R, "meta.json"))) if os.path.exists(os.path.join(R, "meta.json")) else {}
|
||||
rows = res["rows"]
|
||||
issues = res.get("issues", {})
|
||||
|
||||
# safe.directory: a checkout owned by another user (a container) is still ours to read
|
||||
commit = sh("git", "-c", "safe.directory=*", "rev-parse", "HEAD") or os.environ.get("GITHUB_SHA", "unknown")
|
||||
lib_dirty = sh("git", "-c", "safe.directory=*", "status", "--porcelain", "--", "crates", "Cargo.toml")
|
||||
h5dump_v = sh("h5dump", "--version").replace("h5dump: ", "")
|
||||
meta = {
|
||||
"date": datetime.datetime.now(datetime.timezone.utc).strftime("%Y-%m-%d %H:%M UTC"),
|
||||
"commit": commit + (" (library sources modified)" if lib_dirty else ""),
|
||||
"reference": f"h5py {h5py.__version__} / HDF5 {h5py.version.hdf5_version}",
|
||||
}
|
||||
json.dump(meta, open(os.path.join(R, "report-meta.json"), "w"), indent=1)
|
||||
|
||||
pins = []
|
||||
for ln in open(os.path.join(HERE, "corpus.txt")):
|
||||
if ln.strip() and not ln.lstrip().startswith("#"):
|
||||
name, url, rev, root, *_ = ln.split()
|
||||
pins.append((name, url, rev, root))
|
||||
|
||||
by_corpus = collections.defaultdict(collections.Counter)
|
||||
for r in rows:
|
||||
by_corpus[r["corpus"]][r["class"]] += 1
|
||||
total = collections.Counter(r["class"] for r in rows)
|
||||
|
||||
|
||||
def ex_list(files, n=3):
|
||||
s = ", ".join(f"`{f}`" for f in files[:n])
|
||||
return s + (f" (+{len(files) - n} more)" if len(files) > n else "")
|
||||
|
||||
|
||||
# --- known causes that are not clawhdf5 bugs --------------------------------
|
||||
def is_h5py_be_vlen(i):
|
||||
"""h5py returns the elements of a VL sequence of a big-endian base type
|
||||
with their file (big-endian) bytes but a native-endian dtype."""
|
||||
return (i["kind"] == "mismatch" and i["key"] in ("values", "attr-values")
|
||||
and (i.get("ref_dtype") == "object") and (i.get("ours_dtype") or "").startswith("vlen(")
|
||||
and ">" in (i.get("ours_dtype") or ""))
|
||||
|
||||
|
||||
known = collections.defaultdict(list)
|
||||
for r in rows:
|
||||
if r["class"] != "mismatch":
|
||||
continue
|
||||
iss = issues.get(r["file"], [])
|
||||
if iss and all(is_h5py_be_vlen(i) for i in iss):
|
||||
known["h5py-be-vlen"].append(r["file"])
|
||||
|
||||
|
||||
# --- the CVE corpus: clawhdf5 vs h5dump vs h5py ------------------------------
|
||||
def side(run, name):
|
||||
p = os.path.join(R, "runs", run, name)
|
||||
if not os.path.exists(p + ".rc"):
|
||||
return None
|
||||
rc = int(open(p + ".rc").read().strip() or -1)
|
||||
err = open(p + ".err", errors="replace").read()
|
||||
try:
|
||||
j = json.load(open(p + ".json"))
|
||||
except Exception: # noqa: BLE001
|
||||
j = None
|
||||
return rc, err, j
|
||||
|
||||
|
||||
def outcome(s, rust=False):
|
||||
"""-> (bucket, text). bucket in read / error / panic / crash / hang / oom."""
|
||||
if s is None:
|
||||
return "missing", "not run"
|
||||
rc, err, j = s
|
||||
if rc in (137, 124):
|
||||
return "hang", "hang (killed at timeout)"
|
||||
if "memory allocation of" in err or "MemoryError" in err or "bad_alloc" in err or "Cannot allocate" in err:
|
||||
return "oom", "out of memory"
|
||||
if rust and (rc == 101 or "PANIC:" in err):
|
||||
return "panic", "panic"
|
||||
if "overflowed its stack" in err:
|
||||
return "crash", "stack overflow"
|
||||
if rc == 139:
|
||||
return "crash", "SIGSEGV"
|
||||
if rc == 134:
|
||||
return "crash", "SIGABRT" + (" (heap corruption)" if ("corrupted" in err or "free()" in err) else "")
|
||||
if rc > 128:
|
||||
return "crash", f"signal {rc - 128}"
|
||||
if j is None:
|
||||
return ("error", "error exit") if rc in (0, 1) else ("crash", f"exit {rc}")
|
||||
if "open_error" in j:
|
||||
return "error", "open error"
|
||||
objs = j.get("objects", [])
|
||||
ne = sum(1 for o in objs for k in ("error", "attrs_error", "list_error") if k in o)
|
||||
ne += sum(1 for o in objs for a in (o.get("attrs") or {}).values() if "error" in a)
|
||||
return "read", f"read {len(objs)} obj" + (f", {ne} errors" if ne else "")
|
||||
|
||||
|
||||
def h5dump_outcome(s):
|
||||
if s is None:
|
||||
return "missing", "not run"
|
||||
rc, err, _ = s
|
||||
if rc in (137, 124):
|
||||
return "hang", "hang (killed at timeout)"
|
||||
if "memory allocation" in err or "Cannot allocate" in err:
|
||||
return "oom", "out of memory"
|
||||
if rc == 139:
|
||||
return "crash", "SIGSEGV"
|
||||
if rc == 134:
|
||||
return "crash", "SIGABRT" + (" (heap corruption)" if ("corrupted" in err or "free()" in err) else "")
|
||||
if rc > 128:
|
||||
return "crash", f"signal {rc - 128}"
|
||||
return ("read", "ok") if rc == 0 else ("error", "error exit")
|
||||
|
||||
|
||||
cve_rows = []
|
||||
buckets = {"clawhdf5": collections.Counter(), "h5dump": collections.Counter(), "h5py": collections.Counter()}
|
||||
ours_panic = {r["file"] for r in rows if r["class"] == "panic"}
|
||||
for r in rows:
|
||||
if r["corpus"] != "cve_hdf5":
|
||||
continue
|
||||
run = r["file"].replace("/", "__")
|
||||
o = outcome(side(run, "ours"), rust=True)
|
||||
if o[0] == "read" and r["file"] in ours_panic:
|
||||
o = ("panic", "caught panic")
|
||||
p = outcome(side(run, "ref"))
|
||||
d = h5dump_outcome(side(run, "h5dump"))
|
||||
buckets["clawhdf5"][o[0]] += 1
|
||||
buckets["h5py"][p[0]] += 1
|
||||
buckets["h5dump"][d[0]] += 1
|
||||
cve_rows.append((r["file"].split("/", 1)[1], d[1], p[1], o[1], r["class"]))
|
||||
|
||||
# --- render -----------------------------------------------------------------
|
||||
L = []
|
||||
w = L.append
|
||||
w("# clawhdf5 conformance report")
|
||||
w("")
|
||||
w("Every HDF5 file of eight public corpora (pinned by commit) is read twice — by")
|
||||
w("clawhdf5 (`conformance/probe`, the same `clawhdf5-format` calls the facade")
|
||||
w("makes) and by h5py/libhdf5 (`conformance/ref.py`) — and the two readings are")
|
||||
w("compared object by object: the set of hard-linked objects, each dataset's and")
|
||||
w("attribute's shape, and a SHA-256 of its values in a canonical encoding. The")
|
||||
w("CVE corpus is also run through `h5dump`. Each side runs under a timeout and an")
|
||||
w("address-space limit, so a hang, crash or runaway allocation is recorded, not")
|
||||
w("fatal. This file is generated by `conformance/run.sh`; do not edit it by hand.")
|
||||
w("")
|
||||
w("## Run")
|
||||
w("")
|
||||
w("| | |")
|
||||
w("|---|---|")
|
||||
w(f"| date | {meta['date']} |")
|
||||
w(f"| clawhdf5 commit | `{meta['commit']}` |")
|
||||
w(f"| machine | `{platform.node()}`: {cpu_model()}, {os.cpu_count()} CPUs, {mem_gib()}, {platform.system()} {platform.release()} {platform.machine()} |")
|
||||
w(f"| command | `{os.environ.get('CONFORMANCE_CMD', 'conformance/run.sh')}` |")
|
||||
w(f"| rustc | {sh('rustc', '-V')} |")
|
||||
w(f"| reference | h5py {h5py.__version__}, HDF5 {h5py.version.hdf5_version}, numpy {numpy.__version__}, hdf5plugin {HDF5PLUGIN}, Python {platform.python_version()} |")
|
||||
w(f"| h5dump | {h5dump_v} (CVE corpus only) |")
|
||||
if meta_run:
|
||||
w(f"| limits | {meta_run.get('timeout_s')} s timeout (SIGKILL), {int(meta_run.get('mem_kb', 0)) // 1024} MiB address space, per process; {meta_run.get('jobs')} files in parallel |")
|
||||
w(f"| runtime | {meta_run.get('probe_seconds')} s probing + comparing ({meta_run.get('build_seconds')} s fetch/build before it) |")
|
||||
w("")
|
||||
w("## Results")
|
||||
w("")
|
||||
w("A file's class is the first that applies:")
|
||||
w("")
|
||||
w("- **panic / hang / crash / oom** — clawhdf5 panicked (caught per object or not), hit the timeout, died on a signal, or failed an allocation. The CI gate fails on any of these.")
|
||||
w("- **h5py-cannot-read** — libhdf5 could not open the file (or itself crashed or hung). Nothing to compare against; most are the deliberately malformed CVE reproducers.")
|
||||
w("- **our-error** — clawhdf5 returned an error for something h5py reads.")
|
||||
w("- **mismatch** — both read it, but the shapes, values, object set or attribute set differ.")
|
||||
w("- **ok** — every object h5py reads, clawhdf5 reads identically.")
|
||||
w("")
|
||||
w("| corpus | files | " + " | ".join(CLASSES) + " |")
|
||||
w("|---" * (len(CLASSES) + 2) + "|")
|
||||
for c in sorted(by_corpus):
|
||||
cnt = by_corpus[c]
|
||||
w(f"| {c} | {sum(cnt.values())} | " + " | ".join(str(cnt.get(k, 0)) for k in CLASSES) + " |")
|
||||
w(f"| **all** | **{len(rows)}** | " + " | ".join(f"**{total.get(k, 0)}**" for k in CLASSES) + " |")
|
||||
w("")
|
||||
n_known = sum(len(v) for v in known.values())
|
||||
if n_known:
|
||||
w(f"{n_known} of the {total.get('mismatch', 0)} mismatches are a known h5py bug, not ours (see *Known not-our-bug*).")
|
||||
w("")
|
||||
w("Corpora (fetched by `conformance/fetch-corpus.sh` into the gitignored `conformance/.cache/`):")
|
||||
w("")
|
||||
w("| corpus | source | commit |")
|
||||
w("|---|---|---|")
|
||||
for name, url, rev, root in pins:
|
||||
w(f"| {name} | {url.removesuffix('.git')}" + ("" if root == "." else f" (`{root}`)") + f" | `{rev[:12]}` |")
|
||||
w("")
|
||||
|
||||
w("## Panics, hangs, crashes, out-of-memory")
|
||||
w("")
|
||||
if not res["panics"]:
|
||||
w("None.")
|
||||
else:
|
||||
for p in res["panics"]:
|
||||
w(f"- `{p['file']}` [{p['class']}] {p['detail']}")
|
||||
w("")
|
||||
|
||||
w("## Our-error root causes")
|
||||
w("")
|
||||
w("Grouped by normalised error message. *files* counts files whose class this cause affects.")
|
||||
w("")
|
||||
w("| files | objects | error | examples |")
|
||||
w("|---:|---:|---|---|")
|
||||
for k, v in res["root_causes"].items():
|
||||
w(f"| {v['files']} | {v['count']} | `{k.replace('|', '/')}` | {ex_list(v['file_list'])} |")
|
||||
w("")
|
||||
w("## Mismatch root causes")
|
||||
w("")
|
||||
w("| files | objects | cause | examples |")
|
||||
w("|---:|---:|---|---|")
|
||||
for k, v in res["mismatch_causes"].items():
|
||||
w(f"| {v['files']} | {v['count']} | `{k.replace('|', '/')}` | {ex_list(v['file_list'])} |")
|
||||
w("")
|
||||
|
||||
w("## CVE corpus: clawhdf5 vs h5dump vs h5py")
|
||||
w("")
|
||||
w(f"The {len(cve_rows)} files of [HDFGroup/cve_hdf5](https://github.com/HDFGroup/cve_hdf5) — reproducers for")
|
||||
w("published libhdf5 CVEs and fuzzer finds. *read* = produced output (possibly with per-object")
|
||||
w("errors), *error* = refused cleanly. h5dump exits non-zero on any error anywhere in a file, so")
|
||||
w("its read/error split is not comparable with the other two rows; the panic, crash, hang and oom")
|
||||
w("columns are.")
|
||||
w("")
|
||||
w("| tool | read | error | panic | crash | hang | oom |")
|
||||
w("|---|---:|---:|---:|---:|---:|---:|")
|
||||
for tool, label in (("clawhdf5", "clawhdf5"), ("h5dump", f"h5dump {h5dump_v.split()[-1] if h5dump_v else ''}"),
|
||||
("h5py", f"h5py {h5py.__version__} / HDF5 {h5py.version.hdf5_version}")):
|
||||
b = buckets[tool]
|
||||
w(f"| {label} | " + " | ".join(str(b.get(k, 0)) for k in ("read", "error", "panic", "crash", "hang", "oom")) + " |")
|
||||
w("")
|
||||
w("<details><summary>Per-file outcomes</summary>")
|
||||
w("")
|
||||
w("| file | h5dump | h5py | clawhdf5 | class |")
|
||||
w("|---|---|---|---|---|")
|
||||
for f, d, p, o, cls in cve_rows:
|
||||
w(f"| {f} | {d} | {p} | {o} | {cls} |")
|
||||
w("")
|
||||
w("</details>")
|
||||
w("")
|
||||
|
||||
w("## Known not-our-bug")
|
||||
w("")
|
||||
w("- **h5py big-endian variable-length sequences.** h5py returns the elements of a VL sequence")
|
||||
w(" whose base type is big-endian with the file's big-endian bytes but a native (little-endian)")
|
||||
w(" numpy dtype, so the values it reports are byte-swapped garbage; `h5dump` prints the values")
|
||||
w(" clawhdf5 reads. Reproducer: `h5py.vlen_dtype(np.dtype('>f4'))` dataset holding `[1.0, 2.0]`")
|
||||
w(" reads back in h5py as `[4.6e-41, 9.0e-44]`. Affected here: "
|
||||
+ (ex_list(sorted(known["h5py-be-vlen"]), 10) if known["h5py-be-vlen"] else "none") + ".")
|
||||
w("- **Non-IEEE floats and partial-precision integers (N-Bit).** libhdf5 converts a float whose")
|
||||
w(" bit layout is not IEEE (e.g. `H5Tset_precision` for the N-Bit filter) or an integer with a")
|
||||
w(" bit offset / reduced precision into the plain numpy type of the same size. The probe")
|
||||
w(" compares such values as converted numbers, not raw file bytes (before 2026-09-25 it compared")
|
||||
w(" raw bytes, which reported every N-Bit float dataset as a mismatch).")
|
||||
if res["incomparable"]:
|
||||
w("- **Types h5py widens.** Where h5py reads a type into a numpy type of a different size")
|
||||
w(" (FP8 -> float16, bfloat16 -> float32, x87 long double -> float128) the values are not")
|
||||
w(" compared (shape and presence still are): "
|
||||
+ ", ".join(f"{k} ({n}x)" for k, n in res["incomparable"]) + ".")
|
||||
w("- **References** are compared by presence only (`R`), not by target.")
|
||||
w("")
|
||||
if res.get("ref_only_errors"):
|
||||
w("## Objects h5py fails on but clawhdf5 reads")
|
||||
w("")
|
||||
for k, n in res["ref_only_errors"][:15]:
|
||||
w(f"- {n} x `{k}`")
|
||||
w("")
|
||||
w("## Reproduce")
|
||||
w("")
|
||||
w("```sh")
|
||||
w("# needs: Rust, python3 with h5py numpy hdf5plugin (conformance/requirements.txt), h5dump (hdf5-tools), git")
|
||||
w("CLAWHDF5_PYTHON=/path/to/venv/bin/python conformance/run.sh")
|
||||
w("```")
|
||||
w("")
|
||||
w("The corpus (about 450 MB of sparse checkouts) is cached in `conformance/.cache/`; results for")
|
||||
w("every file, both sides' raw JSON and stderr, are in `conformance/.cache/results/`.")
|
||||
w("`conformance/baseline.json` holds the ok files the nightly CI job (`.gitea/workflows/conformance.yml`)")
|
||||
w("must keep; `conformance/run.sh --update-baseline` rewrites it.")
|
||||
|
||||
with open(OUT_MD, "w") as fh:
|
||||
fh.write("\n".join(L) + "\n")
|
||||
@@ -0,0 +1,6 @@
|
||||
# The reference side of the conformance sweep. Pinned so the nightly job and a
|
||||
# local run compare against the same libhdf5 (h5py wheels bundle it).
|
||||
h5py==3.16.0
|
||||
numpy==2.5.3
|
||||
hdf5plugin==7.1.0
|
||||
netCDF4==1.7.4
|
||||
Executable
+88
@@ -0,0 +1,88 @@
|
||||
#!/usr/bin/env bash
|
||||
# conformance/run.sh — the clawhdf5 conformance sweep, end to end.
|
||||
#
|
||||
# fetch the pinned corpora (cached) -> build the probe -> probe every file
|
||||
# with clawhdf5 and with h5py (and h5dump for the CVE corpus), each under a
|
||||
# timeout and a memory limit -> compare -> write CONFORMANCE.md -> check the
|
||||
# result against conformance/baseline.json.
|
||||
#
|
||||
# Usage: conformance/run.sh [--no-fetch] [--no-report] [--update-baseline]
|
||||
#
|
||||
# Environment:
|
||||
# CLAWHDF5_PYTHON python with h5py, numpy, hdf5plugin (default: repo .venv, then python3)
|
||||
# CONFORMANCE_CACHE corpus / build / results cache (default: conformance/.cache)
|
||||
# CONFORMANCE_OUT results directory (default: $CONFORMANCE_CACHE/results)
|
||||
# CONFORMANCE_REPORT report path (default: CONFORMANCE.md at the repo root)
|
||||
# JOBS parallel files (default: nproc)
|
||||
# CONFORMANCE_PROBE use this prebuilt probe binary instead of building one
|
||||
# TMO / MEM_KB per-process timeout in seconds (20) / address-space limit in KiB (4 GiB)
|
||||
#
|
||||
# Exit status: 0 = gate passed; 1 = a panic/hang/crash/oom in clawhdf5, or the
|
||||
# ok count fell below the baseline, or a baseline-ok file regressed; 2 = setup error.
|
||||
set -euo pipefail
|
||||
HERE="$(cd "$(dirname "$0")" && pwd)"
|
||||
ROOT="$(cd "$HERE/.." && pwd)"
|
||||
FETCH=1 REPORT=1 UPDATE=0
|
||||
for a in "$@"; do
|
||||
case "$a" in
|
||||
--no-fetch) FETCH=0 ;;
|
||||
--no-report) REPORT=0 ;;
|
||||
--update-baseline) UPDATE=1 ;;
|
||||
-h|--help) sed -n '2,23p' "$0"; exit 0 ;;
|
||||
*) echo "unknown argument: $a" >&2; exit 2 ;;
|
||||
esac
|
||||
done
|
||||
|
||||
export PATH="$HOME/.cargo/bin:$PATH"
|
||||
CACHE="${CONFORMANCE_CACHE:-$HERE/.cache}"
|
||||
mkdir -p "$CACHE"; CACHE="$(cd "$CACHE" && pwd)"
|
||||
OUT="${CONFORMANCE_OUT:-$CACHE/results}"
|
||||
REPORT_PATH="${CONFORMANCE_REPORT:-$ROOT/CONFORMANCE.md}"
|
||||
JOBS="${JOBS:-$(nproc 2>/dev/null || echo 4)}"
|
||||
if [ -n "${CLAWHDF5_PYTHON:-}" ]; then PY="$CLAWHDF5_PYTHON"
|
||||
elif [ -x "$ROOT/.venv/bin/python" ]; then PY="$ROOT/.venv/bin/python"
|
||||
else PY="$(command -v python3)"; fi
|
||||
export PY TMO="${TMO:-20}" MEM_KB="${MEM_KB:-4194304}"
|
||||
command -v h5dump >/dev/null || { echo "error: h5dump not found (install hdf5-tools)" >&2; exit 2; }
|
||||
"$PY" -c 'import h5py, numpy, hdf5plugin' || { echo "error: $PY lacks h5py/numpy/hdf5plugin" >&2; exit 2; }
|
||||
|
||||
t0=$(date +%s)
|
||||
[ "$FETCH" = 1 ] && bash "$HERE/fetch-corpus.sh" "$CACHE"
|
||||
C="$CACHE/corpus"
|
||||
[ -d "$C" ] || { echo "error: no corpus in $C (run without --no-fetch)" >&2; exit 2; }
|
||||
|
||||
if [ -n "${CONFORMANCE_PROBE:-}" ]; then
|
||||
export PROBE="$CONFORMANCE_PROBE" # a prebuilt probe, e.g. an older one for a before/after
|
||||
else
|
||||
echo "== building the probe"
|
||||
CARGO_TARGET_DIR="${CARGO_TARGET_DIR:-$CACHE/target}" \
|
||||
cargo build -q --release --manifest-path "$HERE/probe/Cargo.toml"
|
||||
export PROBE="${CARGO_TARGET_DIR:-$CACHE/target}/release/conformance-probe"
|
||||
fi
|
||||
t1=$(date +%s)
|
||||
|
||||
rm -rf "$OUT"; mkdir -p "$OUT"
|
||||
"$PY" "$HERE/list_files.py" "$C" > "$OUT/files.txt"
|
||||
echo "== probing $(wc -l <"$OUT/files.txt") files, $JOBS at a time (timeout ${TMO}s, limit $((MEM_KB / 1024)) MiB)"
|
||||
export C OUT HERE
|
||||
# The shell's "Segmentation fault (core dumped)" notices go to probe.log; the
|
||||
# signals themselves are recorded in each side's .rc.
|
||||
xargs -a "$OUT/files.txt" -d '\n' -P "$JOBS" -I{} bash -c '
|
||||
f="$1"; d="$OUT/runs/${f//\//__}"
|
||||
case "$f" in cve_hdf5/*) export WITH_H5DUMP=1 ;; esac
|
||||
"$HERE/run_one.sh" "$C/$f" "$d"' _ {} 2>"$OUT/probe.log"
|
||||
echo "== comparing"
|
||||
"$PY" "$HERE/compare.py" "$OUT" >/dev/null
|
||||
t2=$(date +%s)
|
||||
cat > "$OUT/meta.json" <<EOF
|
||||
{"build_seconds": $((t1 - t0)), "probe_seconds": $((t2 - t1)), "jobs": $JOBS, "timeout_s": $TMO, "mem_kb": $MEM_KB}
|
||||
EOF
|
||||
export CONFORMANCE_CMD="${CONFORMANCE_CMD:-conformance/run.sh${*:+ $*}}"
|
||||
if [ "$REPORT" = 1 ]; then
|
||||
"$PY" "$HERE/report.py" "$OUT" "$REPORT_PATH" "$C"
|
||||
echo "== wrote $REPORT_PATH"
|
||||
fi
|
||||
if [ "$UPDATE" = 1 ]; then
|
||||
"$PY" "$HERE/check.py" "$OUT" "$HERE/baseline.json" --update
|
||||
fi
|
||||
"$PY" "$HERE/check.py" "$OUT" "$HERE/baseline.json"
|
||||
Executable
+27
@@ -0,0 +1,27 @@
|
||||
#!/usr/bin/env bash
|
||||
# run_one.sh <file> <outdir>
|
||||
#
|
||||
# Probe one file with clawhdf5 (PROBE) and with h5py (PY ref.py), and with
|
||||
# h5dump too when WITH_H5DUMP is set. Each side runs under a timeout (TMO
|
||||
# seconds, SIGKILL) and an address-space limit (MEM_KB), with core dumps off.
|
||||
# Writes <outdir>/<side>.{json,err,rc}; rc 137 = killed by the timeout.
|
||||
set -u
|
||||
f="$1"; out="$2"; mkdir -p "$out"
|
||||
HERE="$(cd "$(dirname "$0")" && pwd)"
|
||||
: "${PROBE:?PROBE must name the conformance-probe binary}"
|
||||
: "${PY:?PY must name a python with h5py}"
|
||||
TMO="${TMO:-20}"
|
||||
MEM_KB="${MEM_KB:-4194304}"
|
||||
run() { # name cmd...
|
||||
local name=$1; shift
|
||||
( ulimit -v "$MEM_KB"; ulimit -c 0; RUST_BACKTRACE=1 exec timeout -s KILL "$TMO" "$@" ) \
|
||||
>"$out/$name.json" 2>"$out/$name.err"
|
||||
echo $? >"$out/$name.rc"
|
||||
}
|
||||
run ours "$PROBE" "$f"
|
||||
run ref "$PY" "$HERE/ref.py" "$f"
|
||||
if [ -n "${WITH_H5DUMP:-}" ]; then
|
||||
run h5dump h5dump "$f"
|
||||
: >"$out/h5dump.json" # h5dump's text dump is not compared, only its exit status
|
||||
fi
|
||||
exit 0
|
||||
@@ -1,7 +1,8 @@
|
||||
[package]
|
||||
name = "clawhdf5-accel"
|
||||
version = "2.6.0"
|
||||
version = "2.7.0"
|
||||
edition = "2024"
|
||||
rust-version.workspace = true
|
||||
description = "SIMD-accelerated operations for rustyhdf5"
|
||||
license = "MIT"
|
||||
repository = "https://git.redclaw.dev/quantumclaw/clawhdf5"
|
||||
|
||||
@@ -25,6 +25,55 @@ unsafe fn hsum_256(v: __m256) -> f32 {
|
||||
_mm_cvtss_f32(result)
|
||||
}
|
||||
|
||||
/// AVX2 dot product of two `i8` slices, widened to `i32`.
|
||||
///
|
||||
/// Each 16-byte half is sign-extended to sixteen `i16` lanes and multiplied
|
||||
/// pairwise with `madd_epi16`, which sums adjacent products straight into
|
||||
/// eight `i32` lanes — the widening that an autovectorised scalar loop does
|
||||
/// in several shuffles is one instruction here. A pair sum is at most
|
||||
/// `2 * 127 * 127`, far inside `i32`.
|
||||
///
|
||||
/// # Safety
|
||||
/// Caller must verify is_x86_feature_detected!("avx2").
|
||||
// SAFETY: Caller must have verified AVX2 via is_x86_feature_detected!.
|
||||
#[target_feature(enable = "avx2")]
|
||||
pub unsafe fn dot_i8(a: &[i8], b: &[i8]) -> i32 {
|
||||
// SAFETY: Caller guarantees AVX2 is available per the # Safety contract;
|
||||
// every load reads 32 bytes at an index checked against `len` first.
|
||||
unsafe {
|
||||
assert_eq!(a.len(), b.len());
|
||||
let len = a.len();
|
||||
let mut i = 0;
|
||||
let mut acc0 = _mm256_setzero_si256();
|
||||
let mut acc1 = _mm256_setzero_si256();
|
||||
|
||||
while i + 32 <= len {
|
||||
let va = _mm256_loadu_si256(a.as_ptr().add(i).cast());
|
||||
let vb = _mm256_loadu_si256(b.as_ptr().add(i).cast());
|
||||
let a_lo = _mm256_cvtepi8_epi16(_mm256_castsi256_si128(va));
|
||||
let b_lo = _mm256_cvtepi8_epi16(_mm256_castsi256_si128(vb));
|
||||
let a_hi = _mm256_cvtepi8_epi16(_mm256_extracti128_si256(va, 1));
|
||||
let b_hi = _mm256_cvtepi8_epi16(_mm256_extracti128_si256(vb, 1));
|
||||
acc0 = _mm256_add_epi32(acc0, _mm256_madd_epi16(a_lo, b_lo));
|
||||
acc1 = _mm256_add_epi32(acc1, _mm256_madd_epi16(a_hi, b_hi));
|
||||
i += 32;
|
||||
}
|
||||
|
||||
// Horizontal sum of the eight i32 lanes.
|
||||
let v = _mm256_add_epi32(acc0, acc1);
|
||||
let s128 = _mm_add_epi32(_mm256_castsi256_si128(v), _mm256_extracti128_si256(v, 1));
|
||||
let s64 = _mm_add_epi32(s128, _mm_unpackhi_epi64(s128, s128));
|
||||
let s32 = _mm_add_epi32(s64, _mm_shuffle_epi32(s64, 0b01));
|
||||
let mut sum = _mm_cvtsi128_si32(s32);
|
||||
|
||||
while i < len {
|
||||
sum += i32::from(a[i]) * i32::from(b[i]);
|
||||
i += 1;
|
||||
}
|
||||
sum
|
||||
}
|
||||
}
|
||||
|
||||
/// AVX2 dot product for f32 slices.
|
||||
///
|
||||
/// # Safety
|
||||
|
||||
@@ -122,6 +122,36 @@ pub fn dot_product(a: &[f32], b: &[f32]) -> f32 {
|
||||
}
|
||||
}
|
||||
|
||||
/// Dot product of two `i8` slices, widened to `i32`.
|
||||
///
|
||||
/// The kernel behind int8-quantised vector search. On x86-64 it uses the AVX2
|
||||
/// path whenever AVX2 is present (including on AVX-512 machines, where it is
|
||||
/// what the f32 kernels use too on a default build). On aarch64 it uses the
|
||||
/// ARMv8.2 `SDOT` instruction when the CPU has the dot-product extension, and
|
||||
/// plain NEON otherwise.
|
||||
pub fn dot_i8(a: &[i8], b: &[i8]) -> i32 {
|
||||
match detect_backend() {
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
Backend::Neon => {
|
||||
if std::arch::is_aarch64_feature_detected!("dotprod") {
|
||||
// SAFETY: the dotprod extension was just detected at runtime.
|
||||
unsafe { neon::dot_i8_dotprod(a, b) }
|
||||
} else {
|
||||
// SAFETY: NEON is always available on aarch64.
|
||||
unsafe { neon::dot_i8(a, b) }
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
// SAFETY: both variants imply AVX2 was detected at runtime (the
|
||||
// AVX-512 backend is only selected on CPUs that also have AVX2).
|
||||
Backend::Avx2 | Backend::Avx512 if is_x86_feature_detected!("avx2") => unsafe {
|
||||
avx2::dot_i8(a, b)
|
||||
},
|
||||
_ => scalar::dot_i8(a, b),
|
||||
}
|
||||
}
|
||||
|
||||
/// Compute the L2 norm (magnitude) of a vector.
|
||||
pub fn vector_norm(v: &[f32]) -> f32 {
|
||||
dot_product(v, v).sqrt()
|
||||
@@ -713,3 +743,78 @@ mod tests {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod dot_i8_tests {
|
||||
use super::*;
|
||||
|
||||
fn codes(n: usize, seed: u64) -> Vec<i8> {
|
||||
let mut state = seed;
|
||||
(0..n)
|
||||
.map(|_| {
|
||||
state = state
|
||||
.wrapping_mul(6_364_136_223_846_793_005)
|
||||
.wrapping_add(1_442_695_040_888_963_407);
|
||||
// Full range, including the extremes.
|
||||
((state >> 56) as u8) as i8
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn dispatched_kernel_matches_scalar_exactly() {
|
||||
// Integer arithmetic: the SIMD path must agree bit for bit, at every
|
||||
// length — including ones that are not multiples of the 32-byte block,
|
||||
// which exercise the tail.
|
||||
for len in [0, 1, 7, 31, 32, 33, 63, 64, 100, 384, 385, 1536] {
|
||||
let a = codes(len, 1 + len as u64);
|
||||
let b = codes(len, 1000 + len as u64);
|
||||
assert_eq!(dot_i8(&a, &b), scalar::dot_i8(&a, &b), "len {len}");
|
||||
}
|
||||
}
|
||||
|
||||
/// Dispatch only ever takes one path on a given CPU, so on a machine with
|
||||
/// the dot-product extension the plain-NEON kernel would otherwise go
|
||||
/// untested. Check each aarch64 kernel against scalar directly.
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
#[test]
|
||||
fn every_aarch64_kernel_matches_scalar_exactly() {
|
||||
for len in [0, 1, 7, 15, 16, 17, 31, 32, 33, 63, 64, 100, 384, 385, 1536] {
|
||||
let a = codes(len, 7 + len as u64);
|
||||
let b = codes(len, 7000 + len as u64);
|
||||
let want = scalar::dot_i8(&a, &b);
|
||||
// SAFETY: NEON is always available on aarch64.
|
||||
assert_eq!(unsafe { neon::dot_i8(&a, &b) }, want, "neon, len {len}");
|
||||
if std::arch::is_aarch64_feature_detected!("dotprod") {
|
||||
// SAFETY: the dotprod extension was just detected.
|
||||
assert_eq!(
|
||||
unsafe { neon::dot_i8_dotprod(&a, &b) },
|
||||
want,
|
||||
"dotprod, len {len}"
|
||||
);
|
||||
}
|
||||
}
|
||||
// The extremes, through both kernels.
|
||||
let lo = vec![-128i8; 4096];
|
||||
let hi = vec![127i8; 4096];
|
||||
// SAFETY: NEON is always available on aarch64.
|
||||
assert_eq!(unsafe { neon::dot_i8(&lo, &lo) }, 4096 * 128 * 128);
|
||||
// SAFETY: NEON is always available on aarch64.
|
||||
assert_eq!(unsafe { neon::dot_i8(&lo, &hi) }, -4096 * 128 * 127);
|
||||
if std::arch::is_aarch64_feature_detected!("dotprod") {
|
||||
// SAFETY: the dotprod extension was just detected.
|
||||
assert_eq!(unsafe { neon::dot_i8_dotprod(&lo, &lo) }, 4096 * 128 * 128);
|
||||
// SAFETY: the dotprod extension was just detected.
|
||||
assert_eq!(unsafe { neon::dot_i8_dotprod(&lo, &hi) }, -4096 * 128 * 127);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn extremes_do_not_overflow() {
|
||||
// -128 * -128 is the largest product; a long run of it must still fit.
|
||||
let a = vec![-128i8; 4096];
|
||||
assert_eq!(dot_i8(&a, &a), 4096 * 128 * 128);
|
||||
let b = vec![127i8; 4096];
|
||||
assert_eq!(dot_i8(&a, &b), -4096 * 128 * 127);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -180,3 +180,130 @@ pub fn checksum_fletcher32(data: &[u8]) -> u32 {
|
||||
|
||||
(sum2 << 16) | sum1
|
||||
}
|
||||
|
||||
/// NEON dot product of two `i8` slices, widened to `i32`, for any aarch64 CPU.
|
||||
///
|
||||
/// `vmull_s8` multiplies eight lanes into `i16` — even `-128 * -128` is 16 384,
|
||||
/// inside `i16` — and `vpadalq_s16` adds adjacent pairs of those into `i32`
|
||||
/// accumulators, so nothing can overflow before the final horizontal sum.
|
||||
///
|
||||
/// CPUs with the ARMv8.2 dot-product extension should use
|
||||
/// [`dot_i8_dotprod`], which does the multiply and the accumulate in one
|
||||
/// instruction.
|
||||
///
|
||||
/// # Safety
|
||||
/// Caller must ensure aarch64 target (NEON always available).
|
||||
// SAFETY: NEON is always available on aarch64 targets; caller guarantees aarch64.
|
||||
#[target_feature(enable = "neon")]
|
||||
pub unsafe fn dot_i8(a: &[i8], b: &[i8]) -> i32 {
|
||||
assert_eq!(a.len(), b.len());
|
||||
let len = a.len();
|
||||
let mut i = 0;
|
||||
let mut acc0 = vdupq_n_s32(0);
|
||||
let mut acc1 = vdupq_n_s32(0);
|
||||
|
||||
while i + 16 <= len {
|
||||
// SAFETY: NEON is available per the # Safety contract, and both
|
||||
// 16-byte loads start at an index checked against `len` above.
|
||||
unsafe {
|
||||
let va = vld1q_s8(a.as_ptr().add(i));
|
||||
let vb = vld1q_s8(b.as_ptr().add(i));
|
||||
acc0 = vpadalq_s16(acc0, vmull_s8(vget_low_s8(va), vget_low_s8(vb)));
|
||||
acc1 = vpadalq_s16(acc1, vmull_high_s8(va, vb));
|
||||
}
|
||||
i += 16;
|
||||
}
|
||||
|
||||
let mut sum = vaddvq_s32(vaddq_s32(acc0, acc1));
|
||||
while i < len {
|
||||
sum += i32::from(a[i]) * i32::from(b[i]);
|
||||
i += 1;
|
||||
}
|
||||
sum
|
||||
}
|
||||
|
||||
/// One `SDOT`: for each of the four `i32` lanes of `acc`, add the dot
|
||||
/// product of the corresponding four `i8` pairs from `a` and `b`.
|
||||
///
|
||||
/// Written as inline assembly because the `vdotq_s32` intrinsic is still
|
||||
/// behind the unstable `stdarch_neon_dotprod` feature; inline assembly is
|
||||
/// stable on aarch64.
|
||||
///
|
||||
/// # Safety
|
||||
/// Caller must ensure the CPU supports the `dotprod` extension.
|
||||
#[inline]
|
||||
#[target_feature(enable = "neon,dotprod")]
|
||||
unsafe fn sdot(acc: int32x4_t, a: int8x16_t, b: int8x16_t) -> int32x4_t {
|
||||
let mut acc = acc;
|
||||
// SAFETY: `dotprod` is enabled for this function and the caller
|
||||
// guarantees the CPU supports it. The instruction reads only its three
|
||||
// vector registers and touches no memory.
|
||||
unsafe {
|
||||
std::arch::asm!(
|
||||
"sdot {acc:v}.4s, {a:v}.16b, {b:v}.16b",
|
||||
acc = inout(vreg) acc,
|
||||
a = in(vreg) a,
|
||||
b = in(vreg) b,
|
||||
options(pure, nomem, nostack),
|
||||
);
|
||||
}
|
||||
acc
|
||||
}
|
||||
|
||||
/// NEON dot product of two `i8` slices using the ARMv8.2 dot-product
|
||||
/// extension (`SDOT`): sixteen multiply-accumulates per instruction, straight
|
||||
/// into `i32` lanes.
|
||||
///
|
||||
/// Present on the cores this crate actually runs on — Cortex-A76 and later
|
||||
/// (Raspberry Pi 5, current Android phones), Neoverse-N1 (Graviton2, Ampere
|
||||
/// Altra), and every Apple Silicon generation.
|
||||
///
|
||||
/// # Safety
|
||||
/// Caller must verify `is_aarch64_feature_detected!("dotprod")`.
|
||||
// SAFETY: caller has verified the dotprod extension at runtime.
|
||||
#[target_feature(enable = "neon,dotprod")]
|
||||
pub unsafe fn dot_i8_dotprod(a: &[i8], b: &[i8]) -> i32 {
|
||||
assert_eq!(a.len(), b.len());
|
||||
let len = a.len();
|
||||
let mut i = 0;
|
||||
let mut acc0 = vdupq_n_s32(0);
|
||||
let mut acc1 = vdupq_n_s32(0);
|
||||
|
||||
// Two independent accumulators so consecutive SDOTs are not serialised on
|
||||
// one register.
|
||||
while i + 32 <= len {
|
||||
// SAFETY: dotprod is available per the # Safety contract, and every
|
||||
// 16-byte load starts at an index checked against `len` above.
|
||||
unsafe {
|
||||
acc0 = sdot(
|
||||
acc0,
|
||||
vld1q_s8(a.as_ptr().add(i)),
|
||||
vld1q_s8(b.as_ptr().add(i)),
|
||||
);
|
||||
acc1 = sdot(
|
||||
acc1,
|
||||
vld1q_s8(a.as_ptr().add(i + 16)),
|
||||
vld1q_s8(b.as_ptr().add(i + 16)),
|
||||
);
|
||||
}
|
||||
i += 32;
|
||||
}
|
||||
if i + 16 <= len {
|
||||
// SAFETY: as above; the load is bounds-checked by this condition.
|
||||
unsafe {
|
||||
acc0 = sdot(
|
||||
acc0,
|
||||
vld1q_s8(a.as_ptr().add(i)),
|
||||
vld1q_s8(b.as_ptr().add(i)),
|
||||
);
|
||||
}
|
||||
i += 16;
|
||||
}
|
||||
|
||||
let mut sum = vaddvq_s32(vaddq_s32(acc0, acc1));
|
||||
while i < len {
|
||||
sum += i32::from(a[i]) * i32::from(b[i]);
|
||||
i += 1;
|
||||
}
|
||||
sum
|
||||
}
|
||||
|
||||
@@ -140,3 +140,33 @@ fn f16_to_f32_soft(h: u16) -> f32 {
|
||||
|
||||
f32::from_bits(f32_bits)
|
||||
}
|
||||
|
||||
/// Dot product of two `i8` slices, widened to `i32`.
|
||||
///
|
||||
/// `dim` terms of at most `127 * 127` fit an `i32` for any realistic
|
||||
/// dimension (over 130 000 terms before overflow is possible).
|
||||
pub fn dot_i8(a: &[i8], b: &[i8]) -> i32 {
|
||||
assert_eq!(a.len(), b.len());
|
||||
// Four independent accumulators over 32-lane blocks: the widening product
|
||||
// has to sit in a fixed-length chunk for the vectoriser to see it, and the
|
||||
// separate accumulators keep it off one dependency chain.
|
||||
const LANE: usize = 8;
|
||||
let (a_blocks, a_tail) = a.as_chunks::<{ LANE * 4 }>();
|
||||
let (b_blocks, b_tail) = b.as_chunks::<{ LANE * 4 }>();
|
||||
let mut acc = [0i32; 4];
|
||||
for (x, y) in a_blocks.iter().zip(b_blocks) {
|
||||
for (lane, slot) in acc.iter_mut().enumerate() {
|
||||
let mut sum = 0i32;
|
||||
for k in 0..LANE {
|
||||
sum += i32::from(x[lane * LANE + k]) * i32::from(y[lane * LANE + k]);
|
||||
}
|
||||
*slot += sum;
|
||||
}
|
||||
}
|
||||
let tail: i32 = a_tail
|
||||
.iter()
|
||||
.zip(b_tail)
|
||||
.map(|(&x, &y)| i32::from(x) * i32::from(y))
|
||||
.sum();
|
||||
acc[0] + acc[1] + acc[2] + acc[3] + tail
|
||||
}
|
||||
|
||||
@@ -1,7 +1,8 @@
|
||||
[package]
|
||||
name = "clawhdf5-agent"
|
||||
version = "2.6.0"
|
||||
version = "2.7.0"
|
||||
edition = "2024"
|
||||
rust-version.workspace = true
|
||||
description = "HDF5-backed persistent memory store for on-device AI agents"
|
||||
license = "MIT"
|
||||
repository = "https://git.redclaw.dev/quantumclaw/clawhdf5"
|
||||
@@ -10,14 +11,18 @@ keywords = ["agent", "memory", "hdf5", "vector-search", "embedding"]
|
||||
categories = ["database", "science", "algorithms"]
|
||||
|
||||
[dependencies]
|
||||
clawhdf5-format = { path = "../clawhdf5-format", version = "2.6.0", features = ["parallel", "fast-checksum"] }
|
||||
clawhdf5 = { path = "../clawhdf5", version = "2.6.0" }
|
||||
clawhdf5-io = { path = "../clawhdf5-io", version = "2.6.0", features = ["mmap"] }
|
||||
clawhdf5-accel = { path = "../clawhdf5-accel", version = "2.6.0" }
|
||||
clawhdf5-ann = { path = "../clawhdf5-ann", version = "2.6.0", optional = true }
|
||||
clawhdf5-gpu = { path = "../clawhdf5-gpu", version = "2.6.0", optional = true, default-features = false }
|
||||
clawhdf5-format = { path = "../clawhdf5-format", version = "2.7.0", features = ["parallel", "fast-checksum"] }
|
||||
clawhdf5 = { path = "../clawhdf5", version = "2.7.0" }
|
||||
clawhdf5-io = { path = "../clawhdf5-io", version = "2.7.0", features = ["mmap"] }
|
||||
clawhdf5-accel = { path = "../clawhdf5-accel", version = "2.7.0" }
|
||||
clawhdf5-ann = { path = "../clawhdf5-ann", version = "2.7.0", optional = true }
|
||||
clawhdf5-gpu = { path = "../clawhdf5-gpu", version = "2.7.0", optional = true, default-features = false }
|
||||
serde = { workspace = true }
|
||||
byteorder = "1"
|
||||
# Signed checkpoints (MemoryConfig-independent; see `signing`). Pure Rust.
|
||||
ed25519-dalek = { version = "2", features = ["rand_core"] }
|
||||
sha2 = "0.10"
|
||||
rand_core = { version = "0.6", features = ["getrandom"] }
|
||||
half = { workspace = true, optional = true }
|
||||
rayon = { version = "1", optional = true }
|
||||
matrixmultiply = { version = "0.3", optional = true }
|
||||
@@ -44,6 +49,10 @@ harness = false
|
||||
name = "memory_bench"
|
||||
harness = false
|
||||
|
||||
[[bench]]
|
||||
name = "multimodal_bench"
|
||||
harness = false
|
||||
|
||||
[features]
|
||||
default = ["float16", "hnsw", "parallel"]
|
||||
float16 = ["half"]
|
||||
@@ -59,7 +68,6 @@ zstd = ["clawhdf5/zstd"]
|
||||
# `--no-default-features` (plus re-enabling other defaults) to force the exact
|
||||
# linear cosine scan.
|
||||
hnsw = ["clawhdf5-ann"]
|
||||
agent = []
|
||||
gpu = ["clawhdf5-gpu/gpu-wgpu"]
|
||||
fast-math = ["matrixmultiply"]
|
||||
accelerate = ["accelerate-src", "cblas-sys"]
|
||||
|
||||
@@ -0,0 +1,107 @@
|
||||
//! Multi-modal memory search benchmarks (`clawhdf5_agent::multimodal`).
|
||||
//!
|
||||
//! Covers `MultiModalStore::search_cross_modal` (every embedding of every
|
||||
//! record, whatever its modality) and, for comparison,
|
||||
//! `MultiModalStore::search_by_modality` restricted to one modality.
|
||||
//!
|
||||
//! Corpus: N records (1K and 10K), each carrying two 384-dim embeddings —
|
||||
//! a text embedding of its caption plus one embedding of its primary modality,
|
||||
//! cycling Image / Audio / Video — so a cross-modal query scores 2N vectors.
|
||||
//! All data comes from a fixed-seed LCG, so every run sees the same corpus.
|
||||
//!
|
||||
//! Run: `cargo bench -p clawhdf5-agent --bench multimodal_bench`
|
||||
|
||||
use std::collections::HashMap;
|
||||
|
||||
use clawhdf5_agent::multimodal::{
|
||||
MediaRef, ModalEmbedding, Modality, MultiModalRecord, MultiModalStore,
|
||||
};
|
||||
use criterion::{BenchmarkId, Criterion, criterion_group, criterion_main};
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Simple deterministic PRNG (LCG), same as the other agent benches
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
struct Rng(u32);
|
||||
|
||||
impl Rng {
|
||||
fn new(seed: u32) -> Self {
|
||||
Self(seed)
|
||||
}
|
||||
fn next_u32(&mut self) -> u32 {
|
||||
self.0 = self.0.wrapping_mul(1103515245).wrapping_add(12345);
|
||||
self.0 >> 16
|
||||
}
|
||||
fn next_f32(&mut self) -> f32 {
|
||||
self.next_u32() as f32 / 65536.0 - 0.5
|
||||
}
|
||||
}
|
||||
|
||||
fn make_vec(rng: &mut Rng, dim: usize) -> Vec<f32> {
|
||||
(0..dim).map(|_| rng.next_f32()).collect()
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Corpus
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const DIM: usize = 384;
|
||||
const K: usize = 10;
|
||||
|
||||
const MEDIA: [(Modality, &str, &str); 3] = [
|
||||
(Modality::Image, "image/png", "clip-vit-base"),
|
||||
(Modality::Audio, "audio/wav", "clap-base"),
|
||||
(Modality::Video, "video/mp4", "xclip-base"),
|
||||
];
|
||||
|
||||
fn build_store(n: usize, seed: u32) -> MultiModalStore {
|
||||
let mut rng = Rng::new(seed);
|
||||
let mut store = MultiModalStore::new();
|
||||
for i in 0..n {
|
||||
let (modality, mime, model) = &MEDIA[i % MEDIA.len()];
|
||||
let embeddings = vec![
|
||||
ModalEmbedding::new(Modality::Text, make_vec(&mut rng, DIM), "minilm-l6"),
|
||||
ModalEmbedding::new(modality.clone(), make_vec(&mut rng, DIM), *model),
|
||||
];
|
||||
store.add_record(MultiModalRecord {
|
||||
id: 0,
|
||||
primary_modality: modality.clone(),
|
||||
text_content: Some(format!("{modality} memory {i}")),
|
||||
media_ref: Some(MediaRef::path(format!("/media/{i}"), *mime)),
|
||||
embeddings,
|
||||
observation: None,
|
||||
timestamp: 1_700_000_000.0 + i as f64,
|
||||
metadata: HashMap::new(),
|
||||
});
|
||||
}
|
||||
store
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Benchmarks
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn multimodal_search_benches(c: &mut Criterion) {
|
||||
let query = make_vec(&mut Rng::new(99), DIM);
|
||||
|
||||
let mut group = c.benchmark_group("multimodal_search");
|
||||
group.sample_size(50);
|
||||
|
||||
for (label, n) in [("1k", 1_000usize), ("10k", 10_000)] {
|
||||
let store = build_store(n, 42);
|
||||
assert_eq!(store.count(), n);
|
||||
|
||||
group.bench_with_input(BenchmarkId::new("cross_modal", label), &n, |b, _| {
|
||||
b.iter(|| store.search_cross_modal(&query, K));
|
||||
});
|
||||
|
||||
group.bench_with_input(BenchmarkId::new("by_modality_image", label), &n, |b, _| {
|
||||
b.iter(|| store.search_by_modality(&Modality::Image, &query, K));
|
||||
});
|
||||
}
|
||||
|
||||
group.finish();
|
||||
}
|
||||
|
||||
criterion_group!(multimodal_benches, multimodal_search_benches);
|
||||
criterion_main!(multimodal_benches);
|
||||
@@ -119,6 +119,9 @@ mod tests {
|
||||
wal_enabled: false,
|
||||
wal_max_entries: 500,
|
||||
quantized_index: false,
|
||||
hnsw_m: 16,
|
||||
hnsw_ef_construction: 64,
|
||||
hnsw_ef_search: 0,
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -88,8 +88,11 @@ impl BM25Index {
|
||||
/// Search the index for a query, returning the top `k` results
|
||||
/// as `(doc_id, score)` pairs sorted by score descending.
|
||||
///
|
||||
/// Uses Block-Max WAND for early termination when remaining documents
|
||||
/// cannot beat the current top-k threshold.
|
||||
/// Scores every matching document exhaustively, then keeps the top `k`.
|
||||
/// There is no early termination (WAND, MaxScore): the store's hot path
|
||||
/// is [`scores`](Self::scores), because score fusion normalises over the
|
||||
/// whole matching set and so needs every score, which no pruning scheme
|
||||
/// can skip. This method is for BM25-only callers.
|
||||
pub fn search(&self, query: &str, k: usize) -> Vec<(usize, f32)> {
|
||||
if k == 0 {
|
||||
return Vec::new();
|
||||
@@ -561,8 +564,9 @@ mod tests {
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn wand_returns_same_results_as_exhaustive() {
|
||||
// WAND-style search should produce same scores as exhaustive
|
||||
fn top_k_search_matches_ranking_every_score() {
|
||||
// `search` must agree with ranking the full `scores` set — the
|
||||
// bounded heap is an optimisation over sorting, not an approximation.
|
||||
let docs: Vec<String> = (0..100)
|
||||
.map(|i| {
|
||||
if i % 3 == 0 {
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
//! In-memory cache for memory entries, sessions, and knowledge graph.
|
||||
|
||||
use crate::vector_search;
|
||||
use clawhdf5_format::float16::round_to_f16;
|
||||
|
||||
/// Every entry's embedding, in one contiguous `[N x dim]` buffer.
|
||||
///
|
||||
@@ -149,6 +150,11 @@ pub struct MemoryCache {
|
||||
pub norms: Vec<f32>,
|
||||
/// Hebbian activation weights (default 1.0 per entry).
|
||||
pub activation_weights: Vec<f32>,
|
||||
/// Round every embedding to IEEE half precision as it enters the cache,
|
||||
/// so the cache holds exactly what a `float16` store writes to disk. Set
|
||||
/// it with [`MemoryCache::set_half_precision`], which also rounds the
|
||||
/// rows already held.
|
||||
pub half_precision: bool,
|
||||
}
|
||||
|
||||
impl MemoryCache {
|
||||
@@ -164,9 +170,44 @@ impl MemoryCache {
|
||||
embedding_dim,
|
||||
norms: Vec::new(),
|
||||
activation_weights: Vec::new(),
|
||||
half_precision: false,
|
||||
}
|
||||
}
|
||||
|
||||
/// Switch half-precision rounding on or off. Turning it on rounds every
|
||||
/// embedding already held (and recomputes norms where one changed) —
|
||||
/// e.g. a `float16` store whose last checkpoint predates half-precision
|
||||
/// storage and so is still `f32` on disk.
|
||||
pub fn set_half_precision(&mut self, on: bool) {
|
||||
self.half_precision = on;
|
||||
if !on {
|
||||
return;
|
||||
}
|
||||
for i in 0..self.embeddings.len() {
|
||||
let row = &self.embeddings[i];
|
||||
if row
|
||||
.iter()
|
||||
.all(|&v| round_to_f16(v).to_bits() == v.to_bits())
|
||||
{
|
||||
continue;
|
||||
}
|
||||
let rounded: Vec<f32> = row.iter().map(|&v| round_to_f16(v)).collect();
|
||||
self.norms[i] = vector_search::compute_norm(&rounded);
|
||||
self.embeddings.set(i, &rounded);
|
||||
}
|
||||
}
|
||||
|
||||
/// The embedding as the cache will hold it: rounded to half precision
|
||||
/// when [`Self::half_precision`] is on, otherwise unchanged.
|
||||
fn stored_form(&self, mut embedding: Vec<f32>) -> Vec<f32> {
|
||||
if self.half_precision {
|
||||
for v in &mut embedding {
|
||||
*v = round_to_f16(*v);
|
||||
}
|
||||
}
|
||||
embedding
|
||||
}
|
||||
|
||||
/// Kept for callers that used to have to re-flatten after a bulk load.
|
||||
/// The buffer is always flat now, so there is nothing to rebuild.
|
||||
#[deprecated(note = "embeddings are stored flat; this is a no-op")]
|
||||
@@ -202,6 +243,7 @@ impl MemoryCache {
|
||||
tags: String,
|
||||
) -> usize {
|
||||
let idx = self.chunks.len();
|
||||
let embedding = self.stored_form(embedding);
|
||||
let norm = vector_search::compute_norm(&embedding);
|
||||
self.chunks.push(chunk);
|
||||
self.embeddings.push(&embedding);
|
||||
@@ -240,6 +282,7 @@ impl MemoryCache {
|
||||
session_id: String,
|
||||
) {
|
||||
if idx < self.chunks.len() {
|
||||
let embedding = self.stored_form(embedding);
|
||||
let norm = vector_search::compute_norm(&embedding);
|
||||
self.chunks[idx] = chunk;
|
||||
self.embeddings.set(idx, &embedding);
|
||||
@@ -439,4 +482,60 @@ mod tests {
|
||||
.reset_from(2, vec![vec![1.0, 2.0], vec![3.0, 4.0]]);
|
||||
assert_eq!(cache.embeddings.as_flat(), vec![1.0, 2.0, 3.0, 4.0]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn set_half_precision_rounds_existing_rows_and_their_norms() {
|
||||
// A store with float16 set whose checkpoint is still f32 on disk
|
||||
// loads full-precision rows; switching rounding on must bring them to
|
||||
// exactly what the next checkpoint will write.
|
||||
let mut cache = MemoryCache::new(3);
|
||||
cache.push(
|
||||
"a".into(),
|
||||
vec![0.1, 0.2, 0.3],
|
||||
"c".into(),
|
||||
0.0,
|
||||
"s".into(),
|
||||
"".into(),
|
||||
);
|
||||
cache.push(
|
||||
"b".into(),
|
||||
vec![0.5, 0.25, 1.0],
|
||||
"c".into(),
|
||||
0.0,
|
||||
"s".into(),
|
||||
"".into(),
|
||||
);
|
||||
let exact_norm = cache.norms[0];
|
||||
|
||||
cache.set_half_precision(true);
|
||||
let row0: Vec<f32> = [0.1f32, 0.2, 0.3]
|
||||
.iter()
|
||||
.map(|&v| round_to_f16(v))
|
||||
.collect();
|
||||
assert_eq!(&cache.embeddings[0], row0.as_slice());
|
||||
assert_eq!(cache.norms[0], vector_search::compute_norm(&row0));
|
||||
assert_ne!(cache.norms[0], exact_norm);
|
||||
// Already representable: untouched.
|
||||
assert_eq!(&cache.embeddings[1], &[0.5, 0.25, 1.0]);
|
||||
|
||||
// New rows are rounded as they arrive, and updates too.
|
||||
cache.push(
|
||||
"c".into(),
|
||||
vec![0.1, 0.0, 0.0],
|
||||
"c".into(),
|
||||
0.0,
|
||||
"s".into(),
|
||||
"".into(),
|
||||
);
|
||||
assert_eq!(cache.embeddings[2][0], round_to_f16(0.1));
|
||||
cache.update(
|
||||
2,
|
||||
"c".into(),
|
||||
vec![0.3, 0.0, 0.0],
|
||||
"c".into(),
|
||||
0.0,
|
||||
"s".into(),
|
||||
);
|
||||
assert_eq!(cache.embeddings[2][0], round_to_f16(0.3));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -144,9 +144,44 @@ pub struct ConsolidationStats {
|
||||
|
||||
pub struct ImportanceScorer;
|
||||
|
||||
/// Sum of squares, in 8-wide lanes so it vectorises.
|
||||
fn sum_of_squares(a: &[f32]) -> f32 {
|
||||
let (blocks, tail) = a.as_chunks::<8>();
|
||||
let mut acc = [0.0f32; 8];
|
||||
for b in blocks {
|
||||
for i in 0..8 {
|
||||
acc[i] += b[i] * b[i];
|
||||
}
|
||||
}
|
||||
acc.iter().sum::<f32>() + tail.iter().map(|x| x * x).sum::<f32>()
|
||||
}
|
||||
|
||||
/// `(a · b, |b|²)` in one pass over equal-length slices, in 8-wide lanes.
|
||||
fn dot_and_norm2(a: &[f32], b: &[f32]) -> (f32, f32) {
|
||||
let (a_blocks, a_tail) = a.as_chunks::<8>();
|
||||
let (b_blocks, b_tail) = b.as_chunks::<8>();
|
||||
let mut dot = [0.0f32; 8];
|
||||
let mut nb = [0.0f32; 8];
|
||||
for (x, y) in a_blocks.iter().zip(b_blocks) {
|
||||
for i in 0..8 {
|
||||
dot[i] += x[i] * y[i];
|
||||
nb[i] += y[i] * y[i];
|
||||
}
|
||||
}
|
||||
let mut d = dot.iter().sum::<f32>();
|
||||
let mut n = nb.iter().sum::<f32>();
|
||||
for (x, y) in a_tail.iter().zip(b_tail) {
|
||||
d += x * y;
|
||||
n += y * y;
|
||||
}
|
||||
(d, n)
|
||||
}
|
||||
|
||||
impl ImportanceScorer {
|
||||
/// Cosine similarity between two embedding slices.
|
||||
/// Returns 0.0 if either norm is zero.
|
||||
/// Returns 0.0 if either norm is zero. The reference that
|
||||
/// [`Self::score_surprise`] is tested against.
|
||||
#[cfg(test)]
|
||||
fn cosine_similarity(a: &[f32], b: &[f32]) -> f32 {
|
||||
let len = a.len().min(b.len());
|
||||
if len == 0 {
|
||||
@@ -167,13 +202,54 @@ impl ImportanceScorer {
|
||||
|
||||
/// Novelty score: 1.0 − max cosine similarity against all existing records.
|
||||
/// Returns 1.0 when there are no existing memories.
|
||||
///
|
||||
/// Same result as the reference cosine similarity against each record, but the
|
||||
/// new embedding's norm is computed once rather than per record, each
|
||||
/// record costs one fused pass (dot product and its norm together) rather
|
||||
/// than three, and a large working set is scored in parallel. Every insert
|
||||
/// scores against the whole working tier, so this is what an unbounded
|
||||
/// working tier pays for: at 100K records it was the difference between a
|
||||
/// benchmark finishing and not (`BENCHMARKS.md`, "Consolidation Efficiency").
|
||||
pub fn score_surprise(embedding: &[f32], existing_memories: &[&MemoryRecord]) -> f32 {
|
||||
if existing_memories.is_empty() {
|
||||
return 1.0;
|
||||
}
|
||||
let query_norm2 = sum_of_squares(embedding);
|
||||
let similarity = |r: &&MemoryRecord| -> f32 {
|
||||
let other = &r.embedding;
|
||||
let len = embedding.len().min(other.len());
|
||||
if len == 0 {
|
||||
return 0.0;
|
||||
}
|
||||
let (dot, other_norm2) = dot_and_norm2(&embedding[..len], &other[..len]);
|
||||
// A shorter record compares against the query's matching prefix.
|
||||
let q2 = if len == embedding.len() {
|
||||
query_norm2
|
||||
} else {
|
||||
sum_of_squares(&embedding[..len])
|
||||
};
|
||||
if q2 == 0.0 || other_norm2 == 0.0 {
|
||||
return 0.0;
|
||||
}
|
||||
dot / (q2.sqrt() * other_norm2.sqrt())
|
||||
};
|
||||
#[cfg(feature = "parallel")]
|
||||
let max_sim = if existing_memories.len() >= 4096 {
|
||||
use rayon::prelude::*;
|
||||
existing_memories
|
||||
.par_iter()
|
||||
.map(similarity)
|
||||
.reduce(|| f32::NEG_INFINITY, f32::max)
|
||||
} else {
|
||||
existing_memories
|
||||
.iter()
|
||||
.map(similarity)
|
||||
.fold(f32::NEG_INFINITY, f32::max)
|
||||
};
|
||||
#[cfg(not(feature = "parallel"))]
|
||||
let max_sim = existing_memories
|
||||
.iter()
|
||||
.map(|r| Self::cosine_similarity(embedding, &r.embedding))
|
||||
.map(similarity)
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
(1.0 - max_sim).clamp(0.0, 1.0)
|
||||
}
|
||||
@@ -471,6 +547,54 @@ impl ConsolidationEngine {
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn score_surprise_matches_the_reference_cosine() {
|
||||
let mut x = 0x2545_F491_4F6C_DD1Du64;
|
||||
let mut next = || {
|
||||
x ^= x << 13;
|
||||
x ^= x >> 7;
|
||||
x ^= x << 17;
|
||||
(x >> 40) as f32 / (1u64 << 24) as f32 - 0.5
|
||||
};
|
||||
let make = |id: u64, v: Vec<f32>| MemoryRecord {
|
||||
id,
|
||||
chunk: String::new(),
|
||||
embedding: v,
|
||||
tier: MemoryTier::Working,
|
||||
importance: 0.0,
|
||||
access_count: 0,
|
||||
last_accessed: 0.0,
|
||||
created_at: 0.0,
|
||||
source: MemorySource::User,
|
||||
};
|
||||
// Ordinary rows, a shorter one, an empty one and a zero vector; and
|
||||
// enough rows to take the parallel path too.
|
||||
for n in [5usize, 5000] {
|
||||
let mut recs: Vec<MemoryRecord> = (0..n as u64)
|
||||
.map(|i| make(i, (0..37).map(|_| next()).collect()))
|
||||
.collect();
|
||||
recs.push(make(9_000, (0..20).map(|_| next()).collect()));
|
||||
recs.push(make(9_001, Vec::new()));
|
||||
recs.push(make(9_002, vec![0.0; 37]));
|
||||
let refs: Vec<&MemoryRecord> = recs.iter().collect();
|
||||
for _ in 0..5 {
|
||||
let q: Vec<f32> = (0..37).map(|_| next()).collect();
|
||||
let expected = (1.0
|
||||
- refs
|
||||
.iter()
|
||||
.map(|r| ImportanceScorer::cosine_similarity(&q, &r.embedding))
|
||||
.fold(f32::NEG_INFINITY, f32::max))
|
||||
.clamp(0.0, 1.0);
|
||||
let got = ImportanceScorer::score_surprise(&q, &refs);
|
||||
assert!((got - expected).abs() < 1e-5, "n={n}: {got} vs {expected}");
|
||||
}
|
||||
}
|
||||
assert_eq!(
|
||||
ImportanceScorer::score_surprise(&[0.0; 4], &[&make(1, vec![1.0; 4])]),
|
||||
1.0
|
||||
);
|
||||
}
|
||||
|
||||
// Helper: build a simple normalised embedding of given dimension.
|
||||
fn unit_vec(dim: usize, hot: usize) -> Vec<f32> {
|
||||
let mut v = vec![0.0f32; dim];
|
||||
|
||||
@@ -65,31 +65,34 @@ pub fn hybrid_search_fused(
|
||||
) -> Vec<(usize, f32)> {
|
||||
// Get raw scores from both systems. Request all results so normalization
|
||||
// covers the full distribution.
|
||||
// Use parallel search when rayon feature is enabled and vector count > 10K.
|
||||
let vec_scores = {
|
||||
#[cfg(feature = "parallel")]
|
||||
{
|
||||
if vectors.count() > 10_000 {
|
||||
vector_search::parallel_cosine_batch(
|
||||
query_embedding,
|
||||
vectors,
|
||||
tombstones,
|
||||
vectors.count(),
|
||||
)
|
||||
} else {
|
||||
vector_search::cosine_similarity_batch(query_embedding, vectors, tombstones)
|
||||
}
|
||||
}
|
||||
#[cfg(not(feature = "parallel"))]
|
||||
{
|
||||
vector_search::cosine_similarity_batch(query_embedding, vectors, tombstones)
|
||||
}
|
||||
};
|
||||
let vec_scores = exact_vector_scores(query_embedding, vectors, tombstones);
|
||||
let kw_scores = bm25_index.scores(query_text);
|
||||
|
||||
fuse(vec_scores, kw_scores, fusion, k)
|
||||
}
|
||||
|
||||
/// Cosine similarity of `query_embedding` to every vector whose `skip` byte is
|
||||
/// 0 (a tombstone, or any other exclusion mask). Parallel above 10K vectors
|
||||
/// when the `parallel` feature is on.
|
||||
pub fn exact_vector_scores(
|
||||
query_embedding: &[f32],
|
||||
vectors: &(impl crate::vector_search::VectorSet + Sync + ?Sized),
|
||||
skip: &[u8],
|
||||
) -> Vec<(usize, f32)> {
|
||||
#[cfg(feature = "parallel")]
|
||||
{
|
||||
if vectors.count() > 10_000 {
|
||||
return vector_search::parallel_cosine_batch(
|
||||
query_embedding,
|
||||
vectors,
|
||||
skip,
|
||||
vectors.count(),
|
||||
);
|
||||
}
|
||||
}
|
||||
vector_search::cosine_similarity_batch(query_embedding, vectors, skip)
|
||||
}
|
||||
|
||||
/// Merge pre-computed vector-similarity and keyword scores into a single ranking.
|
||||
///
|
||||
/// Both score sets are independently min-max normalized to [0, 1] and combined
|
||||
|
||||
@@ -163,12 +163,13 @@ fn levenshtein(a: &str, b: &str) -> usize {
|
||||
/// entities-slice-index map, and an entity-id -> relation-indices map (edges
|
||||
/// touching that entity as either source or target).
|
||||
///
|
||||
/// Built fresh per traversal call rather than cached on `KnowledgeCache`:
|
||||
/// entities/relations are plain `pub` `Vec`s that get pushed to directly
|
||||
/// (e.g. `schema.rs`'s load path bypasses `add_entity`/`add_relation`), so a
|
||||
/// persistent index would need extra bookkeeping to avoid drifting stale. A
|
||||
/// one-off O(V+E) build per call is still a large win over the O(V·E) (BFS)
|
||||
/// / O(steps·active·E) (spreading activation) scans it replaces.
|
||||
/// Cached on `KnowledgeCache` and checked against a fingerprint of the graph
|
||||
/// on every use ([`graph_fingerprint`]). entities/relations are plain `pub`
|
||||
/// `Vec`s that get changed directly (e.g. `schema.rs`'s load path bypasses
|
||||
/// `add_entity`/`add_relation`), so the cache cannot rely on being told about
|
||||
/// changes; the fingerprint notices any of them. Rebuilding it on every
|
||||
/// traversal instead made a 2-hop BFS over 1K entities 6.5x slower than the
|
||||
/// scan it replaced (24 -> 155 µs; `BENCHMARKS.md`, "Knowledge Graph").
|
||||
struct AdjacencyIndex {
|
||||
entity_index: HashMap<u64, usize>,
|
||||
by_entity: HashMap<u64, Vec<usize>>,
|
||||
@@ -204,6 +205,45 @@ impl AdjacencyIndex {
|
||||
}
|
||||
}
|
||||
|
||||
/// A hash of everything [`AdjacencyIndex`] depends on — each entity's id and
|
||||
/// position, each relation's endpoints and position. One linear pass, no
|
||||
/// allocation: far cheaper than building the index, which hashes the same
|
||||
/// values into two maps.
|
||||
fn graph_fingerprint(entities: &[Entity], relations: &[Relation]) -> u64 {
|
||||
// splitmix64-style mixing; order matters, so positions are covered.
|
||||
fn mix(h: u64, v: u64) -> u64 {
|
||||
let mut z = (h ^ v).wrapping_add(0x9E37_79B9_7F4A_7C15);
|
||||
z = (z ^ (z >> 30)).wrapping_mul(0xBF58_476D_1CE4_E5B9);
|
||||
z = (z ^ (z >> 27)).wrapping_mul(0x94D0_49BB_1331_11EB);
|
||||
z ^ (z >> 31)
|
||||
}
|
||||
let mut h = mix(entities.len() as u64, relations.len() as u64);
|
||||
for e in entities {
|
||||
h = mix(h, e.id);
|
||||
}
|
||||
for r in relations {
|
||||
h = mix(mix(h, r.src), r.tgt);
|
||||
}
|
||||
h
|
||||
}
|
||||
|
||||
/// The cached [`AdjacencyIndex`] and the fingerprint it was built for.
|
||||
/// Cloning a `KnowledgeCache` starts the clone with an empty cache.
|
||||
#[derive(Default)]
|
||||
struct AdjacencyCache(std::sync::Mutex<Option<(u64, std::sync::Arc<AdjacencyIndex>)>>);
|
||||
|
||||
impl Clone for AdjacencyCache {
|
||||
fn clone(&self) -> Self {
|
||||
Self::default()
|
||||
}
|
||||
}
|
||||
|
||||
impl std::fmt::Debug for AdjacencyCache {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.write_str("AdjacencyCache")
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// KnowledgeCache
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -216,6 +256,7 @@ pub struct KnowledgeCache {
|
||||
pub alias_strings: Vec<String>,
|
||||
pub alias_entity_ids: Vec<i64>,
|
||||
next_entity_id: u64,
|
||||
adjacency: AdjacencyCache,
|
||||
}
|
||||
|
||||
impl KnowledgeCache {
|
||||
@@ -226,6 +267,7 @@ impl KnowledgeCache {
|
||||
alias_strings: Vec::new(),
|
||||
alias_entity_ids: Vec::new(),
|
||||
next_entity_id: 0,
|
||||
adjacency: AdjacencyCache::default(),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -236,9 +278,29 @@ impl KnowledgeCache {
|
||||
alias_strings: Vec::new(),
|
||||
alias_entity_ids: Vec::new(),
|
||||
next_entity_id: next_id,
|
||||
adjacency: AdjacencyCache::default(),
|
||||
}
|
||||
}
|
||||
|
||||
/// The adjacency index for the graph as it is now: the cached one if the
|
||||
/// graph's fingerprint still matches, otherwise rebuilt and cached.
|
||||
fn adjacency_index(&self) -> std::sync::Arc<AdjacencyIndex> {
|
||||
let fp = graph_fingerprint(&self.entities, &self.relations);
|
||||
let mut slot = self
|
||||
.adjacency
|
||||
.0
|
||||
.lock()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
if let Some((cached_fp, idx)) = slot.as_ref()
|
||||
&& *cached_fp == fp
|
||||
{
|
||||
return idx.clone();
|
||||
}
|
||||
let idx = std::sync::Arc::new(AdjacencyIndex::build(&self.entities, &self.relations));
|
||||
*slot = Some((fp, idx.clone()));
|
||||
idx
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------------
|
||||
// Entity management
|
||||
// -----------------------------------------------------------------------
|
||||
@@ -397,7 +459,7 @@ impl KnowledgeCache {
|
||||
/// together with their discovered depth. The seed entity itself is NOT
|
||||
/// included. Traversal follows both outgoing and incoming relation edges.
|
||||
pub fn bfs_neighbors(&self, entity_id: u64, max_depth: usize) -> Vec<(Entity, usize)> {
|
||||
let idx = AdjacencyIndex::build(&self.entities, &self.relations);
|
||||
let idx = self.adjacency_index();
|
||||
let mut visited: HashSet<u64> = HashSet::new();
|
||||
let mut queue: VecDeque<(u64, usize)> = VecDeque::new();
|
||||
let mut results: Vec<(Entity, usize)> = Vec::new();
|
||||
@@ -502,7 +564,7 @@ impl KnowledgeCache {
|
||||
min_activation: f32,
|
||||
max_steps: usize,
|
||||
) -> Vec<(u64, f32)> {
|
||||
let idx = AdjacencyIndex::build(&self.entities, &self.relations);
|
||||
let idx = self.adjacency_index();
|
||||
let mut activation: HashMap<u64, f32> = HashMap::new();
|
||||
|
||||
// Initialise seeds with activation 1.0.
|
||||
@@ -631,6 +693,51 @@ impl Default for KnowledgeCache {
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn cached_adjacency_sees_direct_changes_to_the_graph() {
|
||||
// The index is cached across traversals, but entities/relations are
|
||||
// pub Vecs anyone can edit; every kind of edit must be seen.
|
||||
let mut kg = KnowledgeCache::new();
|
||||
let a = kg.add_entity("a", "t", -1);
|
||||
let b = kg.add_entity("b", "t", -1);
|
||||
let c = kg.add_entity("c", "t", -1);
|
||||
kg.add_relation(a, b, "r", 1.0);
|
||||
let ids = |kg: &KnowledgeCache| -> Vec<u64> {
|
||||
let mut v: Vec<u64> = kg.bfs_neighbors(a, 3).iter().map(|(e, _)| e.id).collect();
|
||||
v.sort();
|
||||
v
|
||||
};
|
||||
assert_eq!(ids(&kg), vec![b]);
|
||||
assert_eq!(ids(&kg), vec![b], "cached index reused");
|
||||
|
||||
// Pushed directly, bypassing add_relation.
|
||||
kg.relations.push(Relation {
|
||||
src: b,
|
||||
tgt: c,
|
||||
..Relation::default()
|
||||
});
|
||||
assert_eq!(ids(&kg), vec![b, c]);
|
||||
|
||||
// Rewired in place: same lengths, different edge.
|
||||
kg.relations[1].tgt = a;
|
||||
assert_eq!(ids(&kg), vec![b]);
|
||||
|
||||
// Removed and replaced: same lengths again.
|
||||
kg.relations.pop();
|
||||
kg.relations.push(Relation {
|
||||
src: a,
|
||||
tgt: c,
|
||||
..Relation::default()
|
||||
});
|
||||
assert_eq!(ids(&kg), vec![b, c]);
|
||||
let act: Vec<u64> = kg
|
||||
.spreading_activation(&[a], 0.5, 0.0, 2)
|
||||
.iter()
|
||||
.map(|(id, _)| *id)
|
||||
.collect();
|
||||
assert!(act.contains(&c));
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------------
|
||||
// Original tests — must remain passing
|
||||
// -----------------------------------------------------------------------
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
//! ZeroClaw agent memory HDF5 backend.
|
||||
//! Agent memory stored in a single HDF5 file.
|
||||
//!
|
||||
//! Provides persistent memory storage for AI agents using HDF5 files.
|
||||
//! All data is cached in-memory for fast access and flushed to disk
|
||||
@@ -36,6 +36,7 @@ pub mod reranker;
|
||||
pub mod schema;
|
||||
pub mod search;
|
||||
pub mod session;
|
||||
pub mod signing;
|
||||
pub mod storage;
|
||||
mod store_lock;
|
||||
pub mod temporal;
|
||||
@@ -63,25 +64,22 @@ use std::path::{Path, PathBuf};
|
||||
use cache::MemoryCache;
|
||||
#[cfg(feature = "hnsw")]
|
||||
use clawhdf5_ann::{DistanceMetric, HnswIndex, Storage};
|
||||
use clawhdf5_format::float16::round_to_f16;
|
||||
use ephemeral::{EphemeralConfig, EphemeralStore};
|
||||
|
||||
/// HNSW construction parameters used for the agent's vector index. Cosine is the
|
||||
/// agent's similarity metric, so the index is built with cosine distance.
|
||||
#[cfg(feature = "hnsw")]
|
||||
const HNSW_M: usize = 16;
|
||||
#[cfg(feature = "hnsw")]
|
||||
const HNSW_EF_CONSTRUCTION: usize = 64;
|
||||
// EphemeralEntry and EphemeralStats are part of the crate public API via
|
||||
// the `ephemeral` module; they are not needed directly in lib.rs internals.
|
||||
#[allow(unused_imports)]
|
||||
pub use ephemeral::{EphemeralEntry, EphemeralStats};
|
||||
use knowledge::KnowledgeCache;
|
||||
use memory_strategy::{Exchange, MemoryStrategy, StrategyOutput};
|
||||
use session::SessionCache;
|
||||
pub use search::SearchOptions;
|
||||
pub use session::{SessionCache, SessionEntry};
|
||||
|
||||
// --- Error type ---
|
||||
|
||||
#[derive(Debug)]
|
||||
#[non_exhaustive]
|
||||
pub enum MemoryError {
|
||||
Io(std::io::Error),
|
||||
Hdf5(String),
|
||||
@@ -89,6 +87,14 @@ pub enum MemoryError {
|
||||
NotFound(String),
|
||||
/// Another `HDF5Memory` (in this or another process) has the store open.
|
||||
Locked(String),
|
||||
/// A record the store cannot hold as given, e.g. an embedding value
|
||||
/// outside the half-precision range of a `float16` store.
|
||||
InvalidEntry(String),
|
||||
/// The store's checkpoints are signed and no signing key is set, so a
|
||||
/// checkpoint would leave it unsigned. Set the key with
|
||||
/// [`HDF5Memory::set_signing_key`], or drop the signature on purpose with
|
||||
/// [`HDF5Memory::remove_signature`].
|
||||
SigningKeyRequired(String),
|
||||
}
|
||||
|
||||
impl std::fmt::Display for MemoryError {
|
||||
@@ -99,6 +105,8 @@ impl std::fmt::Display for MemoryError {
|
||||
MemoryError::Schema(e) => write!(f, "schema error: {e}"),
|
||||
MemoryError::NotFound(e) => write!(f, "not found: {e}"),
|
||||
MemoryError::Locked(e) => write!(f, "store is locked: {e}"),
|
||||
MemoryError::InvalidEntry(e) => write!(f, "invalid entry: {e}"),
|
||||
MemoryError::SigningKeyRequired(e) => write!(f, "signing key required: {e}"),
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -130,6 +138,19 @@ pub struct MemoryConfig {
|
||||
pub embedding_dim: usize,
|
||||
pub chunk_size: usize,
|
||||
pub overlap: usize,
|
||||
/// Store embeddings as IEEE half precision (numpy `float16`): half the
|
||||
/// bytes of the embeddings dataset on disk. Every embedding is rounded to
|
||||
/// the nearest half as it enters the store, in memory as well as on disk,
|
||||
/// so search results are the same before and after a reopen. Values must
|
||||
/// lie within ±65504; a save outside that is `MemoryError::InvalidEntry`.
|
||||
/// Fixed when the store is created (persisted in `/meta`).
|
||||
///
|
||||
/// **On by default for new stores**: on the full LongMemEval haystack with
|
||||
/// real MiniLM embeddings every retrieval metric matched `f32`, and at
|
||||
/// 100K records the file is 48% smaller (`BENCHMARKS.md`). Existing
|
||||
/// stores keep the setting they were created with. Set it to `false` for
|
||||
/// full-precision embeddings, e.g. for unnormalised vectors that may
|
||||
/// exceed the half-precision range.
|
||||
pub float16: bool,
|
||||
pub compression: bool,
|
||||
pub compression_level: u32,
|
||||
@@ -140,16 +161,39 @@ pub struct MemoryConfig {
|
||||
pub wal_enabled: bool,
|
||||
pub wal_max_entries: usize,
|
||||
/// Store the vector index's own copy of the embeddings as int8 rather than
|
||||
/// f32, a quarter of the memory.
|
||||
/// f32, a quarter of the memory. **On by default** for new stores.
|
||||
///
|
||||
/// The index's copy is the single largest part of a loaded store's
|
||||
/// footprint. Quantised distances are approximate, so the candidate pool
|
||||
/// is re-scored against the cache's exact embeddings before fusion, which
|
||||
/// restores recall; what it costs is throughput — roughly 13% of queries
|
||||
/// per second and 16% of build time at 100K x 384. See `BENCHMARKS.md`.
|
||||
/// holds recall at the f32 index's level. It is also faster, not slower:
|
||||
/// at equal recall, 1.63x the queries per second on x86-64 (AVX2) and
|
||||
/// 1.18x on a Raspberry Pi 5 (NEON `SDOT`), with builds 1.8x and 2.3x
|
||||
/// faster. See `BENCHMARKS.md`.
|
||||
///
|
||||
/// Persisted with the store. Stores written before this setting existed
|
||||
/// have no stored value and open as `false`, so reopening an old store
|
||||
/// never changes how its index is held.
|
||||
///
|
||||
/// Has no effect without the `hnsw` feature.
|
||||
pub quantized_index: bool,
|
||||
/// HNSW graph degree. Higher means a denser graph: better recall, more
|
||||
/// memory and slower builds. Clamped to at least 2 when the index is
|
||||
/// built, since a graph with fewer connections is not one.
|
||||
///
|
||||
/// Has no effect without the `hnsw` feature.
|
||||
pub hnsw_m: usize,
|
||||
/// Candidate list size while building the HNSW graph. Higher means a
|
||||
/// better graph and a slower build; it does not affect query cost.
|
||||
///
|
||||
/// Has no effect without the `hnsw` feature.
|
||||
pub hnsw_ef_construction: usize,
|
||||
/// Candidate list size for a query, trading throughput for recall. `0`
|
||||
/// keeps the default, which scales with the requested `k`
|
||||
/// (`max(k * 8, 64)`) so that fusion still sees a useful pool.
|
||||
///
|
||||
/// Has no effect without the `hnsw` feature.
|
||||
pub hnsw_ef_search: usize,
|
||||
}
|
||||
|
||||
impl MemoryConfig {
|
||||
@@ -162,7 +206,7 @@ impl MemoryConfig {
|
||||
embedding_dim,
|
||||
chunk_size: 512,
|
||||
overlap: 50,
|
||||
float16: false,
|
||||
float16: true,
|
||||
compression: false,
|
||||
compression_level: 0,
|
||||
compact_threshold: 0.3,
|
||||
@@ -171,7 +215,10 @@ impl MemoryConfig {
|
||||
created_at,
|
||||
wal_enabled: true,
|
||||
wal_max_entries: 500,
|
||||
quantized_index: false,
|
||||
quantized_index: true,
|
||||
hnsw_m: 16,
|
||||
hnsw_ef_construction: 64,
|
||||
hnsw_ef_search: 0,
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -278,6 +325,12 @@ pub struct HDF5Memory {
|
||||
activations_dirty: bool,
|
||||
/// Opened with [`HDF5Memory::open_read_only`]: nothing may reach the disk.
|
||||
read_only: bool,
|
||||
/// Key that signs every checkpoint; never persisted. See
|
||||
/// [`HDF5Memory::set_signing_key`].
|
||||
signing_key: Option<signing::SigningKey>,
|
||||
/// Checkpoints of this store are signed: the file on disk is, or a key
|
||||
/// has been set. A checkpoint without a key is then refused.
|
||||
signed: bool,
|
||||
/// A WAL that `open()` could not read and moved aside; see
|
||||
/// [`HDF5Memory::quarantined_wal`].
|
||||
quarantined_wal: Option<PathBuf>,
|
||||
@@ -297,7 +350,8 @@ impl HDF5Memory {
|
||||
/// Create a new HDF5 memory file with the given configuration.
|
||||
pub fn create(config: MemoryConfig) -> Result<Self> {
|
||||
let lock = store_lock::StoreLock::acquire(&config.path)?;
|
||||
let cache = MemoryCache::new(config.embedding_dim);
|
||||
let mut cache = MemoryCache::new(config.embedding_dim);
|
||||
cache.set_half_precision(config.float16);
|
||||
let sessions = SessionCache::new();
|
||||
let knowledge = KnowledgeCache::new();
|
||||
|
||||
@@ -332,6 +386,8 @@ impl HDF5Memory {
|
||||
bm25_filter: bm25::TokenFilter::default(),
|
||||
activations_dirty: false,
|
||||
read_only: false,
|
||||
signing_key: None,
|
||||
signed: false,
|
||||
quarantined_wal: None,
|
||||
_lock: Some(lock),
|
||||
})
|
||||
@@ -510,6 +566,8 @@ impl HDF5Memory {
|
||||
bm25_filter: bm25::TokenFilter::default(),
|
||||
activations_dirty: false,
|
||||
read_only,
|
||||
signing_key: None,
|
||||
signed: checkpoint.signed,
|
||||
quarantined_wal,
|
||||
_lock: lock,
|
||||
})
|
||||
@@ -670,6 +728,39 @@ impl HDF5Memory {
|
||||
}
|
||||
}
|
||||
|
||||
/// Sign every checkpoint from now on with `key` (Ed25519). The key is
|
||||
/// never written anywhere; set it again after every `open`. Once a store
|
||||
/// is signed, a checkpoint without the key is refused
|
||||
/// ([`MemoryError::SigningKeyRequired`]) rather than silently leaving it
|
||||
/// unsigned. Setting a different key re-signs the store under that key
|
||||
/// from the next checkpoint; a verifier trusting the old key will then
|
||||
/// reject it, which is the point. Call [`AgentMemory::flush_wal`] to sign
|
||||
/// right away.
|
||||
pub fn set_signing_key(&mut self, key: signing::SigningKey) {
|
||||
self.signing_key = Some(key);
|
||||
self.signed = true;
|
||||
}
|
||||
|
||||
/// Stop signing: the next checkpoint writes the store unsigned. The
|
||||
/// deliberate way out of [`MemoryError::SigningKeyRequired`].
|
||||
pub fn remove_signature(&mut self) {
|
||||
self.signing_key = None;
|
||||
self.signed = false;
|
||||
}
|
||||
|
||||
/// Checkpoints of this store are signed (on disk, or from the next
|
||||
/// checkpoint because a key has been set).
|
||||
pub fn is_signed(&self) -> bool {
|
||||
self.signed
|
||||
}
|
||||
|
||||
/// Check the checkpoint at `path` against the public key the caller
|
||||
/// trusts; see [`signing::verify_store`]. Reads the file only: it works
|
||||
/// on a store another process has open.
|
||||
pub fn verify(path: &Path, trusted: &signing::VerifyingKey) -> Result<signing::VerifyReport> {
|
||||
signing::verify_store(path, trusted)
|
||||
}
|
||||
|
||||
/// Flush current state to disk and truncate the WAL.
|
||||
///
|
||||
/// Every code path that persists the full cache to the .h5 file must
|
||||
@@ -685,10 +776,28 @@ impl HDF5Memory {
|
||||
// Record which WAL prefix this checkpoint contains, so a crash before
|
||||
// the truncate below can't replay those entries a second time.
|
||||
let wal_applied = self.wal.as_ref().map(|w| w.mark());
|
||||
let signature = match &self.signing_key {
|
||||
Some(key) => Some(signing::sign(
|
||||
key,
|
||||
&self.config,
|
||||
&self.cache,
|
||||
&self.sessions,
|
||||
&self.knowledge,
|
||||
wal_applied,
|
||||
)),
|
||||
None if self.signed => {
|
||||
return Err(MemoryError::SigningKeyRequired(format!(
|
||||
"{} is signed; set its signing key before a checkpoint \
|
||||
(saves so far are held in the WAL or in memory)",
|
||||
self.config.path.display()
|
||||
)));
|
||||
}
|
||||
None => None,
|
||||
};
|
||||
// Written before the .h5 so a crash in between leaves a sidecar whose
|
||||
// generation matches no checkpoint (ignored), never the reverse.
|
||||
let ann_generation = self.persist_vector_index();
|
||||
storage::write_to_disk_with_meta(
|
||||
storage::write_to_disk_signed(
|
||||
&self.config.path,
|
||||
&self.config,
|
||||
&self.cache,
|
||||
@@ -697,7 +806,9 @@ impl HDF5Memory {
|
||||
&schema::CheckpointMeta {
|
||||
wal_applied,
|
||||
ann_generation,
|
||||
signed: signature.is_some(),
|
||||
},
|
||||
signature.as_ref(),
|
||||
)?;
|
||||
if let Some(ref mut w) = self.wal {
|
||||
w.truncate()?;
|
||||
@@ -826,6 +937,32 @@ impl HDF5Memory {
|
||||
// the index length drifts from the cache length (covering any mutation path
|
||||
// that doesn't call a hook, e.g. consolidation pushes).
|
||||
|
||||
/// Graph degree for the index, never below the 2 the builder requires:
|
||||
/// a config value of 0 or 1 would otherwise panic inside `clawhdf5-ann`.
|
||||
#[cfg(feature = "hnsw")]
|
||||
fn hnsw_m(&self) -> usize {
|
||||
self.config.hnsw_m.max(2)
|
||||
}
|
||||
|
||||
/// Build-time candidate list size, never below the graph degree — a
|
||||
/// smaller one cannot fill a node's connections.
|
||||
#[cfg(feature = "hnsw")]
|
||||
fn hnsw_ef_construction(&self) -> usize {
|
||||
self.config.hnsw_ef_construction.max(self.hnsw_m())
|
||||
}
|
||||
|
||||
/// Query-time candidate list size for a `k`-result search. `0` means the
|
||||
/// default, which scales with `k`.
|
||||
#[cfg(feature = "hnsw")]
|
||||
pub(crate) fn hnsw_ef_search(&self, k: usize) -> usize {
|
||||
let default = (k * 8).max(64);
|
||||
if self.config.hnsw_ef_search == 0 {
|
||||
default
|
||||
} else {
|
||||
self.config.hnsw_ef_search.max(k)
|
||||
}
|
||||
}
|
||||
|
||||
/// How the index should store its copy of the vectors, per the config.
|
||||
#[cfg(feature = "hnsw")]
|
||||
fn index_storage(&self) -> Storage {
|
||||
@@ -856,8 +993,8 @@ impl HDF5Memory {
|
||||
let rows: Vec<Vec<f32>> = self.cache.embeddings.iter().map(<[f32]>::to_vec).collect();
|
||||
let mut index = HnswIndex::build_with(
|
||||
&rows,
|
||||
HNSW_M,
|
||||
HNSW_EF_CONSTRUCTION,
|
||||
self.hnsw_m(),
|
||||
self.hnsw_ef_construction(),
|
||||
DistanceMetric::Cosine,
|
||||
self.index_storage(),
|
||||
);
|
||||
@@ -962,6 +1099,18 @@ impl HDF5Memory {
|
||||
&self.config
|
||||
}
|
||||
|
||||
/// The sessions recorded in this store.
|
||||
pub fn sessions(&self) -> &SessionCache {
|
||||
&self.sessions
|
||||
}
|
||||
|
||||
/// Mutable access to the sessions, e.g. to add many at once. Changes
|
||||
/// reach the disk at the next checkpoint (any flushing call, such as
|
||||
/// [`HDF5Memory::flush_wal`] or `save_batch`), not immediately.
|
||||
pub fn sessions_mut(&mut self) -> &mut SessionCache {
|
||||
&mut self.sessions
|
||||
}
|
||||
|
||||
/// Get a reference to the knowledge cache.
|
||||
pub fn knowledge(&self) -> &KnowledgeCache {
|
||||
&self.knowledge
|
||||
@@ -1024,7 +1173,29 @@ impl HDF5Memory {
|
||||
/// Upsert: if an active entry with the same tags (key) exists, update it in-place.
|
||||
/// Otherwise append a new entry. Use this for key-based memory stores where
|
||||
/// the same key should not create duplicates.
|
||||
/// A `float16` store holds embeddings as IEEE half precision, which has no
|
||||
/// finite value beyond ±65504. Refuse such an embedding rather than
|
||||
/// silently store infinity. (Values that are already infinite or NaN are
|
||||
/// stored as they are, as in an `f32` store.)
|
||||
fn check_embedding(&self, embedding: &[f32]) -> Result<()> {
|
||||
if !self.config.float16 {
|
||||
return Ok(());
|
||||
}
|
||||
let overflow = embedding
|
||||
.iter()
|
||||
.enumerate()
|
||||
.find(|&(_, &v)| v.is_finite() && round_to_f16(v).is_infinite());
|
||||
match overflow {
|
||||
None => Ok(()),
|
||||
Some((i, v)) => Err(MemoryError::InvalidEntry(format!(
|
||||
"embedding[{i}] = {v} is outside the half-precision range (±65504) \
|
||||
of this float16 store"
|
||||
))),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn save_or_update(&mut self, entry: MemoryEntry) -> Result<usize> {
|
||||
self.check_embedding(&entry.embedding)?;
|
||||
if let Some(existing_idx) = self.cache.find_by_tags(&entry.tags) {
|
||||
if let Some(ref mut w) = self.wal {
|
||||
let wal_entry = wal::WalEntry {
|
||||
@@ -1080,6 +1251,7 @@ impl HDF5Memory {
|
||||
|
||||
impl AgentMemory for HDF5Memory {
|
||||
fn save(&mut self, entry: MemoryEntry) -> Result<usize> {
|
||||
self.check_embedding(&entry.embedding)?;
|
||||
if let Some(ref mut w) = self.wal {
|
||||
let wal_entry = wal::WalEntry {
|
||||
entry_type: wal::WalEntryType::Save,
|
||||
@@ -1121,6 +1293,10 @@ impl AgentMemory for HDF5Memory {
|
||||
}
|
||||
|
||||
fn save_batch(&mut self, entries: Vec<MemoryEntry>) -> Result<Vec<usize>> {
|
||||
// All or nothing: check every entry before storing any.
|
||||
for entry in &entries {
|
||||
self.check_embedding(&entry.embedding)?;
|
||||
}
|
||||
let mut indices = Vec::with_capacity(entries.len());
|
||||
for entry in entries {
|
||||
let idx = self.cache.push(
|
||||
@@ -1281,6 +1457,9 @@ impl HDF5Memory {
|
||||
})?;
|
||||
let view = memory_strategy::CacheStoreView::new(&self.cache, &self.knowledge);
|
||||
let output = strat.evaluate(&exchange, &view);
|
||||
for e in &output.entries {
|
||||
self.check_embedding(&e.embedding)?;
|
||||
}
|
||||
for e in &output.entries {
|
||||
self.cache.push(
|
||||
e.chunk.clone(),
|
||||
@@ -1305,6 +1484,35 @@ impl HDF5Memory {
|
||||
}
|
||||
|
||||
impl HDF5Memory {
|
||||
/// Delete many records with a single checkpoint, where
|
||||
/// [`AgentMemory::delete`] checkpoints once per record.
|
||||
///
|
||||
/// All or nothing: if any id is out of range or already deleted (or
|
||||
/// repeated), nothing is deleted and `MemoryError::NotFound` is returned.
|
||||
/// Unlike `delete`, this never auto-compacts, so the records stay in the
|
||||
/// store as tombstones (their indices unchanged) until [`AgentMemory::compact`]
|
||||
/// is called — importers use it to carry over records that were already
|
||||
/// deleted in the source.
|
||||
pub fn delete_batch(&mut self, ids: &[usize]) -> Result<()> {
|
||||
let mut seen = std::collections::HashSet::with_capacity(ids.len());
|
||||
for &id in ids {
|
||||
if self.cache.tombstones.get(id).copied() != Some(0) || !seen.insert(id) {
|
||||
return Err(MemoryError::NotFound(format!(
|
||||
"entry {id} not found or already deleted"
|
||||
)));
|
||||
}
|
||||
}
|
||||
if ids.is_empty() {
|
||||
return Ok(());
|
||||
}
|
||||
for &id in ids {
|
||||
self.cache.mark_deleted(id);
|
||||
self.hnsw_on_delete(id);
|
||||
self.bm25_on_delete(id);
|
||||
}
|
||||
self.flush()
|
||||
}
|
||||
|
||||
pub fn tick_session(&mut self) -> Result<()> {
|
||||
let d = self.config.decay_factor;
|
||||
for w in self.cache.activation_weights.iter_mut() {
|
||||
@@ -1365,6 +1573,16 @@ impl HDF5Memory {
|
||||
let mut promoted = 0;
|
||||
|
||||
for key in candidates {
|
||||
// Check before taking, so a rejected entry stays in the ephemeral
|
||||
// tier rather than being lost.
|
||||
if let Some(emb) = self
|
||||
.ephemeral
|
||||
.as_ref()
|
||||
.and_then(|s| s.get_entry(&key))
|
||||
.and_then(|e| e.embedding.as_deref())
|
||||
{
|
||||
self.check_embedding(emb)?;
|
||||
}
|
||||
let entry = match self
|
||||
.ephemeral
|
||||
.as_mut()
|
||||
@@ -1493,6 +1711,79 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn delete_batch_tombstones_without_compacting() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let path = dir.path().join("test.h5");
|
||||
let mut mem = HDF5Memory::create(make_config(&dir)).unwrap();
|
||||
mem.save_batch(
|
||||
(0..4)
|
||||
.map(|i| make_entry(&format!("record {i}"), &[i as f32, 1.0, 0.0, 0.0]))
|
||||
.collect(),
|
||||
)
|
||||
.unwrap();
|
||||
// 3 of 4 is far past compact_threshold (0.3): delete() would compact.
|
||||
mem.delete_batch(&[0, 1, 3]).unwrap();
|
||||
assert_eq!(mem.count(), 4);
|
||||
assert_eq!(mem.count_active(), 1);
|
||||
drop(mem);
|
||||
|
||||
let mut mem = HDF5Memory::open(&path).unwrap();
|
||||
assert_eq!(mem.cache.tombstones, vec![1, 1, 0, 1]);
|
||||
let hits = mem.hybrid_search(&[0.0, 1.0, 0.0, 0.0], "record", 0.5, 0.5, 10);
|
||||
assert!(
|
||||
hits.iter().all(|r| r.index == 2),
|
||||
"tombstoned record returned"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn delete_batch_is_all_or_nothing() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let mut mem = HDF5Memory::create(make_config(&dir)).unwrap();
|
||||
mem.save_batch(vec![
|
||||
make_entry("a", &[1.0, 0.0, 0.0, 0.0]),
|
||||
make_entry("b", &[0.0, 1.0, 0.0, 0.0]),
|
||||
])
|
||||
.unwrap();
|
||||
for bad in [&[0, 5][..], &[1, 1][..]] {
|
||||
assert!(matches!(
|
||||
mem.delete_batch(bad),
|
||||
Err(MemoryError::NotFound(_))
|
||||
));
|
||||
assert_eq!(mem.count_active(), 2, "{bad:?} deleted something");
|
||||
}
|
||||
mem.delete_batch(&[]).unwrap();
|
||||
assert_eq!(mem.count_active(), 2);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sessions_mut_add_at_keeps_timestamp_across_reopen() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let path = dir.path().join("test.h5");
|
||||
let mut mem = HDF5Memory::create(make_config(&dir)).unwrap();
|
||||
mem.sessions_mut()
|
||||
.add_at("s-old", 2, 7, "discord", "old summary", 1.7e15);
|
||||
mem.flush_wal().unwrap();
|
||||
drop(mem);
|
||||
|
||||
let mem = HDF5Memory::open_read_only(&path).unwrap();
|
||||
let s = mem.sessions();
|
||||
assert_eq!(s.len(), 1);
|
||||
let e = &s.entries[0];
|
||||
assert_eq!(
|
||||
(
|
||||
e.id.as_str(),
|
||||
e.start_idx,
|
||||
e.end_idx,
|
||||
e.channel.as_str(),
|
||||
e.ts
|
||||
),
|
||||
("s-old", 2, 7, "discord", 1.7e15)
|
||||
);
|
||||
assert_eq!(s.summaries[0], "old summary");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn create_new_file() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
|
||||
@@ -1,10 +1,12 @@
|
||||
//! OpenClaw Integration Layer.
|
||||
//! A Markdown-oriented memory backend over [`crate::HDF5Memory`].
|
||||
//!
|
||||
//! Bridge between OpenClaw agent gateway (Markdown + sqlite-vec) and the
|
||||
//! clawhdf5 HDF5-backed memory backend. Provides:
|
||||
//! Named for OpenClaw, whose workspace memory is Markdown, but **not an
|
||||
//! OpenClaw plugin**: nothing here registers with OpenClaw, and the
|
||||
//! integration it was written for never worked (see `docs/openclaw.md`).
|
||||
//! Provides:
|
||||
//!
|
||||
//! - [`MemoryBackend`] — the trait OpenClaw implements against.
|
||||
//! - [`ClawhdfBackend`] — concrete HDF5-backed implementation.
|
||||
//! - [`MemoryBackend`] — search / read back / write / ingest / export.
|
||||
//! - [`ClawhdfBackend`] — the HDF5-backed implementation.
|
||||
//! - [`MarkdownParser`] — splits Markdown into [`MarkdownSection`] records.
|
||||
//! - [`MarkdownExporter`] — renders sections back to Markdown text.
|
||||
|
||||
@@ -13,9 +15,8 @@ use std::path::{Path, PathBuf};
|
||||
use std::time::{SystemTime, UNIX_EPOCH};
|
||||
|
||||
use crate::{
|
||||
AgentMemory, HDF5Memory, MemoryConfig, MemoryEntry,
|
||||
confidence::{ConfidenceConfig, ScoredResult, reject_low_confidence},
|
||||
reranker::{ReRankConfig, RerankInput, rerank},
|
||||
AgentMemory, HDF5Memory, MemoryConfig, MemoryEntry, SearchOptions,
|
||||
confidence::ConfidenceConfig, reranker::ReRankConfig,
|
||||
};
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────────────────
|
||||
@@ -62,7 +63,8 @@ pub struct BackendStats {
|
||||
// MemoryBackend trait
|
||||
// ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
/// Interface that OpenClaw uses to interact with a memory backend.
|
||||
/// A Markdown-oriented memory backend: search, read back by path, write,
|
||||
/// ingest and export.
|
||||
///
|
||||
/// Implementors provide persistent storage, full-text + vector search,
|
||||
/// Markdown ingestion / export, and statistics.
|
||||
@@ -319,7 +321,7 @@ impl MarkdownExporter {
|
||||
///
|
||||
/// # Path mapping
|
||||
///
|
||||
/// OpenClaw addresses memories by file path (e.g. `"memory/user.md"`).
|
||||
/// Memories are addressed by file path (e.g. `"memory/user.md"`).
|
||||
/// Internally every [`MemoryEntry`] stores the originating path as its
|
||||
/// `source_channel`. Section sub-paths are stored as
|
||||
/// `"<path>::<heading>"`.
|
||||
@@ -422,7 +424,7 @@ impl ClawhdfBackend {
|
||||
|
||||
// ── Compaction & Consolidation hooks (7.6) ────────────────────────────
|
||||
|
||||
/// Run a compaction cycle — called by OpenClaw during session compaction.
|
||||
/// Run a compaction cycle (decay, compaction, WAL flush).
|
||||
///
|
||||
/// Sequence:
|
||||
/// 1. `tick_session()` — apply Hebbian decay to all activation weights.
|
||||
@@ -524,71 +526,27 @@ impl ClawhdfBackend {
|
||||
|
||||
impl MemoryBackend for ClawhdfBackend {
|
||||
/// Search using hybrid vector + BM25 retrieval, then re-rank and
|
||||
/// confidence-filter.
|
||||
/// confidence-filter — [`HDF5Memory::search`] with both stages on.
|
||||
fn search(
|
||||
&mut self,
|
||||
query_text: &str,
|
||||
query_embedding: &[f32],
|
||||
k: usize,
|
||||
) -> Vec<MemorySearchResult> {
|
||||
// 1. Hybrid retrieval (vector + BM25, fused by score).
|
||||
let candidates = k.saturating_mul(3).max(10);
|
||||
let raw = self.memory.hybrid_search_with(
|
||||
query_embedding,
|
||||
query_text,
|
||||
crate::hybrid::DEFAULT_FUSION,
|
||||
candidates,
|
||||
);
|
||||
|
||||
if raw.is_empty() {
|
||||
return Vec::new();
|
||||
}
|
||||
|
||||
let now = Self::now_secs();
|
||||
|
||||
// 2. Re-rank using temporal recency, source authority, Hebbian weight.
|
||||
let rerank_inputs: Vec<RerankInput> = raw
|
||||
.iter()
|
||||
.map(|r| RerankInput {
|
||||
index: r.index,
|
||||
timestamp: r.timestamp,
|
||||
source_channel: r.source_channel.clone(),
|
||||
raw_activation: r.activation,
|
||||
relevance: r.score,
|
||||
})
|
||||
.collect();
|
||||
|
||||
let reranked = rerank(&rerank_inputs, &self.rerank_config, now);
|
||||
|
||||
// 3. Confidence rejection.
|
||||
let scored: Vec<ScoredResult> = reranked
|
||||
.iter()
|
||||
.map(|r| ScoredResult {
|
||||
index: r.index,
|
||||
score: r.combined_score,
|
||||
})
|
||||
.collect();
|
||||
|
||||
let confident = reject_low_confidence(&scored, &self.confidence_config);
|
||||
|
||||
// 4. Map back to MemorySearchResult; preserve raw text via index lookup.
|
||||
let raw_by_idx: HashMap<usize, &crate::SearchResult> =
|
||||
raw.iter().map(|r| (r.index, r)).collect();
|
||||
|
||||
confident
|
||||
let options = SearchOptions::new(k)
|
||||
.with_rerank(self.rerank_config)
|
||||
.with_confidence(self.confidence_config.clone())
|
||||
.at_time(Self::now_secs());
|
||||
self.memory
|
||||
.search(query_embedding, query_text, &options)
|
||||
.into_iter()
|
||||
.take(k)
|
||||
.filter_map(|sr| {
|
||||
let r = raw_by_idx.get(&sr.index)?;
|
||||
let path = r.source_channel.clone();
|
||||
Some(MemorySearchResult {
|
||||
text: r.chunk.clone(),
|
||||
score: sr.score,
|
||||
path: path.clone(),
|
||||
.map(|r| MemorySearchResult {
|
||||
text: r.chunk,
|
||||
score: r.score,
|
||||
path: r.source_channel.clone(),
|
||||
line_range: None,
|
||||
timestamp: Some(r.timestamp),
|
||||
source: path,
|
||||
})
|
||||
source: r.source_channel,
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
//!
|
||||
//! Records the origin, authorship, and a content hash of every memory chunk
|
||||
//! so the system can detect *accidental* corruption and trace data lineage.
|
||||
//! The hash is unkeyed (see [`fnv1a_64`]) — this is not a tamper-evidence or
|
||||
//! The hash is unkeyed (FNV-1a) — this is not a tamper-evidence or
|
||||
//! authenticity guarantee.
|
||||
|
||||
use std::collections::HashMap;
|
||||
|
||||
@@ -15,6 +15,9 @@ use crate::session::SessionCache;
|
||||
use crate::wal::WalMark;
|
||||
|
||||
pub const SCHEMA_VERSION: &str = "1.0";
|
||||
/// Writer-version tag stored in `/meta` as `edgehdf5_version`. Kept for file
|
||||
/// compatibility; despite the name it has nothing to do with ZeroClaw, which
|
||||
/// does not use clawhdf5.
|
||||
pub const ZEROCLAW_VERSION: &str = "0.8.0";
|
||||
|
||||
/// `/meta` attributes holding the [`WalMark`] of the WAL prefix already folded
|
||||
@@ -23,6 +26,7 @@ pub const ZEROCLAW_VERSION: &str = "0.8.0";
|
||||
const WAL_APPLIED_LEN_ATTR: &str = "wal_applied_len";
|
||||
const WAL_APPLIED_CRC_ATTR: &str = "wal_applied_crc";
|
||||
const ANN_GENERATION_ATTR: &str = "ann_generation";
|
||||
const SIG_VERSION_ATTR: &str = "sig_version";
|
||||
|
||||
/// Build a complete HDF5 file from the in-memory state.
|
||||
pub fn build_hdf5_file(
|
||||
@@ -46,7 +50,7 @@ pub fn build_hdf5_file_with_mark(
|
||||
) -> Result<Vec<u8>, MemoryError> {
|
||||
let meta = CheckpointMeta {
|
||||
wal_applied,
|
||||
ann_generation: None,
|
||||
..CheckpointMeta::default()
|
||||
};
|
||||
build_hdf5_file_with_meta(config, cache, sessions, knowledge, &meta)
|
||||
}
|
||||
@@ -61,6 +65,10 @@ pub struct CheckpointMeta {
|
||||
/// one left over from another checkpoint can never be attached to records
|
||||
/// it wasn't built from.
|
||||
pub ann_generation: Option<u64>,
|
||||
/// The checkpoint carries an Ed25519 signature (see [`crate::signing`]).
|
||||
/// Read-only: whether a checkpoint is *written* signed is decided by the
|
||||
/// signature passed to [`build_hdf5_file_signed`].
|
||||
pub signed: bool,
|
||||
}
|
||||
|
||||
/// [`build_hdf5_file`] with checkpoint bookkeeping.
|
||||
@@ -70,6 +78,19 @@ pub fn build_hdf5_file_with_meta(
|
||||
sessions: &SessionCache,
|
||||
knowledge: &KnowledgeCache,
|
||||
checkpoint: &CheckpointMeta,
|
||||
) -> Result<Vec<u8>, MemoryError> {
|
||||
build_hdf5_file_signed(config, cache, sessions, knowledge, checkpoint, None)
|
||||
}
|
||||
|
||||
/// [`build_hdf5_file_with_meta`], plus a signed manifest of the contents
|
||||
/// (see [`crate::signing`]).
|
||||
pub fn build_hdf5_file_signed(
|
||||
config: &MemoryConfig,
|
||||
cache: &MemoryCache,
|
||||
sessions: &SessionCache,
|
||||
knowledge: &KnowledgeCache,
|
||||
checkpoint: &CheckpointMeta,
|
||||
signature: Option<&crate::signing::StoredSignature>,
|
||||
) -> Result<Vec<u8>, MemoryError> {
|
||||
let wal_applied = checkpoint.wal_applied;
|
||||
let mut builder = clawhdf5::FileBuilder::new();
|
||||
@@ -108,6 +129,15 @@ pub fn build_hdf5_file_with_meta(
|
||||
"quantized_index",
|
||||
AttrValue::I64(config.quantized_index.into()),
|
||||
);
|
||||
meta.set_attr("hnsw_m", AttrValue::I64(config.hnsw_m as i64));
|
||||
meta.set_attr(
|
||||
"hnsw_ef_construction",
|
||||
AttrValue::I64(config.hnsw_ef_construction as i64),
|
||||
);
|
||||
meta.set_attr(
|
||||
"hnsw_ef_search",
|
||||
AttrValue::I64(config.hnsw_ef_search as i64),
|
||||
);
|
||||
meta.set_attr(
|
||||
"edgehdf5_version",
|
||||
AttrValue::String(ZEROCLAW_VERSION.into()),
|
||||
@@ -121,11 +151,42 @@ pub fn build_hdf5_file_with_meta(
|
||||
// round trip through every reader.
|
||||
meta.set_attr(ANN_GENERATION_ATTR, AttrValue::I64(generation as i64));
|
||||
}
|
||||
if let Some(sig) = signature {
|
||||
use crate::signing::to_hex;
|
||||
let m = &sig.manifest;
|
||||
meta.set_attr(
|
||||
SIG_VERSION_ATTR,
|
||||
AttrValue::I64(crate::signing::MANIFEST_VERSION),
|
||||
);
|
||||
meta.set_attr("sig_algorithm", AttrValue::String("ed25519".into()));
|
||||
meta.set_attr("sig_public_key", AttrValue::String(to_hex(&sig.public_key)));
|
||||
meta.set_attr("sig_signature", AttrValue::String(to_hex(&sig.signature)));
|
||||
meta.set_attr("sig_record_count", AttrValue::I64(m.record_count as i64));
|
||||
meta.set_attr(
|
||||
"sig_records_root",
|
||||
AttrValue::String(to_hex(&m.records_root)),
|
||||
);
|
||||
meta.set_attr("sig_settings", AttrValue::String(to_hex(&m.settings)));
|
||||
meta.set_attr("sig_sessions", AttrValue::String(to_hex(&m.sessions)));
|
||||
meta.set_attr("sig_graph", AttrValue::String(to_hex(&m.graph)));
|
||||
}
|
||||
// Need at least one dataset in the group for it to be a proper group
|
||||
meta.create_dataset("_marker").with_u8_data(&[1]).compact();
|
||||
let finished_meta = meta.finish();
|
||||
builder.add_group(finished_meta);
|
||||
|
||||
// /integrity: the signed per-record hashes, so verification can say
|
||||
// which records changed.
|
||||
if let Some(sig) = signature {
|
||||
let mut group = builder.create_group("integrity");
|
||||
let flat: Vec<u8> = sig.record_hashes.iter().flatten().copied().collect();
|
||||
group
|
||||
.create_dataset("record_hashes")
|
||||
.with_u8_data(&flat)
|
||||
.with_shape(&[sig.record_hashes.len() as u64, 32]);
|
||||
builder.add_group(group.finish());
|
||||
}
|
||||
|
||||
// /memory group
|
||||
build_memory_group(&mut builder, config, cache)?;
|
||||
|
||||
@@ -150,20 +211,27 @@ fn build_memory_group(
|
||||
// chunks: fixed-length string array
|
||||
write_string_dataset(&mut group, "chunks", &cache.chunks);
|
||||
|
||||
// embeddings: f32 [N x D]
|
||||
// embeddings: [N x D], f32 — or IEEE half precision for a `float16`
|
||||
// store. The cache already holds half-rounded values then, so this
|
||||
// conversion is exact and a reopened store sees the same numbers.
|
||||
let n = cache.embeddings.len() as u64;
|
||||
let d = cache.embedding_dim as u64;
|
||||
let flat = cache.flat_embeddings();
|
||||
{
|
||||
let ds = group
|
||||
.create_dataset("embeddings")
|
||||
.with_f32_data(flat)
|
||||
.with_shape(&[n, d]);
|
||||
let ds = group.create_dataset("embeddings");
|
||||
let elem_bytes: u64 = if config.float16 {
|
||||
ds.with_f16_data(flat);
|
||||
2
|
||||
} else {
|
||||
ds.with_f32_data(flat);
|
||||
4
|
||||
};
|
||||
ds.with_shape(&[n, d]);
|
||||
|
||||
// Chunk size tuning: target ~256KB per chunk for optimal I/O
|
||||
if n > 0 && d > 0 {
|
||||
let target_chunk_bytes: u64 = 256 * 1024;
|
||||
let rows_per_chunk = (target_chunk_bytes / (d * 4)).max(1).min(n);
|
||||
let rows_per_chunk = (target_chunk_bytes / (d * elem_bytes)).max(1).min(n);
|
||||
ds.with_chunks(&[rows_per_chunk, d]);
|
||||
|
||||
// Compression. Shuffle is applied automatically (auto-shuffle
|
||||
@@ -409,10 +477,34 @@ fn write_string_dataset(
|
||||
}
|
||||
}
|
||||
|
||||
/// `/meta`'s attributes, failing if any of them cannot be read.
|
||||
///
|
||||
/// `Group::attrs` leaves out an attribute it cannot decode. For the store's
|
||||
/// settings that would silently fall back to defaults (e.g. `float16`, the
|
||||
/// WAL mark), so an unreadable attribute is an error here, as it was before
|
||||
/// `attrs` became tolerant.
|
||||
fn meta_attrs(
|
||||
file: &clawhdf5::File,
|
||||
) -> Result<std::collections::HashMap<String, AttrValue>, MemoryError> {
|
||||
let meta = file
|
||||
.group("meta")
|
||||
.map_err(|e| MemoryError::Schema(format!("missing /meta group: {e}")))?;
|
||||
let (attrs, errors) = meta
|
||||
.attrs_with_errors()
|
||||
.map_err(|e| MemoryError::Schema(format!("cannot read /meta attrs: {e}")))?;
|
||||
if let Some(e) = errors.first() {
|
||||
return Err(MemoryError::Schema(format!(
|
||||
"cannot read /meta attrs: {} unreadable, first: {e}",
|
||||
errors.len()
|
||||
)));
|
||||
}
|
||||
Ok(attrs)
|
||||
}
|
||||
|
||||
/// Validate an HDF5 file has the correct schema and load all data.
|
||||
/// Read the checkpoint's [`WalMark`] from `/meta`, if it has one.
|
||||
pub fn read_wal_mark(file: &clawhdf5::File) -> Option<WalMark> {
|
||||
let attrs = file.group("meta").ok()?.attrs().ok()?;
|
||||
let attrs = meta_attrs(file).ok()?;
|
||||
let len = match attrs.get(WAL_APPLIED_LEN_ATTR)? {
|
||||
AttrValue::I64(v) => u64::try_from(*v).ok()?,
|
||||
_ => return None,
|
||||
@@ -424,19 +516,75 @@ pub fn read_wal_mark(file: &clawhdf5::File) -> Option<WalMark> {
|
||||
Some(WalMark { len, crc })
|
||||
}
|
||||
|
||||
/// Read a checkpoint's signature, if it has one. A signature whose
|
||||
/// attributes are present but malformed is an error, not "unsigned".
|
||||
pub fn read_signature(
|
||||
file: &clawhdf5::File,
|
||||
) -> Result<Option<crate::signing::StoredSignature>, MemoryError> {
|
||||
use crate::signing::{Manifest, StoredSignature, from_hex};
|
||||
let attrs = meta_attrs(file)?;
|
||||
let version = match attrs.get(SIG_VERSION_ATTR) {
|
||||
None => return Ok(None),
|
||||
Some(AttrValue::I64(v)) => *v,
|
||||
Some(_) => return Err(MemoryError::Schema("malformed sig_version".into())),
|
||||
};
|
||||
if version != crate::signing::MANIFEST_VERSION {
|
||||
return Err(MemoryError::Schema(format!(
|
||||
"unsupported signature version {version}"
|
||||
)));
|
||||
}
|
||||
fn hex<const N: usize>(
|
||||
attrs: &std::collections::HashMap<String, AttrValue>,
|
||||
name: &str,
|
||||
) -> Result<[u8; N], MemoryError> {
|
||||
match attrs.get(name) {
|
||||
Some(AttrValue::String(s)) => from_hex::<N>(s),
|
||||
_ => None,
|
||||
}
|
||||
.ok_or_else(|| MemoryError::Schema(format!("malformed or missing {name}")))
|
||||
}
|
||||
let record_count = match attrs.get("sig_record_count") {
|
||||
Some(AttrValue::I64(v)) if *v >= 0 => *v as u64,
|
||||
_ => return Err(MemoryError::Schema("malformed sig_record_count".into())),
|
||||
};
|
||||
let group = file
|
||||
.group("integrity")
|
||||
.map_err(|e| MemoryError::Schema(format!("signed checkpoint without /integrity: {e}")))?;
|
||||
let flat = read_u8_dataset(&group, "record_hashes")?;
|
||||
if flat.len() % 32 != 0 {
|
||||
return Err(MemoryError::Schema(
|
||||
"/integrity/record_hashes is not a whole number of hashes".into(),
|
||||
));
|
||||
}
|
||||
let record_hashes = flat.as_chunks::<32>().0.to_vec();
|
||||
Ok(Some(StoredSignature {
|
||||
manifest: Manifest {
|
||||
record_count,
|
||||
records_root: hex::<32>(&attrs, "sig_records_root")?,
|
||||
settings: hex::<32>(&attrs, "sig_settings")?,
|
||||
sessions: hex::<32>(&attrs, "sig_sessions")?,
|
||||
graph: hex::<32>(&attrs, "sig_graph")?,
|
||||
},
|
||||
record_hashes,
|
||||
public_key: hex::<32>(&attrs, "sig_public_key")?,
|
||||
signature: hex::<64>(&attrs, "sig_signature")?,
|
||||
}))
|
||||
}
|
||||
|
||||
/// Read the checkpoint bookkeeping from `/meta`.
|
||||
pub fn read_checkpoint_meta(file: &clawhdf5::File) -> CheckpointMeta {
|
||||
let ann_generation = file
|
||||
.group("meta")
|
||||
let ann_generation =
|
||||
meta_attrs(file)
|
||||
.ok()
|
||||
.and_then(|g| g.attrs().ok())
|
||||
.and_then(|attrs| match attrs.get(ANN_GENERATION_ATTR) {
|
||||
Some(AttrValue::I64(v)) => Some(*v as u64),
|
||||
_ => None,
|
||||
});
|
||||
let signed = meta_attrs(file).is_ok_and(|attrs| attrs.contains_key(SIG_VERSION_ATTR));
|
||||
CheckpointMeta {
|
||||
wal_applied: read_wal_mark(file),
|
||||
ann_generation,
|
||||
signed,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -444,12 +592,7 @@ pub fn validate_and_load(
|
||||
file: &clawhdf5::File,
|
||||
) -> Result<(MemoryConfig, MemoryCache, SessionCache, KnowledgeCache), MemoryError> {
|
||||
// Read /meta group attributes
|
||||
let meta = file
|
||||
.group("meta")
|
||||
.map_err(|e| MemoryError::Schema(format!("missing /meta group: {e}")))?;
|
||||
let attrs = meta
|
||||
.attrs()
|
||||
.map_err(|e| MemoryError::Schema(format!("cannot read /meta attrs: {e}")))?;
|
||||
let attrs = meta_attrs(file)?;
|
||||
|
||||
let schema_version = match attrs.get("schema_version") {
|
||||
Some(AttrValue::String(s)) => s.clone(),
|
||||
@@ -488,11 +631,32 @@ pub fn validate_and_load(
|
||||
wal_max_entries: optional_i64_attr(&attrs, "wal_max_entries")
|
||||
.and_then(|v| usize::try_from(v).ok())
|
||||
.unwrap_or(500),
|
||||
// `false`, not the new-store default: a store written before this
|
||||
// setting existed was built with an f32 index, and reopening it must
|
||||
// not silently change that.
|
||||
quantized_index: optional_bool_attr(&attrs, "quantized_index", false),
|
||||
hnsw_m: optional_i64_attr(&attrs, "hnsw_m")
|
||||
.and_then(|v| usize::try_from(v).ok())
|
||||
.unwrap_or(16),
|
||||
hnsw_ef_construction: optional_i64_attr(&attrs, "hnsw_ef_construction")
|
||||
.and_then(|v| usize::try_from(v).ok())
|
||||
.unwrap_or(64),
|
||||
hnsw_ef_search: optional_i64_attr(&attrs, "hnsw_ef_search")
|
||||
.and_then(|v| usize::try_from(v).ok())
|
||||
.unwrap_or(0),
|
||||
};
|
||||
|
||||
// Load /memory group
|
||||
let memory_cache = load_memory_group(file, embedding_dim)?;
|
||||
let mut memory_cache = load_memory_group(file, embedding_dim)?;
|
||||
// A float16 store's cache holds half-rounded embeddings. Embeddings read
|
||||
// from an f16 dataset already are; a float16 store whose last checkpoint
|
||||
// predates half-precision storage is still f32 on disk and is rounded
|
||||
// here.
|
||||
if config.float16 && embeddings_are_f16(file) {
|
||||
memory_cache.half_precision = true;
|
||||
} else {
|
||||
memory_cache.set_half_precision(config.float16);
|
||||
}
|
||||
|
||||
// Load /sessions group
|
||||
let session_cache = load_sessions_group(file)?;
|
||||
@@ -735,6 +899,13 @@ fn read_string_dataset_from_group(
|
||||
.map_err(|e| MemoryError::Hdf5(format!("cannot read strings from {name}: {e}")))
|
||||
}
|
||||
|
||||
/// Whether `/memory/embeddings` is stored as IEEE half precision.
|
||||
fn embeddings_are_f16(file: &clawhdf5::File) -> bool {
|
||||
file.dataset("memory/embeddings")
|
||||
.and_then(|ds| ds.dtype())
|
||||
.is_ok_and(|dt| matches!(dt, clawhdf5::DType::Other(ref s) if s == "float16"))
|
||||
}
|
||||
|
||||
fn read_f32_dataset(group: &clawhdf5::Group<'_>, name: &str) -> Result<Vec<f32>, MemoryError> {
|
||||
let ds = group
|
||||
.dataset(name)
|
||||
|
||||
@@ -2,18 +2,107 @@
|
||||
|
||||
use std::path::Path;
|
||||
|
||||
use std::collections::HashSet;
|
||||
|
||||
use crate::bm25;
|
||||
use crate::confidence::{ConfidenceConfig, ScoredResult, reject_low_confidence};
|
||||
use crate::hybrid;
|
||||
use crate::reranker::{ReRankConfig, RerankInput, rerank};
|
||||
use crate::{HDF5Memory, MAX_ACTIVATION_WEIGHT, MemoryError, Result, SearchResult};
|
||||
|
||||
/// Options for [`HDF5Memory::search`].
|
||||
///
|
||||
/// [`SearchOptions::new`] is plain hybrid search with the tuned default
|
||||
/// fusion — the same as `hybrid_search_with(.., hybrid::DEFAULT_FUSION, k)`.
|
||||
/// Every stage beyond that is opt-in.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct SearchOptions {
|
||||
/// Number of results to return.
|
||||
pub k: usize,
|
||||
/// How the vector and keyword stages are combined.
|
||||
pub fusion: hybrid::Fusion,
|
||||
/// Only consider records whose `source_channel` is one of these. The
|
||||
/// filter applies *before* ranking, so a filtered search still returns up
|
||||
/// to `k` results and scores are normalised over the records it can
|
||||
/// return. `None` searches everything; an empty list matches nothing.
|
||||
pub source_channels: Option<Vec<String>>,
|
||||
/// Re-rank a candidate pool by retrieval relevance, recency, source
|
||||
/// authority and activation — the pipeline the OpenClaw backend runs.
|
||||
pub rerank: Option<ReRankConfig>,
|
||||
/// Candidates retrieved for re-ranking; 0 means `max(3k, 10)`.
|
||||
pub rerank_pool: usize,
|
||||
/// Drop low-confidence results (after re-ranking, when that is on).
|
||||
pub confidence: Option<ConfidenceConfig>,
|
||||
/// The time recency is measured from, in seconds since the epoch.
|
||||
/// `None` uses the system clock.
|
||||
pub now: Option<f64>,
|
||||
}
|
||||
|
||||
impl SearchOptions {
|
||||
pub fn new(k: usize) -> Self {
|
||||
Self {
|
||||
k,
|
||||
fusion: hybrid::DEFAULT_FUSION,
|
||||
source_channels: None,
|
||||
rerank: None,
|
||||
rerank_pool: 0,
|
||||
confidence: None,
|
||||
now: None,
|
||||
}
|
||||
}
|
||||
|
||||
pub fn with_fusion(mut self, fusion: hybrid::Fusion) -> Self {
|
||||
self.fusion = fusion;
|
||||
self
|
||||
}
|
||||
|
||||
/// Search only records from these source channels.
|
||||
pub fn with_sources<S: Into<String>>(mut self, channels: impl IntoIterator<Item = S>) -> Self {
|
||||
self.source_channels = Some(channels.into_iter().map(Into::into).collect());
|
||||
self
|
||||
}
|
||||
|
||||
pub fn with_rerank(mut self, config: ReRankConfig) -> Self {
|
||||
self.rerank = Some(config);
|
||||
self
|
||||
}
|
||||
|
||||
pub fn with_confidence(mut self, config: ConfidenceConfig) -> Self {
|
||||
self.confidence = Some(config);
|
||||
self
|
||||
}
|
||||
|
||||
/// Measure recency from `now` (seconds since the epoch) instead of the
|
||||
/// system clock — for reproducible results and tests.
|
||||
pub fn at_time(mut self, now: f64) -> Self {
|
||||
self.now = Some(now);
|
||||
self
|
||||
}
|
||||
}
|
||||
|
||||
impl Default for SearchOptions {
|
||||
fn default() -> Self {
|
||||
Self::new(10)
|
||||
}
|
||||
}
|
||||
|
||||
impl HDF5Memory {
|
||||
/// Vector + keyword scoring stage of [`HDF5Memory::hybrid_search`].
|
||||
/// Vector + keyword scoring stage of [`HDF5Memory::search`].
|
||||
///
|
||||
/// Without the `hnsw` feature this is a full linear cosine scan (the exact
|
||||
/// previous behaviour, also used as the correctness oracle in tests). With
|
||||
/// `hnsw` enabled and an index available, the vector candidates come from an
|
||||
/// approximate-nearest-neighbour search over an over-fetched pool, then merge
|
||||
/// with BM25 via the shared [`hybrid::merge_vector_keyword`].
|
||||
///
|
||||
/// `exclude`, when given, marks records that must not be returned (1 =
|
||||
/// excluded; it covers tombstones too). The index is over-fetched in
|
||||
/// proportion to how much the mask removes. Surfacing `pool` candidates
|
||||
/// costs the index roughly `pool × M` distance evaluations, while an exact
|
||||
/// scan of the allowed records costs one each — so whenever that scan is
|
||||
/// the cheaper of the two it is used instead, and it is also the fallback
|
||||
/// if the pool comes back with too few allowed hits (the allowed records
|
||||
/// sit away from the query). A filtered search never comes back short.
|
||||
#[cfg(feature = "hnsw")]
|
||||
fn vector_keyword_search(
|
||||
&mut self,
|
||||
@@ -22,14 +111,32 @@ impl HDF5Memory {
|
||||
bm25: &bm25::BM25Index,
|
||||
fusion: hybrid::Fusion,
|
||||
k: usize,
|
||||
exclude: Option<&[u8]>,
|
||||
) -> Vec<(usize, f32)> {
|
||||
self.ensure_hnsw_fresh();
|
||||
let n = self.cache.len();
|
||||
// Over-fetch so the merge sees a useful vector pool. `ef` is
|
||||
// configurable, but the pool the fusion stage sees is not tied to it:
|
||||
// a caller lowering `ef` for speed should not silently narrow what
|
||||
// fusion has to work with.
|
||||
let mut pool = (k * 8).max(64);
|
||||
let mut allowed = n;
|
||||
if let Some(ex) = exclude {
|
||||
allowed = ex.iter().filter(|&&e| e == 0).count();
|
||||
if allowed == 0 {
|
||||
return Vec::new();
|
||||
}
|
||||
// Expect `pool` allowed hits if the filter is independent of the
|
||||
// query's neighbourhood.
|
||||
pool = pool.saturating_mul(n).div_ceil(allowed);
|
||||
if allowed <= pool.saturating_mul(self.hnsw_m()) {
|
||||
return self.exact_masked_search(query_embedding, query_text, bm25, fusion, k, ex);
|
||||
}
|
||||
}
|
||||
match self.hnsw.as_ref() {
|
||||
Some(index) if !index.is_empty() && index.dimension() == query_embedding.len() => {
|
||||
// Over-fetch so the merge sees a useful vector pool; cosine
|
||||
// distance from the index converts back to similarity (1 - d).
|
||||
let pool = (k * 8).max(64);
|
||||
let candidates = index.search(query_embedding, pool, pool);
|
||||
let ef = self.hnsw_ef_search(k).max(pool);
|
||||
let candidates = index.search(query_embedding, pool, ef);
|
||||
// A quantised index returns approximate distances, and no
|
||||
// amount of `ef` fixes that — the loss is in the distances,
|
||||
// not the graph. Re-score the pool against the cache's exact
|
||||
@@ -38,6 +145,7 @@ impl HDF5Memory {
|
||||
let exact = index.storage() == clawhdf5_ann::Storage::Int8;
|
||||
let vec_scores: Vec<(usize, f32)> = candidates
|
||||
.into_iter()
|
||||
.filter(|(id, _)| exclude.is_none_or(|ex| ex[*id] == 0))
|
||||
.map(|(id, dist)| {
|
||||
let score = if exact {
|
||||
crate::vector_search::cosine_similarity(
|
||||
@@ -52,10 +160,54 @@ impl HDF5Memory {
|
||||
.collect();
|
||||
// Fusion normalises over every keyword match, so it needs all
|
||||
// the scores — but not ranked.
|
||||
let kw_scores = bm25.scores(query_text);
|
||||
let mut kw_scores = bm25.scores(query_text);
|
||||
if let Some(ex) = exclude {
|
||||
if vec_scores.len() < k.min(allowed) {
|
||||
// The allowed records are not where the index looked.
|
||||
return self.exact_masked_search(
|
||||
query_embedding,
|
||||
query_text,
|
||||
bm25,
|
||||
fusion,
|
||||
k,
|
||||
ex,
|
||||
);
|
||||
}
|
||||
kw_scores.retain(|(id, _)| ex[*id] == 0);
|
||||
}
|
||||
hybrid::fuse(vec_scores, kw_scores, fusion, k)
|
||||
}
|
||||
_ => hybrid::hybrid_search_fused(
|
||||
_ => match exclude {
|
||||
Some(ex) => {
|
||||
self.exact_masked_search(query_embedding, query_text, bm25, fusion, k, ex)
|
||||
}
|
||||
None => hybrid::hybrid_search_fused(
|
||||
query_embedding,
|
||||
query_text,
|
||||
&self.cache.embeddings,
|
||||
&self.cache.chunks,
|
||||
&self.cache.tombstones,
|
||||
bm25,
|
||||
fusion,
|
||||
k,
|
||||
),
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(not(feature = "hnsw"))]
|
||||
fn vector_keyword_search(
|
||||
&mut self,
|
||||
query_embedding: &[f32],
|
||||
query_text: &str,
|
||||
bm25: &bm25::BM25Index,
|
||||
fusion: hybrid::Fusion,
|
||||
k: usize,
|
||||
exclude: Option<&[u8]>,
|
||||
) -> Vec<(usize, f32)> {
|
||||
match exclude {
|
||||
Some(ex) => self.exact_masked_search(query_embedding, query_text, bm25, fusion, k, ex),
|
||||
None => hybrid::hybrid_search_fused(
|
||||
query_embedding,
|
||||
query_text,
|
||||
&self.cache.embeddings,
|
||||
@@ -68,25 +220,33 @@ impl HDF5Memory {
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(not(feature = "hnsw"))]
|
||||
fn vector_keyword_search(
|
||||
&mut self,
|
||||
/// Exact hybrid search over the records `exclude` leaves (0 = allowed).
|
||||
fn exact_masked_search(
|
||||
&self,
|
||||
query_embedding: &[f32],
|
||||
query_text: &str,
|
||||
bm25: &bm25::BM25Index,
|
||||
fusion: hybrid::Fusion,
|
||||
k: usize,
|
||||
exclude: &[u8],
|
||||
) -> Vec<(usize, f32)> {
|
||||
hybrid::hybrid_search_fused(
|
||||
query_embedding,
|
||||
query_text,
|
||||
&self.cache.embeddings,
|
||||
&self.cache.chunks,
|
||||
&self.cache.tombstones,
|
||||
bm25,
|
||||
fusion,
|
||||
k,
|
||||
)
|
||||
let vec_scores =
|
||||
hybrid::exact_vector_scores(query_embedding, &self.cache.embeddings, exclude);
|
||||
let mut kw_scores = bm25.scores(query_text);
|
||||
kw_scores.retain(|(id, _)| exclude.get(*id) == Some(&0));
|
||||
hybrid::fuse(vec_scores, kw_scores, fusion, k)
|
||||
}
|
||||
|
||||
/// The exclusion mask for a source-channel filter: 1 for a tombstoned
|
||||
/// record or one from a channel not in `channels`.
|
||||
fn source_mask(&self, channels: &[String]) -> Vec<u8> {
|
||||
let allowed: HashSet<&str> = channels.iter().map(String::as_str).collect();
|
||||
self.cache
|
||||
.source_channels
|
||||
.iter()
|
||||
.zip(&self.cache.tombstones)
|
||||
.map(|(ch, &t)| u8::from(t != 0 || !allowed.contains(ch.as_str())))
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Perform hybrid search combining cosine vector similarity and BM25 keyword search.
|
||||
@@ -121,12 +281,53 @@ impl HDF5Memory {
|
||||
fusion: hybrid::Fusion,
|
||||
k: usize,
|
||||
) -> Vec<SearchResult> {
|
||||
self.search(
|
||||
query_embedding,
|
||||
query_text,
|
||||
&SearchOptions::new(k).with_fusion(fusion),
|
||||
)
|
||||
}
|
||||
|
||||
/// Hybrid search with optional source filtering, re-ranking and
|
||||
/// confidence rejection — see [`SearchOptions`].
|
||||
///
|
||||
/// Stages, in order: vector + keyword retrieval over the records the
|
||||
/// source filter allows; fusion; scaling by Hebbian activation; re-ranking
|
||||
/// (if on) of a `rerank_pool` of candidates; confidence rejection (if on);
|
||||
/// the top `k`. The records returned with a positive score get their
|
||||
/// Hebbian boost.
|
||||
pub fn search(
|
||||
&mut self,
|
||||
query_embedding: &[f32],
|
||||
query_text: &str,
|
||||
options: &SearchOptions,
|
||||
) -> Vec<SearchResult> {
|
||||
let k = options.k;
|
||||
let fetch = match options.rerank {
|
||||
Some(_) if options.rerank_pool > 0 => options.rerank_pool.max(k),
|
||||
Some(_) => k.saturating_mul(3).max(10),
|
||||
None => k,
|
||||
};
|
||||
let exclude = options
|
||||
.source_channels
|
||||
.as_deref()
|
||||
.map(|channels| self.source_mask(channels));
|
||||
|
||||
// The keyword index lives for the life of the store and is updated
|
||||
// incrementally. Take it out for the duration of the call so the
|
||||
// vector stage can borrow `self` mutably, then put it back.
|
||||
self.ensure_bm25_fresh();
|
||||
let bm25 = self.bm25.take().expect("ensure_bm25_fresh leaves an index");
|
||||
let scored = self.vector_keyword_search(query_embedding, query_text, &bm25, fusion, k);
|
||||
let scored = self.vector_keyword_search(
|
||||
query_embedding,
|
||||
query_text,
|
||||
&bm25,
|
||||
options.fusion,
|
||||
fetch,
|
||||
exclude.as_deref(),
|
||||
);
|
||||
self.bm25 = Some(bm25);
|
||||
|
||||
let mut results: Vec<SearchResult> = scored
|
||||
.into_iter()
|
||||
.map(|(idx, score)| {
|
||||
@@ -150,6 +351,25 @@ impl HDF5Memory {
|
||||
.then(a.index.cmp(&b.index))
|
||||
});
|
||||
|
||||
if let Some(config) = &options.rerank {
|
||||
results = Self::rerank_results(results, config, options.now);
|
||||
}
|
||||
if let Some(config) = &options.confidence {
|
||||
let scored: Vec<ScoredResult> = results
|
||||
.iter()
|
||||
.map(|r| ScoredResult {
|
||||
index: r.index,
|
||||
score: r.score,
|
||||
})
|
||||
.collect();
|
||||
let keep: HashSet<usize> = reject_low_confidence(&scored, config)
|
||||
.into_iter()
|
||||
.map(|r| r.index)
|
||||
.collect();
|
||||
results.retain(|r| keep.contains(&r.index));
|
||||
}
|
||||
results.truncate(k);
|
||||
|
||||
// Only reinforce records that actually matched. When fewer than `k`
|
||||
// records are relevant, the rest of the list is zero-score filler;
|
||||
// boosting it would teach the store that arbitrary records are
|
||||
@@ -160,11 +380,45 @@ impl HDF5Memory {
|
||||
.map(|r| r.index)
|
||||
.collect();
|
||||
self.apply_hebbian_boost(&hit_indices);
|
||||
self.bm25 = Some(bm25);
|
||||
|
||||
results
|
||||
}
|
||||
|
||||
/// Reorder by the re-ranker's combined score, which also becomes each
|
||||
/// result's `score`.
|
||||
fn rerank_results(
|
||||
results: Vec<SearchResult>,
|
||||
config: &ReRankConfig,
|
||||
now: Option<f64>,
|
||||
) -> Vec<SearchResult> {
|
||||
let now = now.unwrap_or_else(|| {
|
||||
std::time::SystemTime::now()
|
||||
.duration_since(std::time::UNIX_EPOCH)
|
||||
.map(|d| d.as_secs_f64())
|
||||
.unwrap_or(0.0)
|
||||
});
|
||||
let inputs: Vec<RerankInput> = results
|
||||
.iter()
|
||||
.map(|r| RerankInput {
|
||||
index: r.index,
|
||||
timestamp: r.timestamp,
|
||||
source_channel: r.source_channel.clone(),
|
||||
raw_activation: r.activation,
|
||||
relevance: r.score,
|
||||
})
|
||||
.collect();
|
||||
let mut by_index: std::collections::HashMap<usize, SearchResult> =
|
||||
results.into_iter().map(|r| (r.index, r)).collect();
|
||||
rerank(&inputs, config, now)
|
||||
.into_iter()
|
||||
.filter_map(|rr| {
|
||||
let mut r = by_index.remove(&rr.index)?;
|
||||
r.score = rr.combined_score;
|
||||
Some(r)
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Reinforce the records a query returned. The new weights are persisted by
|
||||
/// the next checkpoint (any write that flushes, `flush_wal`, or drop) — not
|
||||
/// by rewriting the whole store inside the query, which is what made
|
||||
|
||||
@@ -33,7 +33,7 @@ impl SessionCache {
|
||||
self.entries.is_empty()
|
||||
}
|
||||
|
||||
/// Add a new session with its summary.
|
||||
/// Add a new session with its summary, timestamped now.
|
||||
pub fn add(
|
||||
&mut self,
|
||||
id: &str,
|
||||
@@ -47,6 +47,21 @@ impl SessionCache {
|
||||
.unwrap_or_default()
|
||||
.as_secs_f64()
|
||||
* 1_000_000.0; // microseconds
|
||||
self.add_at(id, start_idx, end_idx, channel, summary, ts);
|
||||
}
|
||||
|
||||
/// Add a session with an explicit timestamp (Unix **microseconds**, the
|
||||
/// unit [`SessionEntry::ts`] uses) — for importers carrying sessions over
|
||||
/// from another store, whose original time should be kept.
|
||||
pub fn add_at(
|
||||
&mut self,
|
||||
id: &str,
|
||||
start_idx: usize,
|
||||
end_idx: usize,
|
||||
channel: &str,
|
||||
summary: &str,
|
||||
ts: f64,
|
||||
) {
|
||||
self.entries.push(SessionEntry {
|
||||
id: id.to_string(),
|
||||
start_idx: start_idx as u64,
|
||||
|
||||
@@ -0,0 +1,419 @@
|
||||
//! Ed25519-signed checkpoints.
|
||||
//!
|
||||
//! When a signing key is set ([`crate::HDF5Memory::set_signing_key`]), every
|
||||
//! checkpoint writes a signed manifest of the store: a SHA-256 per memory
|
||||
//! record rolled into a Merkle root, plus hashes of the store's settings, its
|
||||
//! sessions and its knowledge graph. [`verify_store`] recomputes all of it from
|
||||
//! the file and checks the signature against a public key the caller trusts,
|
||||
//! so any change to the checkpointed file — a record's text or embedding, a
|
||||
//! setting, a session, a graph edge, made through this crate or any other HDF5
|
||||
//! tool — is detected, and the per-record hashes say which records changed.
|
||||
//!
|
||||
//! What it does not cover: saves still only in the WAL (made since the last
|
||||
//! checkpoint). [`VerifyReport::wal_entries_unsigned`] counts them.
|
||||
//!
|
||||
//! The hashes cover exactly what the file persists, in the form the loader
|
||||
//! returns it, so a store verifies after any number of reopen/checkpoint
|
||||
//! cycles. Derived data (L2 norms, the vector index) is not covered; it is
|
||||
//! recomputed from covered data.
|
||||
|
||||
use ed25519_dalek::{Signature, Signer, Verifier};
|
||||
pub use ed25519_dalek::{SigningKey, VerifyingKey};
|
||||
use sha2::{Digest, Sha256};
|
||||
|
||||
use crate::MemoryConfig;
|
||||
use crate::cache::MemoryCache;
|
||||
use crate::knowledge::KnowledgeCache;
|
||||
use crate::session::SessionCache;
|
||||
use crate::wal::WalMark;
|
||||
|
||||
/// Version of the manifest encoding; part of what is signed.
|
||||
pub const MANIFEST_VERSION: i64 = 1;
|
||||
|
||||
type Hash = [u8; 32];
|
||||
|
||||
/// The hashes a signature covers.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct Manifest {
|
||||
pub record_count: u64,
|
||||
/// Merkle root over the per-record hashes.
|
||||
pub records_root: Hash,
|
||||
/// Settings persisted in `/meta`, plus the checkpoint's WAL mark.
|
||||
pub settings: Hash,
|
||||
pub sessions: Hash,
|
||||
pub graph: Hash,
|
||||
}
|
||||
|
||||
impl Manifest {
|
||||
/// The exact bytes that are signed.
|
||||
pub fn signed_bytes(&self) -> Vec<u8> {
|
||||
let mut m = Vec::with_capacity(160);
|
||||
m.extend_from_slice(b"clawhdf5-agent signed checkpoint\0");
|
||||
m.extend_from_slice(&MANIFEST_VERSION.to_le_bytes());
|
||||
m.extend_from_slice(&self.record_count.to_le_bytes());
|
||||
m.extend_from_slice(&self.records_root);
|
||||
m.extend_from_slice(&self.settings);
|
||||
m.extend_from_slice(&self.sessions);
|
||||
m.extend_from_slice(&self.graph);
|
||||
m
|
||||
}
|
||||
}
|
||||
|
||||
/// A signature as stored in a checkpoint.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct StoredSignature {
|
||||
pub manifest: Manifest,
|
||||
pub record_hashes: Vec<Hash>,
|
||||
pub public_key: [u8; 32],
|
||||
pub signature: [u8; 64],
|
||||
}
|
||||
|
||||
/// Build the manifest (and per-record hashes) for the state about to be
|
||||
/// checkpointed, and sign it.
|
||||
pub fn sign(
|
||||
key: &SigningKey,
|
||||
config: &MemoryConfig,
|
||||
cache: &MemoryCache,
|
||||
sessions: &SessionCache,
|
||||
knowledge: &KnowledgeCache,
|
||||
wal_applied: Option<WalMark>,
|
||||
) -> StoredSignature {
|
||||
let (manifest, record_hashes) = manifest(config, cache, sessions, knowledge, wal_applied);
|
||||
let signature = key.sign(&manifest.signed_bytes()).to_bytes();
|
||||
StoredSignature {
|
||||
manifest,
|
||||
record_hashes,
|
||||
public_key: key.verifying_key().to_bytes(),
|
||||
signature,
|
||||
}
|
||||
}
|
||||
|
||||
/// Compute the manifest of a store's state.
|
||||
pub fn manifest(
|
||||
config: &MemoryConfig,
|
||||
cache: &MemoryCache,
|
||||
sessions: &SessionCache,
|
||||
knowledge: &KnowledgeCache,
|
||||
wal_applied: Option<WalMark>,
|
||||
) -> (Manifest, Vec<Hash>) {
|
||||
let record_hashes: Vec<Hash> = (0..cache.len()).map(|i| record_hash(cache, i)).collect();
|
||||
let manifest = Manifest {
|
||||
record_count: cache.len() as u64,
|
||||
records_root: merkle_root(&record_hashes),
|
||||
settings: settings_hash(config, wal_applied),
|
||||
sessions: sessions_hash(sessions),
|
||||
graph: graph_hash(knowledge),
|
||||
};
|
||||
(manifest, record_hashes)
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Canonical encoding
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// A SHA-256 over length-prefixed fields, so no two different field lists
|
||||
/// hash the same bytes.
|
||||
struct Fields(Sha256);
|
||||
|
||||
impl Fields {
|
||||
fn new(domain: &str) -> Self {
|
||||
let mut h = Sha256::new();
|
||||
h.update((domain.len() as u64).to_le_bytes());
|
||||
h.update(domain.as_bytes());
|
||||
Self(h)
|
||||
}
|
||||
fn bytes(&mut self, b: &[u8]) -> &mut Self {
|
||||
self.0.update((b.len() as u64).to_le_bytes());
|
||||
self.0.update(b);
|
||||
self
|
||||
}
|
||||
/// Strings as the loader returns them: stored null-padded, so a trailing
|
||||
/// NUL cannot survive a round trip and must not be part of the hash.
|
||||
fn str(&mut self, s: &str) -> &mut Self {
|
||||
self.bytes(s.trim_end_matches('\0').as_bytes())
|
||||
}
|
||||
fn u64(&mut self, v: u64) -> &mut Self {
|
||||
self.0.update(v.to_le_bytes());
|
||||
self
|
||||
}
|
||||
fn f64(&mut self, v: f64) -> &mut Self {
|
||||
self.0.update(v.to_bits().to_le_bytes());
|
||||
self
|
||||
}
|
||||
fn f32(&mut self, v: f32) -> &mut Self {
|
||||
self.0.update(v.to_bits().to_le_bytes());
|
||||
self
|
||||
}
|
||||
fn finish(self) -> Hash {
|
||||
self.0.finalize().into()
|
||||
}
|
||||
}
|
||||
|
||||
/// Everything persisted about record `i`, including its position. The
|
||||
/// embedding is hashed as the cache holds it — for a `float16` store that is
|
||||
/// the half-rounded value the file holds.
|
||||
fn record_hash(cache: &MemoryCache, i: usize) -> Hash {
|
||||
let mut f = Fields::new("clawhdf5-agent/record");
|
||||
f.u64(i as u64).str(&cache.chunks[i]);
|
||||
let emb: Vec<u8> = cache.embeddings[i]
|
||||
.iter()
|
||||
.flat_map(|v| v.to_bits().to_le_bytes())
|
||||
.collect();
|
||||
f.bytes(&emb)
|
||||
.str(&cache.source_channels[i])
|
||||
.f64(cache.timestamps[i])
|
||||
.str(&cache.session_ids[i])
|
||||
.str(&cache.tags[i])
|
||||
.u64(u64::from(cache.tombstones[i]))
|
||||
.f32(cache.activation_weights[i]);
|
||||
f.finish()
|
||||
}
|
||||
|
||||
/// Binary Merkle tree: leaves are the record hashes; a parent hashes its two
|
||||
/// children with a node prefix; an odd node is carried up unchanged.
|
||||
fn merkle_root(leaves: &[Hash]) -> Hash {
|
||||
if leaves.is_empty() {
|
||||
return Fields::new("clawhdf5-agent/merkle-empty").finish();
|
||||
}
|
||||
let mut level: Vec<Hash> = leaves.to_vec();
|
||||
while level.len() > 1 {
|
||||
level = level
|
||||
.chunks(2)
|
||||
.map(|pair| match pair {
|
||||
[l, r] => {
|
||||
let mut h = Sha256::new();
|
||||
h.update([1u8]);
|
||||
h.update(l);
|
||||
h.update(r);
|
||||
h.finalize().into()
|
||||
}
|
||||
[only] => *only,
|
||||
_ => unreachable!(),
|
||||
})
|
||||
.collect();
|
||||
}
|
||||
level[0]
|
||||
}
|
||||
|
||||
fn settings_hash(c: &MemoryConfig, wal_applied: Option<WalMark>) -> Hash {
|
||||
let mut f = Fields::new("clawhdf5-agent/settings");
|
||||
f.str(crate::schema::SCHEMA_VERSION)
|
||||
.str(&c.created_at)
|
||||
.str(&c.agent_id)
|
||||
.str(&c.embedder)
|
||||
.u64(c.embedding_dim as u64)
|
||||
.u64(c.chunk_size as u64)
|
||||
.u64(c.overlap as u64)
|
||||
.u64(u64::from(c.float16))
|
||||
.u64(u64::from(c.compression))
|
||||
.u64(u64::from(c.compression_level))
|
||||
.f32(c.compact_threshold)
|
||||
.f32(c.hebbian_boost)
|
||||
.f32(c.decay_factor)
|
||||
.u64(u64::from(c.wal_enabled))
|
||||
.u64(c.wal_max_entries as u64)
|
||||
.u64(u64::from(c.quantized_index))
|
||||
.u64(c.hnsw_m as u64)
|
||||
.u64(c.hnsw_ef_construction as u64)
|
||||
.u64(c.hnsw_ef_search as u64);
|
||||
// An empty mark is not written to the file, so it must hash as none.
|
||||
match wal_applied.filter(|m| m.len > 0) {
|
||||
Some(m) => f.u64(1).u64(m.len).u64(u64::from(m.crc)),
|
||||
None => f.u64(0),
|
||||
};
|
||||
f.finish()
|
||||
}
|
||||
|
||||
fn sessions_hash(s: &SessionCache) -> Hash {
|
||||
let mut f = Fields::new("clawhdf5-agent/sessions");
|
||||
f.u64(s.entries.len() as u64);
|
||||
for (i, e) in s.entries.iter().enumerate() {
|
||||
f.str(&e.id)
|
||||
.u64(e.start_idx)
|
||||
.u64(e.end_idx)
|
||||
.str(&e.channel)
|
||||
.f64(e.ts)
|
||||
.str(s.summaries.get(i).map(String::as_str).unwrap_or(""));
|
||||
}
|
||||
f.finish()
|
||||
}
|
||||
|
||||
fn graph_hash(k: &KnowledgeCache) -> Hash {
|
||||
let mut f = Fields::new("clawhdf5-agent/graph");
|
||||
f.u64(k.entities.len() as u64);
|
||||
for e in &k.entities {
|
||||
f.u64(e.id)
|
||||
.str(&e.name)
|
||||
.str(&e.entity_type)
|
||||
.u64(e.embedding_idx as u64);
|
||||
}
|
||||
f.u64(k.relations.len() as u64);
|
||||
for r in &k.relations {
|
||||
f.u64(r.src)
|
||||
.u64(r.tgt)
|
||||
.str(&r.relation)
|
||||
.f32(r.weight)
|
||||
.f64(r.ts);
|
||||
}
|
||||
f.u64(k.alias_strings.len() as u64);
|
||||
for (s, id) in k.alias_strings.iter().zip(&k.alias_entity_ids) {
|
||||
f.str(s).u64(*id as u64);
|
||||
}
|
||||
f.finish()
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Verification
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// The outcome of [`verify_store`].
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct VerifyReport {
|
||||
/// The checkpoint carries a signature.
|
||||
pub signed: bool,
|
||||
/// The signature was made by the key the caller trusts.
|
||||
pub key_matches: bool,
|
||||
/// The signature over the stored manifest is valid.
|
||||
pub signature_valid: bool,
|
||||
/// The file's current contents match the signed manifest.
|
||||
pub records_match: bool,
|
||||
pub settings_match: bool,
|
||||
pub sessions_match: bool,
|
||||
pub graph_match: bool,
|
||||
/// Records whose contents differ from what was signed (by position),
|
||||
/// when the stored per-record hashes are themselves authentic.
|
||||
pub changed_records: Vec<usize>,
|
||||
/// Records in the file versus in the signed manifest.
|
||||
pub record_count: u64,
|
||||
pub signed_record_count: u64,
|
||||
/// The public key the checkpoint claims to be signed by.
|
||||
pub public_key: Option<[u8; 32]>,
|
||||
/// Saves in the WAL after the checkpoint: not covered by the signature.
|
||||
pub wal_entries_unsigned: usize,
|
||||
}
|
||||
|
||||
impl VerifyReport {
|
||||
/// Signed by the trusted key, signature valid, and every part of the
|
||||
/// file unchanged since it was signed.
|
||||
pub fn is_valid(&self) -> bool {
|
||||
self.signed
|
||||
&& self.key_matches
|
||||
&& self.signature_valid
|
||||
&& self.records_match
|
||||
&& self.settings_match
|
||||
&& self.sessions_match
|
||||
&& self.graph_match
|
||||
}
|
||||
}
|
||||
|
||||
/// Check a store file against the public key the caller trusts.
|
||||
///
|
||||
/// Reads the checkpoint (not the WAL), recomputes every hash from its
|
||||
/// contents and checks the signature. Never writes.
|
||||
pub fn verify_store(
|
||||
path: &std::path::Path,
|
||||
trusted: &VerifyingKey,
|
||||
) -> Result<VerifyReport, crate::MemoryError> {
|
||||
let file = clawhdf5::File::open(path)
|
||||
.map_err(|e| crate::MemoryError::Hdf5(format!("cannot open {}: {e}", path.display())))?;
|
||||
let (config, cache, sessions, knowledge) = crate::schema::validate_and_load(&file)?;
|
||||
let checkpoint = crate::schema::read_checkpoint_meta(&file);
|
||||
let stored = crate::schema::read_signature(&file)?;
|
||||
let wal_entries_unsigned = count_wal_entries_after(path, checkpoint.wal_applied);
|
||||
|
||||
let (current, current_hashes) = manifest(
|
||||
&config,
|
||||
&cache,
|
||||
&sessions,
|
||||
&knowledge,
|
||||
checkpoint.wal_applied,
|
||||
);
|
||||
|
||||
let Some(stored) = stored else {
|
||||
return Ok(VerifyReport {
|
||||
signed: false,
|
||||
key_matches: false,
|
||||
signature_valid: false,
|
||||
records_match: false,
|
||||
settings_match: false,
|
||||
sessions_match: false,
|
||||
graph_match: false,
|
||||
changed_records: Vec::new(),
|
||||
record_count: current.record_count,
|
||||
signed_record_count: 0,
|
||||
public_key: None,
|
||||
wal_entries_unsigned,
|
||||
});
|
||||
};
|
||||
|
||||
let key_matches = stored.public_key == trusted.to_bytes();
|
||||
let signature_valid = trusted
|
||||
.verify(
|
||||
&stored.manifest.signed_bytes(),
|
||||
&Signature::from_bytes(&stored.signature),
|
||||
)
|
||||
.is_ok();
|
||||
// The stored per-record hashes can localise a change only if they are
|
||||
// the ones that were signed.
|
||||
let hashes_authentic = signature_valid
|
||||
&& stored.record_hashes.len() as u64 == stored.manifest.record_count
|
||||
&& merkle_root(&stored.record_hashes) == stored.manifest.records_root;
|
||||
let changed_records = if hashes_authentic {
|
||||
let n = current_hashes.len().max(stored.record_hashes.len());
|
||||
(0..n)
|
||||
.filter(|&i| current_hashes.get(i) != stored.record_hashes.get(i))
|
||||
.collect()
|
||||
} else {
|
||||
Vec::new()
|
||||
};
|
||||
|
||||
Ok(VerifyReport {
|
||||
signed: true,
|
||||
key_matches,
|
||||
signature_valid,
|
||||
records_match: signature_valid
|
||||
&& current.record_count == stored.manifest.record_count
|
||||
&& current.records_root == stored.manifest.records_root,
|
||||
settings_match: signature_valid && current.settings == stored.manifest.settings,
|
||||
sessions_match: signature_valid && current.sessions == stored.manifest.sessions,
|
||||
graph_match: signature_valid && current.graph == stored.manifest.graph,
|
||||
changed_records,
|
||||
record_count: current.record_count,
|
||||
signed_record_count: stored.manifest.record_count,
|
||||
public_key: Some(stored.public_key),
|
||||
wal_entries_unsigned,
|
||||
})
|
||||
}
|
||||
|
||||
fn count_wal_entries_after(store: &std::path::Path, mark: Option<WalMark>) -> usize {
|
||||
let wal = store.with_extension("h5.wal");
|
||||
if !wal.exists() {
|
||||
return 0;
|
||||
}
|
||||
crate::wal::WalFile::read_entries_for_migration(&wal, mark)
|
||||
.map(|e| e.len())
|
||||
.unwrap_or(0)
|
||||
}
|
||||
|
||||
/// A new random signing key from the operating system's RNG.
|
||||
pub fn generate_key() -> SigningKey {
|
||||
SigningKey::generate(&mut rand_core::OsRng)
|
||||
}
|
||||
|
||||
/// Hex encoding for keys and signatures in attributes and the CLI.
|
||||
pub fn to_hex(bytes: &[u8]) -> String {
|
||||
bytes.iter().map(|b| format!("{b:02x}")).collect()
|
||||
}
|
||||
|
||||
/// Parse hex into exactly `N` bytes.
|
||||
pub fn from_hex<const N: usize>(s: &str) -> Option<[u8; N]> {
|
||||
let s = s.trim();
|
||||
if s.len() != 2 * N {
|
||||
return None;
|
||||
}
|
||||
let mut out = [0u8; N];
|
||||
for (i, byte) in out.iter_mut().enumerate() {
|
||||
*byte = u8::from_str_radix(&s[2 * i..2 * i + 2], 16).ok()?;
|
||||
}
|
||||
Some(out)
|
||||
}
|
||||
@@ -36,7 +36,7 @@ pub fn write_to_disk_with_mark(
|
||||
) -> Result<(), MemoryError> {
|
||||
let meta = schema::CheckpointMeta {
|
||||
wal_applied,
|
||||
ann_generation: None,
|
||||
..schema::CheckpointMeta::default()
|
||||
};
|
||||
write_to_disk_with_meta(path, config, cache, sessions, knowledge, &meta)
|
||||
}
|
||||
@@ -50,7 +50,21 @@ pub fn write_to_disk_with_meta(
|
||||
knowledge: &KnowledgeCache,
|
||||
checkpoint: &schema::CheckpointMeta,
|
||||
) -> Result<(), MemoryError> {
|
||||
let bytes = schema::build_hdf5_file_with_meta(config, cache, sessions, knowledge, checkpoint)?;
|
||||
write_to_disk_signed(path, config, cache, sessions, knowledge, checkpoint, None)
|
||||
}
|
||||
|
||||
/// [`write_to_disk_with_meta`] with a signed manifest of the contents.
|
||||
pub fn write_to_disk_signed(
|
||||
path: &Path,
|
||||
config: &MemoryConfig,
|
||||
cache: &MemoryCache,
|
||||
sessions: &SessionCache,
|
||||
knowledge: &KnowledgeCache,
|
||||
checkpoint: &schema::CheckpointMeta,
|
||||
signature: Option<&crate::signing::StoredSignature>,
|
||||
) -> Result<(), MemoryError> {
|
||||
let bytes =
|
||||
schema::build_hdf5_file_signed(config, cache, sessions, knowledge, checkpoint, signature)?;
|
||||
|
||||
if bytes.is_empty() {
|
||||
return Err(MemoryError::Hdf5("build_hdf5_file produced 0 bytes".into()));
|
||||
@@ -113,13 +127,11 @@ pub type StoreState = (MemoryConfig, MemoryCache, SessionCache, KnowledgeCache);
|
||||
/// [`read_from_disk`], plus the checkpoint's [`WalMark`] (if any) so the
|
||||
/// caller can skip WAL entries this file already contains.
|
||||
pub fn read_from_disk_with_mark(path: &Path) -> Result<(StoreState, Option<WalMark>), MemoryError> {
|
||||
let mmap = clawhdf5_io::MmapReader::open(path).map_err(MemoryError::Io)?;
|
||||
|
||||
// Advise the OS we'll need the whole file for parsing
|
||||
mmap.advise_willneed(0, mmap.len());
|
||||
|
||||
// Parse the HDF5 file from the mmap'd bytes
|
||||
let file = clawhdf5::File::from_bytes(mmap.as_bytes().to_vec())
|
||||
// `File::open` memory-maps the file itself (the facade's `mmap` feature is
|
||||
// on by default). Mapping it here and handing over `as_bytes().to_vec()`
|
||||
// did the same work and then copied the whole store — a second full copy
|
||||
// of the file, live for the whole parse, on top of the mapping.
|
||||
let file = clawhdf5::File::open(path)
|
||||
.map_err(|e| MemoryError::Hdf5(format!("cannot open {}: {e}", path.display())))?;
|
||||
|
||||
let (mut config, cache, sessions, knowledge) = schema::validate_and_load(&file)?;
|
||||
@@ -133,9 +145,7 @@ pub fn read_from_disk_with_mark(path: &Path) -> Result<(StoreState, Option<WalMa
|
||||
pub fn read_from_disk_with_meta(
|
||||
path: &Path,
|
||||
) -> Result<(StoreState, schema::CheckpointMeta), MemoryError> {
|
||||
let mmap = clawhdf5_io::MmapReader::open(path).map_err(MemoryError::Io)?;
|
||||
mmap.advise_willneed(0, mmap.len());
|
||||
let file = clawhdf5::File::from_bytes(mmap.as_bytes().to_vec())
|
||||
let file = clawhdf5::File::open(path)
|
||||
.map_err(|e| MemoryError::Hdf5(format!("cannot open {}: {e}", path.display())))?;
|
||||
let (mut config, cache, sessions, knowledge) = schema::validate_and_load(&file)?;
|
||||
config.path = path.to_path_buf();
|
||||
|
||||
Binary file not shown.
@@ -0,0 +1,300 @@
|
||||
//! `MemoryConfig::float16`: embeddings stored as IEEE half precision.
|
||||
//!
|
||||
//! The setting used to be recorded in `/meta` and otherwise ignored — the
|
||||
//! embeddings dataset was always `f32`. These tests pin what it now does: the
|
||||
//! dataset is `float16`, the in-memory cache holds exactly the values the file
|
||||
//! holds (so search results survive a reopen bit for bit), and a value half
|
||||
//! precision cannot represent is refused rather than stored as infinity.
|
||||
|
||||
use std::path::{Path, PathBuf};
|
||||
|
||||
use clawhdf5_agent::{AgentMemory, HDF5Memory, MemoryConfig, MemoryEntry, MemoryError};
|
||||
use clawhdf5_format::float16::round_to_f16;
|
||||
use tempfile::TempDir;
|
||||
|
||||
const DIM: usize = 64;
|
||||
|
||||
/// Deterministic, embedding-like unit vectors.
|
||||
fn embedding(seed: u64) -> Vec<f32> {
|
||||
let mut x = seed.wrapping_mul(0x9E37_79B9_7F4A_7C15) | 1;
|
||||
let v: Vec<f32> = (0..DIM)
|
||||
.map(|_| {
|
||||
x ^= x << 13;
|
||||
x ^= x >> 7;
|
||||
x ^= x << 17;
|
||||
(x >> 40) as f32 / (1u64 << 24) as f32 - 0.5
|
||||
})
|
||||
.collect();
|
||||
let norm = v.iter().map(|a| a * a).sum::<f32>().sqrt();
|
||||
v.iter().map(|a| a / norm).collect()
|
||||
}
|
||||
|
||||
fn entry(i: u64) -> MemoryEntry {
|
||||
MemoryEntry {
|
||||
chunk: format!("memory number {i} about topic {}", i % 7),
|
||||
embedding: embedding(i),
|
||||
source_channel: "test".into(),
|
||||
timestamp: i as f64,
|
||||
session_id: "s".into(),
|
||||
tags: format!("t{i}"),
|
||||
}
|
||||
}
|
||||
|
||||
fn config(dir: &TempDir, name: &str, float16: bool) -> MemoryConfig {
|
||||
let mut c = MemoryConfig::new(dir.path().join(name), "agent", DIM);
|
||||
c.float16 = float16;
|
||||
c
|
||||
}
|
||||
|
||||
fn embeddings_dtype_and_values(path: &Path) -> (String, Vec<f32>) {
|
||||
let file = clawhdf5::File::open(path).unwrap();
|
||||
let ds = file.dataset("memory/embeddings").unwrap();
|
||||
(format!("{:?}", ds.dtype().unwrap()), ds.read_f32().unwrap())
|
||||
}
|
||||
|
||||
fn search_bits(m: &mut HDF5Memory, q: u64) -> Vec<(usize, u32)> {
|
||||
m.hybrid_search(&embedding(q), "memory topic 3", 0.4, 0.6, 10)
|
||||
.iter()
|
||||
.map(|r| (r.index, r.score.to_bits()))
|
||||
.collect()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn float16_store_writes_half_precision_and_reopens_identically() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
// Two identical stores. Search is not read-only (it boosts the Hebbian
|
||||
// activation of what it returns, and checkpoints persist that), so each
|
||||
// is queried exactly once: one live, one after a checkpoint and reopen.
|
||||
let live_cfg = config(&dir, "live.h5", true);
|
||||
let cfg = config(&dir, "f16.h5", true);
|
||||
let path: PathBuf = cfg.path.clone();
|
||||
|
||||
let mut live = HDF5Memory::create(live_cfg).unwrap();
|
||||
live.save_batch((0..200).map(entry).collect()).unwrap();
|
||||
let mut m = HDF5Memory::create(cfg).unwrap();
|
||||
m.save_batch((0..200).map(entry).collect()).unwrap();
|
||||
drop(m);
|
||||
|
||||
// On disk: a genuine float16 dataset holding the rounded inputs.
|
||||
let (dtype, values) = embeddings_dtype_and_values(&path);
|
||||
assert_eq!(dtype, "Other(\"float16\")");
|
||||
let expected: Vec<u32> = (0..200)
|
||||
.flat_map(|i| embedding(i).into_iter().map(|v| round_to_f16(v).to_bits()))
|
||||
.collect();
|
||||
let got: Vec<u32> = values.iter().map(|v| v.to_bits()).collect();
|
||||
assert_eq!(got, expected);
|
||||
|
||||
// Reopened, the store answers exactly as the live one does: the cache
|
||||
// held the half-rounded values before the checkpoint.
|
||||
let mut reopened = HDF5Memory::open(&path).unwrap();
|
||||
for q in 0..5 {
|
||||
assert_eq!(
|
||||
search_bits(&mut live, 1000 + q),
|
||||
search_bits(&mut reopened, 1000 + q),
|
||||
"query {q}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn float16_halves_the_embeddings_on_disk() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let mut sizes = Vec::new();
|
||||
for float16 in [false, true] {
|
||||
let cfg = config(&dir, &format!("s{float16}.h5"), float16);
|
||||
let path = cfg.path.clone();
|
||||
let mut m = HDF5Memory::create(cfg).unwrap();
|
||||
m.save_batch((0..2000).map(entry).collect()).unwrap();
|
||||
drop(m);
|
||||
sizes.push(std::fs::metadata(&path).unwrap().len());
|
||||
}
|
||||
let embedding_bytes_f32 = (2000 * DIM * 4) as u64;
|
||||
let saved = sizes[0] - sizes[1];
|
||||
// Half of the f32 embeddings, give or take metadata and alignment.
|
||||
assert!(
|
||||
saved.abs_diff(embedding_bytes_f32 / 2) < 16 * 1024,
|
||||
"f32 {} B, f16 {} B, saved {saved} B, expected ~{} B",
|
||||
sizes[0],
|
||||
sizes[1],
|
||||
embedding_bytes_f32 / 2
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn f32_store_is_unchanged() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let cfg = config(&dir, "f32.h5", false);
|
||||
let path = cfg.path.clone();
|
||||
let mut m = HDF5Memory::create(cfg).unwrap();
|
||||
m.save_batch((0..50).map(entry).collect()).unwrap();
|
||||
drop(m);
|
||||
let (dtype, values) = embeddings_dtype_and_values(&path);
|
||||
assert_eq!(dtype, "F32");
|
||||
let expected: Vec<f32> = (0..50).flat_map(embedding).collect();
|
||||
assert_eq!(values, expected);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn out_of_range_values_are_refused_not_stored_as_infinity() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let mut cfg = config(&dir, "range.h5", true);
|
||||
cfg.wal_enabled = true;
|
||||
let path = cfg.path.clone();
|
||||
let mut m = HDF5Memory::create(cfg).unwrap();
|
||||
m.save(entry(1)).unwrap();
|
||||
|
||||
let mut bad = entry(2);
|
||||
bad.embedding[5] = 70_000.0;
|
||||
match m.save(bad.clone()) {
|
||||
Err(MemoryError::InvalidEntry(msg)) => assert!(msg.contains("embedding[5]"), "{msg}"),
|
||||
other => panic!("expected InvalidEntry, got {other:?}"),
|
||||
}
|
||||
assert!(matches!(
|
||||
m.save_or_update(bad.clone()),
|
||||
Err(MemoryError::InvalidEntry(_))
|
||||
));
|
||||
// A batch is all or nothing.
|
||||
assert!(matches!(
|
||||
m.save_batch(vec![entry(3), bad.clone(), entry(4)]),
|
||||
Err(MemoryError::InvalidEntry(_))
|
||||
));
|
||||
assert_eq!(m.count(), 1);
|
||||
|
||||
// The largest finite half, and values that round down to it, are fine.
|
||||
let mut edge = entry(5);
|
||||
edge.embedding[0] = 65504.0;
|
||||
edge.embedding[1] = -65519.0;
|
||||
m.save(edge).unwrap();
|
||||
assert_eq!(m.count(), 2);
|
||||
drop(m);
|
||||
|
||||
// Nothing rejected reached the WAL or the file.
|
||||
let m = HDF5Memory::open(&path).unwrap();
|
||||
assert_eq!(m.count(), 2);
|
||||
|
||||
// An f32 store takes the same value as it always did.
|
||||
let mut m32 = HDF5Memory::create(config(&dir, "range32.h5", false)).unwrap();
|
||||
m32.save(bad).unwrap();
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn wal_replay_rounds_like_a_live_save() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let mut cfg = config(&dir, "wal.h5", true);
|
||||
cfg.wal_enabled = true;
|
||||
cfg.wal_max_entries = 10_000; // keep everything in the WAL
|
||||
let path = cfg.path.clone();
|
||||
let mut m = HDF5Memory::create(cfg).unwrap();
|
||||
for i in 0..30 {
|
||||
m.save(entry(i)).unwrap();
|
||||
}
|
||||
let live = search_bits(&mut m, 77);
|
||||
|
||||
// Crash image: the .h5 is still the empty checkpoint; everything is in
|
||||
// the WAL, which holds the caller's f32 values.
|
||||
let crash = TempDir::new().unwrap();
|
||||
let image = crash.path().join("image.h5");
|
||||
std::fs::copy(&path, &image).unwrap();
|
||||
std::fs::copy(
|
||||
path.with_extension("h5.wal"),
|
||||
image.with_extension("h5.wal"),
|
||||
)
|
||||
.unwrap();
|
||||
drop(m);
|
||||
|
||||
let mut recovered = HDF5Memory::open(&image).unwrap();
|
||||
assert_eq!(recovered.count(), 30);
|
||||
assert_eq!(search_bits(&mut recovered, 77), live);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn new_stores_default_to_float16() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let path = dir.path().join("default.h5");
|
||||
let mut m = HDF5Memory::create(MemoryConfig::new(path.clone(), "agent", DIM)).unwrap();
|
||||
assert!(m.config().float16);
|
||||
m.save_batch((0..10).map(entry).collect()).unwrap();
|
||||
drop(m);
|
||||
assert_eq!(embeddings_dtype_and_values(&path).0, "Other(\"float16\")");
|
||||
assert!(HDF5Memory::open(&path).unwrap().config().float16);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_existing_f32_store_stays_f32() {
|
||||
// Written by the v2.5.0 CLI, with `float16 = 0` in /meta (every agent
|
||||
// store has recorded it). Flipping the default for new stores must not
|
||||
// reach back and round an existing store's embeddings.
|
||||
let dir = TempDir::new().unwrap();
|
||||
let path = dir.path().join("legacy.h5");
|
||||
std::fs::copy(
|
||||
concat!(
|
||||
env!("CARGO_MANIFEST_DIR"),
|
||||
"/tests/fixtures/store_v2_5_0.h5"
|
||||
),
|
||||
&path,
|
||||
)
|
||||
.unwrap();
|
||||
let before = embeddings_dtype_and_values(&path);
|
||||
assert_eq!(before.0, "F32");
|
||||
|
||||
let mut m = HDF5Memory::open(&path).unwrap();
|
||||
assert!(!m.config().float16, "an old store must reopen as f32");
|
||||
let dim = m.config().embedding_dim;
|
||||
let odd: Vec<f32> = (0..dim).map(|i| 0.1 + i as f32 * 1e-4).collect();
|
||||
m.save_batch(vec![MemoryEntry {
|
||||
chunk: "added after the upgrade".into(),
|
||||
embedding: odd.clone(),
|
||||
source_channel: "test".into(),
|
||||
timestamp: 1.0,
|
||||
session_id: "s".into(),
|
||||
tags: String::new(),
|
||||
}])
|
||||
.unwrap();
|
||||
drop(m);
|
||||
|
||||
// Checkpointed: still f32, the old rows untouched and the new one exact.
|
||||
let (dtype, values) = embeddings_dtype_and_values(&path);
|
||||
assert_eq!(dtype, "F32");
|
||||
assert_eq!(&values[..before.1.len()], before.1.as_slice());
|
||||
assert_eq!(&values[before.1.len()..], odd.as_slice());
|
||||
}
|
||||
|
||||
/// `Group::attrs` leaves out an attribute it cannot decode. A store whose
|
||||
/// `float16` setting is unreadable must not open as `float16 = false` (or with
|
||||
/// any other default in place of a setting it has): it is an error.
|
||||
#[test]
|
||||
fn unreadable_meta_attribute_fails_open_instead_of_defaulting() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let path = dir.path().join("store.h5");
|
||||
{
|
||||
let mut m = HDF5Memory::create(config(&dir, "store.h5", true)).unwrap();
|
||||
m.save(entry(1)).unwrap();
|
||||
m.flush_wal().unwrap();
|
||||
}
|
||||
assert!(HDF5Memory::open_read_only(&path).is_ok());
|
||||
|
||||
// Give the `float16` attribute message an unknown version (the name is
|
||||
// at +8 in a version-1 message and +9 in a version-3 one).
|
||||
let mut bytes = std::fs::read(&path).unwrap();
|
||||
let name = b"float16\0";
|
||||
let mut hit = false;
|
||||
let positions: Vec<usize> = (9..bytes.len() - name.len())
|
||||
.filter(|&p| &bytes[p..p + name.len()] == name)
|
||||
.collect();
|
||||
for pos in positions {
|
||||
for (back, version) in [(8, 1u8), (9, 3u8)] {
|
||||
if bytes[pos - back] == version {
|
||||
bytes[pos - back] = 0x7f;
|
||||
hit = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
assert!(hit, "float16 attribute message not found");
|
||||
std::fs::write(&path, &bytes).unwrap();
|
||||
|
||||
match HDF5Memory::open_read_only(&path) {
|
||||
Err(MemoryError::Schema(msg)) => assert!(msg.contains("/meta"), "{msg}"),
|
||||
Err(e) => panic!("unexpected error: {e}"),
|
||||
Ok(_) => panic!("store opened with an unreadable float16 setting"),
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,155 @@
|
||||
//! An agent store is a standard HDF5 file: h5py can open it and read every
|
||||
//! dataset.
|
||||
//!
|
||||
//! It could not: the float datatype's sign-bit position was hard-coded for
|
||||
//! f64, so every f32 dataset (embeddings, norms, activation weights) made
|
||||
//! libhdf5 refuse the file with "sign bit position out of bounds".
|
||||
|
||||
use std::process::Command;
|
||||
|
||||
use clawhdf5_agent::{AgentMemory, HDF5Memory, MemoryConfig, MemoryEntry};
|
||||
|
||||
fn python() -> String {
|
||||
std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string())
|
||||
}
|
||||
|
||||
fn h5py_available() -> bool {
|
||||
Command::new(python())
|
||||
.args(["-c", "import h5py"])
|
||||
.output()
|
||||
.map(|o| o.status.success())
|
||||
.unwrap_or(false)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn h5py_reads_every_dataset_of_an_agent_store() {
|
||||
if !h5py_available() {
|
||||
assert!(
|
||||
std::env::var("CLAWHDF5_REQUIRE_INTEROP").as_deref() != Ok("1"),
|
||||
"CLAWHDF5_REQUIRE_INTEROP=1 but python3 with h5py is not available"
|
||||
);
|
||||
eprintln!("SKIP: python3 with h5py not available");
|
||||
return;
|
||||
}
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
for float16 in [false, true] {
|
||||
let path = dir.path().join(format!("store_{float16}.h5"));
|
||||
let mut cfg = MemoryConfig::new(path.clone(), "agent", 8);
|
||||
cfg.float16 = float16;
|
||||
let mut m = HDF5Memory::create(cfg).unwrap();
|
||||
// save_batch checkpoints, so the records are in the .h5, not the WAL.
|
||||
m.save_batch(
|
||||
(0..20)
|
||||
.map(|i| MemoryEntry {
|
||||
chunk: format!("memory {i}"),
|
||||
embedding: (0..8).map(|j| ((i * 8 + j) as f32).sin()).collect(),
|
||||
source_channel: "test".into(),
|
||||
timestamp: i as f64,
|
||||
session_id: "s".into(),
|
||||
tags: String::new(),
|
||||
})
|
||||
.collect(),
|
||||
)
|
||||
.unwrap();
|
||||
drop(m);
|
||||
|
||||
// Exact expected values, as bits: numpy's sin need not match Rust's
|
||||
// to the last place.
|
||||
let bits = (0..160)
|
||||
.map(|k| (k as f32).sin().to_bits().to_string())
|
||||
.collect::<Vec<_>>()
|
||||
.join(",");
|
||||
let script = format!(
|
||||
r#"
|
||||
import h5py, numpy as np
|
||||
want = np.float16 if {py_bool} else np.float32
|
||||
with h5py.File("{path}", "r") as f:
|
||||
names = []
|
||||
f.visititems(lambda n, o: names.append(n) if isinstance(o, h5py.Dataset) else None)
|
||||
for n in names:
|
||||
f[n][()] # every dataset must decode
|
||||
e = f["memory/embeddings"]
|
||||
assert e.dtype == want, e.dtype
|
||||
assert e.shape == (20, 8), e.shape
|
||||
ref = np.array([{bits}], dtype=np.uint32).view(np.float32).astype(want).reshape(20, 8)
|
||||
assert (e[()] == ref).all()
|
||||
assert f["memory/norms"].dtype == np.float32
|
||||
print(len(names))
|
||||
"#,
|
||||
py_bool = if float16 { "True" } else { "False" },
|
||||
path = path.display()
|
||||
);
|
||||
let out = Command::new(python())
|
||||
.args(["-c", &script])
|
||||
.output()
|
||||
.unwrap();
|
||||
assert!(
|
||||
out.status.success(),
|
||||
"float16={float16}: {}",
|
||||
String::from_utf8_lossy(&out.stderr)
|
||||
);
|
||||
let n: usize = String::from_utf8_lossy(&out.stdout).trim().parse().unwrap();
|
||||
assert!(n >= 10, "only {n} datasets");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_edit_made_with_h5py_breaks_the_signature_and_names_the_record() {
|
||||
if !h5py_available() {
|
||||
assert!(
|
||||
std::env::var("CLAWHDF5_REQUIRE_INTEROP").as_deref() != Ok("1"),
|
||||
"CLAWHDF5_REQUIRE_INTEROP=1 but python3 with h5py is not available"
|
||||
);
|
||||
eprintln!("SKIP: python3 with h5py not available");
|
||||
return;
|
||||
}
|
||||
use clawhdf5_agent::signing::SigningKey;
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let path = dir.path().join("signed.h5");
|
||||
let key = SigningKey::from_bytes(&[42; 32]);
|
||||
let mut m = HDF5Memory::create(MemoryConfig::new(path.clone(), "agent", 8)).unwrap();
|
||||
m.set_signing_key(key.clone());
|
||||
m.save_batch(
|
||||
(0..10)
|
||||
.map(|i| MemoryEntry {
|
||||
chunk: format!("memory {i}"),
|
||||
embedding: (0..8).map(|j| ((i * 8 + j) as f32).cos()).collect(),
|
||||
source_channel: "test".into(),
|
||||
timestamp: i as f64,
|
||||
session_id: "s".into(),
|
||||
tags: String::new(),
|
||||
})
|
||||
.collect(),
|
||||
)
|
||||
.unwrap();
|
||||
drop(m);
|
||||
assert!(
|
||||
HDF5Memory::verify(&path, &key.verifying_key())
|
||||
.unwrap()
|
||||
.is_valid()
|
||||
);
|
||||
|
||||
// Someone edits one timestamp in place with h5py.
|
||||
let script = format!(
|
||||
r#"
|
||||
import h5py
|
||||
with h5py.File("{}", "r+") as f:
|
||||
ts = f["memory/timestamps"]
|
||||
ts[3] = 12345.0
|
||||
"#,
|
||||
path.display()
|
||||
);
|
||||
let out = Command::new(python())
|
||||
.args(["-c", &script])
|
||||
.output()
|
||||
.unwrap();
|
||||
assert!(
|
||||
out.status.success(),
|
||||
"{}",
|
||||
String::from_utf8_lossy(&out.stderr)
|
||||
);
|
||||
|
||||
let r = HDF5Memory::verify(&path, &key.verifying_key()).unwrap();
|
||||
assert!(r.signature_valid && !r.is_valid(), "{r:?}");
|
||||
assert_eq!(r.changed_records, vec![3]);
|
||||
}
|
||||
@@ -234,3 +234,113 @@ fn quantized_index_setting_survives_a_reopen() {
|
||||
let reopened = HDF5Memory::open(&path).unwrap();
|
||||
assert!(reopened.config().quantized_index);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn hnsw_parameters_are_configurable_and_persisted() {
|
||||
// The graph degree and both candidate-list sizes used to be constants, so
|
||||
// a deployment could not trade recall against memory or speed at all.
|
||||
let dir = TempDir::new().unwrap();
|
||||
let path = dir.path().join("mem.h5");
|
||||
let mut config = MemoryConfig::new(path.clone(), "agent", 16);
|
||||
config.hnsw_m = 8;
|
||||
config.hnsw_ef_construction = 32;
|
||||
config.hnsw_ef_search = 128;
|
||||
let mut mem = HDF5Memory::create(config).unwrap();
|
||||
|
||||
let mut seed = 99;
|
||||
let vectors: Vec<Vec<f32>> = (0..300).map(|_| make_vector(&mut seed, 16)).collect();
|
||||
for (i, v) in vectors.iter().enumerate() {
|
||||
mem.save(entry(&format!("c{i}"), v.clone(), "t")).unwrap();
|
||||
}
|
||||
// Still correct with a smaller graph: an exact match must rank first.
|
||||
let top = mem.hybrid_search(&vectors[42], "", 1.0, 0.0, 1);
|
||||
assert_eq!(top[0].index, 42);
|
||||
|
||||
mem.flush_wal().unwrap();
|
||||
drop(mem);
|
||||
let reopened = HDF5Memory::open(&path).unwrap();
|
||||
assert_eq!(reopened.config().hnsw_m, 8);
|
||||
assert_eq!(reopened.config().hnsw_ef_construction, 32);
|
||||
assert_eq!(reopened.config().hnsw_ef_search, 128);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn degenerate_hnsw_parameters_do_not_panic() {
|
||||
// `clawhdf5-ann` asserts m >= 2, so a zero from a config file — or from a
|
||||
// caller who assumed 0 meant "default" — would abort the process inside
|
||||
// the index builder. The store clamps instead.
|
||||
let dir = TempDir::new().unwrap();
|
||||
let mut config = MemoryConfig::new(dir.path().join("mem.h5"), "agent", 8);
|
||||
config.hnsw_m = 0;
|
||||
config.hnsw_ef_construction = 0;
|
||||
config.hnsw_ef_search = 1;
|
||||
let mut mem = HDF5Memory::create(config).unwrap();
|
||||
|
||||
let mut seed = 5;
|
||||
let vectors: Vec<Vec<f32>> = (0..50).map(|_| make_vector(&mut seed, 8)).collect();
|
||||
for (i, v) in vectors.iter().enumerate() {
|
||||
mem.save(entry(&format!("c{i}"), v.clone(), "t")).unwrap();
|
||||
}
|
||||
let results = mem.hybrid_search(&vectors[7], "", 1.0, 0.0, 5);
|
||||
assert_eq!(results[0].index, 7, "exact match should still rank first");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn new_stores_default_to_the_quantized_index() {
|
||||
// int8 is the default because it is smaller and, with an exact re-score,
|
||||
// faster at equal recall on every platform measured (see BENCHMARKS.md).
|
||||
let dir = TempDir::new().unwrap();
|
||||
let config = MemoryConfig::new(dir.path().join("mem.h5"), "agent", 8);
|
||||
assert!(config.quantized_index);
|
||||
|
||||
let path = config.path.clone();
|
||||
let mut mem = HDF5Memory::create(config).unwrap();
|
||||
let mut seed = 3;
|
||||
let vectors: Vec<Vec<f32>> = (0..40).map(|_| make_vector(&mut seed, 8)).collect();
|
||||
for (i, v) in vectors.iter().enumerate() {
|
||||
mem.save(entry(&format!("c{i}"), v.clone(), "t")).unwrap();
|
||||
}
|
||||
assert_eq!(
|
||||
mem.hybrid_search(&vectors[11], "", 1.0, 0.0, 1)[0].index,
|
||||
11
|
||||
);
|
||||
mem.flush_wal().unwrap();
|
||||
drop(mem);
|
||||
assert!(HDF5Memory::open(&path).unwrap().config().quantized_index);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_store_written_before_the_setting_existed_stays_f32() {
|
||||
// `store_v2_5_0.h5` was written by the v2.5.0 CLI, before
|
||||
// `quantized_index` or the HNSW parameters were persisted, so it carries
|
||||
// none of them. Flipping the default for new stores must not reach back
|
||||
// and change how an existing store's index is held.
|
||||
let dir = TempDir::new().unwrap();
|
||||
let path = dir.path().join("legacy.h5");
|
||||
std::fs::copy(
|
||||
concat!(
|
||||
env!("CARGO_MANIFEST_DIR"),
|
||||
"/tests/fixtures/store_v2_5_0.h5"
|
||||
),
|
||||
&path,
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
let bytes = std::fs::read(&path).unwrap();
|
||||
assert!(
|
||||
!bytes.windows(15).any(|w| w == b"quantized_index"),
|
||||
"the fixture must predate the setting, or it tests nothing"
|
||||
);
|
||||
|
||||
let mut mem = HDF5Memory::open(&path).unwrap();
|
||||
assert!(
|
||||
!mem.config().quantized_index,
|
||||
"an old store must reopen with an f32 index"
|
||||
);
|
||||
assert_eq!(mem.config().hnsw_m, 16);
|
||||
assert_eq!(mem.config().hnsw_ef_construction, 64);
|
||||
assert_eq!(mem.count(), 6);
|
||||
// And it still searches: entry 3's own embedding finds it first.
|
||||
let hit = mem.hybrid_search(&[3.0, 1.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0], "", 1.0, 0.0, 1);
|
||||
assert_eq!(hit[0].index, 3);
|
||||
}
|
||||
|
||||
@@ -0,0 +1,344 @@
|
||||
//! `HDF5Memory::search` with `SearchOptions`: source filtering, re-ranking and
|
||||
//! confidence rejection in the store's own search path.
|
||||
|
||||
use std::collections::HashSet;
|
||||
|
||||
use clawhdf5_agent::confidence::ConfidenceConfig;
|
||||
use clawhdf5_agent::reranker::ReRankConfig;
|
||||
use clawhdf5_agent::{AgentMemory, HDF5Memory, MemoryConfig, MemoryEntry, SearchOptions, hybrid};
|
||||
use tempfile::TempDir;
|
||||
|
||||
const DIM: usize = 32;
|
||||
const N: usize = 3000;
|
||||
const CLUSTERS: usize = 20;
|
||||
|
||||
struct Rng(u64);
|
||||
impl Rng {
|
||||
fn next(&mut self) -> u64 {
|
||||
self.0 = self.0.wrapping_add(0x9E37_79B9_7F4A_7C15);
|
||||
let mut z = self.0;
|
||||
z = (z ^ (z >> 30)).wrapping_mul(0xBF58_476D_1CE4_E5B9);
|
||||
z = (z ^ (z >> 27)).wrapping_mul(0x94D0_49BB_1331_11EB);
|
||||
z ^ (z >> 31)
|
||||
}
|
||||
fn unit(&mut self) -> f32 {
|
||||
(self.next() >> 40) as f32 / (1u64 << 24) as f32 - 0.5
|
||||
}
|
||||
}
|
||||
|
||||
fn normalize(v: &mut [f32]) {
|
||||
let n = v.iter().map(|x| x * x).sum::<f32>().sqrt();
|
||||
v.iter_mut().for_each(|x| *x /= n);
|
||||
}
|
||||
|
||||
struct Data {
|
||||
vectors: Vec<Vec<f32>>,
|
||||
cluster: Vec<usize>,
|
||||
centres: Vec<Vec<f32>>,
|
||||
}
|
||||
|
||||
fn data() -> Data {
|
||||
let mut rng = Rng(42);
|
||||
let centres: Vec<Vec<f32>> = (0..CLUSTERS)
|
||||
.map(|_| {
|
||||
let mut c: Vec<f32> = (0..DIM).map(|_| rng.unit()).collect();
|
||||
normalize(&mut c);
|
||||
c
|
||||
})
|
||||
.collect();
|
||||
let mut vectors = Vec::new();
|
||||
let mut cluster = Vec::new();
|
||||
for i in 0..N {
|
||||
let c = i % CLUSTERS;
|
||||
let mut v: Vec<f32> = centres[c].iter().map(|x| x + rng.unit() * 0.3).collect();
|
||||
normalize(&mut v);
|
||||
vectors.push(v);
|
||||
cluster.push(c);
|
||||
}
|
||||
Data {
|
||||
vectors,
|
||||
cluster,
|
||||
centres,
|
||||
}
|
||||
}
|
||||
|
||||
/// Channel of record `i` for a filter keeping `percent`% of the store at
|
||||
/// random (independent of the vectors).
|
||||
fn random_channel(i: usize, rng_seed: u64, percent: u64) -> String {
|
||||
let mut r = Rng(rng_seed ^ (i as u64 * 7919));
|
||||
if r.next() % 100 < percent {
|
||||
"keep".into()
|
||||
} else {
|
||||
"other".into()
|
||||
}
|
||||
}
|
||||
|
||||
fn build(data: &Data, channel: impl Fn(usize) -> String) -> (TempDir, HDF5Memory) {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let mut cfg = MemoryConfig::new(dir.path().join("s.h5"), "agent", DIM);
|
||||
cfg.hebbian_boost = 0.0; // every query sees the same store
|
||||
let mut m = HDF5Memory::create(cfg).unwrap();
|
||||
let entries = data
|
||||
.vectors
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(i, v)| MemoryEntry {
|
||||
chunk: format!("record {i} cluster {}", data.cluster[i]),
|
||||
embedding: v.clone(),
|
||||
source_channel: channel(i),
|
||||
timestamp: i as f64,
|
||||
session_id: "s".into(),
|
||||
tags: format!("t{i}"),
|
||||
})
|
||||
.collect();
|
||||
m.save_batch(entries).unwrap();
|
||||
(dir, m)
|
||||
}
|
||||
|
||||
/// Exact top-k by cosine among the records `allowed` keeps.
|
||||
fn exact_top(data: &Data, q: &[f32], k: usize, allowed: impl Fn(usize) -> bool) -> Vec<usize> {
|
||||
let mut s: Vec<(usize, f32)> = (0..N)
|
||||
.filter(|&i| allowed(i))
|
||||
.map(|i| (i, data.vectors[i].iter().zip(q).map(|(a, b)| a * b).sum()))
|
||||
.collect();
|
||||
s.sort_by(|a, b| b.1.total_cmp(&a.1).then(a.0.cmp(&b.0)));
|
||||
s.into_iter().take(k).map(|(i, _)| i).collect()
|
||||
}
|
||||
|
||||
fn query(data: &Data, i: usize) -> Vec<f32> {
|
||||
let mut rng = Rng(1000 + i as u64);
|
||||
let mut q: Vec<f32> = data.centres[i % CLUSTERS]
|
||||
.iter()
|
||||
.map(|x| x + rng.unit() * 0.3)
|
||||
.collect();
|
||||
normalize(&mut q);
|
||||
q
|
||||
}
|
||||
|
||||
fn vector_only(k: usize) -> SearchOptions {
|
||||
SearchOptions::new(k).with_fusion(hybrid::Fusion::Weighted {
|
||||
vector: 1.0,
|
||||
keyword: 0.0,
|
||||
})
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn source_filter_returns_only_allowed_records_and_a_full_page() {
|
||||
let d = data();
|
||||
// At N = 3000 and k = 10 the index serves a filter only when that is
|
||||
// cheaper than scanning the allowed records: pool = 80 * N / allowed
|
||||
// candidates at ~M = 16 distances each, against `allowed` distances. So
|
||||
// 90% goes through the index, 50% and 1% to the exact scan.
|
||||
for percent in [90, 50, 1] {
|
||||
let (_dir, mut m) = build(&d, |i| random_channel(i, 5, percent));
|
||||
let allowed = |i: usize| random_channel(i, 5, percent) == "keep";
|
||||
let mut hits = 0;
|
||||
for qi in 0..40 {
|
||||
let q = query(&d, qi);
|
||||
let got = m.search(&q, "", &vector_only(10).with_sources(["keep"]));
|
||||
assert_eq!(got.len(), 10, "{percent}%: short page");
|
||||
assert!(got.iter().all(|r| r.source_channel == "keep"));
|
||||
let want: HashSet<usize> = exact_top(&d, &q, 10, allowed).into_iter().collect();
|
||||
hits += got.iter().filter(|r| want.contains(&r.index)).count();
|
||||
}
|
||||
let recall = hits as f64 / 400.0;
|
||||
let floor = if percent == 90 { 0.95 } else { 1.0 };
|
||||
assert!(recall >= floor, "{percent}%: recall@10 {recall}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn filter_away_from_the_query_falls_back_to_an_exact_scan() {
|
||||
// Channel = cluster, and the filter keeps two clusters (10% of the
|
||||
// store) that are not the query's: the index's neighbourhood of the
|
||||
// query holds none of them. The search must still return the exact
|
||||
// top 10 among the allowed records, not a short or empty page.
|
||||
let d = data();
|
||||
let (_dir, mut m) = build(&d, |i| format!("c{}", d.cluster[i]));
|
||||
for qi in 0..20 {
|
||||
let q = query(&d, qi);
|
||||
let a = format!("c{}", (qi + 7) % CLUSTERS);
|
||||
let b = format!("c{}", (qi + 13) % CLUSTERS);
|
||||
let got: Vec<usize> = m
|
||||
.search(
|
||||
&q,
|
||||
"",
|
||||
&vector_only(10).with_sources([a.clone(), b.clone()]),
|
||||
)
|
||||
.iter()
|
||||
.map(|r| r.index)
|
||||
.collect();
|
||||
let want = exact_top(&d, &q, 10, |i| {
|
||||
let c = format!("c{}", d.cluster[i]);
|
||||
c == a || c == b
|
||||
});
|
||||
assert_eq!(got, want, "query {qi}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn filter_edge_cases() {
|
||||
let d = data();
|
||||
let (_dir, mut m) = build(&d, |i| random_channel(i, 9, 50));
|
||||
let q = query(&d, 0);
|
||||
assert!(
|
||||
m.search(
|
||||
&q,
|
||||
"cluster",
|
||||
&SearchOptions::new(10).with_sources(Vec::<String>::new())
|
||||
)
|
||||
.is_empty()
|
||||
);
|
||||
assert!(
|
||||
m.search(
|
||||
&q,
|
||||
"cluster",
|
||||
&SearchOptions::new(10).with_sources(["nope"])
|
||||
)
|
||||
.is_empty()
|
||||
);
|
||||
// Keyword matches from other channels are filtered too.
|
||||
let got = m.search(
|
||||
&q,
|
||||
"record cluster",
|
||||
&SearchOptions::new(50).with_sources(["keep"]),
|
||||
);
|
||||
assert_eq!(got.len(), 50);
|
||||
assert!(got.iter().all(|r| r.source_channel == "keep"));
|
||||
// Deleted records never come back, filtered or not.
|
||||
let first = got[0].index;
|
||||
m.delete(first).unwrap();
|
||||
let again = m.search(
|
||||
&q,
|
||||
"record cluster",
|
||||
&SearchOptions::new(50).with_sources(["keep"]),
|
||||
);
|
||||
assert!(again.iter().all(|r| r.index != first));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn plain_options_equal_hybrid_search_with() {
|
||||
// Two identical stores, so neither query sees the other's boosts.
|
||||
let d = data();
|
||||
let (_a, mut a) = build(&d, |i| random_channel(i, 3, 50));
|
||||
let (_b, mut b) = build(&d, |i| random_channel(i, 3, 50));
|
||||
for qi in 0..10 {
|
||||
let q = query(&d, qi);
|
||||
let x: Vec<(usize, u32)> = a
|
||||
.search(&q, "record cluster 3", &SearchOptions::new(10))
|
||||
.iter()
|
||||
.map(|r| (r.index, r.score.to_bits()))
|
||||
.collect();
|
||||
let y: Vec<(usize, u32)> = b
|
||||
.hybrid_search_with(&q, "record cluster 3", hybrid::DEFAULT_FUSION, 10)
|
||||
.iter()
|
||||
.map(|r| (r.index, r.score.to_bits()))
|
||||
.collect();
|
||||
assert_eq!(x, y);
|
||||
}
|
||||
}
|
||||
|
||||
fn small_store(entries: &[(&str, &str, f64)]) -> (TempDir, HDF5Memory) {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let mut m = HDF5Memory::create(MemoryConfig::new(dir.path().join("r.h5"), "a", 4)).unwrap();
|
||||
m.save_batch(
|
||||
entries
|
||||
.iter()
|
||||
.map(|(chunk, channel, ts)| MemoryEntry {
|
||||
chunk: chunk.to_string(),
|
||||
embedding: vec![1.0, 0.0, 0.0, 0.0],
|
||||
source_channel: channel.to_string(),
|
||||
timestamp: *ts,
|
||||
session_id: "s".into(),
|
||||
tags: String::new(),
|
||||
})
|
||||
.collect(),
|
||||
)
|
||||
.unwrap();
|
||||
(dir, m)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rerank_breaks_relevance_ties_by_recency() {
|
||||
// Identical text and vectors, so retrieval ties; re-ranking must put the
|
||||
// newer record first and report the combined score.
|
||||
let now = 1_000_000.0;
|
||||
let (_d, mut m) = small_store(&[
|
||||
("user prefers dark mode", "chat", now - 30.0 * 86_400.0),
|
||||
("user prefers dark mode", "chat", now - 60.0),
|
||||
]);
|
||||
let q = [1.0, 0.0, 0.0, 0.0];
|
||||
let plain = m.search(&q, "dark mode", &SearchOptions::new(2));
|
||||
assert_eq!(plain[0].index, 0, "ties break by index without re-ranking");
|
||||
let reranked = m.search(
|
||||
&q,
|
||||
"dark mode",
|
||||
&SearchOptions::new(2)
|
||||
.with_rerank(ReRankConfig::default())
|
||||
.at_time(now),
|
||||
);
|
||||
assert_eq!(reranked[0].index, 1);
|
||||
assert!(reranked[0].score > reranked[1].score);
|
||||
assert_ne!(reranked[0].score.to_bits(), plain[0].score.to_bits());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn confidence_rejects_when_nothing_is_good_enough() {
|
||||
let (_d, mut m) = small_store(&[("alpha", "chat", 0.0), ("beta", "chat", 0.0)]);
|
||||
let q = [1.0, 0.0, 0.0, 0.0];
|
||||
let strict = ConfidenceConfig {
|
||||
min_score: 10.0,
|
||||
..ConfidenceConfig::default()
|
||||
};
|
||||
assert!(
|
||||
m.search(&q, "alpha", &SearchOptions::new(2).with_confidence(strict))
|
||||
.is_empty()
|
||||
);
|
||||
let lenient = ConfidenceConfig {
|
||||
min_score: 0.0,
|
||||
min_gap: f32::INFINITY,
|
||||
max_results: 1,
|
||||
};
|
||||
assert_eq!(
|
||||
m.search(&q, "alpha", &SearchOptions::new(2).with_confidence(lenient))
|
||||
.len(),
|
||||
1
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn only_returned_results_are_reinforced() {
|
||||
// With re-ranking, a pool of max(3k, 10) candidates is retrieved; only
|
||||
// the k returned should gain activation.
|
||||
let d = data();
|
||||
let dir = TempDir::new().unwrap();
|
||||
let path = dir.path().join("h.h5");
|
||||
let mut m = HDF5Memory::create(MemoryConfig::new(path, "a", DIM)).unwrap();
|
||||
m.save_batch(
|
||||
(0..200)
|
||||
.map(|i| MemoryEntry {
|
||||
chunk: format!("record {i}"),
|
||||
embedding: d.vectors[i].clone(),
|
||||
source_channel: "chat".into(),
|
||||
timestamp: i as f64,
|
||||
session_id: "s".into(),
|
||||
tags: String::new(),
|
||||
})
|
||||
.collect(),
|
||||
)
|
||||
.unwrap();
|
||||
let q = query(&d, 0);
|
||||
let got = m.search(
|
||||
&q,
|
||||
"record",
|
||||
&SearchOptions::new(3).with_rerank(ReRankConfig::default()),
|
||||
);
|
||||
assert_eq!(got.len(), 3);
|
||||
let returned: HashSet<usize> = got.iter().map(|r| r.index).collect();
|
||||
// A second plain search reports each record's current activation.
|
||||
let all = m.search(&q, "record", &SearchOptions::new(200));
|
||||
for r in &all {
|
||||
let boosted = r.activation > 1.0;
|
||||
assert_eq!(boosted, returned.contains(&r.index), "record {}", r.index);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,330 @@
|
||||
//! Ed25519-signed checkpoints: `HDF5Memory::set_signing_key` and
|
||||
//! `HDF5Memory::verify`.
|
||||
|
||||
use std::path::Path;
|
||||
|
||||
use clawhdf5_agent::signing::{SigningKey, VerifyReport, VerifyingKey};
|
||||
use clawhdf5_agent::storage;
|
||||
use clawhdf5_agent::{AgentMemory, HDF5Memory, MemoryConfig, MemoryEntry, MemoryError, schema};
|
||||
use tempfile::TempDir;
|
||||
|
||||
const DIM: usize = 16;
|
||||
|
||||
fn key(seed: u8) -> SigningKey {
|
||||
SigningKey::from_bytes(&[seed; 32])
|
||||
}
|
||||
|
||||
fn entry(i: usize, chunk: &str) -> MemoryEntry {
|
||||
MemoryEntry {
|
||||
chunk: chunk.to_string(),
|
||||
embedding: (0..DIM)
|
||||
.map(|j| ((i * DIM + j) as f32 * 0.37).sin())
|
||||
.collect(),
|
||||
source_channel: "chat".into(),
|
||||
timestamp: 1_700_000_000.0 + i as f64,
|
||||
session_id: format!("s{}", i % 3),
|
||||
tags: format!("t{i}"),
|
||||
}
|
||||
}
|
||||
|
||||
/// Awkward strings on purpose: they must hash the same after a round trip.
|
||||
const TEXTS: [&str; 6] = [
|
||||
"plain text",
|
||||
"ünïcödé — 日本語 🙂",
|
||||
"",
|
||||
"trailing spaces ",
|
||||
"tab\tand\nnewline",
|
||||
"x",
|
||||
];
|
||||
|
||||
fn signed_store(dir: &TempDir, float16: bool, k: &SigningKey) -> std::path::PathBuf {
|
||||
let mut cfg = MemoryConfig::new(dir.path().join("s.h5"), "agent", DIM);
|
||||
cfg.float16 = float16;
|
||||
let path = cfg.path.clone();
|
||||
let mut m = HDF5Memory::create(cfg).unwrap();
|
||||
m.set_signing_key(k.clone());
|
||||
let entries = (0..30).map(|i| entry(i, TEXTS[i % TEXTS.len()])).collect();
|
||||
m.save_batch(entries).unwrap();
|
||||
// Some graph and a deleted record, so every part of the manifest is used.
|
||||
let a = m.knowledge_mut().add_entity("Alice", "person", 0);
|
||||
let b = m.knowledge_mut().add_entity("Acme", "org", -1);
|
||||
m.knowledge_mut().add_relation(a, b, "works_at", 0.75);
|
||||
m.sessions_mut()
|
||||
.add_at("s0", 0, 9, "chat", "first session", 1_700_000_000.0);
|
||||
m.delete(4).unwrap();
|
||||
m.flush_wal().unwrap();
|
||||
path
|
||||
}
|
||||
|
||||
fn verify(path: &Path, k: &SigningKey) -> VerifyReport {
|
||||
HDF5Memory::verify(path, &k.verifying_key()).unwrap()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_signed_store_verifies_through_reopen_and_checkpoint_cycles() {
|
||||
for float16 in [true, false] {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let k = key(7);
|
||||
let path = signed_store(&dir, float16, &k);
|
||||
let r = verify(&path, &k);
|
||||
assert!(r.is_valid(), "float16={float16}: {r:?}");
|
||||
assert_eq!(r.public_key, Some(k.verifying_key().to_bytes()));
|
||||
assert_eq!(r.record_count, 30);
|
||||
assert!(r.changed_records.is_empty());
|
||||
|
||||
// Reopen, change nothing, checkpoint again (with the key): still valid.
|
||||
for _ in 0..3 {
|
||||
let mut m = HDF5Memory::open(&path).unwrap();
|
||||
assert!(m.is_signed());
|
||||
m.set_signing_key(k.clone());
|
||||
m.flush_wal().unwrap();
|
||||
drop(m);
|
||||
assert!(verify(&path, &k).is_valid());
|
||||
}
|
||||
// And after real changes, re-signed.
|
||||
let mut m = HDF5Memory::open(&path).unwrap();
|
||||
m.set_signing_key(k.clone());
|
||||
m.save(entry(99, "added later")).unwrap();
|
||||
m.hybrid_search(&entry(1, "").embedding, "text", 0.4, 0.6, 5);
|
||||
m.flush_wal().unwrap();
|
||||
drop(m);
|
||||
let r = verify(&path, &k);
|
||||
assert!(r.is_valid(), "{r:?}");
|
||||
assert_eq!(r.record_count, 31);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_signed_store_refuses_to_checkpoint_without_its_key() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let k = key(1);
|
||||
let path = signed_store(&dir, true, &k);
|
||||
|
||||
let mut m = HDF5Memory::open(&path).unwrap();
|
||||
m.save(entry(50, "pending")).unwrap();
|
||||
match m.flush_wal() {
|
||||
Err(MemoryError::SigningKeyRequired(msg)) => assert!(msg.contains("signed"), "{msg}"),
|
||||
other => panic!("expected SigningKeyRequired, got {other:?}"),
|
||||
}
|
||||
// The file is untouched and still valid; the save is still in the WAL.
|
||||
let r = verify(&path, &k);
|
||||
assert!(r.is_valid());
|
||||
assert_eq!(r.wal_entries_unsigned, 1);
|
||||
|
||||
// Supplying the key lets the checkpoint through, signed.
|
||||
m.set_signing_key(k.clone());
|
||||
m.flush_wal().unwrap();
|
||||
drop(m);
|
||||
let r = verify(&path, &k);
|
||||
assert!(r.is_valid());
|
||||
assert_eq!((r.record_count, r.wal_entries_unsigned), (31, 0));
|
||||
|
||||
// Removing the signature on purpose writes it unsigned.
|
||||
let mut m = HDF5Memory::open(&path).unwrap();
|
||||
m.remove_signature();
|
||||
m.flush_wal().unwrap();
|
||||
drop(m);
|
||||
let r = verify(&path, &k);
|
||||
assert!(!r.signed && !r.is_valid());
|
||||
assert!(!HDF5Memory::open(&path).unwrap().is_signed());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_wrong_key_does_not_verify_and_a_new_key_re_signs() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let (a, b) = (key(1), key(2));
|
||||
let path = signed_store(&dir, true, &a);
|
||||
let r = verify(&path, &b);
|
||||
assert!(r.signed && !r.key_matches && !r.signature_valid && !r.is_valid());
|
||||
|
||||
let mut m = HDF5Memory::open(&path).unwrap();
|
||||
m.set_signing_key(b.clone());
|
||||
m.flush_wal().unwrap();
|
||||
drop(m);
|
||||
assert!(verify(&path, &b).is_valid());
|
||||
assert!(!verify(&path, &a).is_valid());
|
||||
}
|
||||
|
||||
/// Rewrite the store with changed contents but the *old* signature — what
|
||||
/// someone with write access to the file, but not the key, can do.
|
||||
fn tamper(path: &Path, change: impl FnOnce(&mut Tampered)) {
|
||||
let file = clawhdf5::File::open(path).unwrap();
|
||||
let (config, cache, sessions, knowledge) = schema::validate_and_load(&file).unwrap();
|
||||
let checkpoint = schema::read_checkpoint_meta(&file);
|
||||
let signature = schema::read_signature(&file).unwrap().unwrap();
|
||||
drop(file);
|
||||
let mut t = Tampered {
|
||||
config,
|
||||
cache,
|
||||
sessions,
|
||||
knowledge,
|
||||
};
|
||||
change(&mut t);
|
||||
storage::write_to_disk_signed(
|
||||
path,
|
||||
&t.config,
|
||||
&t.cache,
|
||||
&t.sessions,
|
||||
&t.knowledge,
|
||||
&checkpoint,
|
||||
Some(&signature),
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
struct Tampered {
|
||||
config: MemoryConfig,
|
||||
cache: clawhdf5_agent::cache::MemoryCache,
|
||||
sessions: clawhdf5_agent::SessionCache,
|
||||
knowledge: clawhdf5_agent::knowledge::KnowledgeCache,
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn every_kind_of_edit_is_detected_and_located() {
|
||||
let k = key(3);
|
||||
type Edit = Box<dyn FnOnce(&mut Tampered)>;
|
||||
type Case = (&'static str, Edit, fn(&VerifyReport) -> bool);
|
||||
let cases: Vec<Case> = vec![
|
||||
(
|
||||
"record text",
|
||||
Box::new(|t: &mut Tampered| t.cache.chunks[7] = "rewritten".into()),
|
||||
|r| !r.records_match && r.changed_records == vec![7],
|
||||
),
|
||||
(
|
||||
"one embedding value",
|
||||
Box::new(|t: &mut Tampered| {
|
||||
let mut e = t.cache.embeddings[12].to_vec();
|
||||
e[3] = 0.5;
|
||||
t.cache.embeddings.set(12, &e);
|
||||
}),
|
||||
|r| r.changed_records == vec![12],
|
||||
),
|
||||
(
|
||||
"undelete",
|
||||
Box::new(|t: &mut Tampered| t.cache.tombstones[4] = 0),
|
||||
|r| r.changed_records == vec![4],
|
||||
),
|
||||
(
|
||||
"timestamp",
|
||||
Box::new(|t: &mut Tampered| t.cache.timestamps[20] += 1.0),
|
||||
|r| r.changed_records == vec![20],
|
||||
),
|
||||
(
|
||||
"record appended",
|
||||
Box::new(|t: &mut Tampered| {
|
||||
t.cache.push(
|
||||
"new".into(),
|
||||
vec![0.1; DIM],
|
||||
"x".into(),
|
||||
1.0,
|
||||
"s".into(),
|
||||
"".into(),
|
||||
);
|
||||
}),
|
||||
|r| !r.records_match && r.changed_records == vec![30] && r.record_count == 31,
|
||||
),
|
||||
(
|
||||
"setting",
|
||||
Box::new(|t: &mut Tampered| t.config.agent_id = "someone-else".into()),
|
||||
|r| !r.settings_match && r.records_match,
|
||||
),
|
||||
(
|
||||
"session summary",
|
||||
Box::new(|t: &mut Tampered| t.sessions.summaries[0] = "edited".into()),
|
||||
|r| !r.sessions_match && r.records_match,
|
||||
),
|
||||
(
|
||||
"graph edge",
|
||||
Box::new(|t: &mut Tampered| t.knowledge.relations[0].weight = 1.0),
|
||||
|r| !r.graph_match && r.records_match,
|
||||
),
|
||||
];
|
||||
for (name, edit, check) in cases {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let path = signed_store(&dir, true, &k);
|
||||
tamper(&path, edit);
|
||||
let r = verify(&path, &k);
|
||||
assert!(
|
||||
r.signed && r.key_matches && r.signature_valid,
|
||||
"{name}: {r:?}"
|
||||
);
|
||||
assert!(!r.is_valid(), "{name}: edit not detected: {r:?}");
|
||||
assert!(check(&r), "{name}: {r:?}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_forged_manifest_fails_the_signature() {
|
||||
// Recomputing the hashes for tampered contents does not help without the
|
||||
// key: the signature no longer matches the manifest.
|
||||
let dir = TempDir::new().unwrap();
|
||||
let k = key(5);
|
||||
let path = signed_store(&dir, true, &k);
|
||||
let file = clawhdf5::File::open(&path).unwrap();
|
||||
let (config, mut cache, sessions, knowledge) = schema::validate_and_load(&file).unwrap();
|
||||
let checkpoint = schema::read_checkpoint_meta(&file);
|
||||
let mut sig = schema::read_signature(&file).unwrap().unwrap();
|
||||
drop(file);
|
||||
cache.chunks[0] = "forged".into();
|
||||
// Re-sign with an attacker key, then splice the victim's public key back.
|
||||
let forged = clawhdf5_agent::signing::sign(
|
||||
&key(66),
|
||||
&config,
|
||||
&cache,
|
||||
&sessions,
|
||||
&knowledge,
|
||||
checkpoint.wal_applied,
|
||||
);
|
||||
sig.manifest = forged.manifest;
|
||||
sig.record_hashes = forged.record_hashes;
|
||||
storage::write_to_disk_signed(
|
||||
&path,
|
||||
&config,
|
||||
&cache,
|
||||
&sessions,
|
||||
&knowledge,
|
||||
&checkpoint,
|
||||
Some(&sig),
|
||||
)
|
||||
.unwrap();
|
||||
let r = verify(&path, &k);
|
||||
assert!(
|
||||
r.key_matches && !r.signature_valid && !r.is_valid(),
|
||||
"{r:?}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_unsigned_store_reports_unsigned() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let mut m = HDF5Memory::create(MemoryConfig::new(dir.path().join("u.h5"), "a", DIM)).unwrap();
|
||||
m.save_batch(vec![entry(0, "hello")]).unwrap();
|
||||
drop(m);
|
||||
let r = HDF5Memory::verify(&dir.path().join("u.h5"), &VerifyingKey::from(&key(1))).unwrap();
|
||||
assert!(!r.signed && !r.is_valid());
|
||||
assert_eq!(r.record_count, 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn nul_bytes_in_text_still_verify() {
|
||||
// Strings are stored null-padded; the hash must follow what a reopened
|
||||
// store actually holds, or an untouched store would fail to verify.
|
||||
let dir = TempDir::new().unwrap();
|
||||
let k = key(9);
|
||||
let mut m = HDF5Memory::create(MemoryConfig::new(dir.path().join("n.h5"), "a", DIM)).unwrap();
|
||||
m.set_signing_key(k.clone());
|
||||
m.save_batch(vec![
|
||||
entry(0, "inner\0nul"),
|
||||
entry(1, "trailing nul\0"),
|
||||
entry(2, "\0leading"),
|
||||
])
|
||||
.unwrap();
|
||||
drop(m);
|
||||
let r = verify(&dir.path().join("n.h5"), &k);
|
||||
assert!(r.is_valid(), "{r:?}");
|
||||
let m = HDF5Memory::open(&dir.path().join("n.h5")).unwrap();
|
||||
eprintln!(
|
||||
"reloaded: {:?}",
|
||||
(0..3).map(|i| m.get_chunk(i)).collect::<Vec<_>>()
|
||||
);
|
||||
}
|
||||
@@ -1,7 +1,8 @@
|
||||
[package]
|
||||
name = "clawhdf5-android"
|
||||
version = "2.6.0"
|
||||
version = "2.7.0"
|
||||
edition = "2024"
|
||||
rust-version.workspace = true
|
||||
description = "Android JNI bridge for edgehdf5-memory HDF5 backend"
|
||||
license = "MIT"
|
||||
|
||||
|
||||
@@ -1,7 +1,8 @@
|
||||
[package]
|
||||
name = "clawhdf5-ann"
|
||||
version = "2.6.0"
|
||||
version = "2.7.0"
|
||||
edition = "2024"
|
||||
rust-version.workspace = true
|
||||
description = "HNSW approximate nearest neighbor index stored as HDF5"
|
||||
license = "MIT"
|
||||
repository = "https://git.redclaw.dev/quantumclaw/clawhdf5"
|
||||
@@ -10,9 +11,9 @@ keywords = ["hdf5", "ann", "hnsw", "nearest-neighbor"]
|
||||
categories = ["algorithms", "science"]
|
||||
|
||||
[dependencies]
|
||||
clawhdf5-format = { path = "../clawhdf5-format", version = "2.6.0" }
|
||||
clawhdf5-io = { path = "../clawhdf5-io", version = "2.6.0" }
|
||||
clawhdf5-accel = { path = "../clawhdf5-accel", version = "2.6.0" }
|
||||
clawhdf5-format = { path = "../clawhdf5-format", version = "2.7.0" }
|
||||
clawhdf5-io = { path = "../clawhdf5-io", version = "2.7.0" }
|
||||
clawhdf5-accel = { path = "../clawhdf5-accel", version = "2.7.0" }
|
||||
rayon = { version = "1", optional = true }
|
||||
|
||||
[features]
|
||||
|
||||
@@ -13,7 +13,7 @@ use clawhdf5_format::filter_pipeline::FilterPipeline;
|
||||
use clawhdf5_format::group_v2::resolve_path_any;
|
||||
use clawhdf5_format::message_type::MessageType;
|
||||
use clawhdf5_format::object_header::ObjectHeader;
|
||||
use clawhdf5_format::signature::find_signature;
|
||||
use clawhdf5_format::signature::split_user_block;
|
||||
use clawhdf5_format::superblock::Superblock;
|
||||
use clawhdf5_io::FileWriter as IoFileWriter;
|
||||
|
||||
@@ -358,31 +358,12 @@ enum Query {
|
||||
Int8(Vec<i8>, f32),
|
||||
}
|
||||
|
||||
/// Sum of products, widened so it cannot overflow: `dim` terms of at most
|
||||
/// `127 * 127`, so `i32` suffices for any realistic dimension.
|
||||
/// Sum of products, widened so it cannot overflow. Runtime-dispatched to the
|
||||
/// same SIMD backend as the f32 kernels, so the two storages are compared on
|
||||
/// equal terms.
|
||||
#[inline]
|
||||
fn dot_i8(a: &[i8], b: &[i8]) -> i32 {
|
||||
// Four independent accumulators over 32-lane blocks: the widening product
|
||||
// has to sit in a fixed-length chunk for the vectoriser to see it, and the
|
||||
// separate accumulators keep it off one dependency chain.
|
||||
const LANE: usize = 8;
|
||||
let (a_blocks, a_tail) = a.as_chunks::<{ LANE * 4 }>();
|
||||
let (b_blocks, b_tail) = b.as_chunks::<{ LANE * 4 }>();
|
||||
let mut acc = [0i32; 4];
|
||||
for (x, y) in a_blocks.iter().zip(b_blocks) {
|
||||
for (lane, slot) in acc.iter_mut().enumerate() {
|
||||
let mut sum = 0i32;
|
||||
for k in 0..LANE {
|
||||
sum += i32::from(x[lane * LANE + k]) * i32::from(y[lane * LANE + k]);
|
||||
}
|
||||
*slot += sum;
|
||||
}
|
||||
}
|
||||
let tail: i32 = a_tail
|
||||
.iter()
|
||||
.zip(b_tail)
|
||||
.map(|(&x, &y)| i32::from(x) * i32::from(y))
|
||||
.sum();
|
||||
acc[0] + acc[1] + acc[2] + acc[3] + tail
|
||||
clawhdf5_accel::dot_i8(a, b)
|
||||
}
|
||||
|
||||
/// Magic for [`HnswIndex::graph_to_bytes`].
|
||||
@@ -880,8 +861,9 @@ impl HnswIndex {
|
||||
/// The HDF5 data must contain the `/ann/vectors`, `/ann/graph_layer_*`,
|
||||
/// and `/ann/config` datasets as produced by [`to_hdf5_bytes`].
|
||||
pub fn load_from_hdf5(data: &[u8]) -> Result<Self, FormatError> {
|
||||
let sig_offset = find_signature(data)?;
|
||||
let sb = Superblock::parse(data, sig_offset)?;
|
||||
// Addresses are relative to the superblock: skip any user block.
|
||||
let (_, data) = split_user_block(data)?;
|
||||
let sb = Superblock::parse(data, 0)?;
|
||||
|
||||
// Read config dataset and its attributes
|
||||
let config_attrs = read_dataset_attrs(data, &sb, "ann/config")?;
|
||||
|
||||
@@ -1,7 +1,8 @@
|
||||
[package]
|
||||
name = "clawhdf5-bench"
|
||||
version = "2.6.0"
|
||||
version = "2.7.0"
|
||||
edition = "2024"
|
||||
rust-version.workspace = true
|
||||
description = "Benchmark harnesses for clawhdf5-agent (Track 8)"
|
||||
license = "MIT"
|
||||
|
||||
@@ -33,6 +34,10 @@ path = "src/bin/consolidation_efficiency.rs"
|
||||
name = "ephemeral_perf"
|
||||
path = "src/bin/ephemeral_perf.rs"
|
||||
|
||||
[[bin]]
|
||||
name = "concurrent_read"
|
||||
path = "src/bin/concurrent_read.rs"
|
||||
|
||||
[[bin]]
|
||||
name = "mpi_io_bench"
|
||||
path = "src/bin/mpi_io_bench.rs"
|
||||
@@ -63,6 +68,10 @@ clawhdf5-io = { path = "../clawhdf5-io" }
|
||||
mpi = { version = "0.8", optional = true }
|
||||
serde = { workspace = true }
|
||||
serde_json = "1"
|
||||
# concurrent_read: size the decode pool (--decode-threads) and evict files
|
||||
# from the page cache (--cold, posix_fadvise). Both pure Rust / bindings only.
|
||||
rayon = "1"
|
||||
libc = "0.2"
|
||||
tempfile = { workspace = true }
|
||||
# Optional: libhdf5 C wrapper for side-by-side comparison (requires system libhdf5).
|
||||
# Enable with: cargo bench -p clawhdf5-bench --features libhdf5-compare
|
||||
|
||||
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,70 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Tabulate concurrent_read JSON results (clawhdf5, h5py threads/processes).
|
||||
|
||||
python compare_concurrent_read.py clawhdf5.json h5py-threads.json h5py-procs.json
|
||||
|
||||
Prints one Markdown table: for each layout, mode and thread count, every
|
||||
tool's MB/s and scaling efficiency, and the first file's MB/s relative to each
|
||||
of the others. Refuses to compare runs whose workload parameters differ.
|
||||
"""
|
||||
|
||||
import json
|
||||
import sys
|
||||
|
||||
COMPARED = ("datasets", "rows", "cols", "chunk", "deflate_level", "slab", "slabs", "seed")
|
||||
|
||||
|
||||
def main(paths):
|
||||
if len(paths) < 2:
|
||||
sys.exit(__doc__)
|
||||
docs = []
|
||||
for p in paths:
|
||||
with open(p) as fh:
|
||||
docs.append(json.load(fh))
|
||||
ref = docs[0]
|
||||
for d, p in zip(docs[1:], paths[1:]):
|
||||
diff = [k for k in COMPARED if d["params"].get(k) != ref["params"].get(k)]
|
||||
if diff:
|
||||
sys.exit(f"{p}: workload differs from {paths[0]} in {', '.join(diff)}")
|
||||
if d["cache"] != ref["cache"]:
|
||||
print(f"warning: {p} ran {d['cache']!r}, {paths[0]} ran {ref['cache']!r}",
|
||||
file=sys.stderr)
|
||||
if d.get("host") != ref.get("host"):
|
||||
print(f"warning: {p} ran on {d.get('host')}, {paths[0]} on {ref.get('host')}",
|
||||
file=sys.stderr)
|
||||
|
||||
names = [d["tool"] for d in docs]
|
||||
for d in docs:
|
||||
extra = f", HDF5 {d['hdf5_version']}" if "hdf5_version" in d else ""
|
||||
print(f"- {d['tool']} {d['version']}{extra}: host {d.get('host')}, "
|
||||
f"{d.get('cpus')} CPUs, cache {d['cache']}, decode threads per read "
|
||||
f"{d.get('decode_threads')}")
|
||||
p = ref["params"]
|
||||
print(f"\n{p['datasets']} datasets of {p['rows']} x {p['cols']} f32, chunks "
|
||||
f"{p['chunk'][0]} x {p['chunk'][1]} (deflate {p['deflate_level']}); "
|
||||
f"`same`: {p['slabs']} slabs of {p['slab']} x {p['slab']}\n")
|
||||
|
||||
index = [{(r["layout"], r["mode"], r["threads"]): r for r in d["results"]} for d in docs]
|
||||
keys = [(r["layout"], r["mode"], r["threads"]) for r in ref["results"]]
|
||||
|
||||
head = ["layout", "mode", "threads"]
|
||||
head += [f"{n} MB/s (eff)" for n in names]
|
||||
head += [f"{names[0]} / {n}" for n in names[1:]]
|
||||
print("| " + " | ".join(head) + " |")
|
||||
print("|---|---|" + "---:|" * (len(head) - 2))
|
||||
for key in keys:
|
||||
cells = [key[0], key[1], str(key[2])]
|
||||
rs = [ix.get(key) for ix in index]
|
||||
for r in rs:
|
||||
if r is None:
|
||||
cells.append("-")
|
||||
else:
|
||||
eff = "-" if r["efficiency"] is None else f"{r['efficiency']:.2f}"
|
||||
cells.append(f"{r['mb_s']:.0f} ({eff})")
|
||||
for r in rs[1:]:
|
||||
cells.append("-" if r is None else f"{rs[0]['mb_s'] / r['mb_s']:.2f}x")
|
||||
print("| " + " | ".join(cells) + " |")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main(sys.argv[1:])
|
||||
@@ -0,0 +1,265 @@
|
||||
#!/usr/bin/env python3
|
||||
"""The concurrent_read workload with h5py, on the files concurrent_read wrote.
|
||||
|
||||
libhdf5 serialises every API call under one global lock, and h5py holds its
|
||||
own global lock around every call as well, so h5py *threads* cannot decode in
|
||||
parallel. h5py users scale with *processes* instead; ``--executor processes``
|
||||
measures that (each worker opens the file itself).
|
||||
|
||||
The workload mirrors ``crates/clawhdf5-bench/src/bin/concurrent_read.rs``:
|
||||
|
||||
* ``distinct``: every dataset read in full once per repetition; worker ``t``
|
||||
of ``T`` reads datasets ``t, t + T, ...``.
|
||||
* ``same``: ``--slabs`` random ``--slab`` x ``--slab`` hyperslabs of ``d00``
|
||||
(slab ``j`` to worker ``j % T``), offsets from the same splitmix64 stream.
|
||||
|
||||
Each worker times itself from a start barrier; a repetition spans the earliest
|
||||
start to the latest finish (CLOCK_MONOTONIC, comparable across processes).
|
||||
Threads share one ``h5py.File`` per repetition; process workers open the file
|
||||
inside the timed region (a few ms against reads of many MiB).
|
||||
|
||||
Generate the files first with the Rust harness (it writes ``manifest.json``),
|
||||
then, for example::
|
||||
|
||||
python concurrent_read_h5py.py --dir DIR --executor threads --json h5py-threads.json
|
||||
python concurrent_read_h5py.py --dir DIR --executor processes --json h5py-procs.json
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import multiprocessing as mp
|
||||
import os
|
||||
import platform
|
||||
import socket
|
||||
import sys
|
||||
import threading
|
||||
import time
|
||||
|
||||
import h5py
|
||||
import numpy as np
|
||||
|
||||
M64 = (1 << 64) - 1
|
||||
|
||||
|
||||
def splitmix64(state):
|
||||
"""Return (new_state, value); the same stream as the Rust harness."""
|
||||
state = (state + 0x9E3779B97F4A7C15) & M64
|
||||
z = state
|
||||
z = ((z ^ (z >> 30)) * 0xBF58476D1CE4E5B9) & M64
|
||||
z = ((z ^ (z >> 27)) * 0x94D049BB133111EB) & M64
|
||||
return state, z ^ (z >> 31)
|
||||
|
||||
|
||||
def value(k, i):
|
||||
"""Element i (row-major) of dataset k, exactly as concurrent_read writes it."""
|
||||
_, noise = splitmix64(i ^ (k << 40))
|
||||
return np.float32((((i >> 6) % 16384) + k) + (noise & 0xFF) / 256.0)
|
||||
|
||||
|
||||
def slab_offsets(seed, count, rows, cols, slab):
|
||||
s = seed
|
||||
out = []
|
||||
for _ in range(count):
|
||||
s, r = splitmix64(s)
|
||||
s, c = splitmix64(s)
|
||||
out.append((r % (rows - slab + 1), c % (cols - slab + 1)))
|
||||
return out
|
||||
|
||||
|
||||
def now():
|
||||
return time.clock_gettime(time.CLOCK_MONOTONIC)
|
||||
|
||||
|
||||
def work(f, mode, t, threads, m, slabs, slab, verify):
|
||||
"""Worker t's share of one repetition on an open h5py.File."""
|
||||
n = m["rows"] * m["cols"]
|
||||
if mode == "distinct":
|
||||
for k in range(t, m["datasets"], threads):
|
||||
got = f[f"d{k:02d}"][...]
|
||||
assert got.size == n
|
||||
if verify:
|
||||
flat = got.reshape(-1)
|
||||
for i in (0, n // 3, n - 1):
|
||||
assert flat[i] == value(k, i), f"d{k:02d}[{i}]"
|
||||
else:
|
||||
ds = f["d00"]
|
||||
cols = m["cols"]
|
||||
for r, c in slabs[t::threads]:
|
||||
got = ds[r : r + slab, c : c + slab]
|
||||
assert got.shape == (slab, slab)
|
||||
if verify:
|
||||
assert got[0, 0] == value(0, r * cols + c)
|
||||
last = (r + slab - 1) * cols + c + slab - 1
|
||||
assert got[-1, -1] == value(0, last)
|
||||
|
||||
|
||||
# ----- process workers ------------------------------------------------------
|
||||
|
||||
_barrier = None
|
||||
|
||||
|
||||
def _init(barrier):
|
||||
global _barrier
|
||||
_barrier = barrier
|
||||
|
||||
|
||||
def _proc_task(task):
|
||||
path, mode, t, threads, m, slabs, slab = task
|
||||
_barrier.wait()
|
||||
start = now()
|
||||
with h5py.File(path, "r") as f:
|
||||
work(f, mode, t, threads, m, slabs, slab, False)
|
||||
return start, now()
|
||||
|
||||
|
||||
def _noop(_):
|
||||
return os.getpid()
|
||||
|
||||
|
||||
def run_threads(path, mode, threads, m, slabs, slab):
|
||||
spans = [None] * threads
|
||||
barrier = threading.Barrier(threads)
|
||||
with h5py.File(path, "r") as f:
|
||||
|
||||
def body(t):
|
||||
barrier.wait()
|
||||
start = now()
|
||||
work(f, mode, t, threads, m, slabs, slab, False)
|
||||
spans[t] = (start, now())
|
||||
|
||||
ts = [threading.Thread(target=body, args=(t,)) for t in range(threads)]
|
||||
for th in ts:
|
||||
th.start()
|
||||
for th in ts:
|
||||
th.join()
|
||||
return max(e for _, e in spans) - min(s for s, _ in spans)
|
||||
|
||||
|
||||
def run_processes(pool, path, mode, threads, m, slabs, slab):
|
||||
tasks = [(path, mode, t, threads, m, slabs, slab) for t in range(threads)]
|
||||
# One task per worker: each blocks in the barrier until all T have
|
||||
# started, so no worker can take a second task.
|
||||
spans = pool.map(_proc_task, tasks, chunksize=1)
|
||||
return max(e for _, e in spans) - min(s for s, _ in spans)
|
||||
|
||||
|
||||
def warm(path):
|
||||
with open(path, "rb") as fh:
|
||||
while fh.read(1 << 24):
|
||||
pass
|
||||
|
||||
|
||||
def evict(path):
|
||||
fd = os.open(path, os.O_RDONLY)
|
||||
try:
|
||||
os.posix_fadvise(fd, 0, 0, os.POSIX_FADV_DONTNEED)
|
||||
finally:
|
||||
os.close(fd)
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser(description=__doc__.split("\n\n")[0])
|
||||
ap.add_argument("--dir", default="concurrent-read-data")
|
||||
ap.add_argument("--executor", choices=["threads", "processes"], default="threads")
|
||||
ap.add_argument("--threads", default="1,2,4,8,16")
|
||||
ap.add_argument("--reps", type=int, default=3)
|
||||
ap.add_argument("--slab", type=int, default=256)
|
||||
ap.add_argument("--slabs", type=int, default=1024)
|
||||
ap.add_argument("--seed", type=int, default=42)
|
||||
ap.add_argument("--cold", action="store_true")
|
||||
ap.add_argument("--modes", default="distinct,same")
|
||||
ap.add_argument("--layouts", default="deflate,contiguous")
|
||||
ap.add_argument("--json")
|
||||
a = ap.parse_args()
|
||||
|
||||
# The Rust harness pins this value (splitmix64_reference).
|
||||
assert splitmix64(42)[1] == 0xBDD732262FEB6E95, "splitmix64 port is wrong"
|
||||
|
||||
try:
|
||||
with open(os.path.join(a.dir, "manifest.json")) as fh:
|
||||
m = json.load(fh)
|
||||
except FileNotFoundError:
|
||||
sys.exit(f"{a.dir}/manifest.json not found: generate the files with "
|
||||
"`cargo run --release -p clawhdf5-bench --bin concurrent_read -- --dir ...` first")
|
||||
threads_list = [int(x) for x in a.threads.split(",")]
|
||||
modes = a.modes.split(",")
|
||||
layouts = a.layouts.split(",")
|
||||
if a.slab < 1 or a.slab > min(m["rows"], m["cols"]):
|
||||
sys.exit(f"--slab must be 1..={min(m['rows'], m['cols'])}")
|
||||
files = dict(m["files"])
|
||||
slabs = slab_offsets(a.seed, a.slabs, m["rows"], m["cols"], a.slab)
|
||||
dataset_bytes = m["rows"] * m["cols"] * 4
|
||||
tool = f"h5py-{a.executor}"
|
||||
|
||||
ctx = mp.get_context("spawn") # never fork a process holding HDF5 state
|
||||
pools = {}
|
||||
if a.executor == "processes":
|
||||
for t in threads_list:
|
||||
pool = ctx.Pool(t, initializer=_init, initargs=(ctx.Barrier(t),))
|
||||
pool.map(_noop, range(t)) # start the workers outside the timing
|
||||
pools[t] = pool
|
||||
|
||||
rows = []
|
||||
print("| layout | mode | threads | MB/s | efficiency | median s |")
|
||||
print("|---|---|---:|---:|---:|---:|")
|
||||
try:
|
||||
for layout in layouts:
|
||||
path = os.path.join(a.dir, files[layout])
|
||||
if not a.cold:
|
||||
warm(path)
|
||||
for mode in modes:
|
||||
with h5py.File(path, "r") as f: # untimed, checked pass
|
||||
work(f, mode, 0, 1, m, slabs, a.slab, True)
|
||||
nbytes = (dataset_bytes * m["datasets"] if mode == "distinct"
|
||||
else a.slab * a.slab * 4 * a.slabs)
|
||||
base = None
|
||||
for t in threads_list:
|
||||
times = []
|
||||
for _ in range(a.reps):
|
||||
if a.cold:
|
||||
evict(path)
|
||||
if a.executor == "threads":
|
||||
times.append(run_threads(path, mode, t, m, slabs, a.slab))
|
||||
else:
|
||||
times.append(run_processes(pools[t], path, mode, t, m, slabs, a.slab))
|
||||
med = sorted(times)[len(times) // 2]
|
||||
mb_s = nbytes / (1 << 20) / med
|
||||
if t == 1:
|
||||
base = mb_s
|
||||
eff = mb_s / (t * base) if base else None
|
||||
print(f"| {layout} | {mode} | {t} | {mb_s:.0f} | "
|
||||
f"{'-' if eff is None else f'{eff:.2f}'} | {med:.4f} |")
|
||||
rows.append({
|
||||
"layout": layout, "mode": mode, "threads": t, "bytes": nbytes,
|
||||
"times_s": times, "median_s": med, "mb_s": mb_s, "efficiency": eff,
|
||||
})
|
||||
finally:
|
||||
for pool in pools.values():
|
||||
pool.terminate()
|
||||
|
||||
if a.json:
|
||||
doc = {
|
||||
"tool": tool,
|
||||
"version": h5py.__version__,
|
||||
"hdf5_version": h5py.version.hdf5_version,
|
||||
"python": platform.python_version(),
|
||||
"host": socket.gethostname(),
|
||||
"cpus": os.cpu_count(),
|
||||
"unix_time": int(time.time()),
|
||||
"cache": ("cold (posix_fadvise DONTNEED before each repetition)"
|
||||
if a.cold else "warm"),
|
||||
"decode_threads": 1,
|
||||
"params": {
|
||||
"datasets": m["datasets"], "rows": m["rows"], "cols": m["cols"],
|
||||
"chunk": m["chunk"], "deflate_level": m["deflate_level"],
|
||||
"mib": dataset_bytes // (1 << 20), "slab": a.slab, "slabs": a.slabs,
|
||||
"seed": a.seed, "reps": a.reps, "dir": a.dir,
|
||||
},
|
||||
"results": rows,
|
||||
}
|
||||
with open(a.json, "w") as fh:
|
||||
json.dump(doc, fh, indent=2)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,523 @@
|
||||
//! Concurrent-read harness: how does decoded read throughput scale with the
|
||||
//! number of threads reading one open file?
|
||||
//!
|
||||
//! libhdf5 (threadsafe build) serialises every API call under one global
|
||||
//! mutex, and h5py holds it too, so threads cannot decode in parallel there.
|
||||
//! A clawhdf5 [`File`] is `Send + Sync`; this harness measures what that buys.
|
||||
//! `crates/clawhdf5-bench/scripts/concurrent_read_h5py.py` runs the same
|
||||
//! workload on the same files with h5py (threads, and processes), and
|
||||
//! `compare_concurrent_read.py` tabulates the JSON both write.
|
||||
//!
|
||||
//! Files (generated on first use, reused while `manifest.json` matches):
|
||||
//!
|
||||
//! * `<dir>/deflate.h5`: `--datasets` datasets `d00`, `d01`, ... of `f32`,
|
||||
//! `--mib` MiB decoded each, shape `[mib * 256, 1024]`, chunks `256 x 256`,
|
||||
//! deflate level 4.
|
||||
//! * `<dir>/contiguous.h5`: the same datasets, contiguous.
|
||||
//!
|
||||
//! Modes, for each layout and each thread count `T` (strong scaling: the total
|
||||
//! work per repetition is fixed, split among the threads):
|
||||
//!
|
||||
//! * `distinct`: every dataset is read in full once; thread `t` reads datasets
|
||||
//! `t, t + T, t + 2T, ...`.
|
||||
//! * `same`: all threads read `d00`, `--slabs` random `--slab` x `--slab`
|
||||
//! hyperslabs in total (slab `j` goes to thread `j % T`). The offsets come
|
||||
//! from a splitmix64 stream seeded with `--seed`, identical in the h5py
|
||||
//! script.
|
||||
//!
|
||||
//! One `File` per layout per repetition is shared by all threads (opened
|
||||
//! fresh each repetition, so no chunk cache carries over). Page cache:
|
||||
//! `warm` (default) reads every file once before timing; `--cold` evicts the
|
||||
//! files from the page cache with `posix_fadvise(POSIX_FADV_DONTNEED)` before
|
||||
//! every repetition (no root needed; it only evicts clean, unmapped pages, so
|
||||
//! it is best effort — the JSON says which was used).
|
||||
//!
|
||||
//! Decode inside one read is itself parallel when clawhdf5-format's `parallel`
|
||||
//! feature is on (it is in this binary, via clawhdf5-agent). `--decode-threads
|
||||
//! N` sizes that rayon pool; `--decode-threads 1` measures the API's own
|
||||
//! thread scaling, comparable with h5py where each call decodes on the
|
||||
//! calling thread.
|
||||
//!
|
||||
//! ```text
|
||||
//! cargo run --release -p clawhdf5-bench --bin concurrent_read -- \
|
||||
//! --dir /data/concurrent-read --json clawhdf5.json
|
||||
//! cargo run --release -p clawhdf5-bench --bin concurrent_read -- \
|
||||
//! --dir /tmp/cr --datasets 4 --mib 1 --threads 1,2 --slabs 16 --reps 1 # smoke
|
||||
//! ```
|
||||
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::sync::Barrier;
|
||||
use std::time::Instant;
|
||||
|
||||
use clawhdf5::{File, FileBuilder, Selection};
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
const COLS: u64 = 1024;
|
||||
const ROWS_PER_MIB: u64 = 256; // 256 rows x 1024 cols x 4 bytes = 1 MiB
|
||||
const CHUNK: u64 = 256;
|
||||
const DEFLATE_LEVEL: u32 = 4;
|
||||
const LAYOUTS: [&str; 2] = ["deflate", "contiguous"];
|
||||
const MANIFEST_VERSION: u32 = 1;
|
||||
|
||||
/// splitmix64 — shared with the h5py script, which must produce the same
|
||||
/// stream (both the data and the hyperslab offsets depend on it).
|
||||
fn splitmix64(state: &mut u64) -> u64 {
|
||||
*state = state.wrapping_add(0x9E37_79B9_7F4A_7C15);
|
||||
let mut z = *state;
|
||||
z = (z ^ (z >> 30)).wrapping_mul(0xBF58_476D_1CE4_E5B9);
|
||||
z = (z ^ (z >> 27)).wrapping_mul(0x94D0_49BB_1331_11EB);
|
||||
z ^ (z >> 31)
|
||||
}
|
||||
|
||||
/// Element `i` (row-major) of dataset `k`: a slowly varying integer part plus
|
||||
/// 8 bits of noise, so deflate has real work to do (about 3.1x) and every value
|
||||
/// is exact in `f32` (< 2^15 with 8 fraction bits), which lets both harnesses
|
||||
/// check what they read against this formula.
|
||||
fn value(k: u64, i: u64) -> f32 {
|
||||
let mut s = i ^ (k << 40);
|
||||
let noise = splitmix64(&mut s) & 0xff;
|
||||
(((i >> 6) % 16384) + k) as f32 + noise as f32 / 256.0
|
||||
}
|
||||
|
||||
#[derive(Serialize, Deserialize, PartialEq, Debug, Clone)]
|
||||
struct Manifest {
|
||||
version: u32,
|
||||
datasets: u64,
|
||||
rows: u64,
|
||||
cols: u64,
|
||||
chunk: [u64; 2],
|
||||
deflate_level: u32,
|
||||
files: Vec<(String, String)>, // (layout, file name)
|
||||
writer: String,
|
||||
}
|
||||
|
||||
fn manifest_for(datasets: u64, mib: u64) -> Manifest {
|
||||
Manifest {
|
||||
version: MANIFEST_VERSION,
|
||||
datasets,
|
||||
rows: mib * ROWS_PER_MIB,
|
||||
cols: COLS,
|
||||
chunk: [CHUNK, CHUNK],
|
||||
deflate_level: DEFLATE_LEVEL,
|
||||
files: LAYOUTS
|
||||
.iter()
|
||||
.map(|l| (l.to_string(), format!("{l}.h5")))
|
||||
.collect(),
|
||||
writer: format!("clawhdf5 {}", env!("CARGO_PKG_VERSION")),
|
||||
}
|
||||
}
|
||||
|
||||
fn dataset_values(k: u64, n: u64) -> Vec<f32> {
|
||||
(0..n).map(|i| value(k, i)).collect()
|
||||
}
|
||||
|
||||
/// Write the files unless `dir` already holds ones matching `want`.
|
||||
fn ensure_files(dir: &Path, want: &Manifest) -> std::io::Result<bool> {
|
||||
let manifest_path = dir.join("manifest.json");
|
||||
if let Ok(text) = std::fs::read_to_string(&manifest_path)
|
||||
&& let Ok(have) = serde_json::from_str::<Manifest>(&text)
|
||||
&& have.version == want.version
|
||||
&& have.datasets == want.datasets
|
||||
&& have.rows == want.rows
|
||||
&& have.cols == want.cols
|
||||
&& have.chunk == want.chunk
|
||||
&& have.deflate_level == want.deflate_level
|
||||
&& have.files == want.files
|
||||
&& want.files.iter().all(|(_, f)| dir.join(f).exists())
|
||||
{
|
||||
return Ok(false);
|
||||
}
|
||||
std::fs::create_dir_all(dir)?;
|
||||
// A stale manifest must not survive a half-written regeneration.
|
||||
let _ = std::fs::remove_file(&manifest_path);
|
||||
let n = want.rows * want.cols;
|
||||
for (layout, file) in &want.files {
|
||||
// One layout at a time keeps the peak memory to about twice one
|
||||
// file's decoded size.
|
||||
let mut b = FileBuilder::new();
|
||||
for k in 0..want.datasets {
|
||||
let ds = b.create_dataset(&format!("d{k:02}"));
|
||||
ds.with_f32_data(&dataset_values(k, n))
|
||||
.with_shape(&[want.rows, want.cols]);
|
||||
if layout == "deflate" {
|
||||
ds.with_chunks(&[CHUNK.min(want.rows), CHUNK])
|
||||
.with_deflate(DEFLATE_LEVEL);
|
||||
}
|
||||
}
|
||||
b.write(dir.join(file)).map_err(std::io::Error::other)?;
|
||||
}
|
||||
std::fs::write(
|
||||
&manifest_path,
|
||||
serde_json::to_string_pretty(want).map_err(std::io::Error::other)?,
|
||||
)?;
|
||||
Ok(true)
|
||||
}
|
||||
|
||||
fn slab_offsets(seed: u64, count: usize, rows: u64, cols: u64, slab: u64) -> Vec<(u64, u64)> {
|
||||
let mut s = seed;
|
||||
(0..count)
|
||||
.map(|_| {
|
||||
let r = splitmix64(&mut s) % (rows - slab + 1);
|
||||
let c = splitmix64(&mut s) % (cols - slab + 1);
|
||||
(r, c)
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Warm the page cache by reading every byte of `path`.
|
||||
fn warm(path: &Path) -> std::io::Result<()> {
|
||||
let mut f = std::fs::File::open(path)?;
|
||||
std::io::copy(&mut f, &mut std::io::sink())?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Ask the kernel to drop `path`'s pages from the page cache.
|
||||
fn evict(path: &Path) -> std::io::Result<()> {
|
||||
use std::os::fd::AsRawFd;
|
||||
let f = std::fs::File::open(path)?;
|
||||
// SAFETY: plain syscall on a valid, open file descriptor.
|
||||
let rc = unsafe { libc::posix_fadvise(f.as_raw_fd(), 0, 0, libc::POSIX_FADV_DONTNEED) };
|
||||
if rc != 0 {
|
||||
return Err(std::io::Error::from_raw_os_error(rc));
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
struct Row {
|
||||
layout: String,
|
||||
mode: String,
|
||||
threads: usize,
|
||||
/// Decoded (selected) bytes read per repetition.
|
||||
bytes: u64,
|
||||
times_s: Vec<f64>,
|
||||
median_s: f64,
|
||||
mb_s: f64,
|
||||
/// `mb_s / (threads * mb_s at threads = 1)`; null without a 1-thread row.
|
||||
efficiency: Option<f64>,
|
||||
}
|
||||
|
||||
struct Args {
|
||||
dir: PathBuf,
|
||||
datasets: u64,
|
||||
mib: u64,
|
||||
threads: Vec<usize>,
|
||||
reps: usize,
|
||||
slab: u64,
|
||||
slabs: usize,
|
||||
seed: u64,
|
||||
cold: bool,
|
||||
decode_threads: usize,
|
||||
modes: Vec<String>,
|
||||
layouts: Vec<String>,
|
||||
json: Option<PathBuf>,
|
||||
}
|
||||
|
||||
const USAGE: &str = "\
|
||||
usage: concurrent_read [--dir DIR] [--datasets N] [--mib N] [--threads 1,2,4,8,16]
|
||||
[--reps N] [--slab N] [--slabs N] [--seed N] [--cold]
|
||||
[--decode-threads N] [--modes distinct,same]
|
||||
[--layouts deflate,contiguous] [--json FILE]";
|
||||
|
||||
fn parse_list<T: std::str::FromStr>(s: &str) -> Result<Vec<T>, String> {
|
||||
s.split(',')
|
||||
.map(|x| x.trim().parse().map_err(|_| format!("bad list item {x:?}")))
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn parse_args() -> Result<Args, String> {
|
||||
let mut a = Args {
|
||||
dir: PathBuf::from("concurrent-read-data"),
|
||||
datasets: 64,
|
||||
mib: 64,
|
||||
threads: vec![1, 2, 4, 8, 16],
|
||||
reps: 3,
|
||||
slab: 256,
|
||||
slabs: 1024,
|
||||
seed: 42,
|
||||
cold: false,
|
||||
decode_threads: 0,
|
||||
modes: vec!["distinct".into(), "same".into()],
|
||||
layouts: LAYOUTS.iter().map(|s| s.to_string()).collect(),
|
||||
json: None,
|
||||
};
|
||||
let mut it = std::env::args().skip(1);
|
||||
while let Some(flag) = it.next() {
|
||||
if flag == "--cold" {
|
||||
a.cold = true;
|
||||
continue;
|
||||
}
|
||||
if flag == "-h" || flag == "--help" {
|
||||
return Err(USAGE.into());
|
||||
}
|
||||
let v = it.next().ok_or(format!("{flag} needs a value\n{USAGE}"))?;
|
||||
let num = |v: &str| {
|
||||
v.parse::<u64>()
|
||||
.map_err(|_| format!("{flag}: bad number {v:?}"))
|
||||
};
|
||||
match flag.as_str() {
|
||||
"--dir" => a.dir = v.into(),
|
||||
"--datasets" => a.datasets = num(&v)?,
|
||||
"--mib" => a.mib = num(&v)?,
|
||||
"--threads" => a.threads = parse_list(&v)?,
|
||||
"--reps" => a.reps = num(&v)? as usize,
|
||||
"--slab" => a.slab = num(&v)?,
|
||||
"--slabs" => a.slabs = num(&v)? as usize,
|
||||
"--seed" => a.seed = num(&v)?,
|
||||
"--decode-threads" => a.decode_threads = num(&v)? as usize,
|
||||
"--modes" => a.modes = parse_list(&v)?,
|
||||
"--layouts" => a.layouts = parse_list(&v)?,
|
||||
"--json" => a.json = Some(v.into()),
|
||||
_ => return Err(format!("unknown flag {flag}\n{USAGE}")),
|
||||
}
|
||||
}
|
||||
if a.datasets == 0 || a.datasets > 100 {
|
||||
return Err("--datasets must be 1..=100".into());
|
||||
}
|
||||
if a.mib == 0 || a.reps == 0 || a.slabs == 0 || a.threads.contains(&0) {
|
||||
return Err("--mib, --reps, --slabs and every --threads value must be > 0".into());
|
||||
}
|
||||
if a.slab == 0 || a.slab > COLS || a.slab > a.mib * ROWS_PER_MIB {
|
||||
return Err(format!(
|
||||
"--slab must be 1..={}",
|
||||
COLS.min(a.mib * ROWS_PER_MIB)
|
||||
));
|
||||
}
|
||||
for m in &a.modes {
|
||||
if m != "distinct" && m != "same" {
|
||||
return Err(format!("unknown mode {m:?}"));
|
||||
}
|
||||
}
|
||||
for l in &a.layouts {
|
||||
if !LAYOUTS.contains(&l.as_str()) {
|
||||
return Err(format!("unknown layout {l:?}"));
|
||||
}
|
||||
}
|
||||
Ok(a)
|
||||
}
|
||||
|
||||
/// One timed repetition: `T` threads on one shared `File`. Returns seconds.
|
||||
fn run_once(
|
||||
path: &Path,
|
||||
mode: &str,
|
||||
threads: usize,
|
||||
m: &Manifest,
|
||||
slabs: &[(u64, u64)],
|
||||
slab: u64,
|
||||
verify: bool,
|
||||
) -> f64 {
|
||||
let file = File::open(path).expect("open");
|
||||
let barrier = Barrier::new(threads + 1); // + the spawning thread
|
||||
let n = m.rows * m.cols;
|
||||
// Each thread times itself from the barrier; the repetition spans the
|
||||
// earliest start to the latest finish (timing on the spawning thread
|
||||
// instead undercounts whenever it is scheduled after the workers ran).
|
||||
let spans: Vec<(Instant, Instant)> = std::thread::scope(|s| {
|
||||
let handles: Vec<_> = (0..threads)
|
||||
.map(|t| {
|
||||
let (file, barrier) = (&file, &barrier);
|
||||
s.spawn(move || {
|
||||
barrier.wait();
|
||||
let start = Instant::now();
|
||||
match mode {
|
||||
"distinct" => {
|
||||
for k in (t as u64..m.datasets).step_by(threads) {
|
||||
let got = file.dataset(&format!("d{k:02}")).unwrap().read_f32();
|
||||
let got = got.unwrap();
|
||||
assert_eq!(got.len() as u64, n);
|
||||
if verify {
|
||||
for i in [0, n / 3, n - 1] {
|
||||
assert_eq!(got[i as usize], value(k, i), "d{k:02}[{i}]");
|
||||
}
|
||||
}
|
||||
std::hint::black_box(got);
|
||||
}
|
||||
}
|
||||
_ => {
|
||||
let ds = file.dataset("d00").unwrap();
|
||||
for &(r, c) in slabs.iter().skip(t).step_by(threads) {
|
||||
let sel = Selection::Hyperslab {
|
||||
start: vec![r, c],
|
||||
stride: vec![1, 1],
|
||||
count: vec![slab, slab],
|
||||
block: vec![1, 1],
|
||||
};
|
||||
let got = ds.read_f32_selection(&sel).unwrap();
|
||||
assert_eq!(got.len() as u64, slab * slab);
|
||||
if verify {
|
||||
let last = (r + slab - 1) * m.cols + c + slab - 1;
|
||||
assert_eq!(got[0], value(0, r * m.cols + c));
|
||||
assert_eq!(*got.last().unwrap(), value(0, last));
|
||||
}
|
||||
std::hint::black_box(got);
|
||||
}
|
||||
}
|
||||
}
|
||||
(start, Instant::now())
|
||||
})
|
||||
})
|
||||
.collect();
|
||||
barrier.wait();
|
||||
handles.into_iter().map(|h| h.join().unwrap()).collect()
|
||||
});
|
||||
let start = spans.iter().map(|s| s.0).min().unwrap();
|
||||
let end = spans.iter().map(|s| s.1).max().unwrap();
|
||||
(end - start).as_secs_f64()
|
||||
}
|
||||
|
||||
fn median(v: &[f64]) -> f64 {
|
||||
let mut s = v.to_vec();
|
||||
s.sort_by(f64::total_cmp);
|
||||
s[s.len() / 2]
|
||||
}
|
||||
|
||||
fn hostname() -> String {
|
||||
std::fs::read_to_string("/proc/sys/kernel/hostname")
|
||||
.map(|s| s.trim().to_string())
|
||||
.unwrap_or_else(|_| "unknown".into())
|
||||
}
|
||||
|
||||
fn main() {
|
||||
let args = match parse_args() {
|
||||
Ok(a) => a,
|
||||
Err(e) => {
|
||||
eprintln!("{e}");
|
||||
std::process::exit(2);
|
||||
}
|
||||
};
|
||||
if cfg!(debug_assertions) {
|
||||
eprintln!("warning: debug build — numbers are meaningless. Use --release.");
|
||||
}
|
||||
if args.decode_threads > 0 {
|
||||
rayon::ThreadPoolBuilder::new()
|
||||
.num_threads(args.decode_threads)
|
||||
.build_global()
|
||||
.expect("configure rayon pool");
|
||||
}
|
||||
|
||||
let manifest = manifest_for(args.datasets, args.mib);
|
||||
let t = Instant::now();
|
||||
match ensure_files(&args.dir, &manifest) {
|
||||
Ok(true) => eprintln!(
|
||||
"generated {} in {:.1} s",
|
||||
args.dir.display(),
|
||||
t.elapsed().as_secs_f64()
|
||||
),
|
||||
Ok(false) => eprintln!("reusing {}", args.dir.display()),
|
||||
Err(e) => {
|
||||
eprintln!("cannot write test files in {}: {e}", args.dir.display());
|
||||
std::process::exit(1);
|
||||
}
|
||||
}
|
||||
let path_of = |layout: &str| args.dir.join(format!("{layout}.h5"));
|
||||
let slabs = slab_offsets(
|
||||
args.seed,
|
||||
args.slabs,
|
||||
manifest.rows,
|
||||
manifest.cols,
|
||||
args.slab,
|
||||
);
|
||||
let dataset_bytes = manifest.rows * manifest.cols * 4;
|
||||
|
||||
let mut rows: Vec<Row> = Vec::new();
|
||||
println!("| layout | mode | threads | MB/s | efficiency | median s |");
|
||||
println!("|---|---|---:|---:|---:|---:|");
|
||||
for layout in &args.layouts {
|
||||
let path = path_of(layout);
|
||||
// Untimed pass: page cache warm (unless --cold), results checked.
|
||||
if !args.cold {
|
||||
warm(&path).expect("warm page cache");
|
||||
}
|
||||
for mode in &args.modes {
|
||||
run_once(&path, mode, 1, &manifest, &slabs, args.slab, true);
|
||||
let bytes = match mode.as_str() {
|
||||
"distinct" => dataset_bytes * manifest.datasets,
|
||||
_ => args.slab * args.slab * 4 * args.slabs as u64,
|
||||
};
|
||||
let mut base: Option<f64> = None;
|
||||
for &threads in &args.threads {
|
||||
let times: Vec<f64> = (0..args.reps)
|
||||
.map(|_| {
|
||||
if args.cold {
|
||||
evict(&path).expect("posix_fadvise");
|
||||
}
|
||||
run_once(&path, mode, threads, &manifest, &slabs, args.slab, false)
|
||||
})
|
||||
.collect();
|
||||
let med = median(×);
|
||||
let mb_s = bytes as f64 / (1 << 20) as f64 / med;
|
||||
if threads == 1 {
|
||||
base = Some(mb_s);
|
||||
}
|
||||
let efficiency = base.map(|b| mb_s / (threads as f64 * b));
|
||||
println!(
|
||||
"| {layout} | {mode} | {threads} | {mb_s:.0} | {} | {med:.4} |",
|
||||
efficiency.map_or("-".into(), |e| format!("{e:.2}"))
|
||||
);
|
||||
rows.push(Row {
|
||||
layout: layout.clone(),
|
||||
mode: mode.clone(),
|
||||
threads,
|
||||
bytes,
|
||||
times_s: times,
|
||||
median_s: med,
|
||||
mb_s,
|
||||
efficiency,
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if let Some(out) = &args.json {
|
||||
let doc = serde_json::json!({
|
||||
"tool": "clawhdf5",
|
||||
"version": env!("CARGO_PKG_VERSION"),
|
||||
"host": hostname(),
|
||||
"cpus": std::thread::available_parallelism().map_or(0, |n| n.get()),
|
||||
"unix_time": std::time::SystemTime::now()
|
||||
.duration_since(std::time::UNIX_EPOCH)
|
||||
.map_or(0, |d| d.as_secs()),
|
||||
"cache": if args.cold { "cold (posix_fadvise DONTNEED before each repetition)" } else { "warm" },
|
||||
"decode_threads": rayon::current_num_threads(),
|
||||
"params": {
|
||||
"datasets": manifest.datasets,
|
||||
"mib": args.mib,
|
||||
"rows": manifest.rows,
|
||||
"cols": manifest.cols,
|
||||
"chunk": manifest.chunk,
|
||||
"deflate_level": manifest.deflate_level,
|
||||
"slab": args.slab,
|
||||
"slabs": args.slabs,
|
||||
"seed": args.seed,
|
||||
"reps": args.reps,
|
||||
"dir": args.dir,
|
||||
},
|
||||
"results": rows,
|
||||
});
|
||||
std::fs::write(out, serde_json::to_string_pretty(&doc).unwrap()).expect("write json");
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn values_are_exact_in_f32() {
|
||||
for k in [0, 7, 63] {
|
||||
for i in [0u64, 1, 4095, 1 << 20, (1 << 24) - 1] {
|
||||
let v = value(k, i);
|
||||
assert_eq!(v, (v as f64) as f32);
|
||||
assert!(v < 32768.0);
|
||||
assert_eq!((v * 256.0).fract(), 0.0);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The h5py script hard-codes this vector to check its splitmix64 port.
|
||||
#[test]
|
||||
fn splitmix64_reference() {
|
||||
let mut s = 42;
|
||||
assert_eq!(splitmix64(&mut s), 0xBDD7_3226_2FEB_6E95);
|
||||
}
|
||||
}
|
||||
@@ -396,7 +396,7 @@ fn run_memory_reduction_benchmark() {
|
||||
println!();
|
||||
println!(
|
||||
"{:>8} {:>10} {:>10} {:>10} {:>12}",
|
||||
"Initial", "Remaining", "Eviction%", "Signal OK?", "BM25 Speedup"
|
||||
"Initial", "Remaining", "Eviction%", "Signal OK?", "Records ÷"
|
||||
);
|
||||
println!("{}", "-".repeat(58));
|
||||
|
||||
@@ -440,7 +440,8 @@ fn run_memory_reduction_benchmark() {
|
||||
// Check all signal records survived
|
||||
let signal_survived = signal_ids.iter().all(|&id| engine.get_by_id(id).is_some());
|
||||
|
||||
// Rough speedup: BM25 scales roughly linearly with record count
|
||||
// How many times fewer records there are. Not a measured speedup —
|
||||
// Part 1 measures search latency before and after.
|
||||
let speedup = before_count as f64 / after_count.max(1) as f64;
|
||||
|
||||
println!(
|
||||
@@ -480,7 +481,7 @@ fn main() {
|
||||
println!(" 3. Reducing search latency proportional to record reduction");
|
||||
println!();
|
||||
println!(
|
||||
"Cycle time scales sub-linearly: 100 records ~microseconds, 100K records ~tens of ms."
|
||||
"Cycle time grows a little faster than linearly: 100 records ~microseconds, 100K records ~tens of ms."
|
||||
);
|
||||
println!("Signal records with Correction source + high access_count survive eviction.");
|
||||
}
|
||||
|
||||
@@ -11,12 +11,14 @@
|
||||
//!
|
||||
//! Configuration matrix:
|
||||
//! - Text lengths: short (50 chars), medium (200 chars), long (1000 chars)
|
||||
//! - Embedding: 384-dim f32 (1536 bytes raw per record)
|
||||
//! - Embedding: 384-dim, stored as float16 (the default for new stores) or
|
||||
//! f32 with `--f32`; "raw" bytes are counted as f32 input either way
|
||||
//! - WAL: enabled and disabled
|
||||
//!
|
||||
//! # Usage
|
||||
//! ```
|
||||
//! cargo run --release --bin footprint_bench
|
||||
//! cargo run --release --bin footprint_bench # float16 stores
|
||||
//! cargo run --release --bin footprint_bench -- --f32 # f32 stores
|
||||
//! ```
|
||||
|
||||
use std::time::Instant;
|
||||
@@ -24,6 +26,9 @@ use std::time::Instant;
|
||||
use clawhdf5_agent::{AgentMemory, HDF5Memory, MemoryConfig, MemoryEntry};
|
||||
use tempfile::TempDir;
|
||||
|
||||
/// `--f32`: build f32 stores instead of the library's float16 default.
|
||||
static F32: std::sync::atomic::AtomicBool = std::sync::atomic::AtomicBool::new(false);
|
||||
|
||||
const EMBEDDING_DIM: usize = 384;
|
||||
|
||||
// Raw bytes per record: 384 f32 embeddings + median text + overhead
|
||||
@@ -152,6 +157,9 @@ fn measure_footprint(
|
||||
config.compression = compression;
|
||||
config.compression_level = if compression { 6 } else { 0 };
|
||||
config.compact_threshold = 0.0;
|
||||
if F32.load(std::sync::atomic::Ordering::Relaxed) {
|
||||
config.float16 = false;
|
||||
}
|
||||
|
||||
let mut memory = HDF5Memory::create(config).expect("HDF5Memory::create failed");
|
||||
|
||||
@@ -241,11 +249,19 @@ fn fmt_n(n: usize) -> String {
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn main() {
|
||||
if std::env::args().skip(1).any(|a| a == "--f32") {
|
||||
F32.store(true, std::sync::atomic::Ordering::Relaxed);
|
||||
}
|
||||
let stored = if F32.load(std::sync::atomic::Ordering::Relaxed) {
|
||||
"f32 (1,536 bytes per record)"
|
||||
} else {
|
||||
"float16 (768 bytes per record; the default for new stores)"
|
||||
};
|
||||
println!("=================================================================");
|
||||
println!(" ClawhDF5 Memory Footprint Benchmark");
|
||||
println!("=================================================================");
|
||||
println!();
|
||||
println!("Embedding: 384-dim f32 = 1,536 bytes raw per record");
|
||||
println!("Embedding: 384-dim, stored as {stored}; raw input counted as f32");
|
||||
println!("Text lengths: short=50 chars, medium=200 chars, long=1000 chars");
|
||||
println!();
|
||||
|
||||
|
||||
@@ -64,6 +64,11 @@ use tempfile::TempDir;
|
||||
|
||||
const EMBEDDING_DIM: usize = 384;
|
||||
|
||||
/// `--float16`: build every per-question store with `MemoryConfig::float16`,
|
||||
/// so embeddings are rounded to half precision as they are saved — exactly
|
||||
/// what such a store searches over.
|
||||
static FLOAT16: std::sync::atomic::AtomicBool = std::sync::atomic::AtomicBool::new(false);
|
||||
|
||||
/// A mode's fusion, as one short string for the reports.
|
||||
fn describe(mode: Mode) -> String {
|
||||
let fusion = match mode.fusion {
|
||||
@@ -431,6 +436,7 @@ fn evaluate_question(
|
||||
let mut config = MemoryConfig::new(dir.path().join("lme.h5"), "lme-bench", EMBEDDING_DIM);
|
||||
config.wal_enabled = false;
|
||||
config.compact_threshold = 0.0;
|
||||
config.float16 = FLOAT16.load(std::sync::atomic::Ordering::Relaxed);
|
||||
|
||||
let mut memory = HDF5Memory::create(config).expect("failed to create HDF5Memory");
|
||||
memory.set_token_filter(mode.tokens);
|
||||
@@ -940,6 +946,10 @@ fn main() {
|
||||
limit = Some(v.parse().expect("--limit must be a positive integer"));
|
||||
}
|
||||
"--sweep" => sweep = true,
|
||||
"--float16" => {
|
||||
FLOAT16.store(true, std::sync::atomic::Ordering::Relaxed);
|
||||
eprintln!("Stores use MemoryConfig::float16 (half-precision embeddings)");
|
||||
}
|
||||
"--rerank-sweep" => {
|
||||
// Re-ranking needs the vector stage to have candidates worth
|
||||
// reordering, so this is an embeddings-only comparison.
|
||||
@@ -971,6 +981,9 @@ fn main() {
|
||||
--rerank-sweep\n\
|
||||
compare re-ranking off, metadata-only (the old\n\
|
||||
behaviour) and blended at several half-lives.\n\
|
||||
--float16\n\
|
||||
build each store with MemoryConfig::float16, to\n\
|
||||
compare retrieval on half-precision embeddings.\n\
|
||||
--sweep instead of the three named modes, sweep vector_weight\n\
|
||||
from 0.0 to 1.0 in 0.1 steps. The 0.7/0.3 default was\n\
|
||||
never searched; this is what searches it."
|
||||
|
||||
@@ -19,6 +19,9 @@
|
||||
//! cargo run --release -p clawhdf5-bench --bin search_harness -- --full # + 100K
|
||||
//! cargo run --release -p clawhdf5-bench --bin search_harness -- --json out.json
|
||||
//! cargo run --release -p clawhdf5-bench --bin search_harness -- --ann-only --uniform
|
||||
//! cargo run --release -p clawhdf5-bench --bin search_harness -- --float16-study --full
|
||||
//! cargo run --release -p clawhdf5-bench --bin search_harness -- --options-study --full
|
||||
//! cargo run --release -p clawhdf5-bench --bin search_harness -- --signing-study --full
|
||||
//! ```
|
||||
|
||||
use std::time::{Duration, Instant};
|
||||
@@ -88,6 +91,9 @@ static UNIFORM: std::sync::atomic::AtomicBool = std::sync::atomic::AtomicBool::n
|
||||
/// the memory) instead of f32, to price the recall it costs.
|
||||
static INT8: std::sync::atomic::AtomicBool = std::sync::atomic::AtomicBool::new(false);
|
||||
|
||||
/// `--f16-first`: in `--float16-study`, run the float16 store first.
|
||||
static F16_FIRST: std::sync::atomic::AtomicBool = std::sync::atomic::AtomicBool::new(false);
|
||||
|
||||
/// `--rerank`: re-score the candidate pool against the exact vectors before
|
||||
/// taking the top K.
|
||||
static RERANK: std::sync::atomic::AtomicBool = std::sync::atomic::AtomicBool::new(false);
|
||||
@@ -230,13 +236,27 @@ struct CountingAllocator;
|
||||
|
||||
static LIVE_BYTES: std::sync::atomic::AtomicI64 = std::sync::atomic::AtomicI64::new(0);
|
||||
|
||||
/// High-water mark of [`LIVE_BYTES`] since it was last reset.
|
||||
///
|
||||
/// Live bytes at a checkpoint cannot see a buffer that was allocated and
|
||||
/// freed in between, and that is exactly the shape of a transient copy —
|
||||
/// which still has to fit in memory while it exists.
|
||||
static PEAK_BYTES: std::sync::atomic::AtomicI64 = std::sync::atomic::AtomicI64::new(0);
|
||||
|
||||
fn note_peak(live: i64) {
|
||||
PEAK_BYTES.fetch_max(live, std::sync::atomic::Ordering::Relaxed);
|
||||
}
|
||||
|
||||
// SAFETY: every method forwards to the system allocator with the same layout
|
||||
// it was given, and only adds bookkeeping around it.
|
||||
unsafe impl std::alloc::GlobalAlloc for CountingAllocator {
|
||||
unsafe fn alloc(&self, layout: std::alloc::Layout) -> *mut u8 {
|
||||
let ptr = unsafe { std::alloc::System.alloc(layout) };
|
||||
if !ptr.is_null() {
|
||||
LIVE_BYTES.fetch_add(layout.size() as i64, std::sync::atomic::Ordering::Relaxed);
|
||||
let live = LIVE_BYTES
|
||||
.fetch_add(layout.size() as i64, std::sync::atomic::Ordering::Relaxed)
|
||||
+ layout.size() as i64;
|
||||
note_peak(live);
|
||||
}
|
||||
ptr
|
||||
}
|
||||
@@ -249,10 +269,9 @@ unsafe impl std::alloc::GlobalAlloc for CountingAllocator {
|
||||
unsafe fn realloc(&self, ptr: *mut u8, layout: std::alloc::Layout, new_size: usize) -> *mut u8 {
|
||||
let new_ptr = unsafe { std::alloc::System.realloc(ptr, layout, new_size) };
|
||||
if !new_ptr.is_null() {
|
||||
LIVE_BYTES.fetch_add(
|
||||
new_size as i64 - layout.size() as i64,
|
||||
std::sync::atomic::Ordering::Relaxed,
|
||||
);
|
||||
let delta = new_size as i64 - layout.size() as i64;
|
||||
let live = LIVE_BYTES.fetch_add(delta, std::sync::atomic::Ordering::Relaxed) + delta;
|
||||
note_peak(live);
|
||||
}
|
||||
new_ptr
|
||||
}
|
||||
@@ -266,6 +285,19 @@ fn heap_bytes() -> u64 {
|
||||
LIVE_BYTES.load(std::sync::atomic::Ordering::Relaxed).max(0) as u64
|
||||
}
|
||||
|
||||
/// Start watching for a new high-water mark from the current live total.
|
||||
fn reset_peak() {
|
||||
PEAK_BYTES.store(
|
||||
LIVE_BYTES.load(std::sync::atomic::Ordering::Relaxed),
|
||||
std::sync::atomic::Ordering::Relaxed,
|
||||
);
|
||||
}
|
||||
|
||||
/// The highest live total seen since [`reset_peak`].
|
||||
fn peak_bytes() -> u64 {
|
||||
PEAK_BYTES.load(std::sync::atomic::Ordering::Relaxed).max(0) as u64
|
||||
}
|
||||
|
||||
fn mib(bytes: u64) -> f64 {
|
||||
bytes as f64 / (1 << 20) as f64
|
||||
}
|
||||
@@ -457,6 +489,394 @@ fn bench_end_to_end(n: usize, json: &mut Vec<serde_json::Value>) {
|
||||
}));
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Signing study: what does an Ed25519-signed checkpoint cost?
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// `--signing-study`: checkpoint time unsigned vs signed, `verify` time, and
|
||||
/// the file-size cost of the stored per-record hashes. Default store
|
||||
/// settings (float16, int8 index). Medians of five checkpoints / three
|
||||
/// verifies.
|
||||
fn signing_study(n: usize) {
|
||||
use clawhdf5_agent::signing::SigningKey;
|
||||
let data = make_dataset(n, 0x516 ^ n as u64);
|
||||
let mut rng = Rng(9);
|
||||
let entries: Vec<MemoryEntry> = data
|
||||
.vectors
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(i, v)| MemoryEntry {
|
||||
chunk: text_for(data.cluster_of[i], i, &mut rng),
|
||||
embedding: v.clone(),
|
||||
source_channel: "bench".into(),
|
||||
timestamp: i as f64,
|
||||
session_id: format!("s{}", i % 50),
|
||||
tags: format!("t{i}"),
|
||||
})
|
||||
.collect();
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let path = dir.path().join("sign.h5");
|
||||
let mut mem = HDF5Memory::create(MemoryConfig::new(path.clone(), "bench", DIM)).unwrap();
|
||||
mem.save_batch(entries).unwrap();
|
||||
std::hint::black_box(mem.hybrid_search(&data.queries[0], "", 1.0, 0.0, K));
|
||||
|
||||
let median = |mut v: Vec<Duration>| {
|
||||
v.sort();
|
||||
v[v.len() / 2]
|
||||
};
|
||||
let checkpoint = |mem: &mut HDF5Memory| {
|
||||
median(
|
||||
(0..5)
|
||||
.map(|_| {
|
||||
let t = Instant::now();
|
||||
mem.flush_wal().unwrap();
|
||||
t.elapsed()
|
||||
})
|
||||
.collect(),
|
||||
)
|
||||
};
|
||||
let unsigned = checkpoint(&mut mem);
|
||||
let unsigned_bytes = std::fs::metadata(&path).unwrap().len();
|
||||
let key = SigningKey::from_bytes(&[7; 32]);
|
||||
mem.set_signing_key(key.clone());
|
||||
let signed = checkpoint(&mut mem);
|
||||
let signed_bytes = std::fs::metadata(&path).unwrap().len();
|
||||
drop(mem);
|
||||
let vk = key.verifying_key();
|
||||
let verify = median(
|
||||
(0..3)
|
||||
.map(|_| {
|
||||
let t = Instant::now();
|
||||
let r = HDF5Memory::verify(&path, &vk).unwrap();
|
||||
let d = t.elapsed();
|
||||
assert!(r.is_valid());
|
||||
d
|
||||
})
|
||||
.collect(),
|
||||
);
|
||||
println!(
|
||||
"| {n} | {:.1} | {:.1} | {:+.1} | {:.1} | {:+.2} |",
|
||||
millis(unsigned),
|
||||
millis(signed),
|
||||
millis(signed) - millis(unsigned),
|
||||
millis(verify),
|
||||
(signed_bytes as f64 - unsigned_bytes as f64) / (1024.0 * 1024.0),
|
||||
);
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Search options study: source filters, re-ranking, confidence rejection
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// `--options-study`: what `HDF5Memory::search`'s options cost and whether a
|
||||
/// filtered search finds the right records. Filters keep 50%, 10% or 1% of
|
||||
/// the store at random, or two whole clusters away from the query (the case
|
||||
/// the index cannot serve, which falls back to an exact scan). Recall is
|
||||
/// vector-only against an exact scan of the allowed records; latency is full
|
||||
/// hybrid search. Hebbian boosting is off.
|
||||
fn options_study(n: usize) {
|
||||
use clawhdf5_agent::SearchOptions;
|
||||
use clawhdf5_agent::confidence::ConfidenceConfig;
|
||||
use clawhdf5_agent::hybrid::Fusion;
|
||||
use clawhdf5_agent::reranker::ReRankConfig;
|
||||
|
||||
let data = make_dataset(n, 0x0B7 ^ n as u64);
|
||||
let n_clusters = data.cluster_of.iter().max().map_or(1, |m| m + 1);
|
||||
let mut rng = Rng(5);
|
||||
let bucket_of: Vec<usize> = (0..n).map(|_| rng.below(100)).collect();
|
||||
let bucket = &bucket_of;
|
||||
let query_texts: Vec<String> = data
|
||||
.query_cluster
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(i, c)| text_for(*c, i, &mut rng))
|
||||
.collect();
|
||||
let exact_top = |q: &[f32], allowed: &dyn Fn(usize) -> bool| -> Vec<usize> {
|
||||
let mut s: Vec<(usize, f32)> = (0..n)
|
||||
.filter(|&i| allowed(i))
|
||||
.map(|i| (i, data.vectors[i].iter().zip(q).map(|(a, b)| a * b).sum()))
|
||||
.collect();
|
||||
s.sort_by(|a, b| b.1.total_cmp(&a.1).then(a.0.cmp(&b.0)));
|
||||
s.into_iter().take(K).map(|(i, _)| i).collect()
|
||||
};
|
||||
|
||||
// Two stores: channel = random bucket, and channel = cluster.
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let mut stores = Vec::new();
|
||||
for by_cluster in [false, true] {
|
||||
let mut rng = Rng(3);
|
||||
let entries: Vec<MemoryEntry> = data
|
||||
.vectors
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(i, v)| MemoryEntry {
|
||||
chunk: text_for(data.cluster_of[i], i, &mut rng),
|
||||
embedding: v.clone(),
|
||||
source_channel: if by_cluster {
|
||||
format!("c{}", data.cluster_of[i])
|
||||
} else {
|
||||
format!("b{}", bucket[i])
|
||||
},
|
||||
timestamp: i as f64,
|
||||
session_id: format!("s{}", i % 50),
|
||||
tags: format!("t{i}"),
|
||||
})
|
||||
.collect();
|
||||
let mut config = MemoryConfig::new(
|
||||
dir.path().join(format!("opt_{by_cluster}.h5")),
|
||||
"bench",
|
||||
DIM,
|
||||
);
|
||||
config.hebbian_boost = 0.0;
|
||||
let mut mem = HDF5Memory::create(config).unwrap();
|
||||
mem.save_batch(entries).unwrap();
|
||||
std::hint::black_box(mem.search(&data.queries[0], "", &SearchOptions::new(K)));
|
||||
stores.push(mem);
|
||||
}
|
||||
|
||||
let vector_only = SearchOptions::new(K).with_fusion(Fusion::Weighted {
|
||||
vector: 1.0,
|
||||
keyword: 0.0,
|
||||
});
|
||||
// (label, store, channels for query i, allowed(i, record))
|
||||
type Case<'a> = (
|
||||
String,
|
||||
usize,
|
||||
Box<dyn Fn(usize) -> Option<Vec<String>> + 'a>,
|
||||
Box<dyn Fn(usize, usize) -> bool + 'a>,
|
||||
);
|
||||
let mut cases: Vec<Case> = vec![(
|
||||
"no filter".into(),
|
||||
0,
|
||||
Box::new(|_| None),
|
||||
Box::new(|_, _| true),
|
||||
)];
|
||||
for pct in [50usize, 10, 1] {
|
||||
cases.push((
|
||||
format!("random {pct}%"),
|
||||
0,
|
||||
Box::new(move |_| Some((0..pct).map(|b| format!("b{b}")).collect())),
|
||||
Box::new(move |_, i| bucket[i] < pct),
|
||||
));
|
||||
}
|
||||
let d = &data;
|
||||
let away = move |qi: usize| {
|
||||
let qc = d.query_cluster[qi];
|
||||
[
|
||||
(qc + n_clusters / 3) % n_clusters,
|
||||
(qc + 2 * n_clusters / 3) % n_clusters,
|
||||
]
|
||||
};
|
||||
cases.push((
|
||||
"2 clusters away from the query".into(),
|
||||
1,
|
||||
Box::new(move |qi| Some(away(qi).iter().map(|c| format!("c{c}")).collect())),
|
||||
Box::new(move |qi, i| away(qi).contains(&d.cluster_of[i])),
|
||||
));
|
||||
|
||||
for (label, store, channels, allowed) in &cases {
|
||||
let mem = &mut stores[*store];
|
||||
let mut hits = 0;
|
||||
let mut kept = 0;
|
||||
for (qi, q) in data.queries.iter().enumerate() {
|
||||
let mut opts = vector_only.clone();
|
||||
opts.source_channels = channels(qi);
|
||||
let got = mem.search(q, "", &opts);
|
||||
let want = exact_top(q, &|i| allowed(qi, i));
|
||||
kept += want.len();
|
||||
hits += got.iter().filter(|r| want.contains(&r.index)).count();
|
||||
}
|
||||
let latency = summarize(
|
||||
(0..N_QUERIES)
|
||||
.map(|qi| {
|
||||
let mut opts = SearchOptions::new(K);
|
||||
opts.source_channels = channels(qi);
|
||||
let t = Instant::now();
|
||||
std::hint::black_box(mem.search(&data.queries[qi], &query_texts[qi], &opts));
|
||||
t.elapsed()
|
||||
})
|
||||
.collect(),
|
||||
);
|
||||
println!(
|
||||
"| {n} | {label} | {:.4} | {:.3} | {:.3} |",
|
||||
hits as f64 / kept.max(1) as f64,
|
||||
millis(latency.p50),
|
||||
millis(latency.p99),
|
||||
);
|
||||
}
|
||||
|
||||
let mem = &mut stores[0];
|
||||
for (label, opts) in [
|
||||
(
|
||||
"re-rank",
|
||||
SearchOptions::new(K).with_rerank(ReRankConfig::default()),
|
||||
),
|
||||
(
|
||||
"re-rank + confidence",
|
||||
SearchOptions::new(K)
|
||||
.with_rerank(ReRankConfig::default())
|
||||
.with_confidence(ConfidenceConfig::default()),
|
||||
),
|
||||
] {
|
||||
let latency = summarize(
|
||||
(0..N_QUERIES)
|
||||
.map(|qi| {
|
||||
let t = Instant::now();
|
||||
std::hint::black_box(mem.search(&data.queries[qi], &query_texts[qi], &opts));
|
||||
t.elapsed()
|
||||
})
|
||||
.collect(),
|
||||
);
|
||||
println!(
|
||||
"| {n} | {label} | — | {:.3} | {:.3} |",
|
||||
millis(latency.p50),
|
||||
millis(latency.p99)
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// float16 study: what does half-precision embedding storage cost?
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// `--float16-study`: the same data in an `f32` store and a `float16` store.
|
||||
/// Reports file size, checkpoint and open time, vector-search recall@10
|
||||
/// against an exact scan of the *original* f32 vectors, how often the two
|
||||
/// stores return the same top 10, and `hybrid_search` latency. Hebbian
|
||||
/// boosting is off, so every query sees the same store.
|
||||
fn float16_study(n: usize) {
|
||||
let data = make_dataset(n, 0xF16 ^ n as u64);
|
||||
let mut rng = Rng(11);
|
||||
let query_texts: Vec<String> = data
|
||||
.query_cluster
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(i, c)| text_for(*c, i, &mut rng))
|
||||
.collect();
|
||||
|
||||
// Exact top K by cosine (the vectors are unit length) on the f32 inputs.
|
||||
let exact: Vec<Vec<usize>> = data
|
||||
.queries
|
||||
.iter()
|
||||
.map(|q| {
|
||||
let mut scored: Vec<(usize, f32)> = data
|
||||
.vectors
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(i, v)| (i, v.iter().zip(q).map(|(a, b)| a * b).sum()))
|
||||
.collect();
|
||||
scored.sort_by(|a, b| b.1.total_cmp(&a.1).then(a.0.cmp(&b.0)));
|
||||
scored.into_iter().take(K).map(|(i, _)| i).collect()
|
||||
})
|
||||
.collect();
|
||||
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let mut per_variant: Vec<(bool, Vec<Vec<usize>>)> = Vec::new();
|
||||
// `--f16-first` swaps the order, to check the numbers do not depend on
|
||||
// which store runs first (page cache, allocator, CPU frequency).
|
||||
let order = if F16_FIRST.load(std::sync::atomic::Ordering::Relaxed) {
|
||||
[true, false]
|
||||
} else {
|
||||
[false, true]
|
||||
};
|
||||
for float16 in order {
|
||||
let path = dir.path().join(format!("f16study_{float16}.h5"));
|
||||
let mut rng = Rng(3);
|
||||
let entries: Vec<MemoryEntry> = data
|
||||
.vectors
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(i, v)| MemoryEntry {
|
||||
chunk: text_for(data.cluster_of[i], i, &mut rng),
|
||||
embedding: v.clone(),
|
||||
source_channel: "bench".into(),
|
||||
timestamp: i as f64,
|
||||
session_id: format!("s{}", i % 50),
|
||||
tags: format!("t{i}"),
|
||||
})
|
||||
.collect();
|
||||
let mut config = MemoryConfig::new(path.clone(), "bench", DIM);
|
||||
config.float16 = float16;
|
||||
config.hebbian_boost = 0.0;
|
||||
let mut mem = HDF5Memory::create(config).unwrap();
|
||||
mem.save_batch(entries).unwrap();
|
||||
// Build the indexes, then time a checkpoint that writes everything.
|
||||
std::hint::black_box(mem.hybrid_search(&data.queries[0], "", 1.0, 0.0, K));
|
||||
let t = Instant::now();
|
||||
mem.flush_wal().unwrap();
|
||||
let checkpoint = t.elapsed();
|
||||
drop(mem);
|
||||
let file_bytes = std::fs::metadata(&path).unwrap().len();
|
||||
|
||||
// Median of three opens.
|
||||
let mut opens: Vec<Duration> = (0..3)
|
||||
.map(|_| {
|
||||
let t = Instant::now();
|
||||
let m = HDF5Memory::open(&path).unwrap();
|
||||
let d = t.elapsed();
|
||||
drop(m);
|
||||
d
|
||||
})
|
||||
.collect();
|
||||
opens.sort();
|
||||
let mut mem = HDF5Memory::open(&path).unwrap();
|
||||
|
||||
// Vector-only search: empty text, all weight on the vector stage.
|
||||
let results: Vec<Vec<usize>> = data
|
||||
.queries
|
||||
.iter()
|
||||
.map(|q| {
|
||||
mem.hybrid_search(q, "", 1.0, 0.0, K)
|
||||
.iter()
|
||||
.map(|r| r.index)
|
||||
.collect()
|
||||
})
|
||||
.collect();
|
||||
let hits: usize = results
|
||||
.iter()
|
||||
.zip(&exact)
|
||||
.map(|(got, want)| got.iter().filter(|i| want.contains(i)).count())
|
||||
.sum();
|
||||
let recall = hits as f64 / (K * data.queries.len()) as f64;
|
||||
|
||||
let latency = summarize(
|
||||
(0..N_QUERIES)
|
||||
.map(|i| {
|
||||
let t = Instant::now();
|
||||
std::hint::black_box(mem.hybrid_search(
|
||||
&data.queries[i],
|
||||
&query_texts[i],
|
||||
0.4,
|
||||
0.6,
|
||||
K,
|
||||
));
|
||||
t.elapsed()
|
||||
})
|
||||
.collect(),
|
||||
);
|
||||
let overlap = match per_variant.first() {
|
||||
Some((_, other)) => {
|
||||
let same: usize = results
|
||||
.iter()
|
||||
.zip(other)
|
||||
.map(|(a, b)| a.iter().filter(|i| b.contains(i)).count())
|
||||
.sum();
|
||||
format!("{:.4}", same as f64 / (K * data.queries.len()) as f64)
|
||||
}
|
||||
None => "—".into(),
|
||||
};
|
||||
println!(
|
||||
"| {n} | {} | {:.1} | {:.0} | {:.1} | {recall:.4} | {overlap} | {:.3} |",
|
||||
if float16 { "float16" } else { "f32" },
|
||||
mib(file_bytes),
|
||||
millis(checkpoint),
|
||||
millis(opens[1]),
|
||||
millis(latency.p50),
|
||||
);
|
||||
per_variant.push((float16, results));
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Fusion study: does capping the keyword candidate pool change the ranking?
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -580,19 +1000,24 @@ fn bench_footprint(n: usize) {
|
||||
let path = mem.config().path.clone();
|
||||
drop(mem);
|
||||
let before_open = heap_bytes();
|
||||
reset_peak();
|
||||
let reopened = HDF5Memory::open(&path).unwrap();
|
||||
let after_open = heap_bytes();
|
||||
let loaded = after_open.saturating_sub(before_open);
|
||||
// Peak over the open, not just what it leaves behind: a buffer allocated
|
||||
// and freed during the parse never shows up in the live total.
|
||||
let peak = peak_bytes().saturating_sub(before_open);
|
||||
drop(reopened);
|
||||
|
||||
let raw = (n * DIM * 4) as u64;
|
||||
println!(
|
||||
"| {n} | {:.0} | {:.0} | {:.0} | {:.0} | {:.0} | {:.2}x |",
|
||||
"| {n} | {:.0} | {:.0} | {:.0} | {:.0} | {:.0} | {:.0} | {:.2}x |",
|
||||
mib(raw),
|
||||
mib(after_entries.saturating_sub(base)),
|
||||
mib(after_store.saturating_sub(after_entries)),
|
||||
mib(after_indexes.saturating_sub(after_store)),
|
||||
mib(loaded),
|
||||
mib(peak),
|
||||
loaded as f64 / raw as f64,
|
||||
);
|
||||
}
|
||||
@@ -611,6 +1036,52 @@ fn main() {
|
||||
}
|
||||
return;
|
||||
}
|
||||
if args.iter().any(|a| a == "--signing-study") {
|
||||
println!("## Signed checkpoints ({DIM}-dim, float16, int8 index)\n");
|
||||
println!(
|
||||
"| N | checkpoint ms, unsigned | checkpoint ms, signed | signing adds ms | verify ms | file MiB added |"
|
||||
);
|
||||
println!("|---:|---:|---:|---:|---:|---:|");
|
||||
for &n in if full {
|
||||
&[1_000, 10_000, 100_000][..]
|
||||
} else {
|
||||
&[1_000, 10_000][..]
|
||||
} {
|
||||
signing_study(n);
|
||||
}
|
||||
return;
|
||||
}
|
||||
if args.iter().any(|a| a == "--options-study") {
|
||||
println!("## Search options ({DIM}-dim, k = {K}, Hebbian boost off)\n");
|
||||
println!("| N | options | filtered recall@10 | p50 ms | p99 ms |");
|
||||
println!("|---:|---|---:|---:|---:|");
|
||||
for &n in if full {
|
||||
&[10_000, 100_000][..]
|
||||
} else {
|
||||
&[10_000][..]
|
||||
} {
|
||||
options_study(n);
|
||||
}
|
||||
return;
|
||||
}
|
||||
if args.iter().any(|a| a == "--f16-first") {
|
||||
F16_FIRST.store(true, std::sync::atomic::Ordering::Relaxed);
|
||||
}
|
||||
if args.iter().any(|a| a == "--float16-study") {
|
||||
println!("## float16 embedding storage ({DIM}-dim, int8 index, Hebbian boost off)\n");
|
||||
println!(
|
||||
"| N | embeddings | file MiB | checkpoint ms | open ms | recall@10 | top-10 overlap with the other | hybrid p50 ms |"
|
||||
);
|
||||
println!("|---:|---|---:|---:|---:|---:|---:|---:|");
|
||||
for &n in if full {
|
||||
&[1_000, 10_000, 100_000][..]
|
||||
} else {
|
||||
&[1_000, 10_000][..]
|
||||
} {
|
||||
float16_study(n);
|
||||
}
|
||||
return;
|
||||
}
|
||||
if args.iter().any(|a| a == "--int8") {
|
||||
INT8.store(true, std::sync::atomic::Ordering::Relaxed);
|
||||
println!("(int8-quantised index vectors)");
|
||||
@@ -644,9 +1115,9 @@ fn main() {
|
||||
if args.iter().any(|a| a == "--footprint") {
|
||||
println!("\n### Resident memory, {DIM}-dim f32\n");
|
||||
println!(
|
||||
"| N | vectors (raw) | entries MiB | store MiB | indexes MiB | reopened MiB | reopened / raw |"
|
||||
"| N | vectors (raw) | entries MiB | store MiB | indexes MiB | reopened MiB | peak during open MiB | reopened / raw |"
|
||||
);
|
||||
println!("|---:|---:|---:|---:|---:|---:|---:|");
|
||||
println!("|---:|---:|---:|---:|---:|---:|---:|---:|");
|
||||
for &n in sizes {
|
||||
bench_footprint(n);
|
||||
}
|
||||
|
||||
@@ -0,0 +1,148 @@
|
||||
//! Keeps the concurrent-read harnesses working: runs `concurrent_read`, the
|
||||
//! h5py script (threads and processes) and the comparison script end to end
|
||||
//! on tiny files. h5py reading the files also checks, element by element at
|
||||
//! spot positions, that both harnesses generate the same data and slabs.
|
||||
//!
|
||||
//! The h5py half is skipped when python3 with h5py is unavailable, unless
|
||||
//! `CLAWHDF5_REQUIRE_INTEROP=1`; `CLAWHDF5_PYTHON` picks the interpreter.
|
||||
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::process::Command;
|
||||
|
||||
fn python() -> String {
|
||||
std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string())
|
||||
}
|
||||
|
||||
fn interop_required() -> bool {
|
||||
std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1")
|
||||
}
|
||||
|
||||
fn python_available() -> bool {
|
||||
Command::new(python())
|
||||
.args(["-c", "import h5py, numpy"])
|
||||
.output()
|
||||
.map(|o| o.status.success())
|
||||
.unwrap_or(false)
|
||||
}
|
||||
|
||||
fn scripts() -> PathBuf {
|
||||
Path::new(env!("CARGO_MANIFEST_DIR")).join("scripts")
|
||||
}
|
||||
|
||||
fn run(cmd: &mut Command) -> String {
|
||||
let out = cmd.output().expect("spawn");
|
||||
assert!(
|
||||
out.status.success(),
|
||||
"{cmd:?} failed\nSTDOUT:\n{}\nSTDERR:\n{}",
|
||||
String::from_utf8_lossy(&out.stdout),
|
||||
String::from_utf8_lossy(&out.stderr)
|
||||
);
|
||||
String::from_utf8_lossy(&out.stdout).into_owned()
|
||||
}
|
||||
|
||||
const SMALL: [&str; 8] = [
|
||||
"--threads",
|
||||
"1,2",
|
||||
"--slabs",
|
||||
"8",
|
||||
"--reps",
|
||||
"1",
|
||||
"--slab",
|
||||
"64",
|
||||
];
|
||||
|
||||
fn results(path: &Path) -> serde_json::Value {
|
||||
serde_json::from_str(&std::fs::read_to_string(path).unwrap()).unwrap()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn harnesses_run_end_to_end_on_tiny_files() {
|
||||
let dir = tempfile::TempDir::new().unwrap();
|
||||
let data = dir.path().join("data");
|
||||
let claw = dir.path().join("claw.json");
|
||||
|
||||
let bin = env!("CARGO_BIN_EXE_concurrent_read");
|
||||
run(Command::new(bin)
|
||||
.arg("--dir")
|
||||
.arg(&data)
|
||||
.args(["--datasets", "3", "--mib", "1"])
|
||||
.args(SMALL)
|
||||
.arg("--json")
|
||||
.arg(&claw));
|
||||
// Second run reuses the files (and exercises --cold).
|
||||
let out = Command::new(bin)
|
||||
.arg("--dir")
|
||||
.arg(&data)
|
||||
.args(["--datasets", "3", "--mib", "1", "--cold"])
|
||||
.args(SMALL)
|
||||
.output()
|
||||
.unwrap();
|
||||
assert!(out.status.success());
|
||||
assert!(String::from_utf8_lossy(&out.stderr).contains("reusing"));
|
||||
|
||||
let doc = results(&claw);
|
||||
assert_eq!(doc["tool"], "clawhdf5");
|
||||
// 2 layouts x 2 modes x 2 thread counts.
|
||||
assert_eq!(doc["results"].as_array().unwrap().len(), 8);
|
||||
for r in doc["results"].as_array().unwrap() {
|
||||
assert!(r["mb_s"].as_f64().unwrap() > 0.0, "{r}");
|
||||
}
|
||||
|
||||
if !python_available() {
|
||||
assert!(
|
||||
!interop_required(),
|
||||
"CLAWHDF5_REQUIRE_INTEROP=1 but {} has no h5py",
|
||||
python()
|
||||
);
|
||||
eprintln!("skipping the h5py half: no h5py in {}", python());
|
||||
return;
|
||||
}
|
||||
let mut jsons = vec![claw];
|
||||
for executor in ["threads", "processes"] {
|
||||
let out = dir.path().join(format!("h5py-{executor}.json"));
|
||||
run(Command::new(python())
|
||||
.arg(scripts().join("concurrent_read_h5py.py"))
|
||||
.arg("--dir")
|
||||
.arg(&data)
|
||||
.args(["--executor", executor])
|
||||
.args(SMALL)
|
||||
.arg("--json")
|
||||
.arg(&out));
|
||||
let doc = results(&out);
|
||||
assert_eq!(doc["tool"], format!("h5py-{executor}"));
|
||||
assert_eq!(doc["results"].as_array().unwrap().len(), 8);
|
||||
jsons.push(out);
|
||||
}
|
||||
let table = run(Command::new(python())
|
||||
.arg(scripts().join("compare_concurrent_read.py"))
|
||||
.args(&jsons));
|
||||
assert!(table.contains("| deflate | same | 2 |"), "{table}");
|
||||
assert!(table.contains("clawhdf5 / h5py-processes"), "{table}");
|
||||
|
||||
// A different workload must not be compared.
|
||||
let other = dir.path().join("other.json");
|
||||
run(Command::new(python())
|
||||
.arg(scripts().join("concurrent_read_h5py.py"))
|
||||
.arg("--dir")
|
||||
.arg(&data)
|
||||
.args([
|
||||
"--threads",
|
||||
"1",
|
||||
"--slabs",
|
||||
"4",
|
||||
"--reps",
|
||||
"1",
|
||||
"--slab",
|
||||
"64",
|
||||
])
|
||||
.arg("--json")
|
||||
.arg(&other));
|
||||
let out = Command::new(python())
|
||||
.arg(scripts().join("compare_concurrent_read.py"))
|
||||
.arg(&jsons[0])
|
||||
.arg(&other)
|
||||
.output()
|
||||
.unwrap();
|
||||
assert!(!out.status.success());
|
||||
assert!(String::from_utf8_lossy(&out.stderr).contains("slabs"));
|
||||
}
|
||||
@@ -1,7 +1,8 @@
|
||||
[package]
|
||||
name = "clawhdf5-cli"
|
||||
version = "2.6.0"
|
||||
version = "2.7.0"
|
||||
edition = "2024"
|
||||
rust-version.workspace = true
|
||||
license = "MIT"
|
||||
description = "CLI for clawhdf5 agent memory — create, save, search, recall, stats"
|
||||
repository = "https://git.redclaw.dev/quantumclaw/clawhdf5"
|
||||
@@ -14,7 +15,7 @@ name = "clawhdf5"
|
||||
path = "src/main.rs"
|
||||
|
||||
[dependencies]
|
||||
clawhdf5-agent = { path = "../clawhdf5-agent", version = "2.6.0" }
|
||||
clawhdf5-agent = { path = "../clawhdf5-agent", version = "2.7.0" }
|
||||
clap = { version = "4", features = ["derive", "env"] }
|
||||
serde_json = "1"
|
||||
serde = { workspace = true }
|
||||
|
||||
+158
-21
@@ -1,15 +1,22 @@
|
||||
use std::path::PathBuf;
|
||||
use std::path::{Path, PathBuf};
|
||||
|
||||
use clap::{Parser, Subcommand};
|
||||
use clawhdf5_agent::signing::{self, SigningKey, VerifyingKey};
|
||||
use clawhdf5_agent::{AgentMemory, HDF5Memory, MemoryConfig, MemoryEntry};
|
||||
|
||||
/// ClawhDF5 — HDF5-backed cognitive memory for AI agents
|
||||
#[derive(Parser)]
|
||||
#[command(name = "clawhdf5", version, about)]
|
||||
struct Cli {
|
||||
/// Path to the .h5 memory file
|
||||
/// Path to the .h5 memory file (not needed for `keygen`)
|
||||
#[arg(short, long, env = "CLAWHDF5_PATH")]
|
||||
path: PathBuf,
|
||||
path: Option<PathBuf>,
|
||||
|
||||
/// File holding an Ed25519 signing key (64 hex characters, from
|
||||
/// `keygen`). Every checkpoint this command makes is then signed; a
|
||||
/// signed store refuses to checkpoint without it.
|
||||
#[arg(long, env = "CLAWHDF5_SIGNING_KEY", global = true)]
|
||||
signing_key: Option<PathBuf>,
|
||||
|
||||
#[command(subcommand)]
|
||||
command: Commands,
|
||||
@@ -28,10 +35,22 @@ enum Commands {
|
||||
/// Enable write-ahead log
|
||||
#[arg(long)]
|
||||
wal: bool,
|
||||
/// Store the vector index's copy of the embeddings as int8, roughly
|
||||
/// halving a loaded store's memory at about 13% fewer queries/second
|
||||
/// Hold the vector index's copy of the embeddings as f32 instead of
|
||||
/// the default int8 (which uses a quarter of the memory and is faster
|
||||
/// at equal recall)
|
||||
#[arg(long)]
|
||||
f32_index: bool,
|
||||
/// Accepted for compatibility; int8 is now the default
|
||||
#[arg(long, hide = true, conflicts_with = "f32_index")]
|
||||
quantized_index: bool,
|
||||
/// Store embeddings as full-precision f32 instead of the default
|
||||
/// half precision (float16: half the bytes, about three significant
|
||||
/// digits, values within ±65504)
|
||||
#[arg(long)]
|
||||
f32: bool,
|
||||
/// Accepted for compatibility; float16 is now the default
|
||||
#[arg(long, hide = true, conflicts_with = "f32")]
|
||||
float16: bool,
|
||||
},
|
||||
/// Save a memory entry (reads JSON from stdin or --json)
|
||||
Save {
|
||||
@@ -79,6 +98,38 @@ enum Commands {
|
||||
/// Destination path
|
||||
dest: PathBuf,
|
||||
},
|
||||
/// Generate an Ed25519 signing key for signed checkpoints
|
||||
Keygen {
|
||||
/// Where to write the secret key (created new, owner-only on Unix)
|
||||
#[arg(long)]
|
||||
out: PathBuf,
|
||||
},
|
||||
/// Verify a signed store against a public key; exit status 2 if not valid
|
||||
Verify {
|
||||
/// The trusted public key: 64 hex characters, or a file holding them
|
||||
#[arg(long)]
|
||||
public_key: String,
|
||||
},
|
||||
}
|
||||
|
||||
fn read_signing_key(path: &Path) -> Result<SigningKey, Box<dyn std::error::Error>> {
|
||||
let text = std::fs::read_to_string(path)
|
||||
.map_err(|e| format!("cannot read signing key {}: {e}", path.display()))?;
|
||||
let bytes = signing::from_hex::<32>(&text)
|
||||
.ok_or_else(|| format!("{} is not a 64-hex-character key", path.display()))?;
|
||||
Ok(SigningKey::from_bytes(&bytes))
|
||||
}
|
||||
|
||||
/// Open for writing, with the signing key applied if one was given.
|
||||
fn open_writable(
|
||||
path: &Path,
|
||||
key: &Option<SigningKey>,
|
||||
) -> Result<HDF5Memory, Box<dyn std::error::Error>> {
|
||||
let mut mem = HDF5Memory::open(path)?;
|
||||
if let Some(k) = key {
|
||||
mem.set_signing_key(k.clone());
|
||||
}
|
||||
Ok(mem)
|
||||
}
|
||||
|
||||
fn main() {
|
||||
@@ -91,24 +142,76 @@ fn main() {
|
||||
}
|
||||
|
||||
fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
||||
if let Commands::Keygen { out } = &cli.command {
|
||||
let key = signing::generate_key();
|
||||
let mut opts = std::fs::OpenOptions::new();
|
||||
opts.write(true).create_new(true);
|
||||
#[cfg(unix)]
|
||||
{
|
||||
use std::os::unix::fs::OpenOptionsExt;
|
||||
opts.mode(0o600);
|
||||
}
|
||||
use std::io::Write;
|
||||
let mut f = opts
|
||||
.open(out)
|
||||
.map_err(|e| format!("cannot create {}: {e}", out.display()))?;
|
||||
writeln!(f, "{}", signing::to_hex(&key.to_bytes()))?;
|
||||
let j = serde_json::json!({
|
||||
"status": "generated",
|
||||
"secret_key_file": out.display().to_string(),
|
||||
"public_key": signing::to_hex(&key.verifying_key().to_bytes()),
|
||||
});
|
||||
println!("{}", serde_json::to_string_pretty(&j)?);
|
||||
return Ok(());
|
||||
}
|
||||
let path = cli
|
||||
.path
|
||||
.clone()
|
||||
.ok_or("--path (or CLAWHDF5_PATH) is required")?;
|
||||
let key = cli
|
||||
.signing_key
|
||||
.as_deref()
|
||||
.map(read_signing_key)
|
||||
.transpose()?;
|
||||
match cli.command {
|
||||
Commands::Create {
|
||||
agent_id,
|
||||
dim,
|
||||
wal,
|
||||
quantized_index,
|
||||
f32_index,
|
||||
quantized_index: _,
|
||||
f32,
|
||||
float16: _,
|
||||
} => {
|
||||
let mut config = MemoryConfig::new(cli.path.clone(), &agent_id, dim);
|
||||
let mut config = MemoryConfig::new(path.clone(), &agent_id, dim);
|
||||
config.wal_enabled = wal;
|
||||
config.quantized_index = quantized_index;
|
||||
let mem = HDF5Memory::create(config)?;
|
||||
// As with --f32-index: only ever switch the library default off.
|
||||
if f32 {
|
||||
config.float16 = false;
|
||||
}
|
||||
let config_float16 = config.float16;
|
||||
// Only ever switch *off* the library default: assigning the flag
|
||||
// outright would force every CLI-created store back to f32 unless
|
||||
// the caller knew to ask for int8.
|
||||
if f32_index {
|
||||
config.quantized_index = false;
|
||||
}
|
||||
let config_quantized = config.quantized_index;
|
||||
let mut mem = HDF5Memory::create(config)?;
|
||||
// Sign straight away, so the store is never on disk unsigned.
|
||||
if let Some(k) = &key {
|
||||
mem.set_signing_key(k.clone());
|
||||
mem.flush_wal()?;
|
||||
}
|
||||
let j = serde_json::json!({
|
||||
"status": "created",
|
||||
"path": cli.path.display().to_string(),
|
||||
"path": path.display().to_string(),
|
||||
"agent_id": agent_id,
|
||||
"embedding_dim": dim,
|
||||
"wal_enabled": wal,
|
||||
"quantized_index": quantized_index,
|
||||
"quantized_index": config_quantized,
|
||||
"float16": config_float16,
|
||||
"signed": mem.is_signed(),
|
||||
"count": mem.count(),
|
||||
});
|
||||
println!("{}", serde_json::to_string_pretty(&j)?);
|
||||
@@ -125,7 +228,7 @@ fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
||||
}
|
||||
};
|
||||
let entry: MemoryEntry = serde_json::from_str(&input)?;
|
||||
let mut mem = HDF5Memory::open(&cli.path)?;
|
||||
let mut mem = open_writable(&path, &key)?;
|
||||
let idx = mem.save(entry)?;
|
||||
let j = serde_json::json!({ "status": "saved", "index": idx, "count": mem.count() });
|
||||
println!("{}", serde_json::to_string(&j)?);
|
||||
@@ -139,7 +242,7 @@ fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
||||
keyword_weight,
|
||||
} => {
|
||||
let emb: Vec<f32> = serde_json::from_str(&embedding)?;
|
||||
let mut mem = HDF5Memory::open(&cli.path)?;
|
||||
let mut mem = open_writable(&path, &key)?;
|
||||
let results = mem.hybrid_search(&emb, &query, vector_weight, keyword_weight, top_k);
|
||||
let j: Vec<serde_json::Value> = results
|
||||
.iter()
|
||||
@@ -157,7 +260,7 @@ fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
||||
}
|
||||
|
||||
Commands::Recall { index } => {
|
||||
let mem = HDF5Memory::open_read_only(&cli.path)?;
|
||||
let mem = HDF5Memory::open_read_only(&path)?;
|
||||
match mem.get_chunk(index) {
|
||||
Some(content) => {
|
||||
let j = serde_json::json!({ "index": index, "chunk": content });
|
||||
@@ -171,22 +274,23 @@ fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
||||
}
|
||||
|
||||
Commands::Stats => {
|
||||
let mem = HDF5Memory::open_read_only(&cli.path)?;
|
||||
let mem = HDF5Memory::open_read_only(&path)?;
|
||||
let cfg = mem.config();
|
||||
let j = serde_json::json!({
|
||||
"path": cli.path.display().to_string(),
|
||||
"path": path.display().to_string(),
|
||||
"agent_id": cfg.agent_id,
|
||||
"embedding_dim": cfg.embedding_dim,
|
||||
"count": mem.count(),
|
||||
"active": mem.count_active(),
|
||||
"wal_enabled": cfg.wal_enabled,
|
||||
"wal_pending": mem.wal_pending_count(),
|
||||
"signed": mem.is_signed(),
|
||||
});
|
||||
println!("{}", serde_json::to_string_pretty(&j)?);
|
||||
}
|
||||
|
||||
Commands::FlushWal => {
|
||||
let mut mem = HDF5Memory::open(&cli.path)?;
|
||||
let mut mem = open_writable(&path, &key)?;
|
||||
let before = mem.wal_pending_count();
|
||||
mem.flush_wal()?;
|
||||
let j = serde_json::json!({
|
||||
@@ -198,7 +302,7 @@ fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
||||
}
|
||||
|
||||
Commands::AgentsMd { output } => {
|
||||
let mem = HDF5Memory::open_read_only(&cli.path)?;
|
||||
let mem = HDF5Memory::open_read_only(&path)?;
|
||||
let md = mem.generate_agents_md();
|
||||
match output {
|
||||
Some(p) => {
|
||||
@@ -210,7 +314,7 @@ fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
||||
}
|
||||
|
||||
Commands::Export => {
|
||||
let mem = HDF5Memory::open_read_only(&cli.path)?;
|
||||
let mem = HDF5Memory::open_read_only(&path)?;
|
||||
for i in 0..mem.count() {
|
||||
if let Some(chunk) = mem.get_chunk(i) {
|
||||
let j = serde_json::json!({ "index": i, "chunk": chunk });
|
||||
@@ -219,11 +323,44 @@ fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
||||
}
|
||||
}
|
||||
|
||||
Commands::Keygen { .. } => unreachable!("handled before opening a store"),
|
||||
|
||||
Commands::Verify { public_key } => {
|
||||
let text = if Path::new(&public_key).is_file() {
|
||||
std::fs::read_to_string(&public_key)?
|
||||
} else {
|
||||
public_key
|
||||
};
|
||||
let bytes = signing::from_hex::<32>(&text)
|
||||
.ok_or("--public-key must be 64 hex characters or a file holding them")?;
|
||||
let trusted = VerifyingKey::from_bytes(&bytes)?;
|
||||
let r = HDF5Memory::verify(&path, &trusted)?;
|
||||
let j = serde_json::json!({
|
||||
"valid": r.is_valid(),
|
||||
"signed": r.signed,
|
||||
"key_matches": r.key_matches,
|
||||
"signature_valid": r.signature_valid,
|
||||
"records_match": r.records_match,
|
||||
"settings_match": r.settings_match,
|
||||
"sessions_match": r.sessions_match,
|
||||
"graph_match": r.graph_match,
|
||||
"changed_records": r.changed_records,
|
||||
"record_count": r.record_count,
|
||||
"signed_record_count": r.signed_record_count,
|
||||
"signed_by": r.public_key.map(|k| signing::to_hex(&k)),
|
||||
"wal_entries_unsigned": r.wal_entries_unsigned,
|
||||
});
|
||||
println!("{}", serde_json::to_string_pretty(&j)?);
|
||||
if !r.is_valid() {
|
||||
std::process::exit(2);
|
||||
}
|
||||
}
|
||||
|
||||
Commands::Snapshot { dest } => {
|
||||
let _result = clawhdf5_agent::storage::snapshot_file(&cli.path, &dest)?;
|
||||
let _result = clawhdf5_agent::storage::snapshot_file(&path, &dest)?;
|
||||
let j = serde_json::json!({
|
||||
"status": "snapshot_created",
|
||||
"source": cli.path.display().to_string(),
|
||||
"source": path.display().to_string(),
|
||||
"dest": dest.display().to_string(),
|
||||
});
|
||||
println!("{}", serde_json::to_string(&j)?);
|
||||
|
||||
@@ -1,7 +1,8 @@
|
||||
[package]
|
||||
name = "clawhdf5-derive"
|
||||
version = "2.6.0"
|
||||
version = "2.7.0"
|
||||
edition = "2024"
|
||||
rust-version.workspace = true
|
||||
description = "Derive macros for rustyhdf5 HDF5 traits"
|
||||
license = "MIT"
|
||||
repository = "https://git.redclaw.dev/quantumclaw/clawhdf5"
|
||||
|
||||
@@ -1,7 +1,8 @@
|
||||
[package]
|
||||
name = "clawhdf5-filters"
|
||||
version = "2.6.0"
|
||||
version = "2.7.0"
|
||||
edition = "2024"
|
||||
rust-version.workspace = true
|
||||
description = "Filter and compression pipeline for clawhdf5"
|
||||
license = "MIT"
|
||||
repository = "https://git.redclaw.dev/quantumclaw/clawhdf5"
|
||||
@@ -25,8 +26,12 @@ name = "compression_bench"
|
||||
harness = false
|
||||
|
||||
[features]
|
||||
default = ["fast-deflate"]
|
||||
# Pure-Rust zlib-rs by default; `fast-deflate` (zlib-ng, C) overrides it.
|
||||
default = ["zlib-rs"]
|
||||
fast-deflate = ["flate2/zlib-ng"]
|
||||
system-zlib = ["flate2/zlib-default"]
|
||||
zlib-rs = ["flate2/zlib-rs"]
|
||||
# `runtime_detection` gives zlib-rs `std`, which it needs to detect and use
|
||||
# SIMD at runtime. flate2 enables it by default, but we build flate2 with
|
||||
# default-features = false, and without it zlib-rs inflates 3.5x slower.
|
||||
zlib-rs = ["flate2/zlib-rs", "flate2/runtime_detection"]
|
||||
apple-compression = []
|
||||
|
||||
@@ -8,16 +8,18 @@ Filter and compression pipeline for clawhdf5.
|
||||
## Features
|
||||
|
||||
- DEFLATE compression/decompression
|
||||
- Fast deflate via zlib-ng (`fast-deflate` feature)
|
||||
- Pure-Rust deflate via zlib-rs (default, `zlib-rs` feature)
|
||||
- zlib-ng instead, if you want it (`fast-deflate` feature; C, needs cmake)
|
||||
- Apple Compression framework support (`apple-compression` feature)
|
||||
|
||||
## Usage
|
||||
|
||||
```rust
|
||||
use clawhdf5_filters::{deflate_decode, deflate_encode};
|
||||
use clawhdf5_filters::{deflate_compress, deflate_decompress};
|
||||
|
||||
let compressed = deflate_encode(&data, 6).unwrap();
|
||||
let decompressed = deflate_decode(&compressed).unwrap();
|
||||
let compressed = deflate_compress(&data, 6).unwrap();
|
||||
// The second argument bounds the output: the expected decompressed size.
|
||||
let decompressed = deflate_decompress(&compressed, data.len()).unwrap();
|
||||
```
|
||||
|
||||
## License
|
||||
|
||||
@@ -1,12 +1,13 @@
|
||||
//! Fast deflate backends: Apple Compression Framework and zlib-ng.
|
||||
//! Deflate backends: Apple Compression Framework, zlib-ng and zlib-rs.
|
||||
//!
|
||||
//! Backend selection priority (decompression & compression):
|
||||
//! 1. Apple Compression Framework (macOS only, `apple-compression` feature)
|
||||
//! 2. flate2 with zlib-ng backend (`fast-deflate` feature) or miniz_oxide (default)
|
||||
//! 2. flate2 with zlib-ng (`fast-deflate`), else zlib-rs (`zlib-rs`, the
|
||||
//! default), else miniz_oxide
|
||||
//!
|
||||
//! The Apple Compression Framework uses hardware-accelerated zlib on Apple Silicon
|
||||
//! and is typically the fastest option on macOS. zlib-ng is the fastest portable
|
||||
//! option and what C HDF5 uses internally.
|
||||
//! and is typically the fastest option on macOS. zlib-rs is a pure-Rust port of
|
||||
//! zlib-ng; see `BENCHMARKS.md` for how the two compare.
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Apple Compression Framework FFI (macOS only)
|
||||
@@ -243,65 +244,117 @@ mod apple {
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Streaming decompression via flate2 (uses zlib-ng when fast-deflate enabled)
|
||||
// One-shot (de)compression via flate2 (whichever backend flate2 was built with)
|
||||
//
|
||||
// The whole input goes to the codec in one call, into an output buffer sized
|
||||
// up front. `flate2::read::ZlibDecoder` / `write::ZlibEncoder` stream through a
|
||||
// 32 KiB buffer instead, which cost zlib-rs up to 3.7x against zlib-ng on a
|
||||
// 1 MB chunk. clawhdf5-format's deflate filter does the same; see
|
||||
// `BENCHMARKS.md`, "Deflate backend".
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Streaming decompress with pre-allocated output buffer.
|
||||
///
|
||||
/// When the output size is known (typical for HDF5 chunks), this avoids
|
||||
/// dynamic reallocation by writing directly into a pre-sized buffer.
|
||||
/// Decompress into a buffer pre-sized to `output_size`, the expected
|
||||
/// decompressed length (known for HDF5 chunks). Output longer than that is an
|
||||
/// error, as is a stream that ends early.
|
||||
pub(crate) fn flate2_decompress_preallocated(
|
||||
data: &[u8],
|
||||
output_size: usize,
|
||||
) -> Result<Vec<u8>, String> {
|
||||
use std::io::Read;
|
||||
let mut decoder = flate2::read::ZlibDecoder::new(data);
|
||||
let mut output = vec![0u8; output_size];
|
||||
let mut total_read = 0;
|
||||
|
||||
loop {
|
||||
match decoder.read(&mut output[total_read..]) {
|
||||
Ok(0) => break,
|
||||
Ok(n) => total_read += n,
|
||||
Err(e) => return Err(e.to_string()),
|
||||
}
|
||||
}
|
||||
output.truncate(total_read);
|
||||
Ok(output)
|
||||
inflate_bounded(data, output_size, output_size)
|
||||
}
|
||||
|
||||
/// Absolute ceiling on decompressed output when the caller has no size hint,
|
||||
/// preventing unbounded allocation from a hostile/corrupted zlib stream.
|
||||
const MAX_DECOMPRESS_SIZE: usize = 256 * 1024 * 1024;
|
||||
|
||||
/// Streaming decompress with dynamic sizing (when output size is unknown).
|
||||
///
|
||||
/// Bounded by [`MAX_DECOMPRESS_SIZE`] since there is no chunk-size hint to
|
||||
/// validate against here — an unbounded `read_to_end` would let a hostile
|
||||
/// zlib stream force arbitrarily large allocation (a "zlib bomb").
|
||||
/// Decompress with no size hint, bounded by [`MAX_DECOMPRESS_SIZE`] so a
|
||||
/// hostile zlib stream cannot force arbitrarily large allocation (a "zlib
|
||||
/// bomb").
|
||||
pub(crate) fn flate2_decompress_streaming(data: &[u8]) -> Result<Vec<u8>, String> {
|
||||
use std::io::Read;
|
||||
let decoder = flate2::read::ZlibDecoder::new(data);
|
||||
let mut result = Vec::new();
|
||||
decoder
|
||||
.take(MAX_DECOMPRESS_SIZE as u64 + 1)
|
||||
.read_to_end(&mut result)
|
||||
.map_err(|e| e.to_string())?;
|
||||
if result.len() > MAX_DECOMPRESS_SIZE {
|
||||
return Err(format!(
|
||||
let hint = data.len().saturating_mul(4).min(1 << 20);
|
||||
inflate_bounded(data, hint, MAX_DECOMPRESS_SIZE).map_err(|e| {
|
||||
if e.ends_with("exceeds size limit") {
|
||||
format!(
|
||||
"decompressed output exceeds {} MiB limit",
|
||||
MAX_DECOMPRESS_SIZE / 1024 / 1024
|
||||
));
|
||||
)
|
||||
} else {
|
||||
e
|
||||
}
|
||||
Ok(result)
|
||||
})
|
||||
}
|
||||
|
||||
/// Compress data using flate2 (zlib-ng when fast-deflate enabled, else miniz_oxide).
|
||||
/// Inflate a zlib stream, starting from `size_hint` bytes of output and
|
||||
/// failing past `limit`.
|
||||
fn inflate_bounded(data: &[u8], size_hint: usize, limit: usize) -> Result<Vec<u8>, String> {
|
||||
use flate2::{Decompress, FlushDecompress, Status};
|
||||
|
||||
// One byte of headroom past the limit distinguishes an over-size stream
|
||||
// from one that legitimately ends exactly at the limit.
|
||||
let max_capacity = limit.saturating_add(1);
|
||||
let mut out = Vec::new();
|
||||
out.try_reserve_exact(size_hint.clamp(1, max_capacity))
|
||||
.map_err(|e| format!("deflate: cannot allocate output: {e}"))?;
|
||||
|
||||
let mut inflater = Decompress::new(true);
|
||||
loop {
|
||||
let (in_before, out_before) = (inflater.total_in(), inflater.total_out());
|
||||
let status = inflater
|
||||
.decompress_vec(
|
||||
&data[in_before as usize..],
|
||||
&mut out,
|
||||
FlushDecompress::Finish,
|
||||
)
|
||||
.map_err(|e| format!("deflate: {e}"))?;
|
||||
if out.len() > limit {
|
||||
return Err("deflate: output exceeds size limit".into());
|
||||
}
|
||||
match status {
|
||||
Status::StreamEnd => return Ok(out),
|
||||
Status::Ok | Status::BufError if out.len() == out.capacity() => {
|
||||
let grow = out.capacity().min(max_capacity - out.capacity()).max(1);
|
||||
out.try_reserve_exact(grow)
|
||||
.map_err(|e| format!("deflate: cannot allocate output: {e}"))?;
|
||||
}
|
||||
Status::Ok | Status::BufError => {
|
||||
if inflater.total_in() as usize >= data.len()
|
||||
|| (inflater.total_in(), inflater.total_out()) == (in_before, out_before)
|
||||
{
|
||||
return Err("deflate: truncated stream".into());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Compress data using flate2 (zlib-ng, zlib-rs or miniz_oxide; see module docs).
|
||||
pub(crate) fn flate2_compress(data: &[u8], level: u32) -> Result<Vec<u8>, String> {
|
||||
use std::io::Write;
|
||||
let mut encoder = flate2::write::ZlibEncoder::new(Vec::new(), flate2::Compression::new(level));
|
||||
encoder.write_all(data).map_err(|e| e.to_string())?;
|
||||
encoder.finish().map_err(|e| e.to_string())
|
||||
use flate2::{Compress, Compression, FlushCompress, Status};
|
||||
|
||||
// zlib's compressBound, plus the zlib header and trailer.
|
||||
let bound = data.len() + (data.len() >> 12) + (data.len() >> 14) + (data.len() >> 25) + 13 + 6;
|
||||
let mut out = Vec::new();
|
||||
out.try_reserve_exact(bound)
|
||||
.map_err(|e| format!("deflate: cannot allocate output: {e}"))?;
|
||||
|
||||
let mut deflater = Compress::new(Compression::new(level), true);
|
||||
loop {
|
||||
let (in_before, out_before) = (deflater.total_in(), deflater.total_out());
|
||||
let status = deflater
|
||||
.compress_vec(&data[in_before as usize..], &mut out, FlushCompress::Finish)
|
||||
.map_err(|e| format!("deflate: {e}"))?;
|
||||
match status {
|
||||
Status::StreamEnd => return Ok(out),
|
||||
Status::Ok | Status::BufError if out.len() == out.capacity() => out
|
||||
.try_reserve(out.capacity().max(4096))
|
||||
.map_err(|e| format!("deflate: cannot allocate output: {e}"))?,
|
||||
Status::Ok | Status::BufError => {
|
||||
if (deflater.total_in(), deflater.total_out()) == (in_before, out_before) {
|
||||
return Err("deflate: encoder made no progress".into());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -312,7 +365,7 @@ pub(crate) fn flate2_compress(data: &[u8], level: u32) -> Result<Vec<u8>, String
|
||||
///
|
||||
/// Selection order:
|
||||
/// 1. Apple Compression Framework (macOS + `apple-compression` feature)
|
||||
/// 2. flate2 (zlib-ng with `fast-deflate`, otherwise miniz_oxide)
|
||||
/// 2. flate2 (zlib-ng with `fast-deflate`, else zlib-rs, else miniz_oxide)
|
||||
///
|
||||
/// When `output_hint` > 0, pre-allocates the output buffer for zero-copy
|
||||
/// decompression (avoids reallocation).
|
||||
@@ -344,7 +397,7 @@ pub fn decompress(data: &[u8], output_hint: usize) -> Result<Vec<u8>, String> {
|
||||
///
|
||||
/// Selection order:
|
||||
/// 1. Apple Compression Framework (macOS + `apple-compression` feature)
|
||||
/// 2. flate2 (zlib-ng with `fast-deflate`, otherwise miniz_oxide)
|
||||
/// 2. flate2 (zlib-ng with `fast-deflate`, else zlib-rs, else miniz_oxide)
|
||||
pub fn compress(data: &[u8], level: u32) -> Result<Vec<u8>, String> {
|
||||
#[cfg(all(target_os = "macos", feature = "apple-compression"))]
|
||||
{
|
||||
@@ -377,9 +430,19 @@ pub fn active_backend() -> &'static str {
|
||||
{
|
||||
"zlib-ng"
|
||||
}
|
||||
// flate2 prefers a C zlib over zlib-rs when both are enabled.
|
||||
#[cfg(all(
|
||||
not(all(target_os = "macos", feature = "apple-compression")),
|
||||
not(feature = "fast-deflate"),
|
||||
feature = "zlib-rs"
|
||||
))]
|
||||
{
|
||||
"zlib-rs"
|
||||
}
|
||||
#[cfg(not(any(
|
||||
all(target_os = "macos", feature = "apple-compression"),
|
||||
feature = "fast-deflate"
|
||||
feature = "fast-deflate",
|
||||
feature = "zlib-rs"
|
||||
)))]
|
||||
{
|
||||
"miniz_oxide"
|
||||
@@ -436,7 +499,7 @@ mod tests {
|
||||
fn backend_name_is_set() {
|
||||
let name = active_backend();
|
||||
assert!(
|
||||
["miniz_oxide", "zlib-ng", "apple-compression"].contains(&name),
|
||||
["miniz_oxide", "zlib-rs", "zlib-ng", "apple-compression"].contains(&name),
|
||||
"unexpected backend: {name}"
|
||||
);
|
||||
}
|
||||
|
||||
@@ -2,12 +2,14 @@
|
||||
//!
|
||||
//! Provides deflate (zlib) decompression/compression with multiple backend options:
|
||||
//!
|
||||
//! - **Default**: `miniz_oxide` (pure Rust, no C dependencies)
|
||||
//! - **`fast-deflate` feature**: `zlib-ng` via flate2 (~2-3x faster, matches C HDF5)
|
||||
//! - **Default (`zlib-rs` feature)**: `zlib-rs` via flate2 (pure Rust, no C
|
||||
//! dependencies)
|
||||
//! - **`fast-deflate` feature**: `zlib-ng` via flate2 (C, built with cmake)
|
||||
//! - **`apple-compression` feature**: Apple Compression Framework on macOS
|
||||
//! (hardware-accelerated on Apple Silicon)
|
||||
//! - With none of the above: `miniz_oxide` (pure Rust, slower)
|
||||
//!
|
||||
//! Backend priority: apple-compression > zlib-ng > miniz_oxide.
|
||||
//! Backend priority: apple-compression > zlib-ng > zlib-rs > miniz_oxide.
|
||||
|
||||
pub mod fast_deflate;
|
||||
|
||||
@@ -115,7 +117,7 @@ mod tests {
|
||||
fn backend_reports_name() {
|
||||
let name = deflate_backend();
|
||||
assert!(
|
||||
["miniz_oxide", "zlib-ng", "apple-compression"].contains(&name),
|
||||
["miniz_oxide", "zlib-rs", "zlib-ng", "apple-compression"].contains(&name),
|
||||
"unexpected backend: {name}"
|
||||
);
|
||||
}
|
||||
|
||||
@@ -1,7 +1,8 @@
|
||||
[package]
|
||||
name = "clawhdf5-format"
|
||||
version = "2.6.0"
|
||||
version = "2.7.0"
|
||||
edition = "2024"
|
||||
rust-version.workspace = true
|
||||
description = "Pure-Rust HDF5 binary format parsing and writing — no C dependencies"
|
||||
license = "MIT"
|
||||
repository = "https://git.redclaw.dev/quantumclaw/clawhdf5"
|
||||
@@ -21,18 +22,33 @@ zstd = { version = "0.13", optional = true }
|
||||
blake3 = { version = "1", optional = true }
|
||||
libaec-sys = { path = "../libaec-sys", version = "0.1", optional = true }
|
||||
pco = { version = "1.0", optional = true }
|
||||
# Pure-Rust Zstandard, for the plugin filters that embed zstd (bitshuffle,
|
||||
# blosc). The `zstd` feature (filter 32015) links libzstd instead.
|
||||
ruzstd = { version = "0.9", optional = true }
|
||||
# bzip2 with its default backend, libbz2-rs-sys: a pure-Rust port of
|
||||
# libbzip2 (no C is compiled, despite the -sys name).
|
||||
bzip2 = { version = "0.6", optional = true }
|
||||
snap = { version = "1", optional = true }
|
||||
|
||||
[target.'cfg(target_os = "linux")'.dependencies]
|
||||
# madvise(MADV_HUGEPAGE) for large read buffers (see src/bulk_alloc.rs).
|
||||
libc = { version = "0.2", default-features = false }
|
||||
|
||||
[dev-dependencies]
|
||||
half = { workspace = true }
|
||||
serde_json = "1"
|
||||
criterion = { workspace = true }
|
||||
clawhdf5-derive = { path = "../clawhdf5-derive", version = "2.6.0" }
|
||||
clawhdf5-derive = { path = "../clawhdf5-derive", version = "2.7.0" }
|
||||
|
||||
[[bench]]
|
||||
name = "bench"
|
||||
harness = false
|
||||
|
||||
[features]
|
||||
default = ["std", "checksum", "deflate", "provenance", "fast-deflate", "system-zlib-decompress"]
|
||||
# Deflate backend: `zlib-rs` (pure Rust) by default. `fast-deflate` selects
|
||||
# zlib-ng instead (C, built with cmake); flate2 prefers a C zlib whenever one
|
||||
# is enabled, so turning it on anywhere in the build overrides the default.
|
||||
default = ["std", "checksum", "deflate", "provenance", "zlib-rs", "system-zlib-decompress", "lzf"]
|
||||
std = []
|
||||
checksum = []
|
||||
deflate = ["flate2"]
|
||||
@@ -42,12 +58,26 @@ fast-checksum = ["crc32fast"]
|
||||
fast-deflate = ["flate2/zlib-ng"]
|
||||
system-zlib = ["flate2/zlib-default"]
|
||||
system-zlib-decompress = []
|
||||
zlib-rs = ["flate2/zlib-rs"]
|
||||
# `runtime_detection` gives zlib-rs `std`, which it needs to detect and use
|
||||
# SIMD at runtime. flate2 enables it by default, but we build flate2 with
|
||||
# default-features = false, and without it zlib-rs inflates 3.5x slower.
|
||||
zlib-rs = ["flate2/zlib-rs", "flate2/runtime_detection"]
|
||||
lz4 = ["lz4_flex"]
|
||||
zstd = ["dep:zstd"]
|
||||
blake3_hash = ["blake3"]
|
||||
szip = ["libaec-sys"]
|
||||
pcodec = ["dep:pco"]
|
||||
# Plugin filters, pure Rust. LZF (32000) is h5py's built-in compression; it
|
||||
# has no dependencies, so it is on by default.
|
||||
lzf = []
|
||||
# Bitshuffle (32008), with its LZ4 and Zstandard modes.
|
||||
bitshuffle = ["lz4_flex", "ruzstd"]
|
||||
# bzip2 (307).
|
||||
bzip2 = ["dep:bzip2", "std"]
|
||||
# Blosc 1 (32001) with its BloscLZ, LZ4, Snappy, Zlib and Zstandard codecs.
|
||||
blosc = ["lz4_flex", "ruzstd", "snap", "deflate", "std"]
|
||||
# Every plugin filter above.
|
||||
plugin-filters = ["lzf", "bitshuffle", "bzip2", "blosc"]
|
||||
|
||||
[[bench]]
|
||||
name = "parallel_decompress_bench"
|
||||
|
||||
@@ -1 +1,4 @@
|
||||
target/
|
||||
corpus/
|
||||
artifacts/
|
||||
coverage/
|
||||
|
||||
Binary file not shown.
@@ -1,15 +1,36 @@
|
||||
#![no_main]
|
||||
use clawhdf5_format::btree_v2::{BTreeV2Header, collect_btree_v2_records};
|
||||
use libfuzzer_sys::fuzz_target;
|
||||
|
||||
fuzz_target!(|data: &[u8]| {
|
||||
for &offset_size in &[4u8, 8] {
|
||||
for &length_size in &[4u8, 8] {
|
||||
let _ = clawhdf5_format::btree_v2::BTreeV2Header::parse(
|
||||
data,
|
||||
0,
|
||||
offset_size,
|
||||
length_size,
|
||||
);
|
||||
if let Ok(header) = BTreeV2Header::parse(data, 0, offset_size, length_size) {
|
||||
let _ = collect_btree_v2_records(data, &header, offset_size, length_size);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Parsing a header requires a valid checksum, which random input almost
|
||||
// never has, so the traversal behind it went unfuzzed — and that is where
|
||||
// a node listing itself as its own child overflowed the stack. Take the
|
||||
// header fields straight from the input instead and walk the rest.
|
||||
let Some((fields, file)) = data.split_first_chunk::<20>() else {
|
||||
return;
|
||||
};
|
||||
let header = BTreeV2Header {
|
||||
tree_type: fields[0],
|
||||
node_size: u32::from_le_bytes([fields[1], fields[2], fields[3], fields[4]]),
|
||||
record_size: u16::from_le_bytes([fields[5], fields[6]]),
|
||||
depth: u16::from_le_bytes([fields[7], fields[8]]),
|
||||
root_node_address: u64::from(u32::from_le_bytes([
|
||||
fields[9], fields[10], fields[11], fields[12],
|
||||
])),
|
||||
num_records_in_root: u16::from_le_bytes([fields[13], fields[14]]),
|
||||
total_records: u64::from(u32::from_le_bytes([
|
||||
fields[15], fields[16], fields[17], fields[18],
|
||||
])),
|
||||
};
|
||||
let offset_size = if fields[19] & 1 == 0 { 4 } else { 8 };
|
||||
let _ = collect_btree_v2_records(file, &header, offset_size, 8);
|
||||
});
|
||||
|
||||
@@ -97,7 +97,7 @@ impl AttributeMessage {
|
||||
return Ok(Cow::Borrowed(bytes));
|
||||
}
|
||||
let (file_data, offset_size) = file.ok_or(FormatError::UnresolvedSharedMessage)?;
|
||||
let shared_ref = shared_message::parse_shared_ref(bytes, offset_size)?;
|
||||
let shared_ref = shared_message::parse_shared_ref_sized(bytes, offset_size, length_size)?;
|
||||
shared_message::resolve_shared_message(
|
||||
file_data,
|
||||
&shared_ref,
|
||||
@@ -362,6 +362,18 @@ fn extract_name(bytes: &[u8]) -> String {
|
||||
String::from_utf8_lossy(&bytes[..end]).into_owned()
|
||||
}
|
||||
|
||||
/// An attribute's datatype gets libhdf5's extra check for a header without
|
||||
/// a checksum (see [`Datatype::check_unused_bits`]).
|
||||
fn check_in_header(
|
||||
attr: AttributeMessage,
|
||||
header: &ObjectHeader,
|
||||
) -> Result<AttributeMessage, FormatError> {
|
||||
if header.version == 1 {
|
||||
attr.datatype.check_unused_bits()?;
|
||||
}
|
||||
Ok(attr)
|
||||
}
|
||||
|
||||
/// Extract all attribute messages from an object header.
|
||||
pub fn extract_attributes(
|
||||
header: &ObjectHeader,
|
||||
@@ -371,7 +383,7 @@ pub fn extract_attributes(
|
||||
for msg in &header.messages {
|
||||
if msg.msg_type == MessageType::Attribute {
|
||||
let attr = AttributeMessage::parse(&msg.data, length_size)?;
|
||||
attrs.push(attr);
|
||||
attrs.push(check_in_header(attr, header)?);
|
||||
}
|
||||
}
|
||||
Ok(attrs)
|
||||
@@ -394,42 +406,81 @@ pub fn find_attribute<'a>(
|
||||
///
|
||||
/// Use this instead of `extract_attributes` when reading files that may use dense storage
|
||||
/// (e.g., objects with many attributes, typically >8).
|
||||
///
|
||||
/// Fails if any attribute cannot be read; see [`extract_attributes_tolerant`]
|
||||
/// to read the others.
|
||||
pub fn extract_attributes_full(
|
||||
file_data: &[u8],
|
||||
header: &ObjectHeader,
|
||||
offset_size: u8,
|
||||
length_size: u8,
|
||||
) -> Result<Vec<AttributeMessage>, FormatError> {
|
||||
extract_attributes_with(file_data, header, offset_size, length_size, &mut Err)
|
||||
}
|
||||
|
||||
/// Like [`extract_attributes_full`], but an attribute that cannot be read
|
||||
/// (a corrupt or unsupported attribute message, or a heap object that cannot
|
||||
/// be located) is left out and its error returned alongside the attributes
|
||||
/// that could be read, instead of failing them all.
|
||||
///
|
||||
/// Errors in the structures that index the attributes (the Attribute Info
|
||||
/// message, the dense-storage heap header or B-tree) still fail the call:
|
||||
/// then it is unknown which attributes exist at all.
|
||||
pub fn extract_attributes_tolerant(
|
||||
file_data: &[u8],
|
||||
header: &ObjectHeader,
|
||||
offset_size: u8,
|
||||
length_size: u8,
|
||||
) -> Result<(Vec<AttributeMessage>, Vec<FormatError>), FormatError> {
|
||||
let mut errors = Vec::new();
|
||||
let attrs = extract_attributes_with(file_data, header, offset_size, length_size, &mut |e| {
|
||||
errors.push(e);
|
||||
Ok(())
|
||||
})?;
|
||||
Ok((attrs, errors))
|
||||
}
|
||||
|
||||
/// Read every attribute; each one that fails goes to `on_error`, which
|
||||
/// either stops the read (returns the error) or skips that attribute.
|
||||
fn extract_attributes_with(
|
||||
file_data: &[u8],
|
||||
header: &ObjectHeader,
|
||||
offset_size: u8,
|
||||
length_size: u8,
|
||||
on_error: &mut dyn FnMut(FormatError) -> Result<(), FormatError>,
|
||||
) -> Result<Vec<AttributeMessage>, FormatError> {
|
||||
let mut attrs = Vec::new();
|
||||
|
||||
// Collect compact attributes (inline in OH)
|
||||
for msg in &header.messages {
|
||||
if msg.msg_type == MessageType::Attribute {
|
||||
if shared_message::is_shared(msg.flags) {
|
||||
let attr = if shared_message::is_shared(msg.flags) {
|
||||
// Shared attribute: resolve the reference to get actual attribute data
|
||||
let shared_ref = shared_message::parse_shared_ref(&msg.data, offset_size)?;
|
||||
let resolved_data = shared_message::resolve_shared_message(
|
||||
shared_message::parse_shared_ref_sized(&msg.data, offset_size, length_size)
|
||||
.and_then(|shared_ref| {
|
||||
shared_message::resolve_shared_message(
|
||||
file_data,
|
||||
&shared_ref,
|
||||
MessageType::Attribute,
|
||||
offset_size,
|
||||
length_size,
|
||||
)?;
|
||||
let attr = AttributeMessage::parse_in_file(
|
||||
&resolved_data,
|
||||
)
|
||||
})
|
||||
.and_then(|resolved| {
|
||||
AttributeMessage::parse_in_file(
|
||||
&resolved,
|
||||
file_data,
|
||||
offset_size,
|
||||
length_size,
|
||||
)?;
|
||||
attrs.push(attr);
|
||||
)
|
||||
})
|
||||
} else {
|
||||
let attr = AttributeMessage::parse_in_file(
|
||||
&msg.data,
|
||||
file_data,
|
||||
offset_size,
|
||||
length_size,
|
||||
)?;
|
||||
attrs.push(attr);
|
||||
AttributeMessage::parse_in_file(&msg.data, file_data, offset_size, length_size)
|
||||
};
|
||||
let attr = attr.and_then(|a| check_in_header(a, header));
|
||||
match attr {
|
||||
Ok(attr) => attrs.push(attr),
|
||||
Err(e) => on_error(e)?,
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -439,9 +490,15 @@ pub fn extract_attributes_full(
|
||||
if let Some(info) = attr_info
|
||||
&& let Some(fh_addr) = info.fractal_heap_address
|
||||
{
|
||||
let dense_attrs =
|
||||
extract_dense_attributes(file_data, &info, fh_addr, offset_size, length_size)?;
|
||||
attrs.extend(dense_attrs);
|
||||
extract_dense_attributes(
|
||||
file_data,
|
||||
&info,
|
||||
fh_addr,
|
||||
offset_size,
|
||||
length_size,
|
||||
&mut attrs,
|
||||
on_error,
|
||||
)?;
|
||||
}
|
||||
|
||||
Ok(attrs)
|
||||
@@ -468,7 +525,9 @@ fn extract_dense_attributes(
|
||||
fh_addr: u64,
|
||||
offset_size: u8,
|
||||
length_size: u8,
|
||||
) -> Result<Vec<AttributeMessage>, FormatError> {
|
||||
attrs: &mut Vec<AttributeMessage>,
|
||||
on_error: &mut dyn FnMut(FormatError) -> Result<(), FormatError>,
|
||||
) -> Result<(), FormatError> {
|
||||
// Parse fractal heap
|
||||
let fh = FractalHeapHeader::parse(file_data, fh_addr as usize, offset_size, length_size)?;
|
||||
|
||||
@@ -482,28 +541,32 @@ fn extract_dense_attributes(
|
||||
let btree_hdr = BTreeV2Header::parse(file_data, btree_addr as usize, offset_size, length_size)?;
|
||||
let records = collect_btree_v2_records(file_data, &btree_hdr, offset_size, length_size)?;
|
||||
|
||||
let mut attrs = Vec::new();
|
||||
for record in &records {
|
||||
// Per HDF5 spec, both type 8 and type 9 records start with heap_id:
|
||||
// Type 8: heap_id(8) + msg_flags(1) + creation_order(4) + hash(4)
|
||||
// Type 9: heap_id(8) + msg_flags(1) + creation_order(4)
|
||||
let id_offset = 0;
|
||||
|
||||
if record.data.len() < id_offset + fh.heap_id_length as usize {
|
||||
let id_len = fh.heap_id_length as usize;
|
||||
let Some(id_bytes) = record.data.get(..id_len) else {
|
||||
on_error(FormatError::UnexpectedEof {
|
||||
expected: id_len,
|
||||
available: record.data.len(),
|
||||
})?;
|
||||
continue;
|
||||
}
|
||||
let id_bytes = &record.data[id_offset..id_offset + fh.heap_id_length as usize];
|
||||
|
||||
// Read attribute message from fractal heap
|
||||
let attr_data = fh.read_managed_object(file_data, id_bytes, offset_size)?;
|
||||
};
|
||||
|
||||
// The data in the heap is a complete attribute message
|
||||
let attr =
|
||||
AttributeMessage::parse_in_file(&attr_data, file_data, offset_size, length_size)?;
|
||||
attrs.push(attr);
|
||||
let attr = fh
|
||||
.read_managed_object(file_data, id_bytes, offset_size)
|
||||
.and_then(|attr_data| {
|
||||
AttributeMessage::parse_in_file(&attr_data, file_data, offset_size, length_size)
|
||||
});
|
||||
match attr {
|
||||
Ok(attr) => attrs.push(attr),
|
||||
Err(e) => on_error(e)?,
|
||||
}
|
||||
}
|
||||
|
||||
Ok(attrs)
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
@@ -523,7 +586,8 @@ mod tests {
|
||||
|
||||
/// Build an f64 LE datatype message.
|
||||
fn build_f64_dt() -> Vec<u8> {
|
||||
let mut buf = build_dt_header(1, 1, [0x00, 0x00, 0x02], 8);
|
||||
// Sign bit 63 (bits 8-15 of the class bits).
|
||||
let mut buf = build_dt_header(1, 1, [0x20, 63, 0x00], 8);
|
||||
let mut props = [0u8; 12];
|
||||
props[2..4].copy_from_slice(&64u16.to_le_bytes()); // bit_precision
|
||||
props[4] = 52; // exp_location
|
||||
|
||||
@@ -172,6 +172,17 @@ fn max_records_leaf(node_size: u32, record_size: u16) -> u64 {
|
||||
((node_size - overhead) / record_size as u32) as u64
|
||||
}
|
||||
|
||||
/// Deepest B-tree v2 accepted. See [`collect_btree_v2_records`].
|
||||
const MAX_DEPTH: u16 = 64;
|
||||
|
||||
/// Take `n` records from the traversal's budget, or refuse the tree.
|
||||
fn spend(budget: &mut usize, n: usize) -> Result<(), FormatError> {
|
||||
*budget = budget
|
||||
.checked_sub(n)
|
||||
.ok_or(FormatError::NestingDepthExceeded)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Collect all records from a B-tree v2 by traversing from the root.
|
||||
pub fn collect_btree_v2_records(
|
||||
file_data: &[u8],
|
||||
@@ -182,6 +193,22 @@ pub fn collect_btree_v2_records(
|
||||
if header.total_records == 0 || header.num_records_in_root == 0 {
|
||||
return Ok(Vec::new());
|
||||
}
|
||||
// Recursion is one frame per level, and the depth is read from the file:
|
||||
// a crafted header claiming 65 535 levels over a node that is its own
|
||||
// child overflowed the stack. 64 matches the fractal heap's guard, and no
|
||||
// real tree comes close — even at the minimum fan-out of two it would
|
||||
// hold more than 2^64 records.
|
||||
if header.depth > MAX_DEPTH {
|
||||
return Err(FormatError::NestingDepthExceeded);
|
||||
}
|
||||
// A valid tree stores each record once, in its own bytes, so it cannot
|
||||
// hold more records than the file has room for. Children are addresses,
|
||||
// though, and nothing makes them distinct: levels whose children all
|
||||
// point at one shared node below reach it fan-out^depth times, which is
|
||||
// millions of records from a few kilobytes. Counting against what the
|
||||
// file could physically contain bounds that without trusting the
|
||||
// header's own `total_records`.
|
||||
let mut budget = file_data.len() / usize::from(header.record_size.max(1));
|
||||
|
||||
let max_leaf_nrec = max_records_leaf(header.node_size, header.record_size);
|
||||
|
||||
@@ -206,6 +233,7 @@ pub fn collect_btree_v2_records(
|
||||
offset_size,
|
||||
length_size,
|
||||
max_leaf_nrec,
|
||||
&mut budget,
|
||||
&mut records,
|
||||
)?;
|
||||
Ok(records)
|
||||
@@ -273,6 +301,7 @@ fn collect_internal_records(
|
||||
offset_size: u8,
|
||||
length_size: u8,
|
||||
max_leaf_nrec: u64,
|
||||
budget: &mut usize,
|
||||
out: &mut Vec<BTreeV2Record>,
|
||||
) -> Result<(), FormatError> {
|
||||
// signature(4) + version(1) + type(1) = 6
|
||||
@@ -294,39 +323,21 @@ fn collect_internal_records(
|
||||
let records_start = pos;
|
||||
pos += records_total;
|
||||
|
||||
// Compute sizes for child pointers
|
||||
// max_records at child depth - for variable-width nrec encoding
|
||||
// Child pointer layout, as libhdf5 computes it (H5B2__hdr_init): the
|
||||
// child's record count is always encoded in the width needed for a
|
||||
// *leaf's* maximum, and — below the first internal level — the child
|
||||
// subtree's total record count in the width needed for the most records
|
||||
// a subtree of that depth can hold.
|
||||
let child_depth = depth - 1;
|
||||
let max_nrec_child = if child_depth == 0 {
|
||||
max_leaf_nrec
|
||||
} else {
|
||||
// For internal nodes at child_depth, the true max_nrec depends on the
|
||||
// node size, record size, and the recursive width of child pointer
|
||||
// entries (which themselves depend on max_nrec at deeper levels).
|
||||
// Computing the exact value requires iterating from the leaf level
|
||||
// upward, as described in the HDF5 spec (III.A.2 "Computing the Size
|
||||
// of B-tree Nodes").
|
||||
//
|
||||
// We use `max_leaf_nrec * 2` as a conservative upper bound. This
|
||||
// over-estimates the nrec encoding width, which means we may read
|
||||
// slightly more bytes per child pointer than strictly necessary, but
|
||||
// never fewer. The over-read bytes are harmless because we only
|
||||
// decode `num_records` entries (the actual count from the node header).
|
||||
//
|
||||
// Known limitation: for very deep trees (depth > 3) with small record
|
||||
// sizes, the true max could exceed this estimate, causing us to
|
||||
// under-allocate the nrec encoding width and misparse child pointers.
|
||||
// In practice, HDF5 B-tree v2 depths rarely exceed 2-3.
|
||||
max_leaf_nrec * 2
|
||||
};
|
||||
let nrec_width = bytes_for_max_records(max_nrec_child);
|
||||
|
||||
// Total records in subtree width (only if depth > 1)
|
||||
let nrec_width = bytes_for_max_records(max_leaf_nrec);
|
||||
let total_nrec_width = if depth > 1 {
|
||||
// Width to hold total records in a subtree
|
||||
// We compute max possible total records at this subtree depth
|
||||
let max_total = header_max_total_records(max_leaf_nrec, depth - 1);
|
||||
bytes_for_max_records(max_total)
|
||||
bytes_for_max_records(cum_max_records(
|
||||
node_size,
|
||||
record_size,
|
||||
offset_size,
|
||||
max_leaf_nrec,
|
||||
child_depth,
|
||||
))
|
||||
} else {
|
||||
0
|
||||
};
|
||||
@@ -350,6 +361,8 @@ fn collect_internal_records(
|
||||
// We collect child[0] records, then record[0], then child[1], etc.
|
||||
for (i, &(child_addr, child_nrec)) in children.iter().enumerate() {
|
||||
if child_depth == 0 {
|
||||
// Before parsing, so a refused tree is not also a large allocation.
|
||||
spend(budget, usize::from(child_nrec))?;
|
||||
let leaf_recs =
|
||||
parse_leaf_records(file_data, child_addr as usize, child_nrec, record_size)?;
|
||||
out.extend(leaf_recs);
|
||||
@@ -364,6 +377,7 @@ fn collect_internal_records(
|
||||
offset_size,
|
||||
length_size,
|
||||
max_leaf_nrec,
|
||||
budget,
|
||||
out,
|
||||
)?;
|
||||
}
|
||||
@@ -393,6 +407,7 @@ fn collect_internal_records(
|
||||
available: file_data.len(),
|
||||
});
|
||||
}
|
||||
spend(budget, 1)?;
|
||||
out.push(BTreeV2Record {
|
||||
data: file_data[rec_start..rec_end].to_vec(),
|
||||
});
|
||||
@@ -402,14 +417,36 @@ fn collect_internal_records(
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Estimate maximum total records at a given depth (for variable-width encoding).
|
||||
fn header_max_total_records(max_leaf_nrec: u64, depth: u16) -> u64 {
|
||||
// Conservative: branching factor * max_leaf at each level
|
||||
let mut total = max_leaf_nrec;
|
||||
for _ in 0..depth {
|
||||
total = total.saturating_mul(max_leaf_nrec.max(2));
|
||||
/// Most records a subtree whose root is at `depth` can hold (libhdf5's
|
||||
/// `cum_max_nrec`): a leaf holds `max_leaf_nrec`; an internal node at depth
|
||||
/// `d` holds `max_nrec(d)` records and `max_nrec(d) + 1` subtrees of depth
|
||||
/// `d - 1`, where `max_nrec(d)` is what fits in a node once each record is
|
||||
/// paired with a child pointer of the width depth `d` needs.
|
||||
fn cum_max_records(
|
||||
node_size: u32,
|
||||
record_size: u16,
|
||||
offset_size: u8,
|
||||
max_leaf_nrec: u64,
|
||||
depth: u16,
|
||||
) -> u64 {
|
||||
// Internal node overhead: signature(4) + version(1) + type(1) + checksum(4).
|
||||
const PREFIX: u64 = 10;
|
||||
let nrec_width = bytes_for_max_records(max_leaf_nrec) as u64;
|
||||
let mut cum = max_leaf_nrec;
|
||||
let mut cum_width = 0u64;
|
||||
for d in 1..=depth {
|
||||
let ptr = u64::from(offset_size) + nrec_width + if d > 1 { cum_width } else { 0 };
|
||||
let max_nrec = u64::from(node_size)
|
||||
.saturating_sub(PREFIX)
|
||||
.saturating_sub(ptr)
|
||||
/ (u64::from(record_size) + ptr).max(1);
|
||||
cum = max_nrec
|
||||
.saturating_add(1)
|
||||
.saturating_mul(cum)
|
||||
.saturating_add(max_nrec);
|
||||
cum_width = bytes_for_max_records(cum) as u64;
|
||||
}
|
||||
total
|
||||
cum
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
@@ -466,6 +503,130 @@ mod tests {
|
||||
buf
|
||||
}
|
||||
|
||||
/// An internal node laid out exactly as `collect_internal_records` will
|
||||
/// read it at `depth`: `records` zeroed records, then `children` pointers,
|
||||
/// all to `child_addr` claiming `child_nrec` records.
|
||||
fn internal_node(
|
||||
depth: u16,
|
||||
node_size: u32,
|
||||
record_size: u16,
|
||||
records: usize,
|
||||
children: usize,
|
||||
child_addr: u64,
|
||||
child_nrec: u64,
|
||||
) -> Vec<u8> {
|
||||
let max_leaf = max_records_leaf(node_size, record_size);
|
||||
let nrec_width = bytes_for_max_records(max_leaf);
|
||||
let total_width = if depth > 1 {
|
||||
bytes_for_max_records(cum_max_records(
|
||||
node_size,
|
||||
record_size,
|
||||
8,
|
||||
max_leaf,
|
||||
depth - 1,
|
||||
))
|
||||
} else {
|
||||
0
|
||||
};
|
||||
let mut buf = b"BTIN".to_vec();
|
||||
buf.extend_from_slice(&[0, 5]);
|
||||
buf.resize(buf.len() + records * record_size as usize, 0);
|
||||
for _ in 0..children {
|
||||
buf.extend_from_slice(&child_addr.to_le_bytes());
|
||||
buf.extend_from_slice(&child_nrec.to_le_bytes()[..nrec_width]);
|
||||
buf.resize(buf.len() + total_width, 0);
|
||||
}
|
||||
buf
|
||||
}
|
||||
|
||||
fn header(depth: u16, root: u64, root_nrec: u16, total: u64) -> BTreeV2Header {
|
||||
BTreeV2Header {
|
||||
tree_type: 5,
|
||||
node_size: 512,
|
||||
record_size: 8,
|
||||
depth,
|
||||
root_node_address: root,
|
||||
num_records_in_root: root_nrec,
|
||||
total_records: total,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_node_that_is_its_own_child_is_rejected_not_recursed() {
|
||||
// One internal node whose two children are itself, under a header
|
||||
// claiming the deepest tree a u16 allows. The layout stops depending
|
||||
// on depth once the subtree-total width saturates, so every level
|
||||
// parses cleanly and recursion runs ~65 000 frames deep: before the
|
||||
// cap this overflowed the stack and aborted the process, from a file
|
||||
// of under 100 bytes.
|
||||
let mut data = internal_node(u16::MAX, 512, 8, 1, 2, 0, 1);
|
||||
data.resize(4096, 0);
|
||||
let result = collect_btree_v2_records(&data, &header(u16::MAX, 0, 1, 1), 8, 8);
|
||||
assert!(result.is_err(), "{result:?}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_shared_subtree_cannot_multiply_the_work() {
|
||||
// A chain of distinct levels, each node's children all pointing at the
|
||||
// single node below, ending in a real leaf. Every node parses and
|
||||
// nothing is cyclic, yet the leaf is reached fan-out^depth times: 62
|
||||
// children over 4 levels is ~15 million leaf visits from a few
|
||||
// kilobytes. A valid tree cannot hold more records than the file has
|
||||
// room for, so that bounds the traversal instead.
|
||||
let (node_size, record_size) = (512u32, 8u16);
|
||||
let fanout = 62usize;
|
||||
let depth = 4u16;
|
||||
let leaf = build_leaf_node(5, &[&[0u8; 8][..]]);
|
||||
|
||||
// Lay out root first, then each lower level, then the leaf.
|
||||
let mut nodes: Vec<Vec<u8>> = Vec::new();
|
||||
let mut addrs = Vec::new();
|
||||
let mut at = 0u64;
|
||||
let mut sizes = Vec::new();
|
||||
for d in (1..=depth).rev() {
|
||||
let n = internal_node(d, node_size, record_size, fanout - 1, fanout, 0, 0);
|
||||
sizes.push(n.len());
|
||||
}
|
||||
for size in &sizes {
|
||||
addrs.push(at);
|
||||
at += *size as u64;
|
||||
}
|
||||
let leaf_addr = at;
|
||||
for (i, d) in (1..=depth).rev().enumerate() {
|
||||
let (child, child_nrec) = if d == 1 {
|
||||
(leaf_addr, 1)
|
||||
} else {
|
||||
(addrs[i + 1], fanout as u64 - 1)
|
||||
};
|
||||
nodes.push(internal_node(
|
||||
d,
|
||||
node_size,
|
||||
record_size,
|
||||
fanout - 1,
|
||||
fanout,
|
||||
child,
|
||||
child_nrec,
|
||||
));
|
||||
}
|
||||
let mut data: Vec<u8> = nodes.concat();
|
||||
data.extend_from_slice(&leaf);
|
||||
data.resize(data.len() + 64, 0);
|
||||
|
||||
let started = std::time::Instant::now();
|
||||
let result =
|
||||
collect_btree_v2_records(&data, &header(depth, 0, fanout as u16 - 1, u64::MAX), 8, 8);
|
||||
assert!(
|
||||
result.is_err(),
|
||||
"expected a refusal, got {} records",
|
||||
result.map_or(0, |r| r.len())
|
||||
);
|
||||
assert!(
|
||||
started.elapsed() < std::time::Duration::from_secs(2),
|
||||
"took {:?}",
|
||||
started.elapsed()
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_header() {
|
||||
let data = build_btree_v2_header(5, 512, 11, 0, 0x1000, 3, 3, 8, 8);
|
||||
@@ -522,4 +683,18 @@ mod tests {
|
||||
let records = collect_btree_v2_records(&header, &hdr, 8, 8).unwrap();
|
||||
assert!(records.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn subtree_capacity_matches_libhdf5() {
|
||||
// A link-name index (11-byte records, 512-byte nodes, 8-byte
|
||||
// addresses): libhdf5's H5B2__hdr_init gives 45 records per leaf,
|
||||
// then cum_max_nrec 1 149 at depth 1 and 26 449 at depth 2 — two
|
||||
// bytes of subtree count in a depth-3 root's child pointers, where
|
||||
// leaf_max^3 = 91 125 would need three.
|
||||
let leaf = max_records_leaf(512, 11);
|
||||
assert_eq!(leaf, 45);
|
||||
assert_eq!(cum_max_records(512, 11, 8, leaf, 0), 45);
|
||||
assert_eq!(cum_max_records(512, 11, 8, leaf, 1), 1_149);
|
||||
assert_eq!(cum_max_records(512, 11, 8, leaf, 2), 26_449);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,79 @@
|
||||
//! Large output buffers backed by transparent huge pages where the OS offers
|
||||
//! them.
|
||||
//!
|
||||
//! A fresh multi-megabyte `Vec` is mapped lazily by the kernel: the first
|
||||
//! write to each 4 KiB page takes a page fault, and the kernel zeroes the page
|
||||
//! before handing it over. For a 64 MiB read that is 16384 faults, and they
|
||||
//! cost far more than the copy that fills the buffer — single-threaded
|
||||
//! contiguous reads ran at about a quarter of h5py's speed because of them.
|
||||
//! numpy (so h5py) avoids this by asking for transparent huge pages
|
||||
//! (`madvise(MADV_HUGEPAGE)`) on every allocation of 4 MiB or more, which
|
||||
//! turns 512 faults into one; this module does the same.
|
||||
//!
|
||||
//! The advice only changes how the pages are backed, never their contents, so
|
||||
//! it is harmless when it cannot be honoured (THP disabled, not Linux, a
|
||||
//! region that is part of the heap): the buffer is then exactly what it would
|
||||
//! have been without it.
|
||||
|
||||
#[cfg(not(feature = "std"))]
|
||||
use alloc::vec::Vec;
|
||||
|
||||
/// Buffers smaller than this are left alone (numpy uses the same threshold).
|
||||
#[cfg(any(target_os = "linux", test))]
|
||||
pub(crate) const HUGE_PAGE_THRESHOLD: usize = 4 << 20;
|
||||
|
||||
/// Advise the kernel to back `[ptr, ptr + len)` with transparent huge pages,
|
||||
/// when `len` is large enough to benefit. Call it before the first write so
|
||||
/// the faults happen at huge-page granularity.
|
||||
#[inline]
|
||||
pub(crate) fn advise_huge_pages(ptr: *const u8, len: usize) {
|
||||
#[cfg(target_os = "linux")]
|
||||
if len >= HUGE_PAGE_THRESHOLD {
|
||||
const PAGE: usize = 4096;
|
||||
let start = (ptr as usize).next_multiple_of(PAGE);
|
||||
let end = (ptr as usize + len) & !(PAGE - 1);
|
||||
if end > start {
|
||||
// SAFETY: `[start, end)` lies inside an allocation of `len` bytes
|
||||
// at `ptr` that the caller owns, and is page aligned as madvise
|
||||
// requires. MADV_HUGEPAGE does not change the memory's contents or
|
||||
// validity; on failure (EINVAL when THP is compiled out, etc.) the
|
||||
// region is simply left as it was, so the result is ignored.
|
||||
unsafe {
|
||||
libc::madvise(start as *mut libc::c_void, end - start, libc::MADV_HUGEPAGE);
|
||||
}
|
||||
}
|
||||
}
|
||||
#[cfg(not(target_os = "linux"))]
|
||||
let _ = (ptr, len);
|
||||
}
|
||||
|
||||
/// `Vec::with_capacity(count)` for a buffer about to be filled in bulk, with
|
||||
/// huge-page advice when it is large (see the module docs).
|
||||
#[inline]
|
||||
pub(crate) fn vec_for_bulk<T>(count: usize) -> Vec<T> {
|
||||
let v: Vec<T> = Vec::with_capacity(count);
|
||||
advise_huge_pages(
|
||||
v.as_ptr().cast::<u8>(),
|
||||
v.capacity().saturating_mul(core::mem::size_of::<T>()),
|
||||
);
|
||||
v
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn bulk_vec_is_an_ordinary_vec() {
|
||||
for count in [0usize, 1, 1000, HUGE_PAGE_THRESHOLD / 4 + 3] {
|
||||
let mut v: Vec<u32> = vec_for_bulk(count);
|
||||
assert!(v.capacity() >= count);
|
||||
v.extend((0..count as u32).map(|i| i.wrapping_mul(2654435761)));
|
||||
assert!(
|
||||
v.iter()
|
||||
.enumerate()
|
||||
.all(|(i, &x)| x == (i as u32).wrapping_mul(2654435761))
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -223,13 +223,32 @@ pub const DEFAULT_CACHE_BYTES: usize = 16 * 1024 * 1024; // 16 MiB
|
||||
/// coordinate map and reduces collision chains compared to power-of-two sizes.
|
||||
pub const DEFAULT_MAX_SLOTS: usize = 521;
|
||||
|
||||
/// Most datasets whose chunk index a [`ChunkCache`] keeps at once.
|
||||
pub const MAX_INDEXED_DATASETS: usize = 64;
|
||||
|
||||
/// Most chunk-index entries, summed over all datasets, a [`ChunkCache`] keeps.
|
||||
/// Least-recently-used datasets' indexes are dropped past this (the dataset
|
||||
/// being read is always kept), so a file with many or huge chunked datasets
|
||||
/// cannot grow the cache without bound.
|
||||
pub const MAX_INDEXED_CHUNKS: usize = 1 << 20;
|
||||
|
||||
/// The dataset key the address-less (legacy) methods use when
|
||||
/// [`ChunkCache::ensure_dataset`] has not been called.
|
||||
#[cfg(feature = "std")]
|
||||
const UNBOUND_DATASET: u64 = u64::MAX;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// LRU entry
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Decompressed chunks are keyed by dataset *and* coordinate: every chunked
|
||||
/// dataset has a chunk at (0, 0, ...), so the coordinate alone is ambiguous.
|
||||
#[cfg(feature = "std")]
|
||||
type SlotKey = (u64, ChunkCoord);
|
||||
|
||||
#[cfg(feature = "std")]
|
||||
struct CachedChunk {
|
||||
coord: ChunkCoord,
|
||||
key: SlotKey,
|
||||
/// Shared so a cache hit is a refcount bump, not a copy of the whole
|
||||
/// (potentially large) decompressed chunk.
|
||||
data: Arc<CacheAlignedBuffer>,
|
||||
@@ -237,21 +256,48 @@ struct CachedChunk {
|
||||
last_access: u64,
|
||||
}
|
||||
|
||||
/// Per-dataset index state.
|
||||
#[cfg(feature = "std")]
|
||||
#[derive(Default)]
|
||||
struct DatasetEntry {
|
||||
/// Chunk coordinate -> ChunkInfo (offset + size in file).
|
||||
index: Option<Arc<HashMap<ChunkCoord, ChunkInfo>>>,
|
||||
/// Pre-built chunk index for O(1) coordinate lookups.
|
||||
chunk_index: Option<Arc<ChunkIndex>>,
|
||||
/// Pre-computed chunk layout for fast assembly.
|
||||
chunk_layout: Option<Arc<ChunkLayout>>,
|
||||
/// Tick of the last use, for dropping the least recently used dataset.
|
||||
last_used: u64,
|
||||
}
|
||||
|
||||
#[cfg(feature = "std")]
|
||||
impl DatasetEntry {
|
||||
fn weight(&self) -> usize {
|
||||
self.index.as_ref().map_or(0, |m| m.len())
|
||||
+ self.chunk_index.as_ref().map_or(0, |c| c.num_chunks())
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// ChunkCache
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// A per-dataset chunk cache with hash-based index and LRU eviction.
|
||||
/// A per-file chunk cache: chunk indexes per dataset, plus an LRU of
|
||||
/// decompressed chunks, all keyed by dataset.
|
||||
///
|
||||
/// # Usage
|
||||
/// A dataset is identified by the address of its chunk index (B-tree, fixed
|
||||
/// or extensible array, ...), which is unique within a file. Every method
|
||||
/// that takes an `addr` works on that dataset only, so threads reading
|
||||
/// different datasets through one shared cache never see each other's
|
||||
/// chunks. The address-less methods (`has_index`, `populate_index`,
|
||||
/// `get_decompressed`, ...) act on the dataset last bound with
|
||||
/// [`Self::ensure_dataset`]; that binding is shared state, so concurrent
|
||||
/// readers must use the `*_in` / `*_for` methods instead (the chunked
|
||||
/// readers in [`crate::chunked_read`] do).
|
||||
///
|
||||
/// ```ignore
|
||||
/// let cache = ChunkCache::new();
|
||||
/// // Pass &cache to read_chunked_data — it will populate the index lazily.
|
||||
/// ```
|
||||
///
|
||||
/// The cache is wrapped in `Mutex` internally so it can be mutated through
|
||||
/// shared references (thread-safe).
|
||||
/// Memory is bounded: decompressed data by `max_bytes`/`max_slots` across
|
||||
/// all datasets, indexes by [`MAX_INDEXED_DATASETS`] and
|
||||
/// [`MAX_INDEXED_CHUNKS`].
|
||||
///
|
||||
/// Only available with the `std` feature because it requires `std::sync::Mutex`.
|
||||
#[cfg(feature = "std")]
|
||||
@@ -261,26 +307,20 @@ pub struct ChunkCache {
|
||||
|
||||
#[cfg(feature = "std")]
|
||||
struct CacheInner {
|
||||
/// Hash index: chunk coordinate -> ChunkInfo (offset + size in file).
|
||||
/// Populated once per dataset on first access.
|
||||
index: Option<HashMap<ChunkCoord, ChunkInfo>>,
|
||||
/// Per-dataset chunk indexes, keyed by chunk-index address.
|
||||
datasets: HashMap<u64, DatasetEntry>,
|
||||
|
||||
/// Address of the dataset (its chunk-index base address) that the cached
|
||||
/// index, chunk index, layout, and decompressed slots currently belong to.
|
||||
/// The cache is shared per file across datasets, so every cached-read entry
|
||||
/// checks this and resets the per-dataset state when the dataset changes —
|
||||
/// otherwise one dataset's chunk index (with its own rank) would be reused
|
||||
/// for another, corrupting reads.
|
||||
index_addr: Option<u64>,
|
||||
/// Dataset the address-less methods act on (see `ensure_dataset`).
|
||||
current: Option<u64>,
|
||||
|
||||
/// LRU cache of decompressed chunk data.
|
||||
slots: Vec<CachedChunk>,
|
||||
|
||||
/// Coordinate -> index into `slots`, for O(1) lookup instead of a linear
|
||||
/// Key -> index into `slots`, for O(1) lookup instead of a linear
|
||||
/// scan. Kept in sync with `slots` on every insert/evict/clear — in
|
||||
/// particular, `slots.swap_remove(i)` moves the last element into slot
|
||||
/// `i`, so the moved element's index entry must be updated too.
|
||||
slot_index: HashMap<ChunkCoord, usize>,
|
||||
slot_index: HashMap<SlotKey, usize>,
|
||||
|
||||
/// Current total bytes of cached decompressed data.
|
||||
current_bytes: usize,
|
||||
@@ -294,17 +334,145 @@ struct CacheInner {
|
||||
/// Monotonic counter for LRU ordering.
|
||||
tick: u64,
|
||||
|
||||
/// Last accessed chunk coordinate (for sequential detection).
|
||||
last_coord: Option<ChunkCoord>,
|
||||
/// Last accessed chunk (for sequential detection).
|
||||
last_coord: Option<SlotKey>,
|
||||
|
||||
/// Access pattern statistics.
|
||||
stats: AccessStats,
|
||||
}
|
||||
|
||||
/// Pre-built chunk index for O(1) coordinate lookups.
|
||||
chunk_index: Option<ChunkIndex>,
|
||||
#[cfg(feature = "std")]
|
||||
impl CacheInner {
|
||||
fn current(&self) -> u64 {
|
||||
self.current.unwrap_or(UNBOUND_DATASET)
|
||||
}
|
||||
|
||||
/// Pre-computed chunk layout for fast assembly.
|
||||
chunk_layout: Option<ChunkLayout>,
|
||||
fn touch(&mut self, addr: u64) -> &mut DatasetEntry {
|
||||
self.tick += 1;
|
||||
let tick = self.tick;
|
||||
let entry = self.datasets.entry(addr).or_default();
|
||||
entry.last_used = tick;
|
||||
entry
|
||||
}
|
||||
|
||||
fn entry(&self, addr: u64) -> Option<&DatasetEntry> {
|
||||
self.datasets.get(&addr)
|
||||
}
|
||||
|
||||
/// Drop least-recently-used datasets' indexes (never `keep`'s) until the
|
||||
/// dataset and chunk-entry budgets hold.
|
||||
fn trim_datasets(&mut self, keep: u64) {
|
||||
loop {
|
||||
let total: usize = self.datasets.values().map(DatasetEntry::weight).sum();
|
||||
if self.datasets.len() <= MAX_INDEXED_DATASETS && total <= MAX_INDEXED_CHUNKS {
|
||||
return;
|
||||
}
|
||||
let victim = self
|
||||
.datasets
|
||||
.iter()
|
||||
.filter(|(a, _)| **a != keep)
|
||||
.min_by_key(|(_, e)| e.last_used)
|
||||
.map(|(a, _)| *a);
|
||||
match victim {
|
||||
Some(a) => {
|
||||
self.datasets.remove(&a);
|
||||
}
|
||||
None => return,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn get_decompressed(&mut self, addr: u64, coord: &[u64]) -> Option<Arc<CacheAlignedBuffer>> {
|
||||
self.tick += 1;
|
||||
let tick = self.tick;
|
||||
|
||||
// Track sequential vs random access
|
||||
let is_sequential = self.last_coord.as_ref().is_some_and(|(prev_addr, prev)| {
|
||||
// Sequential if exactly one dimension changed
|
||||
let changes: usize = prev
|
||||
.iter()
|
||||
.zip(coord.iter())
|
||||
.filter(|(a, b)| a != b)
|
||||
.count();
|
||||
*prev_addr == addr && changes <= 1
|
||||
});
|
||||
if is_sequential {
|
||||
self.stats.sequential_count += 1;
|
||||
} else if self.last_coord.is_some() {
|
||||
self.stats.random_count += 1;
|
||||
}
|
||||
let key: SlotKey = (addr, coord.to_vec());
|
||||
let found = if let Some(&idx) = self.slot_index.get(&key) {
|
||||
self.slots[idx].last_access = tick;
|
||||
Some(Arc::clone(&self.slots[idx].data))
|
||||
} else {
|
||||
None
|
||||
};
|
||||
self.last_coord = Some(key);
|
||||
if let Some(ref data) = found {
|
||||
self.stats.hits += 1;
|
||||
self.stats.bytes_read += data.len() as u64;
|
||||
} else {
|
||||
self.stats.misses += 1;
|
||||
}
|
||||
found
|
||||
}
|
||||
|
||||
fn put_decompressed(
|
||||
&mut self,
|
||||
key: SlotKey,
|
||||
data: Arc<CacheAlignedBuffer>,
|
||||
) -> Arc<CacheAlignedBuffer> {
|
||||
let data_len = data.len();
|
||||
|
||||
// Don't cache if single chunk exceeds budget — still return the data
|
||||
// to the caller, just don't retain it.
|
||||
if data_len > self.max_bytes {
|
||||
return data;
|
||||
}
|
||||
|
||||
// Check if already present
|
||||
self.tick += 1;
|
||||
let tick = self.tick;
|
||||
if let Some(&idx) = self.slot_index.get(&key) {
|
||||
self.slots[idx].last_access = tick;
|
||||
return Arc::clone(&self.slots[idx].data); // already cached
|
||||
}
|
||||
|
||||
// Evict until we have room
|
||||
while self.slots.len() >= self.max_slots
|
||||
|| (self.current_bytes + data_len > self.max_bytes && !self.slots.is_empty())
|
||||
{
|
||||
// Find LRU slot
|
||||
let lru_idx = self
|
||||
.slots
|
||||
.iter()
|
||||
.enumerate()
|
||||
.min_by_key(|(_, s)| s.last_access)
|
||||
.map(|(i, _)| i)
|
||||
.unwrap();
|
||||
let removed = self.slots.swap_remove(lru_idx);
|
||||
self.slot_index.remove(&removed.key);
|
||||
// swap_remove moved the former last element into `lru_idx` (unless
|
||||
// it *was* the last element) — fix up that element's index entry.
|
||||
if lru_idx < self.slots.len() {
|
||||
let moved_key = self.slots[lru_idx].key.clone();
|
||||
self.slot_index.insert(moved_key, lru_idx);
|
||||
}
|
||||
self.current_bytes -= removed.data.len();
|
||||
self.stats.evictions += 1;
|
||||
}
|
||||
|
||||
self.current_bytes += data_len;
|
||||
let new_idx = self.slots.len();
|
||||
self.slot_index.insert(key.clone(), new_idx);
|
||||
self.slots.push(CachedChunk {
|
||||
key,
|
||||
data: Arc::clone(&data),
|
||||
last_access: tick,
|
||||
});
|
||||
data
|
||||
}
|
||||
}
|
||||
|
||||
/// Access pattern statistics tracked by the chunk cache.
|
||||
@@ -356,8 +524,8 @@ impl ChunkCache {
|
||||
pub fn with_capacity(max_bytes: usize, max_slots: usize) -> Self {
|
||||
Self {
|
||||
inner: std::sync::Mutex::new(CacheInner {
|
||||
index: None,
|
||||
index_addr: None,
|
||||
datasets: HashMap::new(),
|
||||
current: None,
|
||||
slots: Vec::with_capacity(max_slots.min(64)),
|
||||
slot_index: HashMap::with_capacity(max_slots.min(64)),
|
||||
current_bytes: 0,
|
||||
@@ -366,340 +534,331 @@ impl ChunkCache {
|
||||
tick: 0,
|
||||
last_coord: None,
|
||||
stats: AccessStats::default(),
|
||||
chunk_index: None,
|
||||
chunk_layout: None,
|
||||
}),
|
||||
}
|
||||
}
|
||||
|
||||
// ----- Index operations -----
|
||||
fn lock(&self) -> std::sync::MutexGuard<'_, CacheInner> {
|
||||
self.inner.lock().unwrap_or_else(|e| e.into_inner())
|
||||
}
|
||||
|
||||
/// The most decompressed bytes this cache will hold.
|
||||
pub fn max_bytes(&self) -> usize {
|
||||
self.inner.lock().map(|g| g.max_bytes).unwrap_or(0)
|
||||
self.lock().max_bytes
|
||||
}
|
||||
|
||||
/// Bind the cache to the dataset at chunk-index address `addr`.
|
||||
// ----- Dataset-keyed operations (safe to use concurrently) -----
|
||||
|
||||
/// The chunk list of the dataset whose chunk index is at `addr`.
|
||||
///
|
||||
/// The cache is shared per file across all of its datasets. If the cache
|
||||
/// currently holds state for a different dataset, all per-dataset state
|
||||
/// (chunk index, chunk-index map, layout, and decompressed slots) is
|
||||
/// dropped so the next access rebuilds it for this dataset. Reading the
|
||||
/// same dataset again is a no-op, preserving the cache's benefit for
|
||||
/// repeated/sequential access. Returns `true` if a reset occurred.
|
||||
/// On the first call for a dataset, `build` scans its chunk index; the
|
||||
/// result is kept (offsets truncated to `rank` for the lookup key), so
|
||||
/// later calls skip the scan. `build` runs without the cache lock held;
|
||||
/// if two threads race to build the same dataset's index, the first
|
||||
/// stored one wins and both return equivalent lists.
|
||||
pub fn chunks_for<E>(
|
||||
&self,
|
||||
addr: u64,
|
||||
rank: usize,
|
||||
build: impl FnOnce() -> Result<Vec<ChunkInfo>, E>,
|
||||
) -> Result<Vec<ChunkInfo>, E> {
|
||||
Ok(self
|
||||
.index_for(addr, rank, build)?
|
||||
.values()
|
||||
.cloned()
|
||||
.collect())
|
||||
}
|
||||
|
||||
fn index_for<E>(
|
||||
&self,
|
||||
addr: u64,
|
||||
rank: usize,
|
||||
build: impl FnOnce() -> Result<Vec<ChunkInfo>, E>,
|
||||
) -> Result<Arc<HashMap<ChunkCoord, ChunkInfo>>, E> {
|
||||
if let Some(index) = self.lock().touch(addr).index.clone() {
|
||||
return Ok(index);
|
||||
}
|
||||
let chunks = build()?;
|
||||
let map: HashMap<ChunkCoord, ChunkInfo> = chunks
|
||||
.into_iter()
|
||||
.map(|ci| (ci.offsets.iter().take(rank).copied().collect(), ci))
|
||||
.collect();
|
||||
let mut inner = self.lock();
|
||||
let entry = inner.touch(addr);
|
||||
let index = Arc::clone(entry.index.get_or_insert_with(|| Arc::new(map)));
|
||||
inner.trim_datasets(addr);
|
||||
Ok(index)
|
||||
}
|
||||
|
||||
/// The pre-computed assembly layout of the dataset at `addr`, building
|
||||
/// its chunk index (via `build`, as in [`Self::chunks_for`]) and layout on
|
||||
/// first use.
|
||||
pub fn chunk_layout_for<E>(
|
||||
&self,
|
||||
addr: u64,
|
||||
rank: usize,
|
||||
build: impl FnOnce() -> Result<Vec<ChunkInfo>, E>,
|
||||
ds_dims: &[usize],
|
||||
chunk_dims: &[usize],
|
||||
elem_size: usize,
|
||||
) -> Result<Arc<ChunkLayout>, E> {
|
||||
let (layout, chunk_index) = {
|
||||
let mut inner = self.lock();
|
||||
let entry = inner.touch(addr);
|
||||
(entry.chunk_layout.clone(), entry.chunk_index.clone())
|
||||
};
|
||||
if let Some(layout) = layout {
|
||||
return Ok(layout);
|
||||
}
|
||||
let chunk_index = match chunk_index {
|
||||
Some(ci) => ci,
|
||||
None => {
|
||||
let index = self.index_for(addr, rank, build)?;
|
||||
let chunks: Vec<ChunkInfo> = index.values().cloned().collect();
|
||||
Arc::new(ChunkIndex::build(&chunks, rank))
|
||||
}
|
||||
};
|
||||
let layout = ChunkLayout::build(&chunk_index, ds_dims, chunk_dims, elem_size);
|
||||
let mut inner = self.lock();
|
||||
let entry = inner.touch(addr);
|
||||
entry.chunk_index.get_or_insert(chunk_index);
|
||||
let layout = Arc::clone(entry.chunk_layout.get_or_insert_with(|| Arc::new(layout)));
|
||||
inner.trim_datasets(addr);
|
||||
Ok(layout)
|
||||
}
|
||||
|
||||
/// Cached decompressed chunk at `coord` of the dataset at `addr`.
|
||||
///
|
||||
/// O(1) lookup; the clone is an `Arc` refcount bump, not a copy of the
|
||||
/// underlying decompressed data.
|
||||
pub fn get_decompressed_in(&self, addr: u64, coord: &[u64]) -> Option<Arc<CacheAlignedBuffer>> {
|
||||
self.lock().get_decompressed(addr, coord)
|
||||
}
|
||||
|
||||
/// Cache decompressed chunk data for `coord` of the dataset at `addr`.
|
||||
/// Returns the `Arc`-shared buffer now cached (or already cached).
|
||||
pub fn put_decompressed_in(
|
||||
&self,
|
||||
addr: u64,
|
||||
coord: ChunkCoord,
|
||||
data: Vec<u8>,
|
||||
) -> Arc<CacheAlignedBuffer> {
|
||||
self.put_decompressed_aligned_in(addr, coord, CacheAlignedBuffer::from_vec(data))
|
||||
}
|
||||
|
||||
/// [`Self::put_decompressed_in`] for an already-aligned buffer.
|
||||
pub fn put_decompressed_aligned_in(
|
||||
&self,
|
||||
addr: u64,
|
||||
coord: ChunkCoord,
|
||||
data: CacheAlignedBuffer,
|
||||
) -> Arc<CacheAlignedBuffer> {
|
||||
let data = Arc::new(data);
|
||||
self.lock().put_decompressed((addr, coord), data)
|
||||
}
|
||||
|
||||
/// Record that the given chunk coordinates of the dataset at `addr` are
|
||||
/// predicted to be accessed soon (bookkeeping only).
|
||||
///
|
||||
/// This does **not** prefetch or pre-decompress anything — it only
|
||||
/// checks whether each coordinate is already in the chunk index and
|
||||
/// updates access-pattern stats accordingly.
|
||||
pub fn prefetch_hint_in(&self, addr: u64, next_coords: &[ChunkCoord]) {
|
||||
let mut inner = self.lock();
|
||||
let Some(index) = inner.entry(addr).and_then(|e| e.index.clone()) else {
|
||||
return;
|
||||
};
|
||||
let known = next_coords
|
||||
.iter()
|
||||
.filter(|c| index.contains_key(*c))
|
||||
.count();
|
||||
inner.stats.sequential_count += known as u64;
|
||||
}
|
||||
|
||||
// ----- Address-less operations on the bound dataset -----
|
||||
|
||||
/// Bind the address-less methods to the dataset at chunk-index address
|
||||
/// `addr`. Returns `true` if this changed the bound dataset.
|
||||
///
|
||||
/// Each dataset's state is kept separately, so switching loses nothing
|
||||
/// and never exposes one dataset's index or chunks to another. The
|
||||
/// binding itself is shared, though: concurrent readers should use the
|
||||
/// `addr`-taking methods rather than bind and then call these.
|
||||
pub fn ensure_dataset(&self, addr: u64) -> bool {
|
||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
||||
if inner.index_addr == Some(addr) {
|
||||
return false;
|
||||
}
|
||||
inner.index = None;
|
||||
inner.chunk_index = None;
|
||||
inner.chunk_layout = None;
|
||||
inner.slots.clear();
|
||||
inner.slot_index.clear();
|
||||
inner.current_bytes = 0;
|
||||
inner.last_coord = None;
|
||||
inner.index_addr = Some(addr);
|
||||
true
|
||||
let mut inner = self.lock();
|
||||
let changed = inner.current != Some(addr);
|
||||
inner.current = Some(addr);
|
||||
changed
|
||||
}
|
||||
|
||||
/// Returns `true` if the chunk index has been built.
|
||||
/// Returns `true` if the bound dataset's chunk index has been built.
|
||||
pub fn has_index(&self) -> bool {
|
||||
self.inner
|
||||
.lock()
|
||||
.unwrap_or_else(|e| e.into_inner())
|
||||
.index
|
||||
.is_some()
|
||||
let inner = self.lock();
|
||||
inner
|
||||
.entry(inner.current())
|
||||
.is_some_and(|e| e.index.is_some())
|
||||
}
|
||||
|
||||
/// Build the chunk index from a pre-collected list of `ChunkInfo`.
|
||||
/// Build the bound dataset's chunk index from a pre-collected list of
|
||||
/// `ChunkInfo`.
|
||||
///
|
||||
/// The `rank` parameter is used to truncate offsets to spatial dims only
|
||||
/// (B-tree v1 stores rank+1 offsets).
|
||||
pub fn populate_index(&self, chunks: &[ChunkInfo], rank: usize) {
|
||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
||||
if inner.index.is_some() {
|
||||
return; // already populated
|
||||
}
|
||||
let mut map = HashMap::with_capacity(chunks.len());
|
||||
|
||||
for ci in chunks {
|
||||
let coord: ChunkCoord = ci.offsets.iter().take(rank).copied().collect();
|
||||
map.insert(coord, ci.clone());
|
||||
}
|
||||
inner.index = Some(map);
|
||||
let addr = self.lock().current();
|
||||
let _ = self.index_for::<core::convert::Infallible>(addr, rank, || Ok(chunks.to_vec()));
|
||||
}
|
||||
|
||||
/// Look up a chunk by its spatial coordinate in the index.
|
||||
/// Look up a chunk by its spatial coordinate in the bound dataset's index.
|
||||
pub fn lookup_index(&self, coord: &[u64]) -> Option<ChunkInfo> {
|
||||
let inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
||||
inner.index.as_ref()?.get(coord).cloned()
|
||||
let inner = self.lock();
|
||||
inner
|
||||
.entry(inner.current())?
|
||||
.index
|
||||
.as_ref()?
|
||||
.get(coord)
|
||||
.cloned()
|
||||
}
|
||||
|
||||
/// Return all indexed chunks as a `Vec<ChunkInfo>` (order unspecified).
|
||||
/// Return all of the bound dataset's indexed chunks (order unspecified).
|
||||
pub fn all_indexed_chunks(&self) -> Option<Vec<ChunkInfo>> {
|
||||
let inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
||||
inner.index.as_ref().map(|m| m.values().cloned().collect())
|
||||
let inner = self.lock();
|
||||
let index = inner.entry(inner.current())?.index.as_ref()?;
|
||||
Some(index.values().cloned().collect())
|
||||
}
|
||||
|
||||
// ----- Chunk index (pre-built coordinate → ChunkInfo map) -----
|
||||
|
||||
/// Returns `true` if the chunk B-tree index has been built.
|
||||
/// Returns `true` if the bound dataset's `ChunkIndex` has been built.
|
||||
pub fn has_chunk_index(&self) -> bool {
|
||||
self.inner
|
||||
.lock()
|
||||
.unwrap_or_else(|e| e.into_inner())
|
||||
.chunk_index
|
||||
.is_some()
|
||||
let inner = self.lock();
|
||||
inner
|
||||
.entry(inner.current())
|
||||
.is_some_and(|e| e.chunk_index.is_some())
|
||||
}
|
||||
|
||||
/// Build and store the chunk B-tree index from a pre-collected list of `ChunkInfo`.
|
||||
/// Build and store the bound dataset's `ChunkIndex`.
|
||||
pub fn populate_chunk_index(&self, chunks: &[ChunkInfo], rank: usize) {
|
||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
||||
if inner.chunk_index.is_some() {
|
||||
return;
|
||||
}
|
||||
inner.chunk_index = Some(ChunkIndex::build(chunks, rank));
|
||||
let built = Arc::new(ChunkIndex::build(chunks, rank));
|
||||
let mut inner = self.lock();
|
||||
let addr = inner.current();
|
||||
inner.touch(addr).chunk_index.get_or_insert(built);
|
||||
inner.trim_datasets(addr);
|
||||
}
|
||||
|
||||
// ----- Chunk layout (pre-computed assembly plan) -----
|
||||
|
||||
/// Returns `true` if the chunk layout has been computed.
|
||||
/// Returns `true` if the bound dataset's chunk layout has been computed.
|
||||
pub fn has_chunk_layout(&self) -> bool {
|
||||
self.inner
|
||||
.lock()
|
||||
.unwrap_or_else(|e| e.into_inner())
|
||||
.chunk_layout
|
||||
.is_some()
|
||||
let inner = self.lock();
|
||||
inner
|
||||
.entry(inner.current())
|
||||
.is_some_and(|e| e.chunk_layout.is_some())
|
||||
}
|
||||
|
||||
/// Build and store the pre-computed chunk layout for fast assembly.
|
||||
/// Build and store the bound dataset's chunk layout (needs its
|
||||
/// `ChunkIndex`; does nothing without one).
|
||||
pub fn populate_chunk_layout(&self, ds_dims: &[usize], chunk_dims: &[usize], elem_size: usize) {
|
||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
||||
if inner.chunk_layout.is_some() {
|
||||
let mut inner = self.lock();
|
||||
let addr = inner.current();
|
||||
let entry = inner.touch(addr);
|
||||
if entry.chunk_layout.is_some() {
|
||||
return;
|
||||
}
|
||||
if let Some(ref idx) = inner.chunk_index {
|
||||
inner.chunk_layout = Some(ChunkLayout::build(idx, ds_dims, chunk_dims, elem_size));
|
||||
if let Some(idx) = entry.chunk_index.clone() {
|
||||
entry.chunk_layout = Some(Arc::new(ChunkLayout::build(
|
||||
&idx, ds_dims, chunk_dims, elem_size,
|
||||
)));
|
||||
}
|
||||
}
|
||||
|
||||
/// Execute a function with a reference to the chunk layout.
|
||||
///
|
||||
/// Returns `None` if the layout hasn't been computed yet.
|
||||
/// Execute a function with a reference to the bound dataset's chunk
|
||||
/// layout. Returns `None` if the layout hasn't been computed yet.
|
||||
pub fn with_chunk_layout<F, R>(&self, f: F) -> Option<R>
|
||||
where
|
||||
F: FnOnce(&ChunkLayout) -> R,
|
||||
{
|
||||
let inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
||||
inner.chunk_layout.as_ref().map(f)
|
||||
let layout = {
|
||||
let inner = self.lock();
|
||||
inner.entry(inner.current())?.chunk_layout.clone()?
|
||||
};
|
||||
Some(f(&layout))
|
||||
}
|
||||
|
||||
// ----- Decompressed data cache (LRU) -----
|
||||
|
||||
/// Try to get cached decompressed data for a chunk coordinate.
|
||||
/// Try to get cached decompressed data for a chunk of the bound dataset.
|
||||
///
|
||||
/// O(1) lookup. Returns an owned copy for API compatibility with callers
|
||||
/// that need a `Vec<u8>`; prefer [`Self::get_decompressed_aligned`] when
|
||||
/// an `Arc`-shared buffer works for the caller, since that avoids the
|
||||
/// copy entirely.
|
||||
/// Returns an owned copy; prefer [`Self::get_decompressed_aligned`] when
|
||||
/// an `Arc`-shared buffer works for the caller.
|
||||
pub fn get_decompressed(&self, coord: &[u64]) -> Option<Vec<u8>> {
|
||||
self.get_decompressed_aligned(coord)
|
||||
.map(|arc| arc.as_slice().to_vec())
|
||||
}
|
||||
|
||||
/// Try to get a reference-counted clone of the aligned buffer for a chunk.
|
||||
///
|
||||
/// O(1) index lookup; the clone is an `Arc` refcount bump, not a copy of
|
||||
/// the underlying decompressed data.
|
||||
/// Reference-counted cached buffer for a chunk of the bound dataset.
|
||||
pub fn get_decompressed_aligned(&self, coord: &[u64]) -> Option<Arc<CacheAlignedBuffer>> {
|
||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
||||
inner.tick += 1;
|
||||
let tick = inner.tick;
|
||||
|
||||
// Track sequential vs random access
|
||||
let is_sequential = inner.last_coord.as_ref().is_some_and(|prev| {
|
||||
// Sequential if exactly one dimension changed
|
||||
let changes: usize = prev
|
||||
.iter()
|
||||
.zip(coord.iter())
|
||||
.filter(|(a, b)| a != b)
|
||||
.count();
|
||||
changes <= 1
|
||||
});
|
||||
if is_sequential {
|
||||
inner.stats.sequential_count += 1;
|
||||
} else if inner.last_coord.is_some() {
|
||||
inner.stats.random_count += 1;
|
||||
}
|
||||
inner.last_coord = Some(coord.to_vec());
|
||||
|
||||
let found = if let Some(&idx) = inner.slot_index.get(coord) {
|
||||
inner.slots[idx].last_access = tick;
|
||||
Some(Arc::clone(&inner.slots[idx].data))
|
||||
} else {
|
||||
None
|
||||
};
|
||||
if let Some(ref data) = found {
|
||||
inner.stats.hits += 1;
|
||||
inner.stats.bytes_read += data.len() as u64;
|
||||
} else {
|
||||
inner.stats.misses += 1;
|
||||
}
|
||||
found
|
||||
let mut inner = self.lock();
|
||||
let addr = inner.current();
|
||||
inner.get_decompressed(addr, coord)
|
||||
}
|
||||
|
||||
/// Insert decompressed chunk data into the LRU cache.
|
||||
///
|
||||
/// The data is stored in a [`CacheAlignedBuffer`] so subsequent reads
|
||||
/// return cache-line-aligned memory. Returns the `Arc`-shared buffer that
|
||||
/// is now cached (or already was), so the caller can reuse it directly
|
||||
/// instead of holding a separate copy of the same data.
|
||||
/// Insert decompressed chunk data for the bound dataset into the LRU
|
||||
/// cache, returning the `Arc`-shared buffer now cached.
|
||||
pub fn put_decompressed(&self, coord: ChunkCoord, data: Vec<u8>) -> Arc<CacheAlignedBuffer> {
|
||||
let aligned = CacheAlignedBuffer::from_vec(data);
|
||||
self.put_decompressed_aligned(coord, aligned)
|
||||
self.put_decompressed_aligned(coord, CacheAlignedBuffer::from_vec(data))
|
||||
}
|
||||
|
||||
/// Insert an already-aligned buffer into the LRU cache.
|
||||
///
|
||||
/// Returns the `Arc`-shared buffer now held by the cache (the one just
|
||||
/// inserted, or the existing cached copy if `coord` was already present).
|
||||
/// Insert an already-aligned buffer for the bound dataset.
|
||||
pub fn put_decompressed_aligned(
|
||||
&self,
|
||||
coord: ChunkCoord,
|
||||
data: CacheAlignedBuffer,
|
||||
) -> Arc<CacheAlignedBuffer> {
|
||||
let data = Arc::new(data);
|
||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
||||
let data_len = data.len();
|
||||
|
||||
// Don't cache if single chunk exceeds budget — still return the data
|
||||
// to the caller, just don't retain it.
|
||||
if data_len > inner.max_bytes {
|
||||
return data;
|
||||
let mut inner = self.lock();
|
||||
let addr = inner.current();
|
||||
inner.put_decompressed((addr, coord), data)
|
||||
}
|
||||
|
||||
// Check if already present
|
||||
inner.tick += 1;
|
||||
let tick = inner.tick;
|
||||
if let Some(&idx) = inner.slot_index.get(&coord) {
|
||||
inner.slots[idx].last_access = tick;
|
||||
return Arc::clone(&inner.slots[idx].data); // already cached
|
||||
/// [`Self::prefetch_hint_in`] for the bound dataset.
|
||||
pub fn prefetch_hint(&self, next_coords: &[ChunkCoord]) {
|
||||
let addr = self.lock().current();
|
||||
self.prefetch_hint_in(addr, next_coords);
|
||||
}
|
||||
|
||||
// Evict until we have room
|
||||
while inner.slots.len() >= inner.max_slots
|
||||
|| (inner.current_bytes + data_len > inner.max_bytes && !inner.slots.is_empty())
|
||||
{
|
||||
// Find LRU slot
|
||||
let lru_idx = inner
|
||||
.slots
|
||||
.iter()
|
||||
.enumerate()
|
||||
.min_by_key(|(_, s)| s.last_access)
|
||||
.map(|(i, _)| i)
|
||||
.unwrap();
|
||||
let removed = inner.slots.swap_remove(lru_idx);
|
||||
inner.slot_index.remove(&removed.coord);
|
||||
// swap_remove moved the former last element into `lru_idx` (unless
|
||||
// it *was* the last element) — fix up that element's index entry.
|
||||
if lru_idx < inner.slots.len() {
|
||||
let moved_coord = inner.slots[lru_idx].coord.clone();
|
||||
inner.slot_index.insert(moved_coord, lru_idx);
|
||||
}
|
||||
inner.current_bytes -= removed.data.len();
|
||||
inner.stats.evictions += 1;
|
||||
}
|
||||
// ----- Whole-cache operations -----
|
||||
|
||||
inner.current_bytes += data_len;
|
||||
let new_idx = inner.slots.len();
|
||||
inner.slot_index.insert(coord.clone(), new_idx);
|
||||
inner.slots.push(CachedChunk {
|
||||
coord,
|
||||
data: Arc::clone(&data),
|
||||
last_access: tick,
|
||||
});
|
||||
data
|
||||
}
|
||||
|
||||
/// Clear the entire cache (index + decompressed data).
|
||||
/// Clear the entire cache (indexes + decompressed data + stats).
|
||||
pub fn clear(&self) {
|
||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
||||
inner.index = None;
|
||||
inner.index_addr = None;
|
||||
let mut inner = self.lock();
|
||||
inner.datasets.clear();
|
||||
inner.current = None;
|
||||
inner.slots.clear();
|
||||
inner.slot_index.clear();
|
||||
inner.current_bytes = 0;
|
||||
inner.tick = 0;
|
||||
inner.last_coord = None;
|
||||
inner.stats = AccessStats::default();
|
||||
inner.chunk_index = None;
|
||||
inner.chunk_layout = None;
|
||||
}
|
||||
|
||||
/// Record that the given chunk coordinates are predicted to be accessed
|
||||
/// soon (bookkeeping only).
|
||||
///
|
||||
/// This does **not** prefetch or pre-decompress anything — it only
|
||||
/// checks whether each coordinate is already in the chunk index and
|
||||
/// updates access-pattern stats accordingly. Real prefetching (e.g.
|
||||
/// background pre-decompression) is not implemented.
|
||||
pub fn prefetch_hint(&self, next_coords: &[ChunkCoord]) {
|
||||
let inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
||||
if inner.index.is_none() {
|
||||
return;
|
||||
}
|
||||
drop(inner);
|
||||
// For each predicted coordinate, verify it exists in the index.
|
||||
// The index is already populated, so this is a no-op for known chunks.
|
||||
// The purpose is to signal intent — callers can pre-decompress if needed.
|
||||
// We touch the stats to record that prefetch hints were issued.
|
||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
||||
for coord in next_coords {
|
||||
let exists = inner
|
||||
.index
|
||||
.as_ref()
|
||||
.map(|idx| idx.contains_key(coord))
|
||||
.unwrap_or(false);
|
||||
if exists {
|
||||
inner.stats.sequential_count += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Return the current access pattern statistics.
|
||||
pub fn access_stats(&self) -> AccessStats {
|
||||
self.inner
|
||||
.lock()
|
||||
.unwrap_or_else(|e| e.into_inner())
|
||||
.stats
|
||||
.clone()
|
||||
self.lock().stats.clone()
|
||||
}
|
||||
|
||||
/// Update the sweep direction label in the access stats.
|
||||
pub fn set_sweep_direction(&self, direction: &'static str) {
|
||||
self.inner
|
||||
.lock()
|
||||
.unwrap_or_else(|e| e.into_inner())
|
||||
.stats
|
||||
.sweep_direction = Some(direction);
|
||||
self.lock().stats.sweep_direction = Some(direction);
|
||||
}
|
||||
|
||||
/// Number of decompressed chunks currently cached.
|
||||
/// Number of decompressed chunks currently cached (all datasets).
|
||||
pub fn cached_chunk_count(&self) -> usize {
|
||||
self.inner
|
||||
.lock()
|
||||
.unwrap_or_else(|e| e.into_inner())
|
||||
.slots
|
||||
.len()
|
||||
self.lock().slots.len()
|
||||
}
|
||||
|
||||
/// Total bytes of decompressed data currently cached.
|
||||
/// Total bytes of decompressed data currently cached (all datasets).
|
||||
pub fn cached_bytes(&self) -> usize {
|
||||
self.inner
|
||||
.lock()
|
||||
.unwrap_or_else(|e| e.into_inner())
|
||||
.current_bytes
|
||||
self.lock().current_bytes
|
||||
}
|
||||
|
||||
/// Number of datasets whose chunk index is currently kept.
|
||||
pub fn indexed_dataset_count(&self) -> usize {
|
||||
self.lock().datasets.len()
|
||||
}
|
||||
}
|
||||
|
||||
@@ -808,6 +967,92 @@ mod tests {
|
||||
assert_eq!(cache.cached_bytes(), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn datasets_sharing_coordinates_stay_separate() {
|
||||
let cache = ChunkCache::new();
|
||||
let a = vec![make_chunk(vec![0, 0], 0x100, 8)];
|
||||
let b = vec![make_chunk(vec![0, 0], 0x900, 8)];
|
||||
let got_a = cache.chunks_for::<()>(1, 1, || Ok(a.clone())).unwrap();
|
||||
let got_b = cache.chunks_for::<()>(2, 1, || Ok(b.clone())).unwrap();
|
||||
assert_eq!(got_a[0].address, 0x100);
|
||||
assert_eq!(got_b[0].address, 0x900);
|
||||
// Built once per dataset: a second lookup doesn't call the builder.
|
||||
let again = cache
|
||||
.chunks_for::<()>(1, 1, || panic!("index rebuilt"))
|
||||
.unwrap();
|
||||
assert_eq!(again[0].address, 0x100);
|
||||
|
||||
cache.put_decompressed_in(1, vec![0], vec![1; 4]);
|
||||
cache.put_decompressed_in(2, vec![0], vec![2; 4]);
|
||||
assert_eq!(
|
||||
cache.get_decompressed_in(1, &[0]).unwrap().as_slice(),
|
||||
&[1; 4]
|
||||
);
|
||||
assert_eq!(
|
||||
cache.get_decompressed_in(2, &[0]).unwrap().as_slice(),
|
||||
&[2; 4]
|
||||
);
|
||||
assert!(cache.get_decompressed_in(3, &[0]).is_none());
|
||||
assert_eq!(cache.cached_chunk_count(), 2);
|
||||
|
||||
// The bound-dataset methods see only the bound dataset.
|
||||
cache.ensure_dataset(2);
|
||||
assert_eq!(cache.lookup_index(&[0]).unwrap().address, 0x900);
|
||||
assert_eq!(cache.get_decompressed(&[0]).unwrap(), vec![2; 4]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn dataset_indexes_are_bounded() {
|
||||
let cache = ChunkCache::new();
|
||||
for addr in 0..(MAX_INDEXED_DATASETS as u64 + 10) {
|
||||
cache
|
||||
.chunks_for::<()>(addr, 1, || Ok(vec![make_chunk(vec![0], addr, 8)]))
|
||||
.unwrap();
|
||||
}
|
||||
assert_eq!(cache.indexed_dataset_count(), MAX_INDEXED_DATASETS);
|
||||
|
||||
// One huge index evicts the others but is itself kept.
|
||||
let huge: Vec<ChunkInfo> = (0..MAX_INDEXED_CHUNKS as u64)
|
||||
.map(|i| make_chunk(vec![i], i, 8))
|
||||
.collect();
|
||||
let got = cache.chunks_for::<()>(9999, 1, || Ok(huge)).unwrap();
|
||||
assert_eq!(got.len(), MAX_INDEXED_CHUNKS);
|
||||
assert_eq!(cache.indexed_dataset_count(), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn concurrent_readers_of_different_datasets_see_their_own_chunks() {
|
||||
let cache = std::sync::Arc::new(ChunkCache::with_capacity(1 << 20, 64));
|
||||
let handles: Vec<_> = (0..8u64)
|
||||
.map(|t| {
|
||||
let cache = std::sync::Arc::clone(&cache);
|
||||
std::thread::spawn(move || {
|
||||
for round in 0..500u64 {
|
||||
let addr = (t + round) % 16;
|
||||
let coord = vec![round % 4];
|
||||
let chunks = cache
|
||||
.chunks_for::<()>(addr, 1, || {
|
||||
Ok((0..4).map(|c| make_chunk(vec![c], addr, 8)).collect())
|
||||
})
|
||||
.unwrap();
|
||||
assert!(chunks.iter().all(|c| c.address == addr));
|
||||
let want = vec![addr as u8; 8];
|
||||
let got = match cache.get_decompressed_in(addr, &coord) {
|
||||
Some(hit) => hit.to_vec(),
|
||||
None => cache
|
||||
.put_decompressed_in(addr, coord, want.clone())
|
||||
.to_vec(),
|
||||
};
|
||||
assert_eq!(got, want);
|
||||
}
|
||||
})
|
||||
})
|
||||
.collect();
|
||||
for h in handles {
|
||||
h.join().unwrap();
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn duplicate_insert_is_noop() {
|
||||
let cache = ChunkCache::new();
|
||||
|
||||
@@ -0,0 +1,200 @@
|
||||
//! Chunk-index linearisation shared by the Fixed Array and Extensible Array
|
||||
//! chunk indexes (reader and writer).
|
||||
//!
|
||||
//! Both indexes store one element per chunk at a *linear* index, and the
|
||||
//! library derives that index from the chunk's scaled coordinates
|
||||
//! (`offset / chunk_dim`) using the dataset's **maximum** dimensions, not its
|
||||
//! current ones (`H5D__farray_idx_get_addr` / `H5D__earray_idx_get_addr`,
|
||||
//! via `layout->max_down_chunks`). A dataset whose current shape is smaller
|
||||
//! than its maxshape therefore has gaps in the index, and laying it out by the
|
||||
//! current shape puts every chunk after the first row in the wrong place.
|
||||
//!
|
||||
//! The Extensible Array adds one more step: its one unlimited dimension has no
|
||||
//! finite chunk count, so the library *swizzles* the coordinates to make that
|
||||
//! dimension the slowest-varying one (`H5VM_swizzle_coords`, which moves
|
||||
//! `coords[unlim_dim]` to the front and shifts the dimensions before it right
|
||||
//! by one) before linearising with `swizzled_max_down_chunks`. When the
|
||||
//! unlimited dimension is already dimension 0 no swizzle happens.
|
||||
|
||||
#[cfg(not(feature = "std"))]
|
||||
extern crate alloc;
|
||||
|
||||
#[cfg(not(feature = "std"))]
|
||||
use alloc::{vec, vec::Vec};
|
||||
|
||||
use crate::error::FormatError;
|
||||
|
||||
/// How a chunk index maps linear element indexes to chunk coordinates.
|
||||
#[derive(Debug, Clone)]
|
||||
pub(crate) struct ChunkGrid {
|
||||
/// Spatial chunk dimensions, in dataset order.
|
||||
chunk_dims: Vec<u64>,
|
||||
/// Chunks per dimension covering the *current* extent, in dataset order.
|
||||
cur_chunks: Vec<u64>,
|
||||
/// Dataset dimension stored at each linearisation position (slowest
|
||||
/// first). The identity except for a swizzled Extensible Array.
|
||||
order: Vec<usize>,
|
||||
/// Linear stride of each linearisation position.
|
||||
down: Vec<u64>,
|
||||
}
|
||||
|
||||
impl ChunkGrid {
|
||||
/// Grid for a Fixed Array index: row-major over the chunk counts of the
|
||||
/// maximum dimensions (`max_dims`, falling back to the current dimensions
|
||||
/// when the dataspace records none).
|
||||
pub(crate) fn fixed_array(
|
||||
cur_dims: &[u64],
|
||||
max_dims: Option<&[u64]>,
|
||||
chunk_dims: &[u64],
|
||||
) -> Result<Self, FormatError> {
|
||||
Self::build(cur_dims, max_dims, chunk_dims, None)
|
||||
}
|
||||
|
||||
/// Grid for an Extensible Array index: like the Fixed Array, but the
|
||||
/// unlimited dimension (the one whose maximum is `H5S_UNLIMITED`) is moved
|
||||
/// to the slowest-varying position first.
|
||||
pub(crate) fn extensible_array(
|
||||
cur_dims: &[u64],
|
||||
max_dims: Option<&[u64]>,
|
||||
chunk_dims: &[u64],
|
||||
) -> Result<Self, FormatError> {
|
||||
let unlim = max_dims.and_then(|m| m.iter().position(|&d| d == u64::MAX));
|
||||
Self::build(cur_dims, max_dims, chunk_dims, unlim)
|
||||
}
|
||||
|
||||
fn build(
|
||||
cur_dims: &[u64],
|
||||
max_dims: Option<&[u64]>,
|
||||
chunk_dims: &[u64],
|
||||
unlim: Option<usize>,
|
||||
) -> Result<Self, FormatError> {
|
||||
let rank = chunk_dims.len();
|
||||
if cur_dims.len() != rank || max_dims.is_some_and(|m| m.len() != rank) {
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"chunk index rank does not match the dataspace".into(),
|
||||
));
|
||||
}
|
||||
if chunk_dims.contains(&0) {
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"chunk dimension is zero".into(),
|
||||
));
|
||||
}
|
||||
let cur_chunks: Vec<u64> = cur_dims
|
||||
.iter()
|
||||
.zip(chunk_dims)
|
||||
.map(|(&d, &c)| d.div_ceil(c))
|
||||
.collect();
|
||||
// Chunk counts of the maximum extent. An unlimited dimension has no
|
||||
// finite count; it only ever sits in the slowest position, where its
|
||||
// count never enters a stride. A (corrupt) maximum smaller than the
|
||||
// current extent is widened so no allocated chunk becomes unreachable.
|
||||
let max_chunks: Vec<u64> = (0..rank)
|
||||
.map(|d| {
|
||||
let max = max_dims.map_or(cur_dims[d], |m| m[d]);
|
||||
if max == u64::MAX {
|
||||
u64::MAX
|
||||
} else {
|
||||
max.div_ceil(chunk_dims[d]).max(cur_chunks[d])
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
|
||||
let mut order: Vec<usize> = (0..rank).collect();
|
||||
if let Some(u) = unlim {
|
||||
order.remove(u);
|
||||
order.insert(0, u);
|
||||
}
|
||||
let mut down = vec![1u64; rank];
|
||||
for p in (0..rank.saturating_sub(1)).rev() {
|
||||
let next = max_chunks[order[p + 1]];
|
||||
if next == u64::MAX {
|
||||
// Only reachable with more than one unlimited dimension, which
|
||||
// neither index type can describe.
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"array chunk index with more than one unlimited dimension".into(),
|
||||
));
|
||||
}
|
||||
down[p] = down[p + 1].checked_mul(next).ok_or_else(|| {
|
||||
FormatError::Overflow("chunk index linear stride overflows u64".into())
|
||||
})?;
|
||||
}
|
||||
Ok(Self {
|
||||
chunk_dims: chunk_dims.to_vec(),
|
||||
cur_chunks,
|
||||
order,
|
||||
down,
|
||||
})
|
||||
}
|
||||
|
||||
/// Dataset-space offsets of the chunk stored at linear `index`, or `None`
|
||||
/// when that chunk lies outside the current extent (the index still has a
|
||||
/// slot for it; the library ignores such chunks on read).
|
||||
pub(crate) fn offsets(&self, index: u64) -> Option<Vec<u64>> {
|
||||
let rank = self.chunk_dims.len();
|
||||
let mut offsets = vec![0u64; rank];
|
||||
let mut rem = index;
|
||||
for p in 0..rank {
|
||||
let d = self.order[p];
|
||||
let scaled = rem / self.down[p];
|
||||
rem %= self.down[p];
|
||||
if scaled >= self.cur_chunks[d] {
|
||||
return None;
|
||||
}
|
||||
offsets[d] = scaled * self.chunk_dims[d];
|
||||
}
|
||||
Some(offsets)
|
||||
}
|
||||
|
||||
/// Linear index of the chunk with scaled coordinates `scaled`
|
||||
/// (`offset / chunk_dim` per dimension, in dataset order).
|
||||
pub(crate) fn linear_index(&self, scaled: &[u64]) -> u64 {
|
||||
self.order
|
||||
.iter()
|
||||
.zip(&self.down)
|
||||
.map(|(&d, &stride)| scaled[d] * stride)
|
||||
.sum()
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn fixed_array_uses_max_dims() {
|
||||
// shape (4, 6), chunks (2, 3), maxshape (20, 10): 10 x 4 chunk grid.
|
||||
let g = ChunkGrid::fixed_array(&[4, 6], Some(&[20, 10]), &[2, 3]).unwrap();
|
||||
assert_eq!(g.offsets(0), Some(vec![0, 0]));
|
||||
assert_eq!(g.offsets(1), Some(vec![0, 3]));
|
||||
assert_eq!(g.offsets(2), None); // column chunk 2 is beyond the extent
|
||||
assert_eq!(g.offsets(4), Some(vec![2, 0]));
|
||||
assert_eq!(g.offsets(5), Some(vec![2, 3]));
|
||||
assert_eq!(g.offsets(8), None); // row chunk 2 is beyond the extent
|
||||
assert_eq!(g.linear_index(&[1, 1]), 5);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn extensible_array_swizzles_unlimited_dim() {
|
||||
// maxshape (10, None): dim 1 is unlimited and becomes slowest.
|
||||
let g = ChunkGrid::extensible_array(&[4, 6], Some(&[10, u64::MAX]), &[2, 3]).unwrap();
|
||||
// max chunks of dim 0 = 5, so index = c1 * 5 + c0.
|
||||
assert_eq!(g.linear_index(&[1, 0]), 1);
|
||||
assert_eq!(g.linear_index(&[0, 1]), 5);
|
||||
assert_eq!(g.offsets(5), Some(vec![0, 3]));
|
||||
assert_eq!(g.offsets(6), Some(vec![2, 3]));
|
||||
assert_eq!(g.offsets(2), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn extensible_array_unlimited_first_is_row_major() {
|
||||
let g = ChunkGrid::extensible_array(&[4, 6], Some(&[u64::MAX, 30]), &[2, 3]).unwrap();
|
||||
// max chunks of dim 1 = 10.
|
||||
assert_eq!(g.linear_index(&[1, 1]), 11);
|
||||
assert_eq!(g.offsets(11), Some(vec![2, 3]));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rejects_two_unlimited_dims_after_the_first() {
|
||||
assert!(ChunkGrid::fixed_array(&[4, 6], Some(&[u64::MAX, u64::MAX]), &[2, 3]).is_err());
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -4,15 +4,17 @@
|
||||
extern crate alloc;
|
||||
|
||||
#[cfg(not(feature = "std"))]
|
||||
use alloc::{vec, vec::Vec};
|
||||
use alloc::{format, vec, vec::Vec};
|
||||
|
||||
use crate::checksum::jenkins_lookup3;
|
||||
use crate::chunk_cache::{CACHE_LINE_SIZE, align_to_cache_line};
|
||||
use crate::chunk_grid::ChunkGrid;
|
||||
use crate::ea_writer;
|
||||
use crate::error::FormatError;
|
||||
use crate::filter_pipeline::{
|
||||
FILTER_DEFLATE, FILTER_FLETCHER32, FILTER_LZ4, FILTER_PCODEC, FILTER_SHUFFLE, FILTER_ZSTD,
|
||||
FilterDescription, FilterPipeline,
|
||||
FILTER_BITSHUFFLE, FILTER_BLOSC, FILTER_BZIP2, FILTER_DEFLATE, FILTER_FLETCHER32, FILTER_LZ4,
|
||||
FILTER_LZF, FILTER_PCODEC, FILTER_PCODEC_NAME, FILTER_SHUFFLE, FILTER_ZSTD, FilterDescription,
|
||||
FilterPipeline,
|
||||
};
|
||||
use crate::filters::compress_chunk;
|
||||
/// Round a file offset up to the next cache-line boundary.
|
||||
@@ -44,8 +46,170 @@ pub struct ChunkOptions {
|
||||
pub lz4: bool,
|
||||
/// Zstandard compression level (1-22), None = no zstd. Filter ID 32015.
|
||||
pub zstd_level: Option<u32>,
|
||||
/// Pcodec lossless numerical compression. Filter ID 32023.
|
||||
/// Pcodec lossless numerical compression. Private, unregistered filter
|
||||
/// ID [`FILTER_PCODEC`] (480): only clawhdf5 can read it.
|
||||
pub pcodec: bool,
|
||||
/// A plugin compression filter (LZF, ...). Takes priority over the
|
||||
/// codecs above. Each needs its cargo feature to be written.
|
||||
pub plugin: Option<PluginFilter>,
|
||||
}
|
||||
|
||||
/// A compression filter from the common HDF5 plugin set, written in the
|
||||
/// format the libhdf5 plugin (h5py / hdf5plugin) reads.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
#[non_exhaustive]
|
||||
pub enum PluginFilter {
|
||||
/// LZF (filter 32000), h5py's built-in `compression="lzf"`. Needs the
|
||||
/// `lzf` feature.
|
||||
Lzf,
|
||||
/// Bitshuffle (filter 32008): a bit transpose of each block of
|
||||
/// `block_size` elements (0 = bitshuffle's default, else a multiple of
|
||||
/// 8), optionally compressed. Needs the `bitshuffle` feature.
|
||||
Bitshuffle {
|
||||
/// Block size in elements; 0 for the default.
|
||||
block_size: u32,
|
||||
/// Compression after the transpose.
|
||||
compression: BitshuffleCompression,
|
||||
},
|
||||
/// bzip2 (filter 307) at block size `level` (1-9). Needs the `bzip2`
|
||||
/// feature.
|
||||
Bzip2 {
|
||||
/// Block size 1-9 (9 = hdf5plugin's default).
|
||||
level: u32,
|
||||
},
|
||||
/// Blosc 1 (filter 32001): `codec` at `level` (0-9; 0 stores), after
|
||||
/// `shuffle`. Needs the `blosc` feature.
|
||||
Blosc {
|
||||
/// The codec inside the Blosc frame.
|
||||
codec: BloscCodec,
|
||||
/// Compression level 0-9 (0 stores the data uncompressed).
|
||||
level: u32,
|
||||
/// The shuffle Blosc applies first.
|
||||
shuffle: BloscShuffle,
|
||||
},
|
||||
}
|
||||
|
||||
/// The codec inside a Blosc frame that clawhdf5 can write. (It reads
|
||||
/// BloscLZ too, but cannot write it.)
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum BloscCodec {
|
||||
/// LZ4.
|
||||
Lz4,
|
||||
/// Snappy.
|
||||
Snappy,
|
||||
/// Zlib, at the Blosc level.
|
||||
Zlib,
|
||||
/// Zstandard (clawhdf5's pure-Rust encoder has one level, about zstd 1).
|
||||
Zstd,
|
||||
}
|
||||
|
||||
/// The shuffle Blosc applies before compressing.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum BloscShuffle {
|
||||
/// None.
|
||||
None,
|
||||
/// Byte shuffle (Blosc's default).
|
||||
Byte,
|
||||
/// Bit shuffle.
|
||||
Bit,
|
||||
}
|
||||
|
||||
/// What bitshuffle compresses its blocks with.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum BitshuffleCompression {
|
||||
/// Transpose only.
|
||||
None,
|
||||
/// LZ4 (bitshuffle's `cname="lz4"`, the common choice).
|
||||
Lz4,
|
||||
/// Zstandard. clawhdf5's pure-Rust encoder has a single level (about
|
||||
/// zstd's level 1); `level` is recorded in the file for other writers.
|
||||
Zstd {
|
||||
/// Level recorded in `cd_values[5]`.
|
||||
level: u32,
|
||||
},
|
||||
}
|
||||
|
||||
impl PluginFilter {
|
||||
/// Whether the filter reorders bytes itself, so the automatic shuffle
|
||||
/// pre-filter would only get in its way.
|
||||
fn shuffles_itself(&self) -> bool {
|
||||
match self {
|
||||
PluginFilter::Lzf => false,
|
||||
PluginFilter::Bitshuffle { .. } => true,
|
||||
PluginFilter::Bzip2 { .. } => false,
|
||||
PluginFilter::Blosc { .. } => true,
|
||||
}
|
||||
}
|
||||
|
||||
/// The pipeline entry for this filter. `chunk_bytes` is one chunk's
|
||||
/// uncompressed size (0 if unknown).
|
||||
fn description(&self, element_size: u32, chunk_bytes: u32) -> FilterDescription {
|
||||
match self {
|
||||
// h5py's lzf_set_local: filter version, liblzf version, chunk
|
||||
// size in bytes. Optional, as h5py flags it: a chunk the filter
|
||||
// cannot shrink may then be stored unfiltered.
|
||||
PluginFilter::Lzf => FilterDescription {
|
||||
filter_id: FILTER_LZF,
|
||||
name: Some("lzf".into()),
|
||||
flags: 1,
|
||||
client_data: vec![4, 0x0105, chunk_bytes],
|
||||
},
|
||||
// bshuf_h5_set_local: version 0.4, element size, block size,
|
||||
// compression (0 none, 2 LZ4, 3 Zstandard), Zstandard level.
|
||||
// hdf5-blosc's blosc_set_local: filter revision 2, Blosc format
|
||||
// 2, type size, chunk size, then level, shuffle, compressor.
|
||||
PluginFilter::Blosc {
|
||||
codec,
|
||||
level,
|
||||
shuffle,
|
||||
} => FilterDescription {
|
||||
filter_id: FILTER_BLOSC,
|
||||
name: Some("blosc".into()),
|
||||
flags: 1,
|
||||
client_data: vec![
|
||||
2,
|
||||
2,
|
||||
element_size,
|
||||
chunk_bytes,
|
||||
(*level).min(9),
|
||||
match shuffle {
|
||||
BloscShuffle::None => 0,
|
||||
BloscShuffle::Byte => 1,
|
||||
BloscShuffle::Bit => 2,
|
||||
},
|
||||
match codec {
|
||||
BloscCodec::Lz4 => 1,
|
||||
BloscCodec::Snappy => 3,
|
||||
BloscCodec::Zlib => 4,
|
||||
BloscCodec::Zstd => 5,
|
||||
},
|
||||
],
|
||||
},
|
||||
PluginFilter::Bzip2 { level } => FilterDescription {
|
||||
filter_id: FILTER_BZIP2,
|
||||
name: Some("bzip2".into()),
|
||||
flags: 1,
|
||||
client_data: vec![(*level).clamp(1, 9)],
|
||||
},
|
||||
PluginFilter::Bitshuffle {
|
||||
block_size,
|
||||
compression,
|
||||
} => {
|
||||
let mut cd = vec![0, 4, element_size, *block_size];
|
||||
match compression {
|
||||
BitshuffleCompression::None => cd.push(0),
|
||||
BitshuffleCompression::Lz4 => cd.push(2),
|
||||
BitshuffleCompression::Zstd { level } => cd.extend([3, *level]),
|
||||
}
|
||||
FilterDescription {
|
||||
filter_id: FILTER_BITSHUFFLE,
|
||||
name: Some("bitshuffle; see https://github.com/kiyo-masui/bitshuffle".into()),
|
||||
flags: 1,
|
||||
client_data: cd,
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Largest chunk the automatic choice produces, in bytes.
|
||||
@@ -90,14 +254,33 @@ impl ChunkOptions {
|
||||
|| self.lz4
|
||||
|| self.zstd_level.is_some()
|
||||
|| self.pcodec
|
||||
|| self.plugin.is_some()
|
||||
}
|
||||
|
||||
/// Build a FilterPipeline from the options.
|
||||
pub fn build_pipeline(&self, element_size: u32) -> Option<FilterPipeline> {
|
||||
self.build_pipeline_for_chunk(element_size, 0)
|
||||
}
|
||||
|
||||
/// Build a FilterPipeline for chunks of `chunk_bytes` uncompressed bytes
|
||||
/// (0 if unknown). Some plugin filters record the chunk size in their
|
||||
/// client data.
|
||||
pub fn build_pipeline_for_chunk(
|
||||
&self,
|
||||
element_size: u32,
|
||||
chunk_bytes: u32,
|
||||
) -> Option<FilterPipeline> {
|
||||
let mut filters = Vec::new();
|
||||
|
||||
let has_compression =
|
||||
self.deflate_level.is_some() || self.zstd_level.is_some() || self.lz4 || self.pcodec;
|
||||
let plugin_shuffles = self
|
||||
.plugin
|
||||
.as_ref()
|
||||
.is_some_and(PluginFilter::shuffles_itself);
|
||||
let has_compression = self.deflate_level.is_some()
|
||||
|| self.zstd_level.is_some()
|
||||
|| self.lz4
|
||||
|| self.pcodec
|
||||
|| (self.plugin.is_some() && !plugin_shuffles);
|
||||
|
||||
// Shuffle before compression. Applied if explicitly requested OR if compression
|
||||
// is active and the caller hasn't disabled it — matches h5py default behavior
|
||||
@@ -111,11 +294,14 @@ impl ChunkOptions {
|
||||
});
|
||||
}
|
||||
|
||||
// Compression filters (mutually exclusive, priority: pcodec > zstd > lz4 > deflate)
|
||||
if self.pcodec {
|
||||
// Compression filters (mutually exclusive, priority: plugin > pcodec >
|
||||
// zstd > lz4 > deflate)
|
||||
if let Some(plugin) = &self.plugin {
|
||||
filters.push(plugin.description(element_size, chunk_bytes));
|
||||
} else if self.pcodec {
|
||||
filters.push(FilterDescription {
|
||||
filter_id: FILTER_PCODEC,
|
||||
name: Some("pcodec".into()),
|
||||
name: Some(FILTER_PCODEC_NAME.into()),
|
||||
flags: 0,
|
||||
client_data: vec![element_size],
|
||||
});
|
||||
@@ -381,39 +567,7 @@ fn serialize_v4_single_chunk(
|
||||
let ndims = chunk_dims.len() as u8 + 1;
|
||||
buf.push(ndims);
|
||||
|
||||
// dim_size_encoded_length: how many bytes per dimension
|
||||
// We need to figure out the minimum encoding width
|
||||
let max_dim = chunk_dims
|
||||
.iter()
|
||||
.map(|&d| d as u64)
|
||||
.chain(core::iter::once(element_size as u64))
|
||||
.max()
|
||||
.unwrap_or(1);
|
||||
let dim_encoded_len: u8 = if max_dim <= 0xFF {
|
||||
1
|
||||
} else if max_dim <= 0xFFFF {
|
||||
2
|
||||
} else {
|
||||
4
|
||||
};
|
||||
buf.push(dim_encoded_len);
|
||||
|
||||
// dimension sizes (chunk dims + element size)
|
||||
for &d in chunk_dims {
|
||||
match dim_encoded_len {
|
||||
1 => buf.push(d as u8),
|
||||
2 => buf.extend_from_slice(&(d as u16).to_le_bytes()),
|
||||
4 => buf.extend_from_slice(&d.to_le_bytes()),
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
// Element size dimension
|
||||
match dim_encoded_len {
|
||||
1 => buf.push(element_size as u8),
|
||||
2 => buf.extend_from_slice(&(element_size as u16).to_le_bytes()),
|
||||
4 => buf.extend_from_slice(&element_size.to_le_bytes()),
|
||||
_ => {}
|
||||
}
|
||||
push_v4_chunk_dims(&mut buf, chunk_dims, element_size);
|
||||
|
||||
// chunk index type = 1 (single chunk)
|
||||
buf.push(1);
|
||||
@@ -443,45 +597,7 @@ fn serialize_v4_fixed_array(
|
||||
element_size: u32,
|
||||
max_bits: u8,
|
||||
) -> Vec<u8> {
|
||||
let mut buf = Vec::new();
|
||||
buf.push(4); // version
|
||||
buf.push(2); // class = chunked
|
||||
|
||||
let flags: u8 = 0x00;
|
||||
buf.push(flags);
|
||||
|
||||
let ndims = chunk_dims.len() as u8 + 1;
|
||||
buf.push(ndims);
|
||||
|
||||
let max_dim = chunk_dims
|
||||
.iter()
|
||||
.map(|&d| d as u64)
|
||||
.chain(core::iter::once(element_size as u64))
|
||||
.max()
|
||||
.unwrap_or(1);
|
||||
let dim_encoded_len: u8 = if max_dim <= 0xFF {
|
||||
1
|
||||
} else if max_dim <= 0xFFFF {
|
||||
2
|
||||
} else {
|
||||
4
|
||||
};
|
||||
buf.push(dim_encoded_len);
|
||||
|
||||
for &d in chunk_dims {
|
||||
match dim_encoded_len {
|
||||
1 => buf.push(d as u8),
|
||||
2 => buf.extend_from_slice(&(d as u16).to_le_bytes()),
|
||||
4 => buf.extend_from_slice(&d.to_le_bytes()),
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
match dim_encoded_len {
|
||||
1 => buf.push(element_size as u8),
|
||||
2 => buf.extend_from_slice(&(element_size as u16).to_le_bytes()),
|
||||
4 => buf.extend_from_slice(&element_size.to_le_bytes()),
|
||||
_ => {}
|
||||
}
|
||||
let mut buf = layout_v4_chunked_prefix(chunk_dims, element_size);
|
||||
|
||||
// chunk index type = 3 (Fixed Array)
|
||||
buf.push(3);
|
||||
@@ -499,107 +615,175 @@ fn serialize_v4_fixed_array(
|
||||
buf
|
||||
}
|
||||
|
||||
/// The part of a v4 chunked layout message before the chunk index type:
|
||||
/// version, class, flags and the chunk dimensions (plus the element size).
|
||||
/// Append a v4 layout's dimension width and its dimensions (the chunk
|
||||
/// dimensions, then the element size). Each takes the fewest bytes that hold
|
||||
/// the largest, as libhdf5 computes it (`H5D__chunk_set_sizes`:
|
||||
/// `(log2(dim) + 8) / 8`); HDF5 2.0.0 refuses any other width.
|
||||
pub(crate) fn push_v4_chunk_dims(buf: &mut Vec<u8>, chunk_dims: &[u32], element_size: u32) {
|
||||
let max_dim = chunk_dims
|
||||
.iter()
|
||||
.copied()
|
||||
.chain(core::iter::once(element_size))
|
||||
.max()
|
||||
.unwrap_or(1)
|
||||
.max(1);
|
||||
let width = (32 - max_dim.leading_zeros()).div_ceil(8) as usize;
|
||||
buf.push(width as u8);
|
||||
for &d in chunk_dims.iter().chain(core::iter::once(&element_size)) {
|
||||
buf.extend_from_slice(&d.to_le_bytes()[..width]);
|
||||
}
|
||||
}
|
||||
|
||||
fn layout_v4_chunked_prefix(chunk_dims: &[u32], element_size: u32) -> Vec<u8> {
|
||||
let mut buf = Vec::new();
|
||||
buf.push(4); // version
|
||||
buf.push(2); // class = chunked
|
||||
|
||||
let flags: u8 = 0x00;
|
||||
buf.push(flags);
|
||||
|
||||
let ndims = chunk_dims.len() as u8 + 1;
|
||||
buf.push(ndims);
|
||||
|
||||
push_v4_chunk_dims(&mut buf, chunk_dims, element_size);
|
||||
buf
|
||||
}
|
||||
|
||||
/// log2 of the elements per Fixed Array data block page (the library's
|
||||
/// default, `H5D_FARRAY_MAX_DBLK_PAGE_NELMTS_BITS`).
|
||||
const FA_PAGE_BITS: u8 = 10;
|
||||
|
||||
pub(crate) fn push_addr(buf: &mut Vec<u8>, addr: u64, offset_size: u8) {
|
||||
match offset_size {
|
||||
4 => buf.extend_from_slice(&(addr as u32).to_le_bytes()),
|
||||
_ => buf.extend_from_slice(&addr.to_le_bytes()),
|
||||
}
|
||||
}
|
||||
|
||||
/// Width of the chunk-size field of a filtered chunk index element. Must
|
||||
/// match the library's `H5D_FARRAY_FILT_COMPUTE_CHUNK_SIZE_LEN` (the EA and
|
||||
/// B-tree v2 indexes use the same formula):
|
||||
/// `1 + ((log2(unfiltered chunk bytes) + 8) / 8)`, capped at 8.
|
||||
pub(crate) fn filtered_chunk_size_len(slots: &[Option<WrittenChunk>]) -> usize {
|
||||
let max_raw = slots
|
||||
.iter()
|
||||
.flatten()
|
||||
.map(|c| c.raw_size)
|
||||
.max()
|
||||
.unwrap_or(1);
|
||||
let log2_val = if max_raw <= 1 {
|
||||
0
|
||||
} else {
|
||||
63 - max_raw.leading_zeros()
|
||||
};
|
||||
(1 + ((log2_val + 8) / 8) as usize).min(8)
|
||||
}
|
||||
|
||||
/// Append one chunk index element: the chunk's address, plus its stored size
|
||||
/// and filter mask when the dataset is filtered. `None` is an unallocated
|
||||
/// chunk (undefined address, zero size and mask).
|
||||
pub(crate) fn push_index_element(
|
||||
buf: &mut Vec<u8>,
|
||||
slot: Option<&WrittenChunk>,
|
||||
offset_size: u8,
|
||||
chunk_size_bytes: Option<usize>,
|
||||
) {
|
||||
match slot {
|
||||
Some(c) => {
|
||||
push_addr(buf, c.address, offset_size);
|
||||
if let Some(n) = chunk_size_bytes {
|
||||
buf.extend_from_slice(&c.compressed_size.to_le_bytes()[..n]);
|
||||
buf.extend_from_slice(&c.filter_mask.to_le_bytes());
|
||||
}
|
||||
}
|
||||
None => {
|
||||
buf.extend(core::iter::repeat_n(0xFF, offset_size as usize));
|
||||
if let Some(n) = chunk_size_bytes {
|
||||
buf.extend(core::iter::repeat_n(0x00, n + 4));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Build a complete Fixed Array at a known absolute address.
|
||||
///
|
||||
/// `slots` holds one entry per element of the array, i.e. per chunk of the
|
||||
/// dataset's *maximum* extent in the order [`crate::chunk_grid`] defines;
|
||||
/// `None` marks a chunk that is not allocated. An array with more elements
|
||||
/// than fit in one page (`2^FA_PAGE_BITS`) gets a paged data block: a
|
||||
/// page-init bitmap after the prefix, then one checksummed page per
|
||||
/// `2^FA_PAGE_BITS` elements, the last one short (`H5FA__dblock_create`).
|
||||
pub fn build_fixed_array_at(
|
||||
chunks: &[WrittenChunk],
|
||||
slots: &[Option<WrittenChunk>],
|
||||
offset_size: u8,
|
||||
length_size: u8,
|
||||
has_filters: bool,
|
||||
fa_base_address: u64,
|
||||
) -> Vec<u8> {
|
||||
let os = offset_size as usize;
|
||||
let num_elements = chunks.len();
|
||||
|
||||
// For filtered chunks, compute chunk_size encoding width.
|
||||
// Must match the HDF5 C library's H5D_FARRAY_FILT_COMPUTE_CHUNK_SIZE_LEN macro:
|
||||
// chunk_size_len = 1 + ((H5VM_log2_gen(chunk.size) + 8) / 8)
|
||||
// where chunk.size is the unfiltered chunk size in bytes (product of all chunk dims).
|
||||
let chunk_size_bytes: usize = if has_filters {
|
||||
let max_raw = chunks.iter().map(|c| c.raw_size).max().unwrap_or(1);
|
||||
let log2_val = if max_raw <= 1 {
|
||||
0
|
||||
} else {
|
||||
63 - max_raw.leading_zeros()
|
||||
};
|
||||
let len = 1 + ((log2_val + 8) / 8) as usize;
|
||||
len.min(8)
|
||||
} else {
|
||||
0
|
||||
};
|
||||
|
||||
let elem_size = if has_filters {
|
||||
os + chunk_size_bytes + 4
|
||||
} else {
|
||||
os
|
||||
};
|
||||
let num_elements = slots.len();
|
||||
|
||||
let chunk_size_bytes = has_filters.then(|| filtered_chunk_size_len(slots));
|
||||
let elem_size = os + chunk_size_bytes.map_or(0, |n| n + 4);
|
||||
let client_id: u8 = if has_filters { 1 } else { 0 };
|
||||
|
||||
// FAHD total size
|
||||
let nelmts_field_size = length_size as usize;
|
||||
let fahd_total_size = 4 + 1 + 1 + 1 + 1 + nelmts_field_size + os + 4;
|
||||
let fahd_total_size = 4 + 1 + 1 + 1 + 1 + length_size as usize + os + 4;
|
||||
let fadb_address = fa_base_address + fahd_total_size as u64;
|
||||
|
||||
// Build FAHD
|
||||
let mut fahd = Vec::with_capacity(fahd_total_size);
|
||||
fahd.extend_from_slice(b"FAHD");
|
||||
fahd.push(0); // version
|
||||
fahd.push(client_id);
|
||||
fahd.push(elem_size as u8);
|
||||
|
||||
// max_nelmts_bits: use 10 as default (page_size = 1024), matching h5py convention
|
||||
let max_bits: u8 = 10;
|
||||
fahd.push(max_bits);
|
||||
|
||||
fahd.push(FA_PAGE_BITS);
|
||||
match length_size {
|
||||
4 => fahd.extend_from_slice(&(num_elements as u32).to_le_bytes()),
|
||||
8 => fahd.extend_from_slice(&(num_elements as u64).to_le_bytes()),
|
||||
_ => fahd.extend_from_slice(&(num_elements as u64).to_le_bytes()),
|
||||
}
|
||||
|
||||
match offset_size {
|
||||
4 => fahd.extend_from_slice(&(fadb_address as u32).to_le_bytes()),
|
||||
8 => fahd.extend_from_slice(&fadb_address.to_le_bytes()),
|
||||
_ => fahd.extend_from_slice(&fadb_address.to_le_bytes()),
|
||||
}
|
||||
|
||||
// Checksum
|
||||
push_addr(&mut fahd, fadb_address, offset_size);
|
||||
let checksum = jenkins_lookup3(&fahd);
|
||||
fahd.extend_from_slice(&checksum.to_le_bytes());
|
||||
|
||||
assert_eq!(fahd.len(), fahd_total_size);
|
||||
|
||||
// Build FADB
|
||||
// FADB prefix
|
||||
let mut fadb = Vec::new();
|
||||
fadb.extend_from_slice(b"FADB");
|
||||
fadb.push(0); // version
|
||||
fadb.push(client_id);
|
||||
push_addr(&mut fadb, fa_base_address, offset_size);
|
||||
|
||||
// header address
|
||||
match offset_size {
|
||||
4 => fadb.extend_from_slice(&(fa_base_address as u32).to_le_bytes()),
|
||||
8 => fadb.extend_from_slice(&fa_base_address.to_le_bytes()),
|
||||
_ => fadb.extend_from_slice(&fa_base_address.to_le_bytes()),
|
||||
let page_nelmts = 1usize << FA_PAGE_BITS;
|
||||
if num_elements <= page_nelmts {
|
||||
// Unpaged: the elements follow the prefix, one checksum over both.
|
||||
for slot in slots {
|
||||
push_index_element(&mut fadb, slot.as_ref(), offset_size, chunk_size_bytes);
|
||||
}
|
||||
|
||||
// Element data
|
||||
for chunk in chunks {
|
||||
match offset_size {
|
||||
4 => fadb.extend_from_slice(&(chunk.address as u32).to_le_bytes()),
|
||||
8 => fadb.extend_from_slice(&chunk.address.to_le_bytes()),
|
||||
_ => fadb.extend_from_slice(&chunk.address.to_le_bytes()),
|
||||
}
|
||||
if has_filters {
|
||||
// Write compressed size using chunk_size_bytes (variable width)
|
||||
let cs_bytes = chunk.compressed_size.to_le_bytes();
|
||||
fadb.extend_from_slice(&cs_bytes[..chunk_size_bytes]);
|
||||
fadb.extend_from_slice(&chunk.filter_mask.to_le_bytes());
|
||||
}
|
||||
}
|
||||
|
||||
// FADB checksum
|
||||
let fadb_checksum = jenkins_lookup3(&fadb);
|
||||
fadb.extend_from_slice(&fadb_checksum.to_le_bytes());
|
||||
} else {
|
||||
// Paged: every page is written, so every page-init bit is set
|
||||
// (MSB-first, as `H5VM_bit_set` packs them). The prefix and bitmap
|
||||
// share a checksum; each page carries its own.
|
||||
let npages = num_elements.div_ceil(page_nelmts);
|
||||
let mut bitmap = vec![0u8; npages.div_ceil(8)];
|
||||
for p in 0..npages {
|
||||
bitmap[p / 8] |= 0x80 >> (p % 8);
|
||||
}
|
||||
fadb.extend_from_slice(&bitmap);
|
||||
let prefix_checksum = jenkins_lookup3(&fadb);
|
||||
fadb.extend_from_slice(&prefix_checksum.to_le_bytes());
|
||||
for page in slots.chunks(page_nelmts) {
|
||||
let start = fadb.len();
|
||||
for slot in page {
|
||||
push_index_element(&mut fadb, slot.as_ref(), offset_size, chunk_size_bytes);
|
||||
}
|
||||
let page_checksum = jenkins_lookup3(&fadb[start..]);
|
||||
fadb.extend_from_slice(&page_checksum.to_le_bytes());
|
||||
}
|
||||
}
|
||||
|
||||
let mut combined = fahd;
|
||||
combined.extend_from_slice(&fadb);
|
||||
@@ -634,7 +818,12 @@ pub fn precompress_chunks(
|
||||
element_size: usize,
|
||||
options: &ChunkOptions,
|
||||
) -> Result<PrecompressedChunks, FormatError> {
|
||||
let pipeline = options.build_pipeline(element_size as u32);
|
||||
let chunk_bytes = chunk_dims
|
||||
.iter()
|
||||
.try_fold(element_size as u64, |acc, &d| acc.checked_mul(d))
|
||||
.and_then(|b| u32::try_from(b).ok())
|
||||
.unwrap_or(0);
|
||||
let pipeline = options.build_pipeline_for_chunk(element_size as u32, chunk_bytes);
|
||||
let has_filters = pipeline.is_some();
|
||||
let pipeline_message = pipeline.as_ref().map(|pl| pl.serialize());
|
||||
|
||||
@@ -667,7 +856,8 @@ pub fn build_chunked_data_from_precompressed(
|
||||
pre: &PrecompressedChunks,
|
||||
base_address: u64,
|
||||
maxshape: Option<&[u64]>,
|
||||
) -> ChunkedDataResult {
|
||||
) -> Result<ChunkedDataResult, FormatError> {
|
||||
let index = ChunkIndexPlan::new(&pre.shape, maxshape, &pre.chunk_dims)?;
|
||||
let offset_size: u8 = 8;
|
||||
let length_size: u8 = 8;
|
||||
let num_chunks = pre.chunks.len();
|
||||
@@ -693,17 +883,18 @@ pub fn build_chunked_data_from_precompressed(
|
||||
}
|
||||
|
||||
let chunk_dims_u32: Vec<u32> = pre.chunk_dims.iter().map(|&d| d as u32).collect();
|
||||
let use_extensible = maxshape.is_some_and(|ms| ms.contains(&u64::MAX));
|
||||
|
||||
let aligned_idx = align_to_cache_line(data_buf.len());
|
||||
if aligned_idx > data_buf.len() {
|
||||
data_buf.resize(aligned_idx, 0u8);
|
||||
}
|
||||
|
||||
let layout_message = if use_extensible {
|
||||
let layout_message = match &index {
|
||||
ChunkIndexPlan::ExtensibleArray(grid) => {
|
||||
let ea_address = base_address + data_buf.len() as u64;
|
||||
let slots = index_slots(grid, &pre.shape, &pre.chunk_dims, &written_chunks, None)?;
|
||||
let ea_bytes = ea_writer::build_extensible_array_at(
|
||||
&written_chunks,
|
||||
&slots,
|
||||
offset_size,
|
||||
length_size,
|
||||
pre.has_filters,
|
||||
@@ -716,7 +907,8 @@ pub fn build_chunked_data_from_precompressed(
|
||||
offset_size,
|
||||
element_size as u32,
|
||||
)
|
||||
} else if num_chunks == 1 {
|
||||
}
|
||||
ChunkIndexPlan::SingleChunk => {
|
||||
let chunk_addr = written_chunks[0].address;
|
||||
let filtered_size = if pre.has_filters {
|
||||
Some(written_chunks[0].compressed_size)
|
||||
@@ -732,10 +924,18 @@ pub fn build_chunked_data_from_precompressed(
|
||||
offset_size,
|
||||
element_size as u32,
|
||||
)
|
||||
} else {
|
||||
}
|
||||
ChunkIndexPlan::FixedArray(grid, nslots) => {
|
||||
let fa_address = base_address + data_buf.len() as u64;
|
||||
let fa_bytes = build_fixed_array_at(
|
||||
let slots = index_slots(
|
||||
grid,
|
||||
&pre.shape,
|
||||
&pre.chunk_dims,
|
||||
&written_chunks,
|
||||
Some(*nslots),
|
||||
)?;
|
||||
let fa_bytes = build_fixed_array_at(
|
||||
&slots,
|
||||
offset_size,
|
||||
length_size,
|
||||
pre.has_filters,
|
||||
@@ -747,15 +947,263 @@ pub fn build_chunked_data_from_precompressed(
|
||||
fa_address,
|
||||
offset_size,
|
||||
element_size as u32,
|
||||
10, // max_nelmts_bits — matches h5py convention
|
||||
FA_PAGE_BITS,
|
||||
)
|
||||
}
|
||||
ChunkIndexPlan::BTreeV2 => {
|
||||
let bt_address = base_address + data_buf.len() as u64;
|
||||
let records: Vec<(Vec<u64>, &WrittenChunk)> = written_chunks
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(i, c)| (scaled_coords(&pre.shape, &pre.chunk_dims, i), c))
|
||||
.collect();
|
||||
let (bt_bytes, node_size) = build_btree_v2_chunk_index_at(
|
||||
pre.shape.len(),
|
||||
&records,
|
||||
offset_size,
|
||||
length_size,
|
||||
pre.has_filters,
|
||||
bt_address,
|
||||
)?;
|
||||
data_buf.extend_from_slice(&bt_bytes);
|
||||
serialize_v4_btree_v2(
|
||||
&chunk_dims_u32,
|
||||
bt_address,
|
||||
offset_size,
|
||||
element_size as u32,
|
||||
node_size,
|
||||
)
|
||||
}
|
||||
};
|
||||
|
||||
ChunkedDataResult {
|
||||
Ok(ChunkedDataResult {
|
||||
data_bytes: data_buf,
|
||||
layout_message,
|
||||
pipeline_message: pre.pipeline_message.clone(),
|
||||
})
|
||||
}
|
||||
|
||||
/// Most slots a Fixed Array index may have before we refuse to build it: its
|
||||
/// data block holds one element per chunk of the *maximum* extent, so a huge
|
||||
/// finite maxshape with small chunks would otherwise exhaust memory.
|
||||
const MAX_FIXED_ARRAY_SLOTS: u64 = 1 << 26;
|
||||
|
||||
/// Which chunk index a dataset gets, following the library's choice in
|
||||
/// `H5D__layout_set_latest_indexing`: version-2 B-tree for more than one
|
||||
/// unlimited dimension, Extensible Array for exactly one, Fixed Array for a
|
||||
/// finite maxshape, Single Chunk when the whole maximum extent is one chunk.
|
||||
enum ChunkIndexPlan {
|
||||
SingleChunk,
|
||||
/// The grid and the number of array elements (chunks of the max extent).
|
||||
FixedArray(ChunkGrid, usize),
|
||||
ExtensibleArray(ChunkGrid),
|
||||
BTreeV2,
|
||||
}
|
||||
|
||||
impl ChunkIndexPlan {
|
||||
fn new(
|
||||
shape: &[u64],
|
||||
maxshape: Option<&[u64]>,
|
||||
chunk_dims: &[u64],
|
||||
) -> Result<Self, FormatError> {
|
||||
let bad = |what: &str| FormatError::ChunkedReadError(format!("maxshape: {what}"));
|
||||
if let Some(ms) = maxshape {
|
||||
if ms.len() != shape.len() {
|
||||
return Err(bad("rank differs from the shape"));
|
||||
}
|
||||
if ms.iter().zip(shape).any(|(&m, &s)| m < s) {
|
||||
return Err(bad("smaller than the shape"));
|
||||
}
|
||||
}
|
||||
let max = maxshape.unwrap_or(shape);
|
||||
let nunlim = max.iter().filter(|&&d| d == u64::MAX).count();
|
||||
match nunlim {
|
||||
0 => {
|
||||
let nslots = max
|
||||
.iter()
|
||||
.zip(chunk_dims)
|
||||
.try_fold(1u64, |acc, (&m, &c)| acc.checked_mul(m.div_ceil(c.max(1))))
|
||||
.filter(|&n| n <= MAX_FIXED_ARRAY_SLOTS)
|
||||
.ok_or_else(|| {
|
||||
bad("too many chunks for a Fixed Array index; \
|
||||
use larger chunks or an unlimited dimension")
|
||||
})?;
|
||||
// A Single Chunk index needs that one chunk to exist; an
|
||||
// empty dataset gets an all-unallocated Fixed Array instead.
|
||||
let empty = shape.contains(&0);
|
||||
if nslots == 1 && !empty {
|
||||
Ok(Self::SingleChunk)
|
||||
} else {
|
||||
let grid = ChunkGrid::fixed_array(shape, Some(max), chunk_dims)?;
|
||||
Ok(Self::FixedArray(grid, nslots as usize))
|
||||
}
|
||||
}
|
||||
1 => Ok(Self::ExtensibleArray(ChunkGrid::extensible_array(
|
||||
shape,
|
||||
Some(max),
|
||||
chunk_dims,
|
||||
)?)),
|
||||
_ => Ok(Self::BTreeV2),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Place each written chunk at its linear index in `grid`. `chunks` are in
|
||||
/// row-major order over the chunks of the current extent (`split_into_chunks`).
|
||||
/// `len` fixes the slot count (Fixed Array); otherwise it is one past the
|
||||
/// highest index used.
|
||||
fn index_slots(
|
||||
grid: &ChunkGrid,
|
||||
shape: &[u64],
|
||||
chunk_dims: &[u64],
|
||||
chunks: &[WrittenChunk],
|
||||
len: Option<usize>,
|
||||
) -> Result<Vec<Option<WrittenChunk>>, FormatError> {
|
||||
let mut placed: Vec<(usize, &WrittenChunk)> = Vec::with_capacity(chunks.len());
|
||||
for (i, chunk) in chunks.iter().enumerate() {
|
||||
let scaled = scaled_coords(shape, chunk_dims, i);
|
||||
let idx = usize::try_from(grid.linear_index(&scaled))
|
||||
.map_err(|_| FormatError::Overflow("chunk index slot".into()))?;
|
||||
placed.push((idx, chunk));
|
||||
}
|
||||
let n = len.unwrap_or_else(|| placed.iter().map(|&(i, _)| i + 1).max().unwrap_or(0));
|
||||
let mut slots = vec![None; n];
|
||||
for (idx, chunk) in placed {
|
||||
*slots
|
||||
.get_mut(idx)
|
||||
.ok_or_else(|| FormatError::Overflow("chunk index slot".into()))? = Some(chunk.clone());
|
||||
}
|
||||
Ok(slots)
|
||||
}
|
||||
|
||||
/// Scaled coordinates (`offset / chunk_dim`) of the `i`-th chunk in the
|
||||
/// row-major order `split_into_chunks` produces over the current extent.
|
||||
fn scaled_coords(shape: &[u64], chunk_dims: &[u64], i: usize) -> Vec<u64> {
|
||||
let rank = shape.len();
|
||||
let mut scaled = vec![0u64; rank];
|
||||
let mut rem = i as u64;
|
||||
for d in (0..rank).rev() {
|
||||
let n = shape[d].div_ceil(chunk_dims[d]);
|
||||
scaled[d] = rem % n;
|
||||
rem /= n;
|
||||
}
|
||||
scaled
|
||||
}
|
||||
|
||||
/// Node size the library gives a chunk index B-tree (`H5D_BT2_NODE_SIZE`),
|
||||
/// with its split and merge percentages.
|
||||
const BT2_NODE_SIZE: u32 = 2048;
|
||||
const BT2_SPLIT_PERCENT: u8 = 100;
|
||||
const BT2_MERGE_PERCENT: u8 = 40;
|
||||
/// B-tree v2 record types for chunk indexes (`H5B2_CDSET_ID`,
|
||||
/// `H5B2_CDSET_FILT_ID`).
|
||||
const BT2_CHUNK_UNFILTERED: u8 = 10;
|
||||
const BT2_CHUNK_FILTERED: u8 = 11;
|
||||
|
||||
/// Build a version-2 B-tree chunk index (the library's index for datasets
|
||||
/// with more than one unlimited dimension) at a known absolute address.
|
||||
///
|
||||
/// `records` are `(scaled coordinates, chunk)` in lexicographic order of the
|
||||
/// coordinates, which is the order the library's comparator
|
||||
/// (`H5VM_vector_cmp_u`) keeps them in. The tree is a single leaf: the
|
||||
/// library's 2048-byte node when the records fit, otherwise a leaf node
|
||||
/// sized to hold them all (the root's record count is 16-bit, so at most
|
||||
/// 65535 chunks). Returns the bytes and the node size the layout message
|
||||
/// must record.
|
||||
fn build_btree_v2_chunk_index_at(
|
||||
rank: usize,
|
||||
records: &[(Vec<u64>, &WrittenChunk)],
|
||||
offset_size: u8,
|
||||
length_size: u8,
|
||||
has_filters: bool,
|
||||
base_address: u64,
|
||||
) -> Result<(Vec<u8>, u32), FormatError> {
|
||||
let os = offset_size as usize;
|
||||
let nrec = u16::try_from(records.len()).map_err(|_| {
|
||||
FormatError::ChunkedReadError(
|
||||
"more than 65535 chunks with more than one unlimited dimension: \
|
||||
use larger chunks"
|
||||
.into(),
|
||||
)
|
||||
})?;
|
||||
let chunk_size_bytes = has_filters.then(|| {
|
||||
let slots: Vec<Option<WrittenChunk>> =
|
||||
records.iter().map(|(_, c)| Some((*c).clone())).collect();
|
||||
filtered_chunk_size_len(&slots)
|
||||
});
|
||||
let record_size = os + chunk_size_bytes.map_or(0, |n| n + 4) + 8 * rank;
|
||||
// Leaf: signature, version, type, records, checksum.
|
||||
let leaf_len = 4 + 1 + 1 + records.len() * record_size + 4;
|
||||
let node_size = u32::try_from(leaf_len)
|
||||
.map_err(|_| FormatError::Overflow("B-tree v2 leaf size".into()))?
|
||||
.max(BT2_NODE_SIZE);
|
||||
let tree_type = if has_filters {
|
||||
BT2_CHUNK_FILTERED
|
||||
} else {
|
||||
BT2_CHUNK_UNFILTERED
|
||||
};
|
||||
|
||||
let hdr_len = 4 + 1 + 1 + 4 + 2 + 2 + 1 + 1 + os + 2 + length_size as usize + 4;
|
||||
let leaf_address = base_address + hdr_len as u64;
|
||||
|
||||
let mut out = Vec::with_capacity(hdr_len + node_size as usize);
|
||||
out.extend_from_slice(b"BTHD");
|
||||
out.push(0); // version
|
||||
out.push(tree_type);
|
||||
out.extend_from_slice(&node_size.to_le_bytes());
|
||||
out.extend_from_slice(&(record_size as u16).to_le_bytes());
|
||||
out.extend_from_slice(&0u16.to_le_bytes()); // depth
|
||||
out.push(BT2_SPLIT_PERCENT);
|
||||
out.push(BT2_MERGE_PERCENT);
|
||||
if records.is_empty() {
|
||||
out.extend(core::iter::repeat_n(0xFF, os));
|
||||
} else {
|
||||
push_addr(&mut out, leaf_address, offset_size);
|
||||
}
|
||||
out.extend_from_slice(&nrec.to_le_bytes());
|
||||
match length_size {
|
||||
4 => out.extend_from_slice(&(records.len() as u32).to_le_bytes()),
|
||||
_ => out.extend_from_slice(&(records.len() as u64).to_le_bytes()),
|
||||
}
|
||||
let sum = jenkins_lookup3(&out);
|
||||
out.extend_from_slice(&sum.to_le_bytes());
|
||||
debug_assert_eq!(out.len(), hdr_len);
|
||||
if records.is_empty() {
|
||||
return Ok((out, node_size));
|
||||
}
|
||||
|
||||
let leaf_start = out.len();
|
||||
out.extend_from_slice(b"BTLF");
|
||||
out.push(0); // version
|
||||
out.push(tree_type);
|
||||
for (scaled, chunk) in records {
|
||||
push_index_element(&mut out, Some(chunk), offset_size, chunk_size_bytes);
|
||||
for &c in scaled {
|
||||
out.extend_from_slice(&c.to_le_bytes());
|
||||
}
|
||||
}
|
||||
let sum = jenkins_lookup3(&out[leaf_start..]);
|
||||
out.extend_from_slice(&sum.to_le_bytes());
|
||||
// The library reads whole nodes; pad the leaf out to the node size.
|
||||
out.resize(leaf_start + node_size as usize, 0);
|
||||
Ok((out, node_size))
|
||||
}
|
||||
|
||||
/// Serialize a v4 layout message for a version-2 B-tree chunk index.
|
||||
fn serialize_v4_btree_v2(
|
||||
chunk_dims: &[u32],
|
||||
btree_address: u64,
|
||||
offset_size: u8,
|
||||
element_size: u32,
|
||||
node_size: u32,
|
||||
) -> Vec<u8> {
|
||||
let mut buf = layout_v4_chunked_prefix(chunk_dims, element_size);
|
||||
buf.push(5); // chunk index type = 5 (version-2 B-tree)
|
||||
buf.extend_from_slice(&node_size.to_le_bytes());
|
||||
buf.push(BT2_SPLIT_PERCENT);
|
||||
buf.push(BT2_MERGE_PERCENT);
|
||||
push_addr(&mut buf, btree_address, offset_size);
|
||||
buf
|
||||
}
|
||||
|
||||
/// Build chunked data with absolute addresses.
|
||||
@@ -790,11 +1238,7 @@ pub fn build_chunked_data_at_ext(
|
||||
maxshape: Option<&[u64]>,
|
||||
) -> Result<ChunkedDataResult, FormatError> {
|
||||
let pre = precompress_chunks(raw_data, shape, chunk_dims, element_size, options)?;
|
||||
Ok(build_chunked_data_from_precompressed(
|
||||
&pre,
|
||||
base_address,
|
||||
maxshape,
|
||||
))
|
||||
build_chunked_data_from_precompressed(&pre, base_address, maxshape)
|
||||
}
|
||||
|
||||
/// Write selected elements into an existing in-memory dataset buffer.
|
||||
@@ -1273,6 +1717,35 @@ mod tests {
|
||||
assert_eq!(pl.filters[1].client_data, vec![3]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn chunk_options_pipeline_lzf() {
|
||||
let options = ChunkOptions {
|
||||
plugin: Some(PluginFilter::Lzf),
|
||||
..Default::default()
|
||||
};
|
||||
assert!(options.is_chunked());
|
||||
let pl = options.build_pipeline_for_chunk(8, 800).unwrap();
|
||||
assert_eq!(pl.filters.len(), 2);
|
||||
assert_eq!(pl.filters[0].filter_id, FILTER_SHUFFLE);
|
||||
assert_eq!(pl.filters[1].filter_id, FILTER_LZF);
|
||||
assert_eq!(pl.filters[1].client_data, vec![4, 0x0105, 800]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn chunk_options_pipeline_bitshuffle_has_no_auto_shuffle() {
|
||||
let options = ChunkOptions {
|
||||
plugin: Some(PluginFilter::Bitshuffle {
|
||||
block_size: 0,
|
||||
compression: BitshuffleCompression::Zstd { level: 5 },
|
||||
}),
|
||||
..Default::default()
|
||||
};
|
||||
let pl = options.build_pipeline(4).unwrap();
|
||||
assert_eq!(pl.filters.len(), 1);
|
||||
assert_eq!(pl.filters[0].filter_id, FILTER_BITSHUFFLE);
|
||||
assert_eq!(pl.filters[0].client_data, vec![0, 4, 4, 0, 3, 5]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn chunk_options_zstd_priority_over_deflate() {
|
||||
let options = ChunkOptions {
|
||||
@@ -1314,6 +1787,7 @@ mod tests {
|
||||
chunk_index_type,
|
||||
single_chunk_filtered_size,
|
||||
single_chunk_filter_mask,
|
||||
..
|
||||
} => {
|
||||
assert_eq!(version, 4);
|
||||
assert_eq!(chunk_index_type, Some(1));
|
||||
@@ -1382,7 +1856,8 @@ mod tests {
|
||||
filter_mask: 0,
|
||||
},
|
||||
];
|
||||
let fa = build_fixed_array_at(&chunks, 8, 8, false, 0x2000);
|
||||
let slots: Vec<_> = chunks.into_iter().map(Some).collect();
|
||||
let fa = build_fixed_array_at(&slots, 8, 8, false, 0x2000);
|
||||
// Should start with FAHD
|
||||
assert_eq!(&fa[0..4], b"FAHD");
|
||||
// FAHD size = 4+1+1+1+1+8+8+4 = 28
|
||||
@@ -1429,7 +1904,8 @@ mod tests {
|
||||
filter_mask: 0,
|
||||
},
|
||||
];
|
||||
let ea = ea_writer::build_extensible_array_at(&chunks, 8, 8, false, 0x2000);
|
||||
let slots: Vec<_> = chunks.into_iter().map(Some).collect();
|
||||
let ea = ea_writer::build_extensible_array_at(&slots, 8, 8, false, 0x2000);
|
||||
assert_eq!(&ea[0..4], b"EAHD");
|
||||
// Find EAIB after EAHD: 12 fixed + 6*8 stats + 8 addr + 4 checksum = 72
|
||||
let aehd_size = 4 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 6 * 8 + 8 + 4;
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
//! HDF5 Data Layout message parsing (message type 0x0008).
|
||||
|
||||
#[cfg(not(feature = "std"))]
|
||||
use alloc::{string::String, vec::Vec};
|
||||
use alloc::{format, string::String, vec::Vec};
|
||||
|
||||
#[cfg(feature = "std")]
|
||||
use std::string::String;
|
||||
@@ -24,6 +24,34 @@ pub struct VdsMapping {
|
||||
pub virtual_selection: Vec<u8>,
|
||||
}
|
||||
|
||||
/// Most dimensions a layout message can list (libhdf5 `H5O_LAYOUT_NDIMS`):
|
||||
/// 32 dataspace dimensions plus the element size.
|
||||
const MAX_LAYOUT_NDIMS: usize = 33;
|
||||
|
||||
/// libhdf5's checks on a chunked layout message's dimensions
|
||||
/// (`H5O__layout_decode`): at most [`MAX_LAYOUT_NDIMS`], no dimension 0, and
|
||||
/// before version 4 at least one dataspace dimension plus the element size.
|
||||
/// A zero chunk dimension used to read the dataset as all fill values.
|
||||
fn check_chunk_dims(dims: Vec<u32>, layout_version: u8) -> Result<Vec<u32>, FormatError> {
|
||||
if dims.len() > MAX_LAYOUT_NDIMS {
|
||||
return Err(FormatError::InvalidChunkDimensions(
|
||||
"dimensionality is too large".into(),
|
||||
));
|
||||
}
|
||||
if layout_version < 4 && dims.len() < 2 {
|
||||
return Err(FormatError::InvalidChunkDimensions(
|
||||
"bad dimensions for chunked storage".into(),
|
||||
));
|
||||
}
|
||||
if let Some(u) = dims.iter().position(|&d| d == 0) {
|
||||
return Err(FormatError::InvalidChunkDimensions(format!(
|
||||
"bad chunk dimension value when parsing layout message - chunk dimension must be \
|
||||
positive: mesg->u.chunk.dim[{u}] = 0"
|
||||
)));
|
||||
}
|
||||
Ok(dims)
|
||||
}
|
||||
|
||||
/// Parsed HDF5 data layout message.
|
||||
#[derive(Debug, Clone, PartialEq)]
|
||||
pub enum DataLayout {
|
||||
@@ -45,7 +73,9 @@ pub enum DataLayout {
|
||||
chunk_dimensions: Vec<u32>,
|
||||
/// B-tree address, or `None` if undefined.
|
||||
btree_address: Option<u64>,
|
||||
/// Layout version (3 or 4).
|
||||
/// Layout version (3 or 4). Version 1/2 messages (HDF5 1.4/1.6-era)
|
||||
/// use the same version-1 B-tree chunk index as version 3 and are
|
||||
/// reported as 3.
|
||||
version: u8,
|
||||
/// Chunk index type (v4 only).
|
||||
chunk_index_type: Option<u8>,
|
||||
@@ -53,6 +83,11 @@ pub enum DataLayout {
|
||||
single_chunk_filtered_size: Option<u64>,
|
||||
/// Filter mask for v4 single chunk with filters.
|
||||
single_chunk_filter_mask: Option<u32>,
|
||||
/// Layout v4 flag bit 0 (`H5D_CHUNK_DONT_FILTER_PARTIAL_CHUNKS`):
|
||||
/// partial edge chunks — those extending past the dataset's current
|
||||
/// extent in some dimension — are stored without the filter pipeline,
|
||||
/// even though their filter mask is 0. Always `false` for v3.
|
||||
dont_filter_partial_edge_chunks: bool,
|
||||
},
|
||||
/// Virtual dataset layout (v4 only).
|
||||
Virtual {
|
||||
@@ -67,21 +102,33 @@ pub enum DataLayout {
|
||||
},
|
||||
}
|
||||
|
||||
/// Version-1 VDS mapping flag: the source file name is stored by an earlier
|
||||
/// entry, whose index follows in place of the name.
|
||||
const VDS_SOURCE_FILE_SHARED: u8 = 0x01;
|
||||
/// Version-1 VDS mapping flag: likewise for the source dataset name.
|
||||
const VDS_SOURCE_DSET_SHARED: u8 = 0x02;
|
||||
/// Version-1 VDS mapping flag: the source is in the virtual file itself
|
||||
/// (`"."`); no file name is stored.
|
||||
const VDS_SOURCE_SAME_FILE: u8 = 0x04;
|
||||
const VDS_ALL_FLAGS: u8 = VDS_SOURCE_FILE_SHARED | VDS_SOURCE_DSET_SHARED | VDS_SOURCE_SAME_FILE;
|
||||
|
||||
/// Parse VDS mappings from global-heap object data.
|
||||
///
|
||||
/// The global-heap block holding a VDS mapping list is laid out as
|
||||
/// (reverse-engineered and validated against HDF5 2.0):
|
||||
/// (`H5D__virtual_store_layout` / `H5D__virtual_load_layout` in libhdf5):
|
||||
///
|
||||
/// ```text
|
||||
/// version(1) · nused(length_size, LE) · entry[nused] · checksum(4)
|
||||
/// ```
|
||||
///
|
||||
/// Each entry is:
|
||||
/// - source file name — a null-terminated string in **block version 0**; in
|
||||
/// **block version 1** a same-file reference is encoded as a single `0x04`
|
||||
/// marker byte (the source file is the virtual file itself) in place of the
|
||||
/// name;
|
||||
/// - source dataset name (null-terminated string);
|
||||
/// - **block version 1 only:** a flags byte. `0x04`: the source is in the
|
||||
/// virtual file itself and no file name is stored; `0x01`/`0x02`: the
|
||||
/// source file/dataset name is that of an earlier entry, whose index
|
||||
/// (`length_size` bytes) is stored instead of the name. libhdf5 2.0 writes
|
||||
/// version 1 when the file's low version bound is 2.0 and it saves space;
|
||||
/// - source file name (null-terminated string, unless flagged above);
|
||||
/// - source dataset name (null-terminated string, unless flagged above);
|
||||
/// - source selection (serialized `H5S` dataspace selection — self-describing
|
||||
/// in length);
|
||||
/// - virtual selection (serialized `H5S` dataspace selection).
|
||||
@@ -107,7 +154,7 @@ pub fn parse_vds_mappings(
|
||||
// `nused` is untrusted; don't pre-allocate from it. Each entry consumes at
|
||||
// least a few bytes, so the loop is naturally bounded by the heap data and
|
||||
// a bogus `nused` simply errors out on the first short read.
|
||||
let mut mappings = Vec::new();
|
||||
let mut mappings: Vec<VdsMapping> = Vec::new();
|
||||
// Reads one self-describing selection at `pos`, returning its raw bytes and
|
||||
// advancing past it — bounds-checked so a corrupt selection can't overrun.
|
||||
let read_selection = |heap_data: &[u8], pos: &mut usize| -> Result<Vec<u8>, FormatError> {
|
||||
@@ -127,17 +174,57 @@ pub fn parse_vds_mappings(
|
||||
Ok(bytes)
|
||||
};
|
||||
|
||||
for _ in 0..nused {
|
||||
// Source file name (with the version-1 same-file marker handled).
|
||||
let source_file = if version >= 1 && heap_data.get(pos) == Some(&0x04) {
|
||||
if version > 1 {
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"unsupported VDS mapping block version".into(),
|
||||
));
|
||||
}
|
||||
for i in 0..nused {
|
||||
// Version 1 prefixes each entry with a flags byte; a name may then be
|
||||
// omitted (same file) or replaced by the index of an earlier entry
|
||||
// holding the same name (`H5D__virtual_load_layout`).
|
||||
let flags = if version >= 1 {
|
||||
let f = *heap_data.get(pos).ok_or(FormatError::UnexpectedEof {
|
||||
expected: pos + 1,
|
||||
available: heap_data.len(),
|
||||
})?;
|
||||
pos += 1;
|
||||
if f & !VDS_ALL_FLAGS != 0 {
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"unknown VDS mapping flags".into(),
|
||||
));
|
||||
}
|
||||
f
|
||||
} else {
|
||||
0
|
||||
};
|
||||
// Index of an earlier entry, for a shared name.
|
||||
let earlier = |pos: &mut usize| -> Result<usize, FormatError> {
|
||||
let idx = read_length(heap_data, *pos, length_size)?;
|
||||
*pos += ls;
|
||||
if idx >= i {
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"VDS mapping shares a name with a later entry".into(),
|
||||
));
|
||||
}
|
||||
Ok(idx as usize)
|
||||
};
|
||||
|
||||
let source_file = if flags & VDS_SOURCE_SAME_FILE != 0 {
|
||||
String::from(".")
|
||||
} else if flags & VDS_SOURCE_FILE_SHARED != 0 {
|
||||
let idx = earlier(&mut pos)?;
|
||||
mappings[idx].source_file.clone()
|
||||
} else {
|
||||
read_null_terminated_string(heap_data, &mut pos)?
|
||||
};
|
||||
|
||||
// Source dataset name.
|
||||
let source_dataset = read_null_terminated_string(heap_data, &mut pos)?;
|
||||
let source_dataset = if flags & VDS_SOURCE_DSET_SHARED != 0 {
|
||||
let idx = earlier(&mut pos)?;
|
||||
mappings[idx].source_dataset.clone()
|
||||
} else {
|
||||
read_null_terminated_string(heap_data, &mut pos)?
|
||||
};
|
||||
|
||||
// Source selection, then virtual selection (both self-describing length).
|
||||
let source_selection = read_selection(heap_data, &mut pos)?;
|
||||
@@ -256,6 +343,7 @@ impl DataLayout {
|
||||
let layout_class = data[1];
|
||||
|
||||
match version {
|
||||
1 | 2 => Self::parse_v1_v2(data, offset_size),
|
||||
3 => Self::parse_v3(data, layout_class, offset_size, length_size),
|
||||
// v5 (emitted by HDF5 1.14+/2.0 with `libver=latest`) uses the same
|
||||
// message structure as v4 — only the version number was bumped.
|
||||
@@ -264,6 +352,87 @@ impl DataLayout {
|
||||
}
|
||||
}
|
||||
|
||||
/// Layout message versions 1 and 2 (HDF5 before 1.6.3):
|
||||
///
|
||||
/// ```text
|
||||
/// version(1) · dimensionality(1) · layout class(1) · reserved(5)
|
||||
/// · address(offset_size) — contiguous and chunked only
|
||||
/// · dimension sizes(4 × dimensionality)
|
||||
/// · compact data size(4) · compact raw data — compact only
|
||||
/// ```
|
||||
///
|
||||
/// The dimension sizes are the dataset's (contiguous/compact) or the
|
||||
/// chunk's (chunked) extent plus a trailing element-size dimension, as in
|
||||
/// version 3's chunked form. libhdf5 ignores them for contiguous storage
|
||||
/// and sizes the data from the dataspace; the product of the stored
|
||||
/// dimensions is that same size, and a disagreement (a dimension that was
|
||||
/// truncated to 32 bits) is caught by the reader's size check rather than
|
||||
/// returning wrong data.
|
||||
fn parse_v1_v2(data: &[u8], offset_size: u8) -> Result<DataLayout, FormatError> {
|
||||
ensure_len(data, 0, 8)?;
|
||||
let dimensionality = data[1] as usize;
|
||||
let layout_class = data[2];
|
||||
// H5O_LAYOUT_NDIMS: 32 dataspace dimensions + the element-size one.
|
||||
if dimensionality > 33 {
|
||||
return Err(FormatError::Overflow(format!(
|
||||
"data layout dimensionality {dimensionality} exceeds 33"
|
||||
)));
|
||||
}
|
||||
let mut p = 8;
|
||||
let os = offset_size as usize;
|
||||
let address = match layout_class {
|
||||
1 | 2 => {
|
||||
ensure_len(data, p, os)?;
|
||||
let a = if is_undefined(data, p, offset_size) {
|
||||
None
|
||||
} else {
|
||||
Some(read_offset(data, p, offset_size)?)
|
||||
};
|
||||
p += os;
|
||||
a
|
||||
}
|
||||
0 => None,
|
||||
_ => return Err(FormatError::InvalidLayoutClass(layout_class)),
|
||||
};
|
||||
ensure_len(data, p, dimensionality * 4)?;
|
||||
let dims: Vec<u32> = data[p..p + dimensionality * 4]
|
||||
.as_chunks::<4>()
|
||||
.0
|
||||
.iter()
|
||||
.map(|c| u32::from_le_bytes(*c))
|
||||
.collect();
|
||||
p += dimensionality * 4;
|
||||
match layout_class {
|
||||
0 => {
|
||||
ensure_len(data, p, 4)?;
|
||||
let size =
|
||||
u32::from_le_bytes([data[p], data[p + 1], data[p + 2], data[p + 3]]) as usize;
|
||||
ensure_len(data, p + 4, size)?;
|
||||
Ok(DataLayout::Compact {
|
||||
data: data[p + 4..p + 4 + size].to_vec(),
|
||||
})
|
||||
}
|
||||
1 => {
|
||||
let size = dims
|
||||
.iter()
|
||||
.try_fold(1u64, |acc, &d| acc.checked_mul(d as u64))
|
||||
.ok_or_else(|| {
|
||||
FormatError::Overflow(format!("contiguous layout size {dims:?}"))
|
||||
})?;
|
||||
Ok(DataLayout::Contiguous { address, size })
|
||||
}
|
||||
_ => Ok(DataLayout::Chunked {
|
||||
chunk_dimensions: check_chunk_dims(dims, 2)?,
|
||||
btree_address: address,
|
||||
version: 3,
|
||||
chunk_index_type: None,
|
||||
single_chunk_filtered_size: None,
|
||||
single_chunk_filter_mask: None,
|
||||
dont_filter_partial_edge_chunks: false,
|
||||
}),
|
||||
}
|
||||
}
|
||||
|
||||
fn parse_v3(
|
||||
data: &[u8],
|
||||
layout_class: u8,
|
||||
@@ -316,12 +485,13 @@ impl DataLayout {
|
||||
p += 4;
|
||||
}
|
||||
Ok(DataLayout::Chunked {
|
||||
chunk_dimensions,
|
||||
chunk_dimensions: check_chunk_dims(chunk_dimensions, 3)?,
|
||||
btree_address,
|
||||
version: 3,
|
||||
chunk_index_type: None,
|
||||
single_chunk_filtered_size: None,
|
||||
single_chunk_filter_mask: None,
|
||||
dont_filter_partial_edge_chunks: false,
|
||||
})
|
||||
}
|
||||
_ => Err(FormatError::InvalidLayoutClass(layout_class)),
|
||||
@@ -364,47 +534,40 @@ impl DataLayout {
|
||||
let dimensionality = data[pos + 1] as usize;
|
||||
let dim_size_encoded_length = data[pos + 2] as usize;
|
||||
let mut p = pos + 3;
|
||||
if dimensionality > MAX_LAYOUT_NDIMS {
|
||||
return Err(FormatError::InvalidChunkDimensions(
|
||||
"dimensionality is too large".into(),
|
||||
));
|
||||
}
|
||||
|
||||
// dimension sizes
|
||||
// Each dimension takes 1 to 8 bytes (libhdf5 writes the
|
||||
// fewest that hold the largest one, so 3, 5, 6 and 7 occur:
|
||||
// a chunk dimension of 70 000 takes 3). libhdf5 refuses 0
|
||||
// and more than 8.
|
||||
if dim_size_encoded_length == 0 || dim_size_encoded_length > 8 {
|
||||
return Err(FormatError::InvalidChunkDimensions(
|
||||
"encoded chunk dimension size is too large".into(),
|
||||
));
|
||||
}
|
||||
ensure_len(data, p, dimensionality * dim_size_encoded_length)?;
|
||||
let mut chunk_dimensions = Vec::with_capacity(dimensionality);
|
||||
for _ in 0..dimensionality {
|
||||
let val = match dim_size_encoded_length {
|
||||
1 => data[p] as u32,
|
||||
2 => u16::from_le_bytes([data[p], data[p + 1]]) as u32,
|
||||
4 => u32::from_le_bytes([data[p], data[p + 1], data[p + 2], data[p + 3]]),
|
||||
8 => {
|
||||
// V4 chunked encodes dimension sizes as 8 bytes, but
|
||||
// our ChunkedStorageV4 stores them as u32. We read only
|
||||
// the low 4 bytes (little-endian). This silently
|
||||
// truncates dimensions > 4 GiB, which are not expected
|
||||
// in practice (HDF5 chunk dimensions are always small).
|
||||
// If the high bytes are non-zero, the file is malformed
|
||||
// or uses dimensions we cannot represent.
|
||||
let high = u32::from_le_bytes([
|
||||
data[p + 4],
|
||||
data[p + 5],
|
||||
data[p + 6],
|
||||
data[p + 7],
|
||||
]);
|
||||
if high != 0 {
|
||||
return Err(FormatError::UnexpectedEof {
|
||||
expected: p + 8,
|
||||
available: data.len(),
|
||||
});
|
||||
}
|
||||
u32::from_le_bytes([data[p], data[p + 1], data[p + 2], data[p + 3]])
|
||||
}
|
||||
_ => {
|
||||
return Err(FormatError::UnexpectedEof {
|
||||
expected: p + dim_size_encoded_length,
|
||||
available: data.len(),
|
||||
});
|
||||
}
|
||||
};
|
||||
let val = data[p..p + dim_size_encoded_length]
|
||||
.iter()
|
||||
.rev()
|
||||
.fold(0u64, |acc, &b| (acc << 8) | u64::from(b));
|
||||
// Chunk dimensions are held as u32; HDF5 2.0 can write
|
||||
// larger ones (layout version 5), which are refused
|
||||
// rather than truncated.
|
||||
let val = u32::try_from(val).map_err(|_| {
|
||||
FormatError::InvalidChunkDimensions(format!(
|
||||
"chunk dimension {val} is larger than 2^32 - 1, which is not supported"
|
||||
))
|
||||
})?;
|
||||
chunk_dimensions.push(val);
|
||||
p += dim_size_encoded_length;
|
||||
}
|
||||
let chunk_dimensions = check_chunk_dims(chunk_dimensions, 4)?;
|
||||
|
||||
// chunk index type
|
||||
ensure_len(data, p, 1)?;
|
||||
@@ -505,6 +668,7 @@ impl DataLayout {
|
||||
chunk_index_type: Some(chunk_index_type),
|
||||
single_chunk_filtered_size,
|
||||
single_chunk_filter_mask,
|
||||
dont_filter_partial_edge_chunks: flags & 0x01 != 0,
|
||||
})
|
||||
}
|
||||
3 => {
|
||||
@@ -539,6 +703,202 @@ impl DataLayout {
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// Version 1/2 header: version, dimensionality, class, reserved(5).
|
||||
fn v1v2_header(version: u8, ndims: u8, class: u8) -> Vec<u8> {
|
||||
vec![version, ndims, class, 0, 0, 0, 0, 0]
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn v2_compact() {
|
||||
let mut buf = v1v2_header(2, 2, 0);
|
||||
// dims (3 elements of 2 bytes) — no address for compact
|
||||
buf.extend_from_slice(&3u32.to_le_bytes());
|
||||
buf.extend_from_slice(&2u32.to_le_bytes());
|
||||
buf.extend_from_slice(&6u32.to_le_bytes()); // compact size (u32 in v1/v2)
|
||||
buf.extend_from_slice(&[1, 0, 2, 0, 3, 0]);
|
||||
assert_eq!(
|
||||
DataLayout::parse(&buf, 8, 8).unwrap(),
|
||||
DataLayout::Compact {
|
||||
data: vec![1, 0, 2, 0, 3, 0]
|
||||
}
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn v1_contiguous_size_from_dimensions() {
|
||||
let mut buf = v1v2_header(1, 3, 1);
|
||||
buf.extend_from_slice(&0x800u32.to_le_bytes()); // 4-byte address
|
||||
for d in [10u32, 20, 4] {
|
||||
buf.extend_from_slice(&d.to_le_bytes());
|
||||
}
|
||||
assert_eq!(
|
||||
DataLayout::parse(&buf, 4, 4).unwrap(),
|
||||
DataLayout::Contiguous {
|
||||
address: Some(0x800),
|
||||
size: 800,
|
||||
}
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn v1_contiguous_undefined_address() {
|
||||
let mut buf = v1v2_header(1, 2, 1);
|
||||
buf.extend_from_slice(&[0xFF; 8]);
|
||||
buf.extend_from_slice(&5u32.to_le_bytes());
|
||||
buf.extend_from_slice(&8u32.to_le_bytes());
|
||||
assert_eq!(
|
||||
DataLayout::parse(&buf, 8, 8).unwrap(),
|
||||
DataLayout::Contiguous {
|
||||
address: None,
|
||||
size: 40,
|
||||
}
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn v1_chunked_maps_to_btree_v1_index() {
|
||||
let mut buf = v1v2_header(1, 3, 2);
|
||||
buf.extend_from_slice(&0x1234u64.to_le_bytes());
|
||||
for d in [50u32, 50, 4] {
|
||||
buf.extend_from_slice(&d.to_le_bytes());
|
||||
}
|
||||
assert_eq!(
|
||||
DataLayout::parse(&buf, 8, 8).unwrap(),
|
||||
DataLayout::Chunked {
|
||||
chunk_dimensions: vec![50, 50, 4],
|
||||
btree_address: Some(0x1234),
|
||||
version: 3,
|
||||
chunk_index_type: None,
|
||||
single_chunk_filtered_size: None,
|
||||
single_chunk_filter_mask: None,
|
||||
dont_filter_partial_edge_chunks: false,
|
||||
}
|
||||
);
|
||||
}
|
||||
|
||||
/// A v3 chunked layout message with these dims (element size last).
|
||||
fn v3_chunked_msg(dims: &[u32]) -> Vec<u8> {
|
||||
let mut buf = vec![3u8, 2, dims.len() as u8];
|
||||
buf.extend_from_slice(&0x1000u64.to_le_bytes());
|
||||
for d in dims {
|
||||
buf.extend_from_slice(&d.to_le_bytes());
|
||||
}
|
||||
buf
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn chunk_dimensions_are_checked_when_the_layout_is_parsed() {
|
||||
assert!(DataLayout::parse(&v3_chunked_msg(&[4, 4, 8]), 8, 8).is_ok());
|
||||
// A zero chunk dimension used to read as all fill values.
|
||||
let err = DataLayout::parse(&v3_chunked_msg(&[4, 0, 8]), 8, 8).unwrap_err();
|
||||
assert!(
|
||||
matches!(&err, FormatError::InvalidChunkDimensions(m) if m.contains("dim[1] = 0")),
|
||||
"{err:?}"
|
||||
);
|
||||
// Only the element-size dimension: libhdf5 "bad dimensions".
|
||||
assert_eq!(
|
||||
DataLayout::parse(&v3_chunked_msg(&[8]), 8, 8).unwrap_err(),
|
||||
FormatError::InvalidChunkDimensions("bad dimensions for chunked storage".into())
|
||||
);
|
||||
assert_eq!(
|
||||
DataLayout::parse(&v3_chunked_msg(&[1; 34]), 8, 8).unwrap_err(),
|
||||
FormatError::InvalidChunkDimensions("dimensionality is too large".into())
|
||||
);
|
||||
// v1/v2 and v4 messages get the zero check too.
|
||||
let mut v1 = v1v2_header(1, 2, 2);
|
||||
v1.extend_from_slice(&0x1000u64.to_le_bytes());
|
||||
v1.extend_from_slice(&0u32.to_le_bytes());
|
||||
v1.extend_from_slice(&8u32.to_le_bytes());
|
||||
assert!(matches!(
|
||||
DataLayout::parse(&v1, 8, 8),
|
||||
Err(FormatError::InvalidChunkDimensions(_))
|
||||
));
|
||||
let mut v4 = vec![4u8, 2, 0, 2, 4];
|
||||
v4.extend_from_slice(&0u32.to_le_bytes());
|
||||
v4.extend_from_slice(&8u32.to_le_bytes());
|
||||
v4.push(3); // fixed array index
|
||||
v4.push(0); // page bits
|
||||
v4.extend_from_slice(&0x1000u64.to_le_bytes());
|
||||
assert!(matches!(
|
||||
DataLayout::parse(&v4, 8, 8),
|
||||
Err(FormatError::InvalidChunkDimensions(_))
|
||||
));
|
||||
}
|
||||
|
||||
/// A v4 chunked layout (fixed array index) whose `dims` are each
|
||||
/// encoded in `width` bytes.
|
||||
fn v4_chunked_msg(width: u8, dims: &[u64]) -> Vec<u8> {
|
||||
let mut m = vec![4u8, 2, 0, dims.len() as u8, width];
|
||||
for &d in dims {
|
||||
m.extend_from_slice(&d.to_le_bytes()[..width.min(8) as usize]);
|
||||
}
|
||||
m.push(3); // fixed array index
|
||||
m.push(0); // page bits
|
||||
m.extend_from_slice(&0x1000u64.to_le_bytes());
|
||||
m
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn v4_chunk_dimensions_take_1_to_8_bytes() {
|
||||
// libhdf5 encodes each dimension in the fewest bytes that hold the
|
||||
// largest: a chunk dimension of 70 000 takes 3, and 3, 5, 6 and 7
|
||||
// were refused ("UnexpectedEof").
|
||||
for width in 1..=8u8 {
|
||||
let dims = [if width >= 3 { 70_000 } else { 200 }, 8];
|
||||
let layout = DataLayout::parse(&v4_chunked_msg(width, &dims), 8, 8)
|
||||
.unwrap_or_else(|e| panic!("width {width}: {e:?}"));
|
||||
assert!(
|
||||
matches!(&layout, DataLayout::Chunked { chunk_dimensions, .. }
|
||||
if chunk_dimensions.iter().map(|&d| u64::from(d)).eq(dims)),
|
||||
"width {width}: {layout:?}"
|
||||
);
|
||||
}
|
||||
// libhdf5 refuses 0 and more than 8 bytes.
|
||||
for width in [0u8, 9] {
|
||||
assert_eq!(
|
||||
DataLayout::parse(&v4_chunked_msg(width, &[4, 8]), 8, 8).unwrap_err(),
|
||||
FormatError::InvalidChunkDimensions(
|
||||
"encoded chunk dimension size is too large".into()
|
||||
)
|
||||
);
|
||||
}
|
||||
// A dimension past u32 cannot be represented and is refused, not
|
||||
// truncated.
|
||||
assert!(matches!(
|
||||
DataLayout::parse(&v4_chunked_msg(5, &[1 << 32, 8]), 8, 8),
|
||||
Err(FormatError::InvalidChunkDimensions(m)) if m.contains("2^32")
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn v1v2_rejects_bad_class_dimensionality_and_truncation() {
|
||||
assert_eq!(
|
||||
DataLayout::parse(&v1v2_header(1, 1, 3), 8, 8).unwrap_err(),
|
||||
FormatError::InvalidLayoutClass(3)
|
||||
);
|
||||
assert!(matches!(
|
||||
DataLayout::parse(&v1v2_header(2, 34, 1), 8, 8).unwrap_err(),
|
||||
FormatError::Overflow(_)
|
||||
));
|
||||
// Chunked, dims cut short.
|
||||
let mut buf = v1v2_header(1, 2, 2);
|
||||
buf.extend_from_slice(&0x10u64.to_le_bytes());
|
||||
buf.extend_from_slice(&7u32.to_le_bytes());
|
||||
assert!(matches!(
|
||||
DataLayout::parse(&buf, 8, 8).unwrap_err(),
|
||||
FormatError::UnexpectedEof { .. }
|
||||
));
|
||||
// Compact, raw data shorter than its declared size.
|
||||
let mut buf = v1v2_header(2, 1, 0);
|
||||
buf.extend_from_slice(&4u32.to_le_bytes());
|
||||
buf.extend_from_slice(&100u32.to_le_bytes());
|
||||
buf.extend_from_slice(&[0; 4]);
|
||||
assert!(matches!(
|
||||
DataLayout::parse(&buf, 8, 8).unwrap_err(),
|
||||
FormatError::UnexpectedEof { .. }
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn v3_compact() {
|
||||
let mut buf = vec![3u8, 0]; // version=3, class=0 (compact)
|
||||
@@ -602,6 +962,7 @@ mod tests {
|
||||
chunk_index_type: None,
|
||||
single_chunk_filtered_size: None,
|
||||
single_chunk_filter_mask: None,
|
||||
dont_filter_partial_edge_chunks: false,
|
||||
}
|
||||
);
|
||||
}
|
||||
@@ -679,10 +1040,35 @@ mod tests {
|
||||
chunk_index_type: Some(1),
|
||||
single_chunk_filtered_size: None,
|
||||
single_chunk_filter_mask: None,
|
||||
dont_filter_partial_edge_chunks: false,
|
||||
}
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn v4_chunked_dont_filter_partial_edge_chunks_flag() {
|
||||
let mut buf = vec![4u8, 2]; // version=4, class=2
|
||||
buf.push(0x01); // flags bit 0 = don't filter partial edge chunks
|
||||
buf.push(2); // dimensionality=2
|
||||
buf.push(4); // dim_size_encoded_length=4
|
||||
buf.extend_from_slice(&5u32.to_le_bytes());
|
||||
buf.extend_from_slice(&4u32.to_le_bytes());
|
||||
buf.push(3); // Fixed Array
|
||||
buf.push(10); // max_dblk_page_nelmts_bits
|
||||
buf.extend_from_slice(&0x3000u64.to_le_bytes());
|
||||
match DataLayout::parse(&buf, 8, 8).unwrap() {
|
||||
DataLayout::Chunked {
|
||||
dont_filter_partial_edge_chunks,
|
||||
btree_address,
|
||||
..
|
||||
} => {
|
||||
assert!(dont_filter_partial_edge_chunks);
|
||||
assert_eq!(btree_address, Some(0x3000));
|
||||
}
|
||||
other => panic!("expected Chunked, got {other:?}"),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn v4_chunked_single_chunk_with_filters() {
|
||||
let mut buf = vec![4u8, 2]; // version=4, class=2
|
||||
@@ -705,6 +1091,7 @@ mod tests {
|
||||
chunk_index_type: Some(1),
|
||||
single_chunk_filtered_size: Some(1024),
|
||||
single_chunk_filter_mask: Some(0),
|
||||
dont_filter_partial_edge_chunks: false,
|
||||
}
|
||||
);
|
||||
}
|
||||
@@ -815,6 +1202,62 @@ mod tests {
|
||||
assert_eq!(v1.iter_linear_1d(8).unwrap(), vec![4, 5, 6, 7]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_vds_mappings_v1_shared_names() {
|
||||
// Written by HDF5 2.0 (h5py, libver=("v200", "v200")) for three
|
||||
// mappings from `a_rather_long_source_file.h5:a_rather_long_dataset_name`
|
||||
// and one from the same file: the entries carry flags 0x00, 0x03, 0x03
|
||||
// and 0x06, so names after the first are stored as entry indices.
|
||||
let blob: &[u8] = &[
|
||||
0x01, 0x04, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x61, 0x5f, 0x72, 0x61,
|
||||
0x74, 0x68, 0x65, 0x72, 0x5f, 0x6c, 0x6f, 0x6e, 0x67, 0x5f, 0x73, 0x6f, 0x75, 0x72,
|
||||
0x63, 0x65, 0x5f, 0x66, 0x69, 0x6c, 0x65, 0x2e, 0x68, 0x35, 0x00, 0x61, 0x5f, 0x72,
|
||||
0x61, 0x74, 0x68, 0x65, 0x72, 0x5f, 0x6c, 0x6f, 0x6e, 0x67, 0x5f, 0x64, 0x61, 0x74,
|
||||
0x61, 0x73, 0x65, 0x74, 0x5f, 0x6e, 0x61, 0x6d, 0x65, 0x00, 0x02, 0x00, 0x00, 0x00,
|
||||
0x03, 0x00, 0x00, 0x00, 0x01, 0x02, 0x01, 0x00, 0x00, 0x00, 0x00, 0x00, 0x01, 0x00,
|
||||
0x01, 0x00, 0x04, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00, 0x00, 0x00, 0x01, 0x02,
|
||||
0x02, 0x00, 0x00, 0x00, 0x00, 0x00, 0x01, 0x00, 0x01, 0x00, 0x01, 0x00, 0x00, 0x00,
|
||||
0x01, 0x00, 0x01, 0x00, 0x04, 0x00, 0x03, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
|
||||
0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03,
|
||||
0x00, 0x00, 0x00, 0x01, 0x02, 0x01, 0x00, 0x00, 0x00, 0x04, 0x00, 0x01, 0x00, 0x01,
|
||||
0x00, 0x04, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00, 0x00, 0x00, 0x01, 0x02, 0x02,
|
||||
0x00, 0x00, 0x00, 0x01, 0x00, 0x01, 0x00, 0x01, 0x00, 0x01, 0x00, 0x00, 0x00, 0x01,
|
||||
0x00, 0x01, 0x00, 0x04, 0x00, 0x03, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
|
||||
0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00,
|
||||
0x00, 0x00, 0x01, 0x02, 0x01, 0x00, 0x00, 0x00, 0x08, 0x00, 0x01, 0x00, 0x01, 0x00,
|
||||
0x04, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00, 0x00, 0x00, 0x01, 0x02, 0x02, 0x00,
|
||||
0x00, 0x00, 0x02, 0x00, 0x01, 0x00, 0x01, 0x00, 0x01, 0x00, 0x00, 0x00, 0x01, 0x00,
|
||||
0x01, 0x00, 0x04, 0x00, 0x06, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x02,
|
||||
0x00, 0x00, 0x00, 0x03, 0x00, 0x00, 0x00, 0x01, 0x02, 0x01, 0x00, 0x00, 0x00, 0x00,
|
||||
0x00, 0x01, 0x00, 0x01, 0x00, 0x04, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00, 0x00,
|
||||
0x00, 0x01, 0x02, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00, 0x01, 0x00, 0x01, 0x00, 0x01,
|
||||
0x00, 0x00, 0x00, 0x01, 0x00, 0x01, 0x00, 0x04, 0x00, 0x8e, 0xa7, 0xea, 0x7a,
|
||||
];
|
||||
let mappings = parse_vds_mappings(blob, 8).unwrap();
|
||||
let names: Vec<(&str, &str)> = mappings
|
||||
.iter()
|
||||
.map(|m| (m.source_file.as_str(), m.source_dataset.as_str()))
|
||||
.collect();
|
||||
let (file, dset) = ("a_rather_long_source_file.h5", "a_rather_long_dataset_name");
|
||||
assert_eq!(
|
||||
names,
|
||||
vec![(file, dset), (file, dset), (file, dset), (".", dset)]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_vds_mappings_v1_forward_reference_is_error() {
|
||||
// Entry 0 claiming to share entry 0's file name must not index past
|
||||
// the entries decoded so far.
|
||||
let mut blob = vec![0x01u8, 1, 0, 0, 0, 0, 0, 0, 0, 0x01];
|
||||
blob.extend_from_slice(&[0u8; 8]);
|
||||
blob.extend_from_slice(b"d\0");
|
||||
assert!(parse_vds_mappings(&blob, 8).is_err());
|
||||
// Unknown flag bits are refused.
|
||||
let blob = [0x01u8, 1, 0, 0, 0, 0, 0, 0, 0, 0x08, b'd', 0];
|
||||
assert!(parse_vds_mappings(&blob, 8).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_vds_mappings_external_v0() {
|
||||
// Block version 0 with an explicit (external) source file name.
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -7,7 +7,9 @@ extern crate alloc;
|
||||
use alloc::{vec, vec::Vec};
|
||||
|
||||
use crate::checksum::jenkins_lookup3;
|
||||
use crate::chunked_write::WrittenChunk;
|
||||
use crate::chunked_write::{
|
||||
WrittenChunk, filtered_chunk_size_len, push_addr, push_index_element, push_v4_chunk_dims,
|
||||
};
|
||||
|
||||
/// Serialize a v4 Extensible Array layout message.
|
||||
pub(crate) fn serialize_v4_extensible_array(
|
||||
@@ -24,45 +26,17 @@ pub(crate) fn serialize_v4_extensible_array(
|
||||
let ndims = chunk_dims.len() as u8 + 1;
|
||||
buf.push(ndims);
|
||||
|
||||
let max_dim = chunk_dims
|
||||
.iter()
|
||||
.map(|&d| d as u64)
|
||||
.chain(core::iter::once(element_size as u64))
|
||||
.max()
|
||||
.unwrap_or(1);
|
||||
let dim_encoded_len: u8 = if max_dim <= 0xFF {
|
||||
1
|
||||
} else if max_dim <= 0xFFFF {
|
||||
2
|
||||
} else {
|
||||
4
|
||||
};
|
||||
buf.push(dim_encoded_len);
|
||||
|
||||
for &d in chunk_dims {
|
||||
match dim_encoded_len {
|
||||
1 => buf.push(d as u8),
|
||||
2 => buf.extend_from_slice(&(d as u16).to_le_bytes()),
|
||||
4 => buf.extend_from_slice(&d.to_le_bytes()),
|
||||
_ => unreachable!("unexpected dim_encoded_len: {dim_encoded_len}"),
|
||||
}
|
||||
}
|
||||
match dim_encoded_len {
|
||||
1 => buf.push(element_size as u8),
|
||||
2 => buf.extend_from_slice(&(element_size as u16).to_le_bytes()),
|
||||
4 => buf.extend_from_slice(&element_size.to_le_bytes()),
|
||||
_ => unreachable!("unexpected dim_encoded_len: {dim_encoded_len}"),
|
||||
}
|
||||
push_v4_chunk_dims(&mut buf, chunk_dims, element_size);
|
||||
|
||||
// chunk index type = 4 (Extensible Array)
|
||||
buf.push(4);
|
||||
|
||||
// EA creation parameters (must match AEHD and HDF5 C library defaults)
|
||||
buf.push(32); // max_nelmts_bits
|
||||
buf.push(4); // idx_blk_elmts
|
||||
buf.push(4); // super_blk_min_data_ptrs
|
||||
buf.push(16); // data_blk_min_elmts
|
||||
buf.push(10); // max_dblk_page_nelmts_bits
|
||||
buf.push(MAX_NELMTS_BITS);
|
||||
buf.push(IDX_BLK_ELMTS);
|
||||
buf.push(SUP_BLK_MIN_DATA_PTRS);
|
||||
buf.push(DATA_BLK_MIN_ELMTS);
|
||||
buf.push(MAX_DBLK_PAGE_NELMTS_BITS);
|
||||
|
||||
// EA header address
|
||||
match offset_size {
|
||||
@@ -74,304 +48,281 @@ pub(crate) fn serialize_v4_extensible_array(
|
||||
buf
|
||||
}
|
||||
|
||||
// EA creation parameters — the HDF5 library's defaults for chunk indexes
|
||||
// (`H5D_EARRAY_*`); the layout message above and the header must agree.
|
||||
const MAX_NELMTS_BITS: u8 = 32;
|
||||
const IDX_BLK_ELMTS: u8 = 4;
|
||||
const SUP_BLK_MIN_DATA_PTRS: u8 = 4;
|
||||
const DATA_BLK_MIN_ELMTS: u8 = 16;
|
||||
const MAX_DBLK_PAGE_NELMTS_BITS: u8 = 10;
|
||||
|
||||
/// One data block of the array: its first element (relative to the end of
|
||||
/// the index block's own elements), element count, and address when it is
|
||||
/// allocated.
|
||||
struct DataBlock {
|
||||
start: usize,
|
||||
nelmts: usize,
|
||||
addr: Option<u64>,
|
||||
}
|
||||
|
||||
/// Build a complete Extensible Array at a known absolute address.
|
||||
///
|
||||
/// For simplicity, we put all elements inline in the index block when the
|
||||
/// number of chunks is small (up to idx_blk_elmts), otherwise use inline +
|
||||
/// direct data blocks.
|
||||
/// `slots[i]` is the element at linear index `i` (see `chunk_grid`); `None`
|
||||
/// marks an unallocated chunk. The first `IDX_BLK_ELMTS` elements live in
|
||||
/// the index block, the rest in data blocks grouped by super block level
|
||||
/// exactly as `H5EA__hdr_init` sizes them: level `u` has `2^(u/2)` data
|
||||
/// blocks of `DATA_BLK_MIN_ELMTS * 2^ceil(u/2)` elements. The data blocks of
|
||||
/// the first levels are addressed straight from the index block; later
|
||||
/// levels go through a super block (EASB). Data blocks larger than a page
|
||||
/// (`2^MAX_DBLK_PAGE_NELMTS_BITS` elements) are paged, with their page-init
|
||||
/// bits kept in the owning super block. Only blocks holding a defined element
|
||||
/// are allocated; the rest keep the undefined address, as in a file the
|
||||
/// library wrote.
|
||||
pub fn build_extensible_array_at(
|
||||
chunks: &[WrittenChunk],
|
||||
slots: &[Option<WrittenChunk>],
|
||||
offset_size: u8,
|
||||
length_size: u8,
|
||||
has_filters: bool,
|
||||
ea_base_address: u64,
|
||||
) -> Vec<u8> {
|
||||
let os = offset_size as usize;
|
||||
let num_elements = chunks.len();
|
||||
|
||||
// Compute element encoding size (same logic as Fixed Array)
|
||||
let chunk_size_bytes: usize = if has_filters {
|
||||
let max_raw = chunks.iter().map(|c| c.raw_size).max().unwrap_or(1);
|
||||
let log2_val = if max_raw <= 1 {
|
||||
0
|
||||
} else {
|
||||
63 - max_raw.leading_zeros()
|
||||
};
|
||||
let len = 1 + ((log2_val + 8) / 8) as usize;
|
||||
len.min(8)
|
||||
} else {
|
||||
0
|
||||
};
|
||||
|
||||
let elem_size = if has_filters {
|
||||
os + chunk_size_bytes + 4
|
||||
} else {
|
||||
os
|
||||
};
|
||||
|
||||
let chunk_size_bytes = has_filters.then(|| filtered_chunk_size_len(slots));
|
||||
let elem_size = os + chunk_size_bytes.map_or(0, |n| n + 4);
|
||||
let client_id: u8 = if has_filters { 1 } else { 0 };
|
||||
let arr_off_size = (MAX_NELMTS_BITS as usize).div_ceil(8);
|
||||
let page_nelmts = 1usize << MAX_DBLK_PAGE_NELMTS_BITS;
|
||||
let idx_blk = IDX_BLK_ELMTS as usize;
|
||||
|
||||
// EA creation parameters — must match HDF5 C library defaults exactly
|
||||
let max_nelmts_bits: u8 = 32;
|
||||
let idx_blk_elmts: u8 = 4;
|
||||
let min_dblk_nelmts: u8 = 16;
|
||||
let super_blk_min_nelmts: u8 = 4;
|
||||
let max_dblk_nelmts_bits: u8 = 10;
|
||||
// Elements past the last defined one are never realised
|
||||
// (`max_idx_set` is one past the highest index ever set).
|
||||
let max_idx_set = slots.iter().rposition(Option::is_some).map_or(0, |i| i + 1);
|
||||
let slots = &slots[..max_idx_set];
|
||||
let defined_in = |start: usize, n: usize| -> bool {
|
||||
let lo = idx_blk.saturating_add(start).min(slots.len());
|
||||
let hi = idx_blk
|
||||
.saturating_add(start)
|
||||
.saturating_add(n)
|
||||
.min(slots.len());
|
||||
slots[lo..hi].iter().any(Option::is_some)
|
||||
};
|
||||
|
||||
// EAHD size: fixed(12) + 6 stats(6*length_size) + addr(offset_size) + checksum(4)
|
||||
// Super block levels: (ndblks, dblk_nelmts, first element).
|
||||
let log2_dmin = (DATA_BLK_MIN_ELMTS as u32).trailing_zeros() as usize;
|
||||
let nsblks = 1 + MAX_NELMTS_BITS as usize - log2_dmin;
|
||||
let ndblk_addrs = 2 * (SUP_BLK_MIN_DATA_PTRS as usize - 1);
|
||||
let mut levels: Vec<(usize, usize, usize)> = Vec::with_capacity(nsblks);
|
||||
let mut start = 0usize;
|
||||
for u in 0..nsblks {
|
||||
let ndblks = 1usize << (u / 2);
|
||||
let nelmts = (DATA_BLK_MIN_ELMTS as usize) << u.div_ceil(2);
|
||||
levels.push((ndblks, nelmts, start));
|
||||
// Saturate: on 32-bit targets the last levels only need to compare
|
||||
// as "beyond the end".
|
||||
start = start.saturating_add(ndblks.saturating_mul(nelmts));
|
||||
}
|
||||
// Levels whose data blocks the index block addresses directly.
|
||||
let mut direct_levels = 0;
|
||||
let mut n = 0;
|
||||
while n < ndblk_addrs {
|
||||
n += levels[direct_levels].0;
|
||||
direct_levels += 1;
|
||||
}
|
||||
let nsblk_addrs = nsblks - direct_levels;
|
||||
|
||||
let dblk_size = |nelmts: usize| -> usize {
|
||||
let prefix = 4 + 1 + 1 + os + arr_off_size + 4;
|
||||
if nelmts > page_nelmts {
|
||||
prefix + (nelmts / page_nelmts) * (page_nelmts * elem_size + 4)
|
||||
} else {
|
||||
prefix + nelmts * elem_size
|
||||
}
|
||||
};
|
||||
let sblk_bitmap_len = |ndblks: usize, nelmts: usize| -> usize {
|
||||
if nelmts > page_nelmts {
|
||||
ndblks * (nelmts / page_nelmts).div_ceil(8)
|
||||
} else {
|
||||
0
|
||||
}
|
||||
};
|
||||
|
||||
// Plan addresses: header, index block, the direct data blocks, then each
|
||||
// allocated super block followed by its allocated data blocks.
|
||||
let aehd_size = 4 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 6 * length_size as usize + os + 4;
|
||||
let aeib_address = ea_base_address + aehd_size as u64;
|
||||
let aeib_size = 4 + 1 + 1 + os + idx_blk * elem_size + ndblk_addrs * os + nsblk_addrs * os + 4;
|
||||
let mut cursor = aeib_address + aeib_size as u64;
|
||||
|
||||
// Determine how many elements go inline vs data blocks
|
||||
let n_inline = (idx_blk_elmts as usize).min(num_elements);
|
||||
let remaining_after_inline = num_elements.saturating_sub(n_inline);
|
||||
let mut ndata_blks = 0u64;
|
||||
let mut data_blk_size = 0u64;
|
||||
let mut nsuper_blks = 0u64;
|
||||
let mut super_blk_size = 0u64;
|
||||
let mut realized = idx_blk as u64;
|
||||
|
||||
// Compute super block layout per HDF5 spec
|
||||
let sblk_min = super_blk_min_nelmts as usize;
|
||||
let log2_dblk_min = if min_dblk_nelmts <= 1 {
|
||||
0
|
||||
} else {
|
||||
(min_dblk_nelmts as u32).trailing_zeros() as usize
|
||||
};
|
||||
let nsblks = (max_nelmts_bits as usize).saturating_sub(log2_dblk_min) + 1;
|
||||
|
||||
// Direct data block addresses (from super blocks 0..sblk_min-1)
|
||||
let mut dblk_sizes: Vec<usize> = Vec::new();
|
||||
for sblk_idx in 0..sblk_min.min(nsblks) {
|
||||
let ndblks = 1usize << (sblk_idx / 2);
|
||||
let dblk_nelmts = (min_dblk_nelmts as usize) * (1 << sblk_idx.div_ceil(2));
|
||||
for _ in 0..ndblks {
|
||||
dblk_sizes.push(dblk_nelmts);
|
||||
let mut plan_dblk = |cursor: &mut u64, start: usize, nelmts: usize| -> DataBlock {
|
||||
let addr = defined_in(start, nelmts).then(|| {
|
||||
let a = *cursor;
|
||||
let size = dblk_size(nelmts) as u64;
|
||||
*cursor += size;
|
||||
ndata_blks += 1;
|
||||
data_blk_size += size;
|
||||
realized += nelmts as u64;
|
||||
a
|
||||
});
|
||||
DataBlock {
|
||||
start,
|
||||
nelmts,
|
||||
addr,
|
||||
}
|
||||
}
|
||||
let n_direct_dblks = dblk_sizes.len();
|
||||
|
||||
// Super block addresses (for super blocks sblk_min..nsblks-1)
|
||||
let n_sblk_addrs = nsblks.saturating_sub(sblk_min);
|
||||
|
||||
// EAIB size
|
||||
let aeib_size = 4
|
||||
+ 1
|
||||
+ 1
|
||||
+ os
|
||||
+ idx_blk_elmts as usize * elem_size
|
||||
+ n_direct_dblks * os
|
||||
+ n_sblk_addrs * os
|
||||
+ 4;
|
||||
|
||||
// Build AEHD
|
||||
let mut aehd = Vec::with_capacity(aehd_size);
|
||||
aehd.extend_from_slice(b"EAHD");
|
||||
aehd.push(0); // version
|
||||
aehd.push(client_id);
|
||||
aehd.push(elem_size as u8);
|
||||
aehd.push(max_nelmts_bits);
|
||||
aehd.push(idx_blk_elmts);
|
||||
aehd.push(min_dblk_nelmts);
|
||||
aehd.push(super_blk_min_nelmts);
|
||||
aehd.push(max_dblk_nelmts_bits);
|
||||
|
||||
// Count data blocks that will have chunks
|
||||
let n_active_dblks: u64 = if remaining_after_inline > 0 {
|
||||
let mut count = 0u64;
|
||||
let mut ci = n_inline;
|
||||
for &sz in &dblk_sizes {
|
||||
if ci < num_elements {
|
||||
count += 1;
|
||||
ci += sz;
|
||||
}
|
||||
}
|
||||
count
|
||||
} else {
|
||||
0
|
||||
};
|
||||
let blk_off_size = (max_nelmts_bits as usize).div_ceil(8);
|
||||
let aedb_header_overhead = 4 + 1 + 1 + os + blk_off_size + 4;
|
||||
let data_blk_total_size: u64 = if remaining_after_inline > 0 {
|
||||
let mut total = 0u64;
|
||||
let mut ci = n_inline;
|
||||
for &sz in &dblk_sizes {
|
||||
if ci < num_elements {
|
||||
total += (aedb_header_overhead + sz * elem_size) as u64;
|
||||
ci += sz;
|
||||
}
|
||||
}
|
||||
total
|
||||
} else {
|
||||
0
|
||||
};
|
||||
let max_idx_set: u64 = if remaining_after_inline > 0 {
|
||||
let mut max_set = idx_blk_elmts as u64;
|
||||
let mut ci = n_inline;
|
||||
for &sz in &dblk_sizes {
|
||||
if ci < num_elements {
|
||||
max_set += sz as u64;
|
||||
ci += sz;
|
||||
}
|
||||
}
|
||||
max_set
|
||||
} else {
|
||||
idx_blk_elmts as u64
|
||||
};
|
||||
|
||||
let mut direct: Vec<DataBlock> = Vec::with_capacity(ndblk_addrs);
|
||||
for &(ndblks, nelmts, first) in &levels[..direct_levels] {
|
||||
for k in 0..ndblks {
|
||||
direct.push(plan_dblk(&mut cursor, first + k * nelmts, nelmts));
|
||||
}
|
||||
}
|
||||
// (super block address, level, its data blocks)
|
||||
let mut supers: Vec<(Option<u64>, usize, Vec<DataBlock>)> = Vec::with_capacity(nsblk_addrs);
|
||||
for (u, &(ndblks, nelmts, first)) in levels.iter().enumerate().skip(direct_levels) {
|
||||
if !defined_in(first, ndblks.saturating_mul(nelmts)) {
|
||||
supers.push((None, u, Vec::new()));
|
||||
continue;
|
||||
}
|
||||
let sb_size =
|
||||
4 + 1 + 1 + os + arr_off_size + sblk_bitmap_len(ndblks, nelmts) + ndblks * os + 4;
|
||||
let sb_addr = cursor;
|
||||
cursor += sb_size as u64;
|
||||
nsuper_blks += 1;
|
||||
super_blk_size += sb_size as u64;
|
||||
let dblks = (0..ndblks)
|
||||
.map(|k| plan_dblk(&mut cursor, first + k * nelmts, nelmts))
|
||||
.collect();
|
||||
supers.push((Some(sb_addr), u, dblks));
|
||||
}
|
||||
|
||||
let slot = |i: usize| slots.get(i).and_then(Option::as_ref);
|
||||
let write_length = |buf: &mut Vec<u8>, val: u64| match length_size {
|
||||
4 => buf.extend_from_slice(&(val as u32).to_le_bytes()),
|
||||
_ => buf.extend_from_slice(&val.to_le_bytes()),
|
||||
};
|
||||
let write_addr = |buf: &mut Vec<u8>, val: u64| match offset_size {
|
||||
4 => buf.extend_from_slice(&(val as u32).to_le_bytes()),
|
||||
_ => buf.extend_from_slice(&val.to_le_bytes()),
|
||||
let write_addr_opt = |buf: &mut Vec<u8>, addr: Option<u64>| match addr {
|
||||
Some(a) => push_addr(buf, a, offset_size),
|
||||
None => buf.extend(core::iter::repeat_n(0xFF, os)),
|
||||
};
|
||||
let block_prefix = |buf: &mut Vec<u8>, sig: &[u8; 4], block_off: usize| {
|
||||
buf.extend_from_slice(sig);
|
||||
buf.push(0); // version
|
||||
buf.push(client_id);
|
||||
push_addr(buf, ea_base_address, offset_size);
|
||||
buf.extend_from_slice(&(block_off as u64).to_le_bytes()[..arr_off_size]);
|
||||
};
|
||||
// Serialise one data block (paged or not) onto `out`.
|
||||
let write_dblk = |out: &mut Vec<u8>, db: &DataBlock| {
|
||||
let at = out.len();
|
||||
block_prefix(out, b"EADB", db.start);
|
||||
let first = idx_blk + db.start;
|
||||
if db.nelmts > page_nelmts {
|
||||
// Paged: the prefix carries only its own checksum; each page
|
||||
// follows with one of its own.
|
||||
let sum = jenkins_lookup3(&out[at..]);
|
||||
out.extend_from_slice(&sum.to_le_bytes());
|
||||
for p in 0..db.nelmts / page_nelmts {
|
||||
let page_at = out.len();
|
||||
for e in 0..page_nelmts {
|
||||
let i = first + p * page_nelmts + e;
|
||||
push_index_element(out, slot(i), offset_size, chunk_size_bytes);
|
||||
}
|
||||
let sum = jenkins_lookup3(&out[page_at..]);
|
||||
out.extend_from_slice(&sum.to_le_bytes());
|
||||
}
|
||||
} else {
|
||||
for i in first..first + db.nelmts {
|
||||
push_index_element(out, slot(i), offset_size, chunk_size_bytes);
|
||||
}
|
||||
let sum = jenkins_lookup3(&out[at..]);
|
||||
out.extend_from_slice(&sum.to_le_bytes());
|
||||
}
|
||||
debug_assert_eq!(out.len() - at, dblk_size(db.nelmts));
|
||||
};
|
||||
|
||||
write_length(&mut aehd, 0);
|
||||
write_length(&mut aehd, 0);
|
||||
write_length(&mut aehd, n_active_dblks);
|
||||
write_length(&mut aehd, data_blk_total_size);
|
||||
write_length(&mut aehd, num_elements as u64);
|
||||
write_length(&mut aehd, max_idx_set);
|
||||
// Header (EAHD). The six statistics are, in order: super blocks, their
|
||||
// bytes, data blocks, their bytes, max index set, elements realised.
|
||||
let mut out = Vec::with_capacity((cursor - ea_base_address) as usize);
|
||||
out.extend_from_slice(b"EAHD");
|
||||
out.push(0); // version
|
||||
out.push(client_id);
|
||||
out.push(elem_size as u8);
|
||||
out.push(MAX_NELMTS_BITS);
|
||||
out.push(IDX_BLK_ELMTS);
|
||||
out.push(DATA_BLK_MIN_ELMTS);
|
||||
out.push(SUP_BLK_MIN_DATA_PTRS);
|
||||
out.push(MAX_DBLK_PAGE_NELMTS_BITS);
|
||||
write_length(&mut out, nsuper_blks);
|
||||
write_length(&mut out, super_blk_size);
|
||||
write_length(&mut out, ndata_blks);
|
||||
write_length(&mut out, data_blk_size);
|
||||
write_length(&mut out, max_idx_set as u64);
|
||||
write_length(&mut out, realized);
|
||||
push_addr(&mut out, aeib_address, offset_size);
|
||||
let sum = jenkins_lookup3(&out);
|
||||
out.extend_from_slice(&sum.to_le_bytes());
|
||||
debug_assert_eq!(out.len(), aehd_size);
|
||||
|
||||
write_addr(&mut aehd, aeib_address);
|
||||
|
||||
let aehd_checksum = jenkins_lookup3(&aehd);
|
||||
aehd.extend_from_slice(&aehd_checksum.to_le_bytes());
|
||||
debug_assert_eq!(aehd.len(), aehd_size);
|
||||
|
||||
// Build AEIB
|
||||
let mut aeib = Vec::with_capacity(aeib_size);
|
||||
aeib.extend_from_slice(b"EAIB");
|
||||
aeib.push(0);
|
||||
aeib.push(client_id);
|
||||
|
||||
match offset_size {
|
||||
4 => aeib.extend_from_slice(&(ea_base_address as u32).to_le_bytes()),
|
||||
8 => aeib.extend_from_slice(&ea_base_address.to_le_bytes()),
|
||||
_ => aeib.extend_from_slice(&ea_base_address.to_le_bytes()),
|
||||
// Index block (EAIB): inline elements, data block and super block
|
||||
// addresses.
|
||||
let ib_start = out.len();
|
||||
out.extend_from_slice(b"EAIB");
|
||||
out.push(0);
|
||||
out.push(client_id);
|
||||
push_addr(&mut out, ea_base_address, offset_size);
|
||||
for i in 0..idx_blk {
|
||||
push_index_element(&mut out, slot(i), offset_size, chunk_size_bytes);
|
||||
}
|
||||
|
||||
// Inline elements
|
||||
#[allow(clippy::needless_range_loop)]
|
||||
for i in 0..idx_blk_elmts as usize {
|
||||
if i < n_inline {
|
||||
write_chunk_element(
|
||||
&mut aeib,
|
||||
&chunks[i],
|
||||
offset_size,
|
||||
has_filters,
|
||||
chunk_size_bytes,
|
||||
);
|
||||
} else {
|
||||
write_undefined_element(&mut aeib, offset_size, has_filters, chunk_size_bytes);
|
||||
for db in &direct {
|
||||
write_addr_opt(&mut out, db.addr);
|
||||
}
|
||||
for (sb_addr, _, _) in &supers {
|
||||
write_addr_opt(&mut out, *sb_addr);
|
||||
}
|
||||
let sum = jenkins_lookup3(&out[ib_start..]);
|
||||
out.extend_from_slice(&sum.to_le_bytes());
|
||||
debug_assert_eq!(out.len() - ib_start, aeib_size);
|
||||
|
||||
// Data block addresses + build data blocks
|
||||
let mut data_blocks_buf = Vec::new();
|
||||
let dblks_base = aeib_address + aeib_size as u64;
|
||||
let mut dblk_cursor = dblks_base;
|
||||
let mut chunk_idx = n_inline;
|
||||
|
||||
for &nelmts in &dblk_sizes {
|
||||
if chunk_idx >= num_elements {
|
||||
match offset_size {
|
||||
4 => aeib.extend_from_slice(&u32::MAX.to_le_bytes()),
|
||||
8 => aeib.extend_from_slice(&u64::MAX.to_le_bytes()),
|
||||
_ => aeib.extend_from_slice(&u64::MAX.to_le_bytes()),
|
||||
for db in direct.iter().filter(|d| d.addr.is_some()) {
|
||||
write_dblk(&mut out, db);
|
||||
}
|
||||
for (sb_addr, u, dblks) in &supers {
|
||||
if sb_addr.is_none() {
|
||||
continue;
|
||||
}
|
||||
|
||||
match offset_size {
|
||||
4 => aeib.extend_from_slice(&(dblk_cursor as u32).to_le_bytes()),
|
||||
8 => aeib.extend_from_slice(&dblk_cursor.to_le_bytes()),
|
||||
_ => aeib.extend_from_slice(&dblk_cursor.to_le_bytes()),
|
||||
}
|
||||
|
||||
// Build EADB
|
||||
let mut aedb = Vec::new();
|
||||
aedb.extend_from_slice(b"EADB");
|
||||
aedb.push(0);
|
||||
aedb.push(client_id);
|
||||
match offset_size {
|
||||
4 => aedb.extend_from_slice(&(ea_base_address as u32).to_le_bytes()),
|
||||
8 => aedb.extend_from_slice(&ea_base_address.to_le_bytes()),
|
||||
_ => aedb.extend_from_slice(&ea_base_address.to_le_bytes()),
|
||||
}
|
||||
|
||||
let blk_off_size = (max_nelmts_bits as usize).div_ceil(8);
|
||||
let blk_off_val = (chunk_idx - n_inline) as u64;
|
||||
aedb.extend_from_slice(&blk_off_val.to_le_bytes()[..blk_off_size]);
|
||||
|
||||
for slot in 0..nelmts {
|
||||
if chunk_idx + slot < num_elements {
|
||||
write_chunk_element(
|
||||
&mut aedb,
|
||||
&chunks[chunk_idx + slot],
|
||||
offset_size,
|
||||
has_filters,
|
||||
chunk_size_bytes,
|
||||
);
|
||||
} else {
|
||||
write_undefined_element(&mut aedb, offset_size, has_filters, chunk_size_bytes);
|
||||
let (ndblks, nelmts, first) = levels[*u];
|
||||
let sb_start = out.len();
|
||||
block_prefix(&mut out, b"EASB", first);
|
||||
if nelmts > page_nelmts {
|
||||
// Page-init bits, `npages` per data block, packed MSB-first
|
||||
// (`H5VM_bit_set`): every page of an allocated data block is
|
||||
// written.
|
||||
let npages = nelmts / page_nelmts;
|
||||
let mut bitmap = vec![0u8; sblk_bitmap_len(ndblks, nelmts)];
|
||||
for (k, db) in dblks.iter().enumerate() {
|
||||
if db.addr.is_some() {
|
||||
for p in 0..npages {
|
||||
let bit = k * npages + p;
|
||||
bitmap[bit / 8] |= 0x80 >> (bit % 8);
|
||||
}
|
||||
}
|
||||
|
||||
let aedb_checksum = jenkins_lookup3(&aedb);
|
||||
aedb.extend_from_slice(&aedb_checksum.to_le_bytes());
|
||||
|
||||
dblk_cursor += aedb.len() as u64;
|
||||
data_blocks_buf.extend_from_slice(&aedb);
|
||||
chunk_idx += nelmts;
|
||||
}
|
||||
|
||||
// Super block addresses (all undefined)
|
||||
for _ in 0..n_sblk_addrs {
|
||||
match offset_size {
|
||||
4 => aeib.extend_from_slice(&u32::MAX.to_le_bytes()),
|
||||
8 => aeib.extend_from_slice(&u64::MAX.to_le_bytes()),
|
||||
_ => aeib.extend_from_slice(&u64::MAX.to_le_bytes()),
|
||||
out.extend_from_slice(&bitmap);
|
||||
}
|
||||
for db in dblks {
|
||||
write_addr_opt(&mut out, db.addr);
|
||||
}
|
||||
let sum = jenkins_lookup3(&out[sb_start..]);
|
||||
out.extend_from_slice(&sum.to_le_bytes());
|
||||
for db in dblks.iter().filter(|d| d.addr.is_some()) {
|
||||
write_dblk(&mut out, db);
|
||||
}
|
||||
}
|
||||
|
||||
let aeib_checksum = jenkins_lookup3(&aeib);
|
||||
aeib.extend_from_slice(&aeib_checksum.to_le_bytes());
|
||||
debug_assert_eq!(aeib.len(), aeib_size);
|
||||
|
||||
let mut combined = aehd;
|
||||
combined.extend_from_slice(&aeib);
|
||||
combined.extend_from_slice(&data_blocks_buf);
|
||||
combined
|
||||
}
|
||||
|
||||
fn write_chunk_element(
|
||||
buf: &mut Vec<u8>,
|
||||
chunk: &WrittenChunk,
|
||||
offset_size: u8,
|
||||
has_filters: bool,
|
||||
chunk_size_bytes: usize,
|
||||
) {
|
||||
match offset_size {
|
||||
4 => buf.extend_from_slice(&(chunk.address as u32).to_le_bytes()),
|
||||
8 => buf.extend_from_slice(&chunk.address.to_le_bytes()),
|
||||
_ => buf.extend_from_slice(&chunk.address.to_le_bytes()),
|
||||
}
|
||||
if has_filters {
|
||||
let cs_bytes = chunk.compressed_size.to_le_bytes();
|
||||
buf.extend_from_slice(&cs_bytes[..chunk_size_bytes]);
|
||||
buf.extend_from_slice(&chunk.filter_mask.to_le_bytes());
|
||||
}
|
||||
}
|
||||
|
||||
fn write_undefined_element(
|
||||
buf: &mut Vec<u8>,
|
||||
offset_size: u8,
|
||||
has_filters: bool,
|
||||
chunk_size_bytes: usize,
|
||||
) {
|
||||
let os = offset_size as usize;
|
||||
// Use extend with repeat to avoid heap-allocating a temporary Vec on each call.
|
||||
buf.extend(core::iter::repeat_n(0xFF, os));
|
||||
if has_filters {
|
||||
buf.extend(core::iter::repeat_n(0x00, chunk_size_bytes));
|
||||
buf.extend_from_slice(&0u32.to_le_bytes());
|
||||
}
|
||||
debug_assert_eq!(out.len() as u64, cursor - ea_base_address);
|
||||
out
|
||||
}
|
||||
|
||||
@@ -80,6 +80,9 @@ pub enum FormatError {
|
||||
InvalidLocalHeapSignature,
|
||||
/// Invalid local heap version.
|
||||
InvalidLocalHeapVersion(u8),
|
||||
/// A local heap's free list points outside its data segment (libhdf5:
|
||||
/// "bad heap free list").
|
||||
InvalidLocalHeapFreeList,
|
||||
/// Invalid B-tree v1 signature.
|
||||
InvalidBTreeSignature,
|
||||
/// Invalid B-tree node type.
|
||||
@@ -117,6 +120,14 @@ pub enum FormatError {
|
||||
/// A message is marked shared but was parsed without access to the file,
|
||||
/// so the reference to the real message could not be followed.
|
||||
UnresolvedSharedMessage,
|
||||
/// A shared-message reference points at an object header that holds no
|
||||
/// (unshared) message of the referenced type (raw message type id).
|
||||
SharedMessageTargetMissing(u16),
|
||||
/// A superblock was parsed at a non-zero offset of the buffer (the file
|
||||
/// has a user block of this many bytes). HDF5 addresses are relative to
|
||||
/// the superblock, so the buffer must start there: see
|
||||
/// `signature::split_user_block`.
|
||||
UserBlockNotStripped(u64),
|
||||
/// A selection does not fit the dataset it was applied to (wrong rank, or
|
||||
/// it reaches past a dimension's extent).
|
||||
SelectionOutOfBounds(String),
|
||||
@@ -190,6 +201,28 @@ pub enum FormatError {
|
||||
DuplicateDatasetName(String),
|
||||
/// Integer overflow in size computation (malformed data protection).
|
||||
Overflow(String),
|
||||
/// An object header that libhdf5 refuses to load (the reason is
|
||||
/// libhdf5's own error text): a misaligned or overrunning message, a
|
||||
/// wrong message count, contradictory message flags, a message of a
|
||||
/// class that cannot be shared flagged shareable, …
|
||||
InvalidObjectHeader(&'static str),
|
||||
/// A datatype message libhdf5 refuses to decode (the reason is
|
||||
/// libhdf5's own error text): size 0, bit fields outside the type,
|
||||
/// an empty enum name, a compound member outside its compound, …
|
||||
InvalidDatatype(String),
|
||||
/// A chunked layout whose chunk dimensions libhdf5 refuses: a zero
|
||||
/// dimension, a rank that does not match the dataspace, an element size
|
||||
/// that is not the datatype's, or a chunk of 4 GiB or more indexed by a
|
||||
/// version-1 B-tree.
|
||||
InvalidChunkDimensions(String),
|
||||
/// The superblock's end-of-file address lies past the end of the file:
|
||||
/// the file was truncated (libhdf5 refuses to open it).
|
||||
TruncatedFile {
|
||||
/// End of file recorded in the superblock (relative to byte 0).
|
||||
stored_eof: u64,
|
||||
/// The file's actual length in bytes.
|
||||
actual_len: u64,
|
||||
},
|
||||
}
|
||||
|
||||
impl fmt::Display for FormatError {
|
||||
@@ -270,6 +303,9 @@ impl fmt::Display for FormatError {
|
||||
FormatError::InvalidLocalHeapSignature => {
|
||||
write!(f, "invalid local heap signature")
|
||||
}
|
||||
FormatError::InvalidLocalHeapFreeList => {
|
||||
write!(f, "bad local heap free list")
|
||||
}
|
||||
FormatError::InvalidLocalHeapVersion(v) => {
|
||||
write!(f, "invalid local heap version: {v}")
|
||||
}
|
||||
@@ -339,6 +375,16 @@ impl fmt::Display for FormatError {
|
||||
FormatError::SelectionOutOfBounds(msg) => {
|
||||
write!(f, "selection out of bounds: {msg}")
|
||||
}
|
||||
FormatError::UserBlockNotStripped(n) => write!(
|
||||
f,
|
||||
"file has a {n}-byte user block: parse the bytes from the superblock on \
|
||||
(signature::split_user_block)"
|
||||
),
|
||||
FormatError::SharedMessageTargetMissing(t) => write!(
|
||||
f,
|
||||
"shared message reference points at an object header with no message of type \
|
||||
{t:#06x}"
|
||||
),
|
||||
FormatError::UnresolvedSharedMessage => write!(
|
||||
f,
|
||||
"message is shared but no file data was available to resolve it"
|
||||
@@ -382,9 +428,17 @@ impl fmt::Display for FormatError {
|
||||
FormatError::InvalidFilterPipelineVersion(v) => {
|
||||
write!(f, "invalid filter pipeline version: {v}")
|
||||
}
|
||||
FormatError::UnsupportedFilter(id) => {
|
||||
write!(f, "unsupported filter: {id}")
|
||||
}
|
||||
FormatError::UnsupportedFilter(id) => match crate::filter_registry::known_filter(*id) {
|
||||
Some((name, Some(feature))) => write!(
|
||||
f,
|
||||
"unsupported filter: {id} ({name}; this build lacks the `{feature}` feature)"
|
||||
),
|
||||
Some((name, None)) => write!(
|
||||
f,
|
||||
"unsupported filter: {id} ({name}, not implemented by clawhdf5)"
|
||||
),
|
||||
None => write!(f, "unsupported filter: {id}"),
|
||||
},
|
||||
FormatError::FilterError(msg) => {
|
||||
write!(f, "filter error: {msg}")
|
||||
}
|
||||
@@ -421,6 +475,25 @@ impl fmt::Display for FormatError {
|
||||
FormatError::Overflow(msg) => {
|
||||
write!(f, "integer overflow: {msg}")
|
||||
}
|
||||
FormatError::InvalidObjectHeader(why) => {
|
||||
write!(f, "corrupt object header: {why}")
|
||||
}
|
||||
FormatError::InvalidDatatype(why) => {
|
||||
write!(f, "invalid datatype: {why}")
|
||||
}
|
||||
FormatError::InvalidChunkDimensions(why) => {
|
||||
write!(f, "invalid chunk dimensions: {why}")
|
||||
}
|
||||
FormatError::TruncatedFile {
|
||||
stored_eof,
|
||||
actual_len,
|
||||
} => {
|
||||
write!(
|
||||
f,
|
||||
"truncated file: the superblock records end of file {stored_eof}, \
|
||||
but the file is {actual_len} bytes"
|
||||
)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -9,9 +9,35 @@ extern crate alloc;
|
||||
#[cfg(not(feature = "std"))]
|
||||
use alloc::{format, vec, vec::Vec};
|
||||
|
||||
use crate::chunk_grid::ChunkGrid;
|
||||
use crate::chunked_read::ChunkInfo;
|
||||
use crate::error::FormatError;
|
||||
|
||||
/// Verify the Jenkins lookup3 checksum stored immediately after
|
||||
/// `data[start..end]`, as every Extensible Array structure carries one.
|
||||
///
|
||||
/// A corrupt chunk index yields addresses pointing at the wrong bytes, so a
|
||||
/// mismatch is an error: otherwise the damage surfaces as plausible data read
|
||||
/// from the wrong chunk.
|
||||
#[cfg(feature = "checksum")]
|
||||
fn verify_checksum(data: &[u8], start: usize, end: usize) -> Result<(), FormatError> {
|
||||
ensure_len(data, end, 4)?;
|
||||
let stored = u32::from_le_bytes([data[end], data[end + 1], data[end + 2], data[end + 3]]);
|
||||
let computed = crate::checksum::jenkins_lookup3(&data[start..end]);
|
||||
if computed != stored {
|
||||
return Err(FormatError::ChecksumMismatch {
|
||||
expected: stored,
|
||||
computed,
|
||||
});
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[cfg(not(feature = "checksum"))]
|
||||
fn verify_checksum(_data: &[u8], _start: usize, _end: usize) -> Result<(), FormatError> {
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Parsed Extensible Array header (AEHD).
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct ExtensibleArrayHeader {
|
||||
@@ -145,6 +171,8 @@ impl ExtensibleArrayHeader {
|
||||
pos += ls; // skip nelmts
|
||||
pos += ls; // skip max_idx_set (6th stats field)
|
||||
let index_block_address = read_offset(d, pos, offset_size)?;
|
||||
pos += offset_size as usize;
|
||||
verify_checksum(file_data, offset, offset + pos)?;
|
||||
|
||||
Ok(ExtensibleArrayHeader {
|
||||
client_id,
|
||||
@@ -176,8 +204,7 @@ fn read_element(
|
||||
offset_size: u8,
|
||||
chunk_byte_size: u64,
|
||||
linear_index: usize,
|
||||
num_chunks_per_dim: &[u64],
|
||||
chunk_dimensions: &[u32],
|
||||
grid: &ChunkGrid,
|
||||
) -> Result<(Option<ChunkInfo>, usize), FormatError> {
|
||||
let os = offset_size as usize;
|
||||
|
||||
@@ -193,7 +220,10 @@ fn read_element(
|
||||
return Ok((None, os));
|
||||
}
|
||||
let address = read_offset(data, pos, offset_size)?;
|
||||
let offsets = index_to_chunk_offsets(linear_index, num_chunks_per_dim, chunk_dimensions);
|
||||
// A slot beyond the current extent is ignored, as the library does.
|
||||
let Some(offsets) = grid.offsets(linear_index as u64) else {
|
||||
return Ok((None, os));
|
||||
};
|
||||
Ok((
|
||||
Some(ChunkInfo {
|
||||
chunk_size: chunk_byte_size as u32,
|
||||
@@ -234,7 +264,9 @@ fn read_element(
|
||||
data[fm_off + 2],
|
||||
data[fm_off + 3],
|
||||
]);
|
||||
let offsets = index_to_chunk_offsets(linear_index, num_chunks_per_dim, chunk_dimensions);
|
||||
let Some(offsets) = grid.offsets(linear_index as u64) else {
|
||||
return Ok((None, elem_total));
|
||||
};
|
||||
Ok((
|
||||
Some(ChunkInfo {
|
||||
chunk_size: chunk_size as u32,
|
||||
@@ -247,28 +279,41 @@ fn read_element(
|
||||
}
|
||||
}
|
||||
|
||||
/// Convert a linear chunk index to N-dimensional chunk offsets in dataset space.
|
||||
fn index_to_chunk_offsets(
|
||||
index: usize,
|
||||
num_chunks_per_dim: &[u64],
|
||||
chunk_dimensions: &[u32],
|
||||
) -> Vec<u64> {
|
||||
let rank = num_chunks_per_dim.len();
|
||||
let mut offsets = vec![0u64; rank];
|
||||
let mut remaining = index as u64;
|
||||
for d in (0..rank).rev() {
|
||||
let nchunks = num_chunks_per_dim[d];
|
||||
if nchunks == 0 {
|
||||
continue;
|
||||
}
|
||||
let chunk_idx = remaining % nchunks;
|
||||
remaining /= nchunks;
|
||||
offsets[d] = chunk_idx * chunk_dimensions[d] as u64;
|
||||
}
|
||||
offsets
|
||||
/// Collect elements from a data block at the given offset.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
/// Layout of super block `u`, per the HDF5 spec: the number of data blocks it
|
||||
/// owns and how many elements each of them holds.
|
||||
///
|
||||
/// `ndblks` and `dblk_nelmts` each double every *other* level, a half-step
|
||||
/// apart, so the blocks grow as 1x16, 1x32, 2x32, 2x64, 4x64 ... for a
|
||||
/// 16-element minimum. Treating either as doubling every level (the previous
|
||||
/// implementation) puts every element after the first data block at the wrong
|
||||
/// index.
|
||||
fn sblk_info(u: usize, data_blk_min_elmts: usize) -> Option<(usize, usize)> {
|
||||
let ndblks = 1usize.checked_shl((u / 2) as u32)?;
|
||||
let dblk_nelmts = 1usize
|
||||
.checked_shl(u.div_ceil(2) as u32)?
|
||||
.checked_mul(data_blk_min_elmts)?;
|
||||
Some((ndblks, dblk_nelmts))
|
||||
}
|
||||
|
||||
/// Collect elements from a data block at the given offset.
|
||||
/// Width of the "offset of the block in the array" field carried by super and
|
||||
/// data blocks (`hdr->arr_off_size`).
|
||||
fn arr_off_size(header: &ExtensibleArrayHeader) -> usize {
|
||||
(header.max_nelmts_bits as usize).div_ceil(8)
|
||||
}
|
||||
|
||||
/// Elements per data block page, once a data block is large enough to be paged.
|
||||
fn page_nelmts(header: &ExtensibleArrayHeader) -> Option<usize> {
|
||||
1usize.checked_shl(u32::from(header.max_dblk_nelmts_bits))
|
||||
}
|
||||
|
||||
/// Read the elements of one data block (EADB).
|
||||
///
|
||||
/// `page_init` is the owning super block's page-init bitmap and `first_page`
|
||||
/// this block's first bit in it; both are only consulted when the block is
|
||||
/// paged. The bitmap lives in the super block, not here — a paged data block
|
||||
/// stores only its prefix, then one slot per page.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
fn read_data_block_elements(
|
||||
file_data: &[u8],
|
||||
@@ -278,119 +323,101 @@ fn read_data_block_elements(
|
||||
offset_size: u8,
|
||||
chunk_byte_size: u64,
|
||||
start_index: usize,
|
||||
num_chunks_per_dim: &[u64],
|
||||
chunk_dimensions: &[u32],
|
||||
grid: &ChunkGrid,
|
||||
page_init: &[u8],
|
||||
first_page: usize,
|
||||
) -> Result<Vec<ChunkInfo>, FormatError> {
|
||||
// AEDB: signature(4) + version(1) + client_id(1) + header_address(offset_size)
|
||||
let db_header_size = 4 + 1 + 1 + offset_size as usize;
|
||||
// EADB: signature(4) + version(1) + client_id(1) + header_address(offset_size)
|
||||
// + block offset(arr_off_size)
|
||||
let db_header_size = 4 + 1 + 1 + offset_size as usize + arr_off_size(header);
|
||||
ensure_len(file_data, db_offset, db_header_size)?;
|
||||
|
||||
let d = &file_data[db_offset..];
|
||||
if &d[0..4] != b"EADB" {
|
||||
if &file_data[db_offset..db_offset + 4] != b"EADB" {
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"invalid Extensible Array data block signature".into(),
|
||||
));
|
||||
}
|
||||
// Skip version(1) + client_id(1) + header_address(offset_size) + block_offset
|
||||
// Block offset is encoded in ceil(max_nelmts_bits/8) bytes
|
||||
let blk_off_size = (header.max_nelmts_bits as usize).div_ceil(8);
|
||||
let mut pos = db_offset + db_header_size + blk_off_size;
|
||||
|
||||
// Check if paged
|
||||
if header.max_nelmts_bits >= usize::BITS as u8 {
|
||||
return Err(FormatError::Overflow(
|
||||
"max_nelmts_bits exceeds usize bit width".into(),
|
||||
));
|
||||
}
|
||||
let page_nelmts = 1usize << header.max_nelmts_bits;
|
||||
let is_paged = nelmts > page_nelmts;
|
||||
let mut pos = db_offset + db_header_size;
|
||||
let page = page_nelmts(header).ok_or_else(|| {
|
||||
FormatError::Overflow("Extensible Array page element count overflows usize".into())
|
||||
})?;
|
||||
|
||||
let mut chunks = Vec::new();
|
||||
|
||||
if !is_paged {
|
||||
for i in 0..nelmts {
|
||||
let read_run = |from: usize,
|
||||
count: usize,
|
||||
first_index: usize,
|
||||
chunks: &mut Vec<ChunkInfo>|
|
||||
-> Result<usize, FormatError> {
|
||||
let mut p = from;
|
||||
for i in 0..count {
|
||||
let (info, consumed) = read_element(
|
||||
file_data,
|
||||
pos,
|
||||
p,
|
||||
header.client_id,
|
||||
header.element_size,
|
||||
offset_size,
|
||||
chunk_byte_size,
|
||||
start_index + i,
|
||||
num_chunks_per_dim,
|
||||
chunk_dimensions,
|
||||
first_index + i,
|
||||
grid,
|
||||
)?;
|
||||
if let Some(ci) = info {
|
||||
chunks.push(ci);
|
||||
}
|
||||
pos += consumed;
|
||||
p += consumed;
|
||||
}
|
||||
} else {
|
||||
// Paged: elements are split into pages of page_nelmts.
|
||||
// After the data block header comes a page bitmap, then each page
|
||||
// has page_nelmts elements followed by a 4-byte checksum.
|
||||
let npages = nelmts.div_ceil(page_nelmts);
|
||||
// Page bitmap: ceil(npages / 8) bytes
|
||||
let bitmap_size = npages.div_ceil(8);
|
||||
// Read bitmap
|
||||
if pos + bitmap_size > file_data.len() {
|
||||
return Err(FormatError::UnexpectedEof {
|
||||
expected: pos + bitmap_size,
|
||||
available: file_data.len(),
|
||||
});
|
||||
}
|
||||
let bitmap = &file_data[pos..pos + bitmap_size];
|
||||
pos += bitmap_size;
|
||||
Ok(p)
|
||||
};
|
||||
|
||||
if nelmts <= page {
|
||||
// Prefix and elements are covered by one checksum.
|
||||
let elem_bytes = if header.client_id == 0 {
|
||||
offset_size as usize
|
||||
} else {
|
||||
header.element_size as usize
|
||||
};
|
||||
|
||||
let mut global_idx = start_index;
|
||||
for page_idx in 0..npages {
|
||||
let byte_idx = page_idx / 8;
|
||||
let bit_idx = page_idx % 8;
|
||||
let page_has_data = (bitmap[byte_idx] >> bit_idx) & 1 != 0;
|
||||
|
||||
let elems_this_page = if page_idx == npages - 1 {
|
||||
let remainder = nelmts % page_nelmts;
|
||||
if remainder == 0 {
|
||||
page_nelmts
|
||||
} else {
|
||||
remainder
|
||||
let end = nelmts
|
||||
.checked_mul(elem_bytes)
|
||||
.and_then(|b| pos.checked_add(b))
|
||||
.ok_or_else(|| FormatError::Overflow("Extensible Array data block span".into()))?;
|
||||
verify_checksum(file_data, db_offset, end)?;
|
||||
read_run(pos, nelmts, start_index, &mut chunks)?;
|
||||
return Ok(chunks);
|
||||
}
|
||||
} else {
|
||||
page_nelmts
|
||||
};
|
||||
|
||||
if page_has_data {
|
||||
for i in 0..elems_this_page {
|
||||
let (info, consumed) = read_element(
|
||||
file_data,
|
||||
pos,
|
||||
header.client_id,
|
||||
header.element_size,
|
||||
offset_size,
|
||||
chunk_byte_size,
|
||||
global_idx + i,
|
||||
num_chunks_per_dim,
|
||||
chunk_dimensions,
|
||||
)?;
|
||||
if let Some(ci) = info {
|
||||
chunks.push(ci);
|
||||
}
|
||||
pos += consumed;
|
||||
}
|
||||
// Skip page checksum (4 bytes)
|
||||
// Paged: the prefix ends with its own checksum, then one slot per page,
|
||||
// each holding `page` elements followed by a checksum. Pages whose bit is
|
||||
// clear were never written; their slot still occupies the file, so stride
|
||||
// over it rather than reading zeros as addresses.
|
||||
verify_checksum(file_data, db_offset, pos)?;
|
||||
pos += 4;
|
||||
let elem_bytes = if header.client_id == 0 {
|
||||
offset_size as usize
|
||||
} else {
|
||||
// Empty page: skip all elements + checksum
|
||||
pos += elems_this_page * elem_bytes + 4;
|
||||
}
|
||||
global_idx += elems_this_page;
|
||||
header.element_size as usize
|
||||
};
|
||||
let page_stride = page
|
||||
.checked_mul(elem_bytes)
|
||||
.and_then(|b| b.checked_add(4))
|
||||
.ok_or_else(|| FormatError::Overflow("Extensible Array page stride".into()))?;
|
||||
let npages = nelmts.div_ceil(page);
|
||||
for p in 0..npages {
|
||||
// One bit per page across the whole super block, packed contiguously
|
||||
// and MSB-first within each byte, as H5VM_bit_get reads it.
|
||||
let bit = first_page + p;
|
||||
let initialised = page_init
|
||||
.get(bit / 8)
|
||||
.is_some_and(|byte| byte & (0x80 >> (bit % 8)) != 0);
|
||||
if initialised {
|
||||
let count = core::cmp::min(page, nelmts - p * page);
|
||||
// Each page carries its own checksum, over a full page's worth of
|
||||
// slots even when the last one holds fewer live elements.
|
||||
verify_checksum(file_data, pos, pos + page * elem_bytes)?;
|
||||
read_run(pos, count, start_index + p * page, &mut chunks)?;
|
||||
}
|
||||
pos = pos
|
||||
.checked_add(page_stride)
|
||||
.ok_or_else(|| FormatError::Overflow("Extensible Array page offset".into()))?;
|
||||
}
|
||||
|
||||
Ok(chunks)
|
||||
@@ -404,53 +431,100 @@ pub fn read_extensible_array_chunks(
|
||||
file_data: &[u8],
|
||||
header: &ExtensibleArrayHeader,
|
||||
dataset_dims: &[u64],
|
||||
max_dims: Option<&[u64]>,
|
||||
chunk_dimensions: &[u32],
|
||||
element_size: u32,
|
||||
offset_size: u8,
|
||||
_length_size: u8,
|
||||
) -> Result<Vec<ChunkInfo>, FormatError> {
|
||||
let rank = chunk_dimensions.len();
|
||||
let os = offset_size as usize;
|
||||
|
||||
let mut num_chunks_per_dim = Vec::with_capacity(rank);
|
||||
for d in 0..rank {
|
||||
let ch_dim = chunk_dimensions[d] as u64;
|
||||
if ch_dim == 0 {
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"chunk dimension is zero".into(),
|
||||
));
|
||||
}
|
||||
let ds_dim = dataset_dims[d];
|
||||
num_chunks_per_dim.push(ds_dim.div_ceil(ch_dim));
|
||||
}
|
||||
// Linear indexes follow the maximum dimensions, with the unlimited
|
||||
// dimension swizzled to the slowest position (see `chunk_grid`).
|
||||
let dims_u64: Vec<u64> = chunk_dimensions.iter().map(|&d| d as u64).collect();
|
||||
let grid = ChunkGrid::extensible_array(dataset_dims, max_dims, &dims_u64)?;
|
||||
let grid = &grid;
|
||||
|
||||
let chunk_byte_size: u64 =
|
||||
chunk_dimensions.iter().map(|&d| d as u64).product::<u64>() * element_size as u64;
|
||||
|
||||
// Parse index block (AEIB)
|
||||
// Parse index block (EAIB): signature(4) + version(1) + client_id(1)
|
||||
// + header address(offset_size), then the inline elements, then the
|
||||
// direct data block addresses, then the super block addresses.
|
||||
let ib_offset = header.index_block_address as usize;
|
||||
let ib_header_size = 4 + 1 + 1 + offset_size as usize; // sig + ver + client + hdr_addr
|
||||
let ib_header_size = 4 + 1 + 1 + os;
|
||||
ensure_len(file_data, ib_offset, ib_header_size)?;
|
||||
|
||||
let ib = &file_data[ib_offset..];
|
||||
if &ib[0..4] != b"EAIB" {
|
||||
if &file_data[ib_offset..ib_offset + 4] != b"EAIB" {
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"invalid Extensible Array index block signature".into(),
|
||||
));
|
||||
}
|
||||
// Skip version(1) + client_id(1) + header_address(offset_size)
|
||||
let mut pos = ib_offset + ib_header_size;
|
||||
|
||||
let mut chunks = Vec::new();
|
||||
let mut global_index = 0usize;
|
||||
let total_elements = header.num_elements as usize;
|
||||
|
||||
// 1. Read inline elements in index block
|
||||
let n_inline = header.idx_blk_elmts as usize;
|
||||
for i in 0..n_inline {
|
||||
if global_index + i >= total_elements {
|
||||
break;
|
||||
let dmin = header.min_dblk_nelmts as usize;
|
||||
if dmin == 0 || !dmin.is_power_of_two() {
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"Extensible Array data block minimum is not a power of two".into(),
|
||||
));
|
||||
}
|
||||
// nsblks = 1 + (max_nelmts_bits - log2(data_blk_min_elmts)), and the index
|
||||
// block holds 2 * (sup_blk_min_data_ptrs - 1) data block addresses.
|
||||
let log2_dmin = dmin.trailing_zeros() as usize;
|
||||
let nsblks = 1 + (header.max_nelmts_bits as usize).saturating_sub(log2_dmin);
|
||||
let ndblk_addrs = 2 * (header.super_blk_min_nelmts as usize).saturating_sub(1);
|
||||
|
||||
// The data blocks listed directly in the index block are the first
|
||||
// `ndblk_addrs` in super-block order, each sized by the level it belongs
|
||||
// to; the super block addresses that follow resume at the next level.
|
||||
let mut direct: Vec<usize> = Vec::with_capacity(ndblk_addrs);
|
||||
let mut level = 0usize;
|
||||
while direct.len() < ndblk_addrs {
|
||||
if level >= nsblks {
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"Extensible Array index block claims more data blocks than the array has".into(),
|
||||
));
|
||||
}
|
||||
let (ndblks, dblk_nelmts) = sblk_info(level, dmin).ok_or_else(|| {
|
||||
FormatError::Overflow("Extensible Array super block layout overflows usize".into())
|
||||
})?;
|
||||
for _ in 0..ndblks {
|
||||
direct.push(dblk_nelmts);
|
||||
}
|
||||
level += 1;
|
||||
}
|
||||
if direct.len() != ndblk_addrs {
|
||||
// A partial level in the index block is not a layout HDF5 produces,
|
||||
// and guessing where the super blocks resume would misplace elements.
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"Extensible Array index block ends mid super block".into(),
|
||||
));
|
||||
}
|
||||
|
||||
// One checksum covers the prefix, every inline element slot, and every
|
||||
// data block and super block address.
|
||||
let elem_bytes = if header.client_id == 0 {
|
||||
os
|
||||
} else {
|
||||
header.element_size as usize
|
||||
};
|
||||
let ib_end = (header.idx_blk_elmts as usize)
|
||||
.checked_mul(elem_bytes)
|
||||
.and_then(|b| pos.checked_add(b))
|
||||
.and_then(|p| {
|
||||
ndblk_addrs
|
||||
.checked_add(nsblks - level)
|
||||
.and_then(|n| n.checked_mul(os).and_then(|b| p.checked_add(b)))
|
||||
})
|
||||
.ok_or_else(|| FormatError::Overflow("Extensible Array index block span".into()))?;
|
||||
verify_checksum(file_data, ib_offset, ib_end)?;
|
||||
|
||||
// 1. Elements stored inline in the index block.
|
||||
let n_inline = (header.idx_blk_elmts as usize).min(total_elements);
|
||||
for i in 0..n_inline {
|
||||
let (info, consumed) = read_element(
|
||||
file_data,
|
||||
pos,
|
||||
@@ -458,174 +532,104 @@ pub fn read_extensible_array_chunks(
|
||||
header.element_size,
|
||||
offset_size,
|
||||
chunk_byte_size,
|
||||
global_index + i,
|
||||
&num_chunks_per_dim,
|
||||
chunk_dimensions,
|
||||
i,
|
||||
grid,
|
||||
)?;
|
||||
if let Some(ci) = info {
|
||||
chunks.push(ci);
|
||||
}
|
||||
pos += consumed;
|
||||
}
|
||||
global_index += n_inline.min(total_elements);
|
||||
|
||||
// If all elements were inline, we're done
|
||||
let mut global_index = n_inline;
|
||||
if global_index >= total_elements {
|
||||
return Ok(chunks);
|
||||
}
|
||||
|
||||
// Compute data block and super block counts
|
||||
let min_dblk = header.min_dblk_nelmts as usize;
|
||||
let sblk_min = header.super_blk_min_nelmts as usize;
|
||||
|
||||
// The first sblk_min super block levels have their data blocks listed directly
|
||||
// in the index block. Compute their sizes.
|
||||
let mut n_direct_dblks = 0usize;
|
||||
let mut dblk_sizes: Vec<usize> = Vec::new();
|
||||
{
|
||||
let mut nelmts = min_dblk;
|
||||
for sb_level in 0..sblk_min {
|
||||
if sb_level >= usize::BITS as usize {
|
||||
return Err(FormatError::Overflow(
|
||||
"sb_level exceeds usize bit width".into(),
|
||||
// 2. Data blocks listed directly in the index block.
|
||||
for &dblk_nelmts in &direct {
|
||||
if global_index >= total_elements {
|
||||
return Ok(chunks);
|
||||
}
|
||||
ensure_len(file_data, pos, os)?;
|
||||
let addr = read_offset(file_data, pos, offset_size)?;
|
||||
pos += os;
|
||||
if !is_undefined_addr(addr, offset_size) {
|
||||
if dblk_nelmts > page_nelmts(header).unwrap_or(usize::MAX) {
|
||||
// Would need a page-init bitmap, which only a super block
|
||||
// carries. HDF5 never pages these small early blocks.
|
||||
return Err(FormatError::ChunkedReadError(
|
||||
"Extensible Array index block references a paged data block".into(),
|
||||
));
|
||||
}
|
||||
let ndblks = 1usize << sb_level;
|
||||
for _ in 0..ndblks {
|
||||
dblk_sizes.push(nelmts);
|
||||
n_direct_dblks += 1;
|
||||
}
|
||||
if sb_level > 0 {
|
||||
nelmts *= 2;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Read direct data block addresses from index block
|
||||
let mut dblk_addrs: Vec<u64> = Vec::with_capacity(n_direct_dblks);
|
||||
for _ in 0..n_direct_dblks {
|
||||
if pos + os > file_data.len() {
|
||||
break;
|
||||
}
|
||||
let addr = read_offset(file_data, pos, offset_size)?;
|
||||
dblk_addrs.push(addr);
|
||||
pos += os;
|
||||
}
|
||||
|
||||
// Read elements from direct data blocks
|
||||
for (i, &addr) in dblk_addrs.iter().enumerate() {
|
||||
if i >= dblk_sizes.len() {
|
||||
break;
|
||||
}
|
||||
let nelmts = dblk_sizes[i];
|
||||
if is_undefined_addr(addr, offset_size) {
|
||||
global_index += nelmts;
|
||||
continue;
|
||||
}
|
||||
let block_chunks = read_data_block_elements(
|
||||
chunks.extend(read_data_block_elements(
|
||||
file_data,
|
||||
addr as usize,
|
||||
nelmts,
|
||||
dblk_nelmts,
|
||||
header,
|
||||
offset_size,
|
||||
chunk_byte_size,
|
||||
global_index,
|
||||
&num_chunks_per_dim,
|
||||
chunk_dimensions,
|
||||
)?;
|
||||
chunks.extend(block_chunks);
|
||||
global_index += nelmts;
|
||||
grid,
|
||||
&[],
|
||||
0,
|
||||
)?);
|
||||
}
|
||||
global_index += dblk_nelmts;
|
||||
}
|
||||
|
||||
// Remaining elements are in super blocks
|
||||
let total_in_ib_and_direct: usize = n_inline + dblk_sizes.iter().sum::<usize>();
|
||||
if total_elements <= total_in_ib_and_direct {
|
||||
return Ok(chunks);
|
||||
}
|
||||
let remaining_elements = total_elements - total_in_ib_and_direct;
|
||||
|
||||
// Compute super block layout
|
||||
let mut sb_addrs: Vec<u64> = Vec::new();
|
||||
let mut sb_infos: Vec<(usize, usize)> = Vec::new();
|
||||
{
|
||||
let mut covered = 0usize;
|
||||
let mut sb_level = sblk_min;
|
||||
let mut nelmts_per_dblk = min_dblk;
|
||||
for lev in 0..sblk_min {
|
||||
if lev > 0 {
|
||||
nelmts_per_dblk *= 2;
|
||||
}
|
||||
}
|
||||
|
||||
while covered < remaining_elements {
|
||||
if sb_level >= usize::BITS as usize {
|
||||
return Err(FormatError::Overflow(
|
||||
"sb_level exceeds usize bit width".into(),
|
||||
));
|
||||
}
|
||||
let ndblks = 1usize << sb_level;
|
||||
nelmts_per_dblk *= 2;
|
||||
let total_in_sb = ndblks * nelmts_per_dblk;
|
||||
sb_infos.push((ndblks, nelmts_per_dblk));
|
||||
covered += total_in_sb;
|
||||
sb_level += 1;
|
||||
}
|
||||
}
|
||||
|
||||
// Read super block addresses from index block
|
||||
for _ in 0..sb_infos.len() {
|
||||
if pos + os > file_data.len() {
|
||||
// 3. Everything else lives in super blocks, one address per remaining
|
||||
// level, starting at the level after the direct data blocks.
|
||||
for u in level..nsblks {
|
||||
if global_index >= total_elements {
|
||||
break;
|
||||
}
|
||||
let addr = read_offset(file_data, pos, offset_size)?;
|
||||
sb_addrs.push(addr);
|
||||
ensure_len(file_data, pos, os)?;
|
||||
let sb_addr = read_offset(file_data, pos, offset_size)?;
|
||||
pos += os;
|
||||
}
|
||||
|
||||
// Process each super block
|
||||
for (sb_idx, &sb_addr) in sb_addrs.iter().enumerate() {
|
||||
let (ndblks, nelmts_per_dblk) = sb_infos[sb_idx];
|
||||
if is_undefined_addr(sb_addr, offset_size) {
|
||||
global_index += ndblks * nelmts_per_dblk;
|
||||
continue;
|
||||
}
|
||||
let sb_chunks = read_super_block(
|
||||
let (ndblks, dblk_nelmts) = sblk_info(u, dmin).ok_or_else(|| {
|
||||
FormatError::Overflow("Extensible Array super block layout overflows usize".into())
|
||||
})?;
|
||||
if !is_undefined_addr(sb_addr, offset_size) {
|
||||
chunks.extend(read_super_block(
|
||||
file_data,
|
||||
sb_addr as usize,
|
||||
ndblks,
|
||||
nelmts_per_dblk,
|
||||
dblk_nelmts,
|
||||
header,
|
||||
offset_size,
|
||||
chunk_byte_size,
|
||||
global_index,
|
||||
&num_chunks_per_dim,
|
||||
chunk_dimensions,
|
||||
)?;
|
||||
chunks.extend(sb_chunks);
|
||||
global_index += ndblks * nelmts_per_dblk;
|
||||
grid,
|
||||
)?);
|
||||
}
|
||||
global_index =
|
||||
global_index.saturating_add(ndblks.checked_mul(dblk_nelmts).ok_or_else(|| {
|
||||
FormatError::Overflow("Extensible Array super block span".into())
|
||||
})?);
|
||||
}
|
||||
|
||||
Ok(chunks)
|
||||
}
|
||||
|
||||
/// Read a super block (AESB) and its data blocks.
|
||||
/// Read a super block (EASB) and the data blocks it owns.
|
||||
///
|
||||
/// On disk: signature(4) + version(1) + client_id(1) + header address
|
||||
/// + block offset + the page-init bitmap for every data block it owns
|
||||
/// + one address per data block + checksum.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
fn read_super_block(
|
||||
file_data: &[u8],
|
||||
sb_offset: usize,
|
||||
ndblks: usize,
|
||||
nelmts_per_dblk: usize,
|
||||
dblk_nelmts: usize,
|
||||
header: &ExtensibleArrayHeader,
|
||||
offset_size: u8,
|
||||
chunk_byte_size: u64,
|
||||
start_index: usize,
|
||||
num_chunks_per_dim: &[u64],
|
||||
chunk_dimensions: &[u32],
|
||||
grid: &ChunkGrid,
|
||||
) -> Result<Vec<ChunkInfo>, FormatError> {
|
||||
let os = offset_size as usize;
|
||||
|
||||
// AESB: signature(4) + version(1) + client_id(1) + header_address(offset_size)
|
||||
let sb_header_size = 4 + 1 + 1 + os;
|
||||
let sb_header_size = 4 + 1 + 1 + os + arr_off_size(header);
|
||||
ensure_len(file_data, sb_offset, sb_header_size)?;
|
||||
|
||||
if &file_data[sb_offset..sb_offset + 4] != b"EASB" {
|
||||
@@ -634,43 +638,56 @@ fn read_super_block(
|
||||
));
|
||||
}
|
||||
|
||||
let mut pos = sb_offset + sb_header_size;
|
||||
|
||||
// Read data block addresses
|
||||
let mut dblk_addrs: Vec<u64> = Vec::with_capacity(ndblks);
|
||||
for _ in 0..ndblks {
|
||||
if pos + os > file_data.len() {
|
||||
return Err(FormatError::UnexpectedEof {
|
||||
expected: pos + os,
|
||||
available: file_data.len(),
|
||||
});
|
||||
}
|
||||
let addr = read_offset(file_data, pos, offset_size)?;
|
||||
dblk_addrs.push(addr);
|
||||
pos += os;
|
||||
}
|
||||
// Page-init bitmap: one bit per page, `npages` bits per data block, packed
|
||||
// contiguously. HDF5 sizes the buffer `ndblks * ceil(npages / 8)`, which
|
||||
// is bigger than the bits need when `npages` is not a multiple of eight.
|
||||
// Zero-sized unless this level's data blocks are paged.
|
||||
let page = page_nelmts(header).ok_or_else(|| {
|
||||
FormatError::Overflow("Extensible Array page element count overflows usize".into())
|
||||
})?;
|
||||
let npages = if dblk_nelmts > page {
|
||||
dblk_nelmts / page
|
||||
} else {
|
||||
0
|
||||
};
|
||||
let per_dblk_bitmap = npages.div_ceil(8);
|
||||
let bitmap_bytes = per_dblk_bitmap
|
||||
.checked_mul(ndblks)
|
||||
.ok_or_else(|| FormatError::Overflow("Extensible Array page bitmap size".into()))?;
|
||||
let bitmap_start = sb_offset + sb_header_size;
|
||||
ensure_len(file_data, bitmap_start, bitmap_bytes)?;
|
||||
let bitmap = &file_data[bitmap_start..bitmap_start + bitmap_bytes];
|
||||
|
||||
let mut pos = bitmap_start + bitmap_bytes;
|
||||
let mut chunks = Vec::new();
|
||||
let mut global_idx = start_index;
|
||||
|
||||
for &addr in &dblk_addrs {
|
||||
if is_undefined_addr(addr, offset_size) {
|
||||
global_idx += nelmts_per_dblk;
|
||||
continue;
|
||||
}
|
||||
let block_chunks = read_data_block_elements(
|
||||
// One checksum covers the prefix, the bitmap and every data block address.
|
||||
let sb_end = ndblks
|
||||
.checked_mul(os)
|
||||
.and_then(|b| pos.checked_add(b))
|
||||
.ok_or_else(|| FormatError::Overflow("Extensible Array super block span".into()))?;
|
||||
verify_checksum(file_data, sb_offset, sb_end)?;
|
||||
|
||||
for i in 0..ndblks {
|
||||
ensure_len(file_data, pos, os)?;
|
||||
let addr = read_offset(file_data, pos, offset_size)?;
|
||||
pos += os;
|
||||
if !is_undefined_addr(addr, offset_size) {
|
||||
chunks.extend(read_data_block_elements(
|
||||
file_data,
|
||||
addr as usize,
|
||||
nelmts_per_dblk,
|
||||
dblk_nelmts,
|
||||
header,
|
||||
offset_size,
|
||||
chunk_byte_size,
|
||||
global_idx,
|
||||
num_chunks_per_dim,
|
||||
chunk_dimensions,
|
||||
)?;
|
||||
chunks.extend(block_chunks);
|
||||
global_idx += nelmts_per_dblk;
|
||||
grid,
|
||||
bitmap,
|
||||
i * npages,
|
||||
)?);
|
||||
}
|
||||
global_idx += dblk_nelmts;
|
||||
}
|
||||
|
||||
Ok(chunks)
|
||||
@@ -679,37 +696,28 @@ fn read_super_block(
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// Stamp the Jenkins checksum a real file would carry over
|
||||
/// `data[start..end]`, writing it at `end`. Hand-built fixtures need this
|
||||
/// now that the reader validates it, exactly as HDF5 writes it.
|
||||
fn stamp_checksum(data: &mut [u8], start: usize, end: usize) {
|
||||
let sum = crate::checksum::jenkins_lookup3(&data[start..end]);
|
||||
data[end..end + 4].copy_from_slice(&sum.to_le_bytes());
|
||||
}
|
||||
#[test]
|
||||
fn index_to_offsets_1d() {
|
||||
let num_chunks = vec![5u64];
|
||||
let chunk_dims = vec![20u32];
|
||||
assert_eq!(index_to_chunk_offsets(0, &num_chunks, &chunk_dims), vec![0]);
|
||||
assert_eq!(
|
||||
index_to_chunk_offsets(1, &num_chunks, &chunk_dims),
|
||||
vec![20]
|
||||
);
|
||||
assert_eq!(
|
||||
index_to_chunk_offsets(4, &num_chunks, &chunk_dims),
|
||||
vec![80]
|
||||
);
|
||||
let g = ChunkGrid::fixed_array(&[100], None, &[20]).unwrap();
|
||||
assert_eq!(g.offsets(0).unwrap(), vec![0]);
|
||||
assert_eq!(g.offsets(1).unwrap(), vec![20]);
|
||||
assert_eq!(g.offsets(4).unwrap(), vec![80]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn index_to_offsets_2d() {
|
||||
let num_chunks = vec![3u64, 2];
|
||||
let chunk_dims = vec![4u32, 3];
|
||||
assert_eq!(
|
||||
index_to_chunk_offsets(0, &num_chunks, &chunk_dims),
|
||||
vec![0, 0]
|
||||
);
|
||||
assert_eq!(
|
||||
index_to_chunk_offsets(1, &num_chunks, &chunk_dims),
|
||||
vec![0, 3]
|
||||
);
|
||||
assert_eq!(
|
||||
index_to_chunk_offsets(2, &num_chunks, &chunk_dims),
|
||||
vec![4, 0]
|
||||
);
|
||||
let g = ChunkGrid::fixed_array(&[10, 6], None, &[4, 3]).unwrap();
|
||||
assert_eq!(g.offsets(0).unwrap(), vec![0, 0]);
|
||||
assert_eq!(g.offsets(1).unwrap(), vec![0, 3]);
|
||||
assert_eq!(g.offsets(2).unwrap(), vec![4, 0]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -734,6 +742,7 @@ mod tests {
|
||||
buf[44..52].copy_from_slice(&5u64.to_le_bytes()); // stat[4] = num_elements
|
||||
buf[52..60].copy_from_slice(&0u64.to_le_bytes()); // stat[5]
|
||||
buf[60..68].copy_from_slice(&0x1000u64.to_le_bytes()); // index_block_address
|
||||
stamp_checksum(&mut buf, 0, 68);
|
||||
|
||||
let hdr = ExtensibleArrayHeader::parse(&buf, 0, os, ls).unwrap();
|
||||
assert_eq!(hdr.client_id, 0);
|
||||
@@ -775,7 +784,7 @@ mod tests {
|
||||
index_block_address: (usize::MAX - 4) as u64,
|
||||
};
|
||||
let buf = vec![0u8; 64];
|
||||
let r = read_extensible_array_chunks(&buf, &header, &[100], &[20], 8, 8, 8);
|
||||
let r = read_extensible_array_chunks(&buf, &header, &[100], None, &[20], 8, 8, 8);
|
||||
assert!(r.is_err());
|
||||
}
|
||||
|
||||
@@ -819,6 +828,7 @@ mod tests {
|
||||
.copy_from_slice(&(num_chunks as u64).to_le_bytes());
|
||||
file_data[aehd_offset + 60..aehd_offset + 68]
|
||||
.copy_from_slice(&(aeib_offset as u64).to_le_bytes());
|
||||
stamp_checksum(&mut file_data, aehd_offset, aehd_offset + 68);
|
||||
// checksum (4 bytes at +68) — not validated
|
||||
|
||||
// Build AEIB at aeib_offset
|
||||
@@ -836,12 +846,37 @@ mod tests {
|
||||
let p = elem_start + i * osv;
|
||||
file_data[p..p + osv].copy_from_slice(&addr.to_le_bytes());
|
||||
}
|
||||
// The index block's checksum covers its prefix, every inline element
|
||||
// slot, and every data block and super block address slot:
|
||||
// ndblk_addrs = 2 * (sup_blk_min_data_ptrs - 1), and the super block
|
||||
// pointers make up the rest of nsblks levels.
|
||||
let sup_ptrs = file_data[aehd_offset + 10] as usize;
|
||||
let dmin = file_data[aehd_offset + 9] as usize;
|
||||
let nsblks = 1 + 10 - dmin.trailing_zeros() as usize;
|
||||
let ndblk_addrs = 2 * (sup_ptrs - 1);
|
||||
// Levels consumed by those direct data blocks (1, 1, 2, 2, ... per level).
|
||||
let mut consumed = 0usize;
|
||||
let mut levels = 0usize;
|
||||
while consumed < ndblk_addrs {
|
||||
consumed += 1 << (levels / 2);
|
||||
levels += 1;
|
||||
}
|
||||
let ib_end = elem_start + num_chunks * osv + (ndblk_addrs + nsblks - levels) * osv;
|
||||
stamp_checksum(&mut file_data, aeib_offset, ib_end);
|
||||
|
||||
let header = ExtensibleArrayHeader::parse(&file_data, aehd_offset, os, ls).unwrap();
|
||||
let ds_dims = vec![40u64]; // 2 chunks × 20 elements
|
||||
let chunk_dims = vec![20u32];
|
||||
let chunks =
|
||||
read_extensible_array_chunks(&file_data, &header, &ds_dims, &chunk_dims, 8, os, ls)
|
||||
let chunks = read_extensible_array_chunks(
|
||||
&file_data,
|
||||
&header,
|
||||
&ds_dims,
|
||||
None,
|
||||
&chunk_dims,
|
||||
8,
|
||||
os,
|
||||
ls,
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(chunks.len(), 2);
|
||||
@@ -885,6 +920,7 @@ mod tests {
|
||||
// idx_blk_addr at offset 12 + 6*8 = 60
|
||||
file_data[aehd_offset + 60..aehd_offset + 68]
|
||||
.copy_from_slice(&(aeib_offset as u64).to_le_bytes());
|
||||
stamp_checksum(&mut file_data, aehd_offset, aehd_offset + 68);
|
||||
|
||||
// AEIB
|
||||
file_data[aeib_offset..aeib_offset + 4].copy_from_slice(b"EAIB");
|
||||
@@ -903,48 +939,62 @@ mod tests {
|
||||
pos += osv;
|
||||
}
|
||||
|
||||
// Direct data block addresses: first sb_level=0 has 1 dblk, sb_level=1 has 1 dblk
|
||||
// Total direct dblks for sblk_min=2: 2^0 + 2^1 = 1 + 2 = 3 (oops)
|
||||
// Actually: sblk_min levels. level 0: 2^0=1 dblk, level 1: 2^1=2 dblks => 3 dblks
|
||||
// But we only have 2 remaining elements.
|
||||
// dblk sizes: level 0: 1 dblk of min_dblk=2; level 1: 2 dblks of 2 each (nelmts doubles at level > 0)
|
||||
// Wait, re-reading the code: at level 0, nelmts=min_dblk=2, 1 dblk.
|
||||
// At level 1, 1 dblk, nelmts still 2 (doubles only at level > 0... but the code says
|
||||
// `if sb_level > 0 { nelmts *= 2 }` after pushing). Let me re-check.
|
||||
// After push at level 0: nelmts=2. Then if 0>0 false, no double. Push 1 dblk of 2.
|
||||
// Level 1: ndblks=2. Push 2 dblks of 2. Then 1>0 true, nelmts=4.
|
||||
// Total: 3 dblks with sizes [2, 2, 2]. Total = 6.
|
||||
// We only need 2 more elements. So only the first dblk has data.
|
||||
let n_direct_dblks = 3;
|
||||
// Direct data block addresses. With sup_blk_min_data_ptrs = 2 the index
|
||||
// block holds 2 * (2 - 1) = 2 of them, which are the data blocks of
|
||||
// super block levels 0 and 1: one of `min_dblk_nelmts` elements, then
|
||||
// one of twice that (ndblks = 2^(u/2), dblk_nelmts = 2^((u+1)/2) * min).
|
||||
// Only the first is allocated here; the rest of the array is empty.
|
||||
let ndblk_addrs = 2 * (sblk_min as usize - 1);
|
||||
file_data[pos..pos + osv].copy_from_slice(&(aedb_offset as u64).to_le_bytes());
|
||||
pos += osv;
|
||||
// 2 more dblk addresses - undefined
|
||||
for _ in 1..n_direct_dblks {
|
||||
for _ in 1..ndblk_addrs {
|
||||
file_data[pos..pos + osv].copy_from_slice(&u64::MAX.to_le_bytes());
|
||||
pos += osv;
|
||||
}
|
||||
// Super block addresses fill the remaining levels; all unallocated.
|
||||
let nsblks = 1 + 10 - (min_dblk_nelmts as usize).trailing_zeros() as usize;
|
||||
let mut consumed = 0usize;
|
||||
let mut levels = 0usize;
|
||||
while consumed < ndblk_addrs {
|
||||
consumed += 1 << (levels / 2);
|
||||
levels += 1;
|
||||
}
|
||||
for _ in 0..(nsblks - levels) {
|
||||
file_data[pos..pos + osv].copy_from_slice(&u64::MAX.to_le_bytes());
|
||||
pos += osv;
|
||||
}
|
||||
stamp_checksum(&mut file_data, aeib_offset, pos);
|
||||
|
||||
// EADB at aedb_offset (min_dblk_nelmts elements)
|
||||
// EADB holding the first data block's `min_dblk_nelmts` elements.
|
||||
file_data[aedb_offset..aedb_offset + 4].copy_from_slice(b"EADB");
|
||||
file_data[aedb_offset + 4] = 0;
|
||||
file_data[aedb_offset + 5] = 0;
|
||||
file_data[aedb_offset + 6..aedb_offset + 14]
|
||||
.copy_from_slice(&(aehd_offset as u64).to_le_bytes());
|
||||
// block_offset: ceil(max_nelmts_bits/8) = ceil(10/8) = 2 bytes
|
||||
// block_offset = 0 for first data block
|
||||
let blk_off_size = (10usize).div_ceil(8); // max_nelmts_bits=10
|
||||
let mut dbpos = aedb_offset + 6 + osv + blk_off_size;
|
||||
// Block offset field: ceil(max_nelmts_bits / 8) bytes, zero here.
|
||||
let blk_off_size = (10usize).div_ceil(8);
|
||||
let db_elems = aedb_offset + 6 + osv + blk_off_size;
|
||||
let mut dbpos = db_elems;
|
||||
for i in 0..min_dblk_nelmts as usize {
|
||||
let addr = base_addr + (idx_blk_elmts as u64 + i as u64) * chunk_byte_size;
|
||||
file_data[dbpos..dbpos + osv].copy_from_slice(&addr.to_le_bytes());
|
||||
dbpos += osv;
|
||||
}
|
||||
stamp_checksum(&mut file_data, aedb_offset, dbpos);
|
||||
|
||||
let header = ExtensibleArrayHeader::parse(&file_data, aehd_offset, os, ls).unwrap();
|
||||
let ds_dims = vec![40u64];
|
||||
let chunk_dims = vec![10u32];
|
||||
let chunks =
|
||||
read_extensible_array_chunks(&file_data, &header, &ds_dims, &chunk_dims, 8, os, ls)
|
||||
let chunks = read_extensible_array_chunks(
|
||||
&file_data,
|
||||
&header,
|
||||
&ds_dims,
|
||||
None,
|
||||
&chunk_dims,
|
||||
8,
|
||||
os,
|
||||
ls,
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(chunks.len(), 4);
|
||||
@@ -967,10 +1017,8 @@ mod tests {
|
||||
#[test]
|
||||
fn read_element_unallocated() {
|
||||
let data = vec![0xFFu8; 16];
|
||||
let num_chunks = vec![5u64];
|
||||
let chunk_dims = vec![10u32];
|
||||
let (info, consumed) =
|
||||
read_element(&data, 0, 0, 8, 8, 80, 0, &num_chunks, &chunk_dims).unwrap();
|
||||
let grid = ChunkGrid::fixed_array(&[50], None, &[10]).unwrap();
|
||||
let (info, consumed) = read_element(&data, 0, 0, 8, 8, 80, 0, &grid).unwrap();
|
||||
assert!(info.is_none());
|
||||
assert_eq!(consumed, 8);
|
||||
}
|
||||
@@ -989,20 +1037,9 @@ mod tests {
|
||||
// Filter mask
|
||||
data[12..16].copy_from_slice(&0u32.to_le_bytes());
|
||||
|
||||
let num_chunks = vec![5u64];
|
||||
let chunk_dims = vec![10u32];
|
||||
let (info, consumed) = read_element(
|
||||
&data,
|
||||
0,
|
||||
1,
|
||||
elem_size as u8,
|
||||
os,
|
||||
80,
|
||||
2,
|
||||
&num_chunks,
|
||||
&chunk_dims,
|
||||
)
|
||||
.unwrap();
|
||||
let grid = ChunkGrid::fixed_array(&[50], None, &[10]).unwrap();
|
||||
let (info, consumed) =
|
||||
read_element(&data, 0, 1, elem_size as u8, os, 80, 2, &grid).unwrap();
|
||||
let ci = info.unwrap();
|
||||
assert_eq!(ci.address, 0x2000);
|
||||
assert_eq!(ci.chunk_size, 120);
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -98,15 +98,50 @@ pub fn parse_fill_value(msg: &HeaderMessage) -> Result<Option<Vec<u8>>, FormatEr
|
||||
|
||||
/// The fill value that applies to a dataset given its header messages. The new
|
||||
/// message wins over the old one when both are present.
|
||||
///
|
||||
/// A *shared* fill value message holds only a reference to the real message,
|
||||
/// which cannot be followed without the file: this returns
|
||||
/// [`FormatError::UnresolvedSharedMessage`] for one (it used to answer "zeros").
|
||||
/// Use [`dataset_fill_value_in`] when the file bytes are at hand.
|
||||
pub fn dataset_fill_value(messages: &[HeaderMessage]) -> Result<Option<Vec<u8>>, FormatError> {
|
||||
fill_value_from(messages, |_| Err(FormatError::UnresolvedSharedMessage))
|
||||
}
|
||||
|
||||
/// [`dataset_fill_value`] for a dataset in `file_data`, following a shared
|
||||
/// fill value message to where it lives: another object header, or the
|
||||
/// file's shared-message (SOHM) heap, as libhdf5 writes it when the file has
|
||||
/// a SOHM index for fill values.
|
||||
pub fn dataset_fill_value_in(
|
||||
file_data: &[u8],
|
||||
messages: &[HeaderMessage],
|
||||
offset_size: u8,
|
||||
length_size: u8,
|
||||
) -> Result<Option<Vec<u8>>, FormatError> {
|
||||
fill_value_from(messages, |msg| {
|
||||
crate::shared_message::message_data_with_sohm(file_data, msg, offset_size, length_size)
|
||||
.map(|data| data.into_owned())
|
||||
})
|
||||
}
|
||||
|
||||
fn fill_value_from(
|
||||
messages: &[HeaderMessage],
|
||||
resolve_shared: impl Fn(&HeaderMessage) -> Result<Vec<u8>, FormatError>,
|
||||
) -> Result<Option<Vec<u8>>, FormatError> {
|
||||
for wanted in [MessageType::FillValue, MessageType::FillValueOld] {
|
||||
if let Some(msg) = messages.iter().find(|m| m.msg_type == wanted) {
|
||||
if crate::shared_message::is_shared(msg.flags) {
|
||||
// A shared fill value is legal but vanishingly rare; treat it
|
||||
// as the default rather than misparsing the reference.
|
||||
return Ok(None);
|
||||
}
|
||||
if let Some(value) = parse_fill_value(msg)? {
|
||||
let value = if crate::shared_message::is_shared(msg.flags) {
|
||||
let data = resolve_shared(msg)?;
|
||||
parse_fill_value(&HeaderMessage {
|
||||
msg_type: msg.msg_type,
|
||||
size: data.len(),
|
||||
flags: msg.flags & !0x02,
|
||||
creation_order: msg.creation_order,
|
||||
data,
|
||||
})?
|
||||
} else {
|
||||
parse_fill_value(msg)?
|
||||
};
|
||||
if let Some(value) = value {
|
||||
return Ok(Some(value));
|
||||
}
|
||||
}
|
||||
@@ -174,7 +209,7 @@ pub fn read_full_with_fill<E: From<FormatError>>(
|
||||
{
|
||||
return Err(FormatError::ExternalDataFilesUnsupported.into());
|
||||
}
|
||||
let fill = dataset_fill_value(messages)?;
|
||||
let fill = dataset_fill_value_in(file_data, messages, offset_size, length_size)?;
|
||||
if !has_storage(layout) {
|
||||
return Ok(filled_dataset(dataspace, elem_size, fill.as_deref())?);
|
||||
}
|
||||
|
||||
@@ -19,8 +19,35 @@ pub const FILTER_SCALEOFFSET: u16 = 6;
|
||||
pub const FILTER_LZ4: u16 = 32004;
|
||||
/// Zstandard compression.
|
||||
pub const FILTER_ZSTD: u16 = 32015;
|
||||
/// Pcodec lossless numerical codec (clawhdf5 internal; not yet HDF5-registered).
|
||||
pub const FILTER_PCODEC: u16 = 32023;
|
||||
/// bzip2 (registered by PyTables; hdf5plugin's `BZip2`).
|
||||
pub const FILTER_BZIP2: u16 = 307;
|
||||
/// LZF — h5py's built-in `compression="lzf"`.
|
||||
pub const FILTER_LZF: u16 = 32000;
|
||||
/// Blosc 1 (hdf5-blosc; hdf5plugin's `Blosc`).
|
||||
pub const FILTER_BLOSC: u16 = 32001;
|
||||
/// Bitshuffle, optionally with LZ4 or Zstandard (hdf5plugin's `Bitshuffle`).
|
||||
pub const FILTER_BITSHUFFLE: u16 = 32008;
|
||||
/// ZFP lossy floating-point compression (hdf5plugin's `Zfp`). Not supported.
|
||||
pub const FILTER_ZFP: u16 = 32013;
|
||||
/// Blosc 2 (hdf5plugin's `Blosc2`).
|
||||
pub const FILTER_BLOSC2: u16 = 32026;
|
||||
/// Pcodec lossless numerical codec — a **private, unregistered** clawhdf5
|
||||
/// filter. Pcodec has no ID in the HDF Group's filter registry (checked
|
||||
/// 2026-09-25, `hdf5_plugins/docs/RegisteredFilterPlugins.md`), so it uses an
|
||||
/// ID from the registry's testing/private range (256–511). No libhdf5 plugin
|
||||
/// decodes it: h5py/libhdf5 report the filter as unavailable. Only clawhdf5
|
||||
/// (with the `pcodec` feature) reads these datasets.
|
||||
pub const FILTER_PCODEC: u16 = 480;
|
||||
/// Filter name written with [`FILTER_PCODEC`].
|
||||
pub const FILTER_PCODEC_NAME: &str = "pcodec (clawhdf5 private)";
|
||||
/// The ID clawhdf5 up to 2.7.0 wrote pcodec under. It is registered to
|
||||
/// Granular BitRound (GBR), whose decode is a pass-through, so libhdf5 with
|
||||
/// that plugin would have returned the compressed bytes as data. Read as
|
||||
/// pcodec only when the filter is named exactly [`FILTER_PCODEC_LEGACY_NAME`],
|
||||
/// the name those versions wrote; never written.
|
||||
pub const FILTER_PCODEC_LEGACY: u16 = 32023;
|
||||
/// The filter name clawhdf5 up to 2.7.0 wrote with [`FILTER_PCODEC_LEGACY`].
|
||||
pub const FILTER_PCODEC_LEGACY_NAME: &str = "pcodec";
|
||||
|
||||
/// Description of a single filter in a pipeline.
|
||||
#[derive(Debug, Clone, PartialEq)]
|
||||
|
||||
@@ -0,0 +1,477 @@
|
||||
//! Filter registry: every filter is looked up here by its HDF5 filter ID.
|
||||
//!
|
||||
//! Two tiers:
|
||||
//!
|
||||
//! * **Built-in filters** — a static table of the filters compiled into this
|
||||
//! build: the HDF5 standard filters (deflate, shuffle, Fletcher32, szip,
|
||||
//! N-Bit, scale-offset) and the plugin filters whose cargo features are
|
||||
//! enabled (LZ4, Zstandard, pcodec, LZF, bitshuffle, bzip2, blosc).
|
||||
//! [`builtin_filters`] lists them.
|
||||
//! * **Registered filters** (`std` only) — codecs the application supplies
|
||||
//! for any other ID with [`register_filter`] (a [`FilterCodec`], or just a
|
||||
//! decoding closure). A registered codec cannot shadow a built-in one,
|
||||
//! except under 32023: that ID belongs to Granular BitRound, and the
|
||||
//! built-in entry there only reads the pcodec chunks clawhdf5 <= 2.7.0
|
||||
//! wrote (filter name `"pcodec"`), so a codec registered for 32023 handles
|
||||
//! every other chunk with that ID, and writes.
|
||||
//!
|
||||
//! An ID in neither tier fails with [`FormatError::UnsupportedFilter`], as it
|
||||
//! always has.
|
||||
//!
|
||||
//! ```
|
||||
//! # #[cfg(feature = "std")] {
|
||||
//! use clawhdf5_format::filter_registry::{self, FilterContext};
|
||||
//! use clawhdf5_format::error::FormatError;
|
||||
//!
|
||||
//! // A toy filter in the private-use range: every byte XORed with 0x5A.
|
||||
//! filter_registry::register_filter(300, |input: &[u8], _ctx: &FilterContext<'_>| {
|
||||
//! Ok::<_, FormatError>(input.iter().map(|b| b ^ 0x5A).collect())
|
||||
//! })
|
||||
//! .unwrap();
|
||||
//! assert!(filter_registry::is_filter_available(300));
|
||||
//! filter_registry::unregister_filter(300);
|
||||
//! # }
|
||||
//! ```
|
||||
|
||||
#[cfg(not(feature = "std"))]
|
||||
extern crate alloc;
|
||||
|
||||
#[cfg(not(feature = "std"))]
|
||||
use alloc::vec::Vec;
|
||||
|
||||
use crate::error::FormatError;
|
||||
use crate::filter_pipeline::FilterDescription;
|
||||
|
||||
/// What a codec is told about the filter it is applying.
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
pub struct FilterContext<'a> {
|
||||
/// The filter as recorded in the dataset's filter pipeline: its ID, name,
|
||||
/// flags and client data (`cd_values`).
|
||||
pub filter: &'a FilterDescription,
|
||||
/// Size in bytes of one dataset element (the datatype's size).
|
||||
pub element_size: usize,
|
||||
/// Decoding only: the most bytes this stage may produce — what entered
|
||||
/// the filter when the chunk was written. 0 means unknown; a decoder then
|
||||
/// falls back to a fixed ceiling. Always 0 when encoding.
|
||||
pub max_output: usize,
|
||||
}
|
||||
|
||||
impl FilterContext<'_> {
|
||||
/// The filter's client data (`cd_values`).
|
||||
pub fn client_data(&self) -> &[u32] {
|
||||
&self.filter.client_data
|
||||
}
|
||||
|
||||
/// The largest output a decoder should allow: [`Self::max_output`], or
|
||||
/// 256 MiB when that is unknown.
|
||||
pub fn output_limit(&self) -> usize {
|
||||
if self.max_output != 0 {
|
||||
self.max_output
|
||||
} else {
|
||||
crate::filters::MAX_DECOMPRESS_SIZE
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// A filter implementation.
|
||||
///
|
||||
/// `decode` undoes the filter (the read direction). `encode` applies it (the
|
||||
/// write direction); the default refuses with
|
||||
/// [`FormatError::UnsupportedFilter`], which is right for a read-only codec.
|
||||
pub trait FilterCodec: Send + Sync {
|
||||
/// Undo the filter on one chunk. The output must not exceed
|
||||
/// [`FilterContext::output_limit`]; the pipeline rejects a larger one.
|
||||
fn decode(&self, input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError>;
|
||||
|
||||
/// Apply the filter to one chunk.
|
||||
fn encode(&self, input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||
let _ = input;
|
||||
Err(FormatError::UnsupportedFilter(ctx.filter.filter_id))
|
||||
}
|
||||
}
|
||||
|
||||
/// Any `Fn(&[u8], &FilterContext) -> Result<Vec<u8>, FormatError>` is a
|
||||
/// decode-only codec.
|
||||
impl<F> FilterCodec for F
|
||||
where
|
||||
F: Fn(&[u8], &FilterContext<'_>) -> Result<Vec<u8>, FormatError> + Send + Sync,
|
||||
{
|
||||
fn decode(&self, input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||
self(input, ctx)
|
||||
}
|
||||
}
|
||||
|
||||
/// Signature of a built-in filter's decoder or encoder.
|
||||
pub type BuiltinFn = fn(&[u8], &FilterContext<'_>) -> Result<Vec<u8>, FormatError>;
|
||||
|
||||
/// A filter compiled into this build.
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
pub struct BuiltinFilter {
|
||||
/// HDF5 filter ID.
|
||||
pub id: u16,
|
||||
/// Human-readable name.
|
||||
pub name: &'static str,
|
||||
/// Decoder.
|
||||
pub(crate) decode: BuiltinFn,
|
||||
/// Encoder, if this build can write the filter.
|
||||
pub(crate) encode: Option<BuiltinFn>,
|
||||
}
|
||||
|
||||
impl BuiltinFilter {
|
||||
/// Whether this build can write the filter as well as read it.
|
||||
pub fn can_encode(&self) -> bool {
|
||||
self.encode.is_some()
|
||||
}
|
||||
|
||||
/// Whether the built-in entry only borrows its ID for some chunks, so a
|
||||
/// registered codec may take the rest: the legacy pcodec entry under
|
||||
/// Granular BitRound's 32023, which claims only chunks named `"pcodec"`.
|
||||
fn is_shared(&self) -> bool {
|
||||
self.id == crate::filter_pipeline::FILTER_PCODEC_LEGACY
|
||||
}
|
||||
|
||||
/// Whether this entry decodes chunks written with `filter`.
|
||||
fn claims(&self, filter: &crate::filter_pipeline::FilterDescription) -> bool {
|
||||
!self.is_shared()
|
||||
|| filter.name.as_deref() == Some(crate::filter_pipeline::FILTER_PCODEC_LEGACY_NAME)
|
||||
}
|
||||
}
|
||||
|
||||
/// The filters compiled into this build, in ID order.
|
||||
pub fn builtin_filters() -> &'static [BuiltinFilter] {
|
||||
crate::filters::BUILTIN_FILTERS
|
||||
}
|
||||
|
||||
/// The built-in filter with this ID, if it is compiled in.
|
||||
pub fn builtin_filter(id: u16) -> Option<&'static BuiltinFilter> {
|
||||
builtin_filters().iter().find(|f| f.id == id)
|
||||
}
|
||||
|
||||
/// Why a filter ID may be missing from this build: the filter's name, and
|
||||
/// the cargo feature that provides it (`None`: clawhdf5 does not implement
|
||||
/// it — register a codec for it with [`register_filter`]). `None` for an ID
|
||||
/// clawhdf5 knows nothing about.
|
||||
pub fn known_filter(id: u16) -> Option<(&'static str, Option<&'static str>)> {
|
||||
Some(match id {
|
||||
1 => ("deflate", Some("deflate")),
|
||||
4 => ("SZIP", Some("szip")),
|
||||
307 => ("bzip2", Some("bzip2")),
|
||||
480 => ("pcodec", Some("pcodec")),
|
||||
32000 => ("LZF", Some("lzf")),
|
||||
32001 => ("Blosc", Some("blosc")),
|
||||
32004 => ("LZ4", Some("lz4")),
|
||||
32008 => ("bitshuffle", Some("bitshuffle")),
|
||||
32013 => ("ZFP", None),
|
||||
32015 => ("Zstandard", Some("zstd")),
|
||||
32019 => ("JPEG", None),
|
||||
32022 => ("BitGroom", None),
|
||||
32023 => ("Granular BitRound", None),
|
||||
32026 => ("Blosc2", None),
|
||||
_ => return None,
|
||||
})
|
||||
}
|
||||
|
||||
/// Whether a chunk filtered with `id` can be decoded: a built-in filter or a
|
||||
/// registered one.
|
||||
pub fn is_filter_available(id: u16) -> bool {
|
||||
if builtin_filter(id).is_some() {
|
||||
return true;
|
||||
}
|
||||
#[cfg(feature = "std")]
|
||||
{
|
||||
registered(id).is_some()
|
||||
}
|
||||
#[cfg(not(feature = "std"))]
|
||||
{
|
||||
false
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(feature = "std")]
|
||||
mod custom {
|
||||
use super::FilterCodec;
|
||||
use std::collections::BTreeMap;
|
||||
use std::sync::{Arc, PoisonError, RwLock};
|
||||
|
||||
pub(super) type Registry = BTreeMap<u16, Arc<dyn FilterCodec>>;
|
||||
|
||||
static REGISTRY: RwLock<Registry> = RwLock::new(BTreeMap::new());
|
||||
|
||||
pub(super) fn with_read<R>(f: impl FnOnce(&Registry) -> R) -> R {
|
||||
// A panic while holding the lock cannot leave the map half-updated
|
||||
// (every update is a single insert/remove), so poisoning is ignored.
|
||||
f(®ISTRY.read().unwrap_or_else(PoisonError::into_inner))
|
||||
}
|
||||
|
||||
pub(super) fn with_write<R>(f: impl FnOnce(&mut Registry) -> R) -> R {
|
||||
f(&mut REGISTRY.write().unwrap_or_else(PoisonError::into_inner))
|
||||
}
|
||||
}
|
||||
|
||||
/// Register a codec for filter `id`, process-wide. It is used for every
|
||||
/// chunk read (and, if it implements [`FilterCodec::encode`], written) with
|
||||
/// that filter ID, by every file.
|
||||
///
|
||||
/// A plain closure `Fn(&[u8], &FilterContext) -> Result<Vec<u8>, FormatError>`
|
||||
/// registers a decoder. Replaces (and returns) an earlier registration for
|
||||
/// the same ID. Fails with [`FormatError::FilterError`] if `id` is a built-in
|
||||
/// filter of this build: those cannot be overridden. The exception is 32023
|
||||
/// (Granular BitRound): with the `pcodec` feature the built-in entry there
|
||||
/// reads only chunks whose filter is named `"pcodec"` (clawhdf5 <= 2.7.0's
|
||||
/// files); a codec registered for 32023 decodes every other chunk with that
|
||||
/// ID and does all the writing.
|
||||
#[cfg(feature = "std")]
|
||||
pub fn register_filter<C>(
|
||||
id: u16,
|
||||
codec: C,
|
||||
) -> Result<Option<std::sync::Arc<dyn FilterCodec>>, FormatError>
|
||||
where
|
||||
C: FilterCodec + 'static,
|
||||
{
|
||||
if let Some(builtin) = builtin_filter(id).filter(|b| !b.is_shared()) {
|
||||
return Err(FormatError::FilterError(format!(
|
||||
"filter {id} ({}) is built in and cannot be re-registered",
|
||||
builtin.name
|
||||
)));
|
||||
}
|
||||
let codec: std::sync::Arc<dyn FilterCodec> = std::sync::Arc::new(codec);
|
||||
Ok(custom::with_write(|r| r.insert(id, codec)))
|
||||
}
|
||||
|
||||
/// Remove the codec registered for `id`. Returns whether one was registered.
|
||||
#[cfg(feature = "std")]
|
||||
pub fn unregister_filter(id: u16) -> bool {
|
||||
custom::with_write(|r| r.remove(&id).is_some())
|
||||
}
|
||||
|
||||
/// The codec registered for `id`, if any.
|
||||
#[cfg(feature = "std")]
|
||||
pub fn registered(id: u16) -> Option<std::sync::Arc<dyn FilterCodec>> {
|
||||
custom::with_read(|r| r.get(&id).cloned())
|
||||
}
|
||||
|
||||
/// Undo filter `ctx.filter` on `input`: the built-in decoder if there is one
|
||||
/// that claims the chunk, else a registered one, else the built-in decoder's
|
||||
/// own refusal or [`FormatError::UnsupportedFilter`].
|
||||
pub(crate) fn decode(input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||
let id = ctx.filter.filter_id;
|
||||
let builtin = builtin_filter(id);
|
||||
if let Some(builtin) = builtin.filter(|b| b.claims(ctx.filter)) {
|
||||
return (builtin.decode)(input, ctx);
|
||||
}
|
||||
#[cfg(feature = "std")]
|
||||
if let Some(codec) = registered(id) {
|
||||
let out = codec.decode(input, ctx)?;
|
||||
// A registered codec is outside our control: hold it to the same
|
||||
// bound the built-in decoders enforce.
|
||||
if out.len() > ctx.output_limit() {
|
||||
return Err(FormatError::DecompressionError(format!(
|
||||
"filter {id}: decoded {} bytes, more than the {} the chunk can hold",
|
||||
out.len(),
|
||||
ctx.output_limit()
|
||||
)));
|
||||
}
|
||||
return Ok(out);
|
||||
}
|
||||
match builtin {
|
||||
Some(builtin) => (builtin.decode)(input, ctx),
|
||||
None => Err(FormatError::UnsupportedFilter(id)),
|
||||
}
|
||||
}
|
||||
|
||||
/// Apply filter `ctx.filter` to `input`.
|
||||
pub(crate) fn encode(input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||
let id = ctx.filter.filter_id;
|
||||
#[cfg(feature = "std")]
|
||||
if builtin_filter(id).is_some_and(|b| b.is_shared())
|
||||
&& let Some(codec) = registered(id)
|
||||
{
|
||||
return codec.encode(input, ctx);
|
||||
}
|
||||
if let Some(builtin) = builtin_filter(id) {
|
||||
return match builtin.encode {
|
||||
Some(encode) => encode(input, ctx),
|
||||
None => Err(FormatError::UnsupportedFilter(id)),
|
||||
};
|
||||
}
|
||||
#[cfg(feature = "std")]
|
||||
if let Some(codec) = registered(id) {
|
||||
return codec.encode(input, ctx);
|
||||
}
|
||||
Err(FormatError::UnsupportedFilter(id))
|
||||
}
|
||||
|
||||
#[cfg(all(test, feature = "std"))]
|
||||
pub(crate) mod tests {
|
||||
use super::*;
|
||||
use crate::filter_pipeline::{FILTER_FLETCHER32, FILTER_SHUFFLE, FilterPipeline};
|
||||
use crate::filters::{compress_chunk, decompress_chunk};
|
||||
|
||||
fn pipeline(id: u16) -> FilterPipeline {
|
||||
FilterPipeline {
|
||||
version: 2,
|
||||
filters: vec![FilterDescription {
|
||||
filter_id: id,
|
||||
name: Some("test".into()),
|
||||
flags: 0,
|
||||
client_data: vec![7],
|
||||
}],
|
||||
}
|
||||
}
|
||||
|
||||
struct Xor;
|
||||
impl FilterCodec for Xor {
|
||||
fn decode(&self, input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||
let k = ctx.client_data()[0] as u8;
|
||||
Ok(input.iter().map(|b| b ^ k).collect())
|
||||
}
|
||||
fn encode(&self, input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||
self.decode(input, ctx)
|
||||
}
|
||||
}
|
||||
|
||||
// Each test uses its own ID: the registry is process-wide and tests run
|
||||
// in parallel.
|
||||
|
||||
#[test]
|
||||
fn unknown_filter_keeps_its_error() {
|
||||
let err = decompress_chunk(b"abc", &pipeline(311), 3, 1).unwrap_err();
|
||||
assert_eq!(err, FormatError::UnsupportedFilter(311));
|
||||
let err = compress_chunk(b"abc", &pipeline(311), 1).unwrap_err();
|
||||
assert_eq!(err, FormatError::UnsupportedFilter(311));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn registered_codec_round_trips_through_the_pipeline() {
|
||||
assert!(!is_filter_available(312));
|
||||
assert!(register_filter(312, Xor).unwrap().is_none());
|
||||
assert!(is_filter_available(312));
|
||||
let data = b"hello, registry".to_vec();
|
||||
let enc = compress_chunk(&data, &pipeline(312), 1).unwrap();
|
||||
assert_ne!(enc, data);
|
||||
assert_eq!(
|
||||
decompress_chunk(&enc, &pipeline(312), data.len(), 1).unwrap(),
|
||||
data
|
||||
);
|
||||
assert!(unregister_filter(312));
|
||||
assert!(!unregister_filter(312));
|
||||
assert_eq!(
|
||||
decompress_chunk(&enc, &pipeline(312), data.len(), 1).unwrap_err(),
|
||||
FormatError::UnsupportedFilter(312)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn closure_registers_a_decoder_only() {
|
||||
register_filter(313, |input: &[u8], _ctx: &FilterContext<'_>| {
|
||||
Ok(input.iter().rev().copied().collect())
|
||||
})
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
decompress_chunk(b"abc", &pipeline(313), 3, 1).unwrap(),
|
||||
b"cba"
|
||||
);
|
||||
assert_eq!(
|
||||
compress_chunk(b"abc", &pipeline(313), 1).unwrap_err(),
|
||||
FormatError::UnsupportedFilter(313)
|
||||
);
|
||||
unregister_filter(313);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn registered_decoder_output_is_bounded() {
|
||||
register_filter(314, |_input: &[u8], _ctx: &FilterContext<'_>| {
|
||||
Ok(vec![0u8; 1000])
|
||||
})
|
||||
.unwrap();
|
||||
let err = decompress_chunk(b"abc", &pipeline(314), 10, 1).unwrap_err();
|
||||
assert!(matches!(err, FormatError::DecompressionError(_)), "{err:?}");
|
||||
unregister_filter(314);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn builtins_cannot_be_overridden() {
|
||||
for id in [FILTER_SHUFFLE, FILTER_FLETCHER32] {
|
||||
let Err(err) = register_filter(id, Xor) else {
|
||||
panic!("built-in filter {id} was re-registered");
|
||||
};
|
||||
assert!(matches!(err, FormatError::FilterError(_)), "{err:?}");
|
||||
}
|
||||
assert!(builtin_filter(FILTER_SHUFFLE).is_some());
|
||||
}
|
||||
|
||||
/// Serialises the tests that register or read filter 32023 (the
|
||||
/// registry is process-wide).
|
||||
pub(crate) static ID_32023: std::sync::Mutex<()> = std::sync::Mutex::new(());
|
||||
|
||||
/// 32023 is Granular BitRound's ID; the `pcodec` build's built-in entry
|
||||
/// there reads only clawhdf5 <= 2.7.0's pcodec chunks (named "pcodec"),
|
||||
/// so a codec can be registered for the rest, and writes with it.
|
||||
#[test]
|
||||
fn a_codec_can_be_registered_for_granular_bitround() {
|
||||
let _guard = ID_32023
|
||||
.lock()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
let named = |name: Option<&str>| FilterPipeline {
|
||||
version: 2,
|
||||
filters: vec![FilterDescription {
|
||||
filter_id: 32023,
|
||||
name: name.map(Into::into),
|
||||
flags: 0,
|
||||
client_data: vec![7],
|
||||
}],
|
||||
};
|
||||
let prev = register_filter(32023, Xor).expect("32023 must be registrable");
|
||||
assert!(prev.is_none());
|
||||
let data = b"granular bitround".to_vec();
|
||||
for name in [None, Some("granular_bitround"), Some("test")] {
|
||||
let pl = named(name);
|
||||
let enc = compress_chunk(&data, &pl, 1).unwrap();
|
||||
assert_ne!(enc, data);
|
||||
assert_eq!(decompress_chunk(&enc, &pl, data.len(), 1).unwrap(), data);
|
||||
}
|
||||
// clawhdf5 <= 2.7.0's pcodec chunks still go to the built-in reader.
|
||||
#[cfg(feature = "pcodec")]
|
||||
{
|
||||
let raw: Vec<u8> = (0..64)
|
||||
.flat_map(|i| (f64::from(i) * 0.5).to_le_bytes())
|
||||
.collect();
|
||||
let comp = crate::filters::pcodec_compress(&raw, 8).unwrap();
|
||||
let mut pl = named(Some("pcodec"));
|
||||
pl.filters[0].client_data = vec![8];
|
||||
assert_eq!(decompress_chunk(&comp, &pl, raw.len(), 8).unwrap(), raw);
|
||||
}
|
||||
assert!(unregister_filter(32023));
|
||||
let pl = named(None);
|
||||
assert!(matches!(
|
||||
decompress_chunk(&data, &pl, data.len(), 1),
|
||||
Err(FormatError::UnsupportedFilter(32023))
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unsupported_filter_error_names_the_filter() {
|
||||
let msg = FormatError::UnsupportedFilter(32026).to_string();
|
||||
assert!(
|
||||
msg.contains("Blosc2") && msg.contains("not implemented"),
|
||||
"{msg}"
|
||||
);
|
||||
let msg = FormatError::UnsupportedFilter(32013).to_string();
|
||||
assert!(msg.contains("ZFP"), "{msg}");
|
||||
let msg = FormatError::UnsupportedFilter(32000).to_string();
|
||||
assert!(msg.contains("LZF") && msg.contains("`lzf`"), "{msg}");
|
||||
assert_eq!(
|
||||
FormatError::UnsupportedFilter(399).to_string(),
|
||||
"unsupported filter: 399"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn builtin_table_is_sorted_and_unique() {
|
||||
let ids: Vec<u16> = builtin_filters().iter().map(|f| f.id).collect();
|
||||
let mut sorted = ids.clone();
|
||||
sorted.sort_unstable();
|
||||
sorted.dedup();
|
||||
assert_eq!(ids, sorted);
|
||||
}
|
||||
}
|
||||
+1089
-136
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,438 @@
|
||||
//! Bitshuffle (HDF5 filter 32008) and the bit transpose it shares with blosc.
|
||||
//!
|
||||
//! **The transform.** A block of `n` elements (`n` a multiple of 8) of
|
||||
//! `es` bytes each is viewed as an `n × 8·es` bit matrix — row *i* is
|
||||
//! element *i*, column `8·j + k` is bit *k* (LSB first) of its byte *j* — and
|
||||
//! transposed: the output is `8·es` rows of `n` bits, row `8·j + k` holding
|
||||
//! bit *k* of byte *j* of every element in order, packed LSB first. That is
|
||||
//! what `bshuf_trans_bit_elem` produces (checked against hdf5plugin's
|
||||
//! library bit for bit).
|
||||
//!
|
||||
//! **The filter** (`bshuf_h5filter.c`). `cd_values`: `[0..2]` bitshuffle
|
||||
//! version, `[2]` element size, `[3]` block size in elements (0 = default:
|
||||
//! 8192 bytes' worth, rounded down to a multiple of 8, at least 128),
|
||||
//! `[4]` compression (0 none, 2 LZ4, 3 Zstandard), `[5]` Zstandard level.
|
||||
//! The chunk is cut into blocks of `block size` elements; the tail shorter
|
||||
//! than a block is transposed as one block rounded down to a multiple of 8
|
||||
//! elements, and the last `n mod 8` elements are stored as they are.
|
||||
//! Uncompressed, that is the whole chunk. Compressed, the chunk starts with a
|
||||
//! 12-byte header — the decoded size (u64 big-endian) and the block size in
|
||||
//! bytes (u32 big-endian) — and each transposed block is stored as a u32
|
||||
//! big-endian length and an LZ4 block / Zstandard frame; the untransposed
|
||||
//! tail follows the last block.
|
||||
|
||||
#[cfg(not(feature = "std"))]
|
||||
extern crate alloc;
|
||||
#[cfg(not(feature = "std"))]
|
||||
use alloc::{format, vec, vec::Vec};
|
||||
|
||||
use crate::error::FormatError;
|
||||
#[cfg(feature = "bitshuffle")]
|
||||
use crate::filter_registry::FilterContext;
|
||||
|
||||
/// Transpose an 8×8 bit matrix packed in a u64 (byte *r* = row *r*, bit *c*
|
||||
/// of that byte = column *c*). An involution.
|
||||
#[inline]
|
||||
fn transpose8(mut x: u64) -> u64 {
|
||||
let t = (x ^ (x >> 7)) & 0x00AA_00AA_00AA_00AA;
|
||||
x = x ^ t ^ (t << 7);
|
||||
let t = (x ^ (x >> 14)) & 0x0000_CCCC_0000_CCCC;
|
||||
x = x ^ t ^ (t << 14);
|
||||
let t = (x ^ (x >> 28)) & 0x0000_0000_F0F0_F0F0;
|
||||
x ^ t ^ (t << 28)
|
||||
}
|
||||
|
||||
/// Bit-transpose one block: `input` and `out` are `n * es` bytes, `n` a
|
||||
/// multiple of 8.
|
||||
pub(crate) fn bitshuffle_block(input: &[u8], out: &mut [u8], n: usize, es: usize) {
|
||||
debug_assert!(n.is_multiple_of(8) && input.len() == n * es && out.len() == n * es);
|
||||
let row = n / 8;
|
||||
for j in 0..es {
|
||||
for g in 0..row {
|
||||
let mut x = 0u64;
|
||||
for t in 0..8 {
|
||||
x |= u64::from(input[(8 * g + t) * es + j]) << (8 * t);
|
||||
}
|
||||
let y = transpose8(x);
|
||||
for k in 0..8 {
|
||||
out[(8 * j + k) * row + g] = (y >> (8 * k)) as u8;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Undo [`bitshuffle_block`].
|
||||
pub(crate) fn bitunshuffle_block(input: &[u8], out: &mut [u8], n: usize, es: usize) {
|
||||
debug_assert!(n.is_multiple_of(8) && input.len() == n * es && out.len() == n * es);
|
||||
let row = n / 8;
|
||||
for j in 0..es {
|
||||
for g in 0..row {
|
||||
let mut y = 0u64;
|
||||
for k in 0..8 {
|
||||
y |= u64::from(input[(8 * j + k) * row + g]) << (8 * k);
|
||||
}
|
||||
let x = transpose8(y);
|
||||
for t in 0..8 {
|
||||
out[(8 * g + t) * es + j] = (x >> (8 * t)) as u8;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// `bshuf_default_block_size`: 8 KiB of elements, a multiple of 8, >= 128.
|
||||
#[cfg(feature = "bitshuffle")]
|
||||
fn default_block_size(es: usize) -> usize {
|
||||
((8192 / es) / 8 * 8).max(128)
|
||||
}
|
||||
|
||||
#[cfg(feature = "bitshuffle")]
|
||||
fn err(msg: &str) -> FormatError {
|
||||
FormatError::DecompressionError(format!("bitshuffle: {msg}"))
|
||||
}
|
||||
|
||||
/// `cd_values[4]`: the compression bitshuffle applies after the transpose.
|
||||
#[cfg(feature = "bitshuffle")]
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
enum Codec {
|
||||
None,
|
||||
Lz4,
|
||||
Zstd,
|
||||
}
|
||||
|
||||
#[cfg(feature = "bitshuffle")]
|
||||
fn codec(cd: &[u32]) -> Result<Codec, FormatError> {
|
||||
match cd.get(4).copied().unwrap_or(0) {
|
||||
0 => Ok(Codec::None),
|
||||
2 => Ok(Codec::Lz4),
|
||||
3 => Ok(Codec::Zstd),
|
||||
other => Err(FormatError::FilterError(format!(
|
||||
"bitshuffle: unknown compression {other}"
|
||||
))),
|
||||
}
|
||||
}
|
||||
|
||||
/// The element counts of the transposed blocks for `size` elements.
|
||||
#[cfg(feature = "bitshuffle")]
|
||||
fn blocks(size: usize, block: usize) -> impl Iterator<Item = usize> {
|
||||
let full = size / block;
|
||||
let last = (size % block) / 8 * 8;
|
||||
core::iter::repeat_n(block, full).chain((last > 0).then_some(last))
|
||||
}
|
||||
|
||||
/// Decode a bitshuffle-filtered chunk.
|
||||
#[cfg(feature = "bitshuffle")]
|
||||
pub(crate) fn bitshuffle_decode(
|
||||
input: &[u8],
|
||||
ctx: &FilterContext<'_>,
|
||||
) -> Result<Vec<u8>, FormatError> {
|
||||
let cd = ctx.client_data();
|
||||
let es = match cd.get(2) {
|
||||
Some(&e) if e != 0 => e as usize,
|
||||
_ => return Err(err("missing element size")),
|
||||
};
|
||||
let codec = codec(cd)?;
|
||||
let limit = ctx.output_limit();
|
||||
if codec == Codec::None {
|
||||
if input.len() > limit {
|
||||
return Err(err("output exceeds the chunk size"));
|
||||
}
|
||||
let block = match cd.get(3) {
|
||||
Some(&b) if b != 0 => b as usize,
|
||||
_ => default_block_size(es),
|
||||
};
|
||||
if !block.is_multiple_of(8) {
|
||||
return Err(err("block size is not a multiple of 8"));
|
||||
}
|
||||
if !input.len().is_multiple_of(es) {
|
||||
return Err(err("chunk is not a whole number of elements"));
|
||||
}
|
||||
let size = input.len() / es;
|
||||
let mut out = vec![0u8; input.len()];
|
||||
let mut pos = 0;
|
||||
for n in blocks(size, block) {
|
||||
let bytes = n * es;
|
||||
bitunshuffle_block(&input[pos..pos + bytes], &mut out[pos..pos + bytes], n, es);
|
||||
pos += bytes;
|
||||
}
|
||||
out[pos..].copy_from_slice(&input[pos..]);
|
||||
return Ok(out);
|
||||
}
|
||||
|
||||
let header = input.get(..12).ok_or_else(|| err("truncated header"))?;
|
||||
let total = u64::from_be_bytes(header[..8].try_into().unwrap());
|
||||
let block_bytes = u32::from_be_bytes(header[8..12].try_into().unwrap()) as usize;
|
||||
let total = usize::try_from(total)
|
||||
.ok()
|
||||
.filter(|&t| t <= limit)
|
||||
.ok_or_else(|| err("decoded size exceeds the chunk size"))?;
|
||||
if !total.is_multiple_of(es) {
|
||||
return Err(err("chunk is not a whole number of elements"));
|
||||
}
|
||||
if block_bytes == 0 || !block_bytes.is_multiple_of(es) {
|
||||
return Err(err("bad block size"));
|
||||
}
|
||||
let block = block_bytes / es;
|
||||
if !block.is_multiple_of(8) {
|
||||
return Err(err("block size is not a multiple of 8"));
|
||||
}
|
||||
let size = total / es;
|
||||
let mut out = vec![0u8; total];
|
||||
let mut tmp = vec![0u8; block_bytes.min(total)];
|
||||
let mut ip = 12usize;
|
||||
let mut op = 0usize;
|
||||
let mut zstd = None;
|
||||
for n in blocks(size, block) {
|
||||
let bytes = n * es;
|
||||
let len = input
|
||||
.get(ip..ip + 4)
|
||||
.map(|b| u32::from_be_bytes(b.try_into().unwrap()) as usize)
|
||||
.ok_or_else(|| err("truncated block header"))?;
|
||||
ip += 4;
|
||||
let comp = input
|
||||
.get(ip..ip.saturating_add(len))
|
||||
.ok_or_else(|| err("truncated block"))?;
|
||||
ip += len;
|
||||
let dst = &mut tmp[..bytes];
|
||||
let got = match codec {
|
||||
Codec::Lz4 => lz4_flex::block::decompress_into(comp, dst)
|
||||
.map_err(|e| err(&format!("lz4: {e}")))?,
|
||||
Codec::Zstd => zstd_decode_into(
|
||||
zstd.get_or_insert_with(ruzstd::decoding::FrameDecoder::new),
|
||||
comp,
|
||||
dst,
|
||||
)?,
|
||||
Codec::None => unreachable!(),
|
||||
};
|
||||
if got != bytes {
|
||||
return Err(err("block decoded to the wrong size"));
|
||||
}
|
||||
bitunshuffle_block(dst, &mut out[op..op + bytes], n, es);
|
||||
op += bytes;
|
||||
}
|
||||
let tail = total - op;
|
||||
let rest = input
|
||||
.get(ip..ip + tail)
|
||||
.ok_or_else(|| err("truncated trailing elements"))?;
|
||||
out[op..].copy_from_slice(rest);
|
||||
Ok(out)
|
||||
}
|
||||
|
||||
/// Decode Zstandard frames into exactly `dst`, failing if they hold more.
|
||||
#[cfg(any(feature = "bitshuffle", feature = "blosc"))]
|
||||
pub(crate) fn zstd_decode_into(
|
||||
decoder: &mut ruzstd::decoding::FrameDecoder,
|
||||
frames: &[u8],
|
||||
dst: &mut [u8],
|
||||
) -> Result<usize, FormatError> {
|
||||
decoder
|
||||
.decode_all(frames, dst)
|
||||
.map_err(|e| FormatError::DecompressionError(format!("zstd: {e}")))
|
||||
}
|
||||
|
||||
/// Compress with ruzstd. It implements one level (roughly zstd's level 1),
|
||||
/// so the requested level only matters to other encoders.
|
||||
#[cfg(any(feature = "bitshuffle", feature = "blosc"))]
|
||||
pub(crate) fn zstd_encode(data: &[u8]) -> Vec<u8> {
|
||||
ruzstd::encoding::compress_to_vec(data, ruzstd::encoding::CompressionLevel::Fastest)
|
||||
}
|
||||
|
||||
/// Encode a chunk with the bitshuffle filter.
|
||||
#[cfg(feature = "bitshuffle")]
|
||||
pub(crate) fn bitshuffle_encode(
|
||||
input: &[u8],
|
||||
ctx: &FilterContext<'_>,
|
||||
) -> Result<Vec<u8>, FormatError> {
|
||||
let cd = ctx.client_data();
|
||||
let es = match cd.get(2) {
|
||||
Some(&e) if e != 0 => e as usize,
|
||||
_ => ctx.element_size.max(1),
|
||||
};
|
||||
let codec = codec(cd)?;
|
||||
let block = match cd.get(3) {
|
||||
Some(&b) if b != 0 => b as usize,
|
||||
_ => default_block_size(es),
|
||||
};
|
||||
let cerr = |m: &str| FormatError::CompressionError(format!("bitshuffle: {m}"));
|
||||
if !block.is_multiple_of(8) {
|
||||
return Err(cerr("block size is not a multiple of 8"));
|
||||
}
|
||||
if !input.len().is_multiple_of(es) {
|
||||
return Err(cerr("chunk is not a whole number of elements"));
|
||||
}
|
||||
let size = input.len() / es;
|
||||
let mut out = Vec::with_capacity(input.len() + 12 + input.len() / 64);
|
||||
if codec != Codec::None {
|
||||
out.extend_from_slice(&(input.len() as u64).to_be_bytes());
|
||||
let block_bytes =
|
||||
u32::try_from(block * es).map_err(|_| cerr("block size does not fit in 32 bits"))?;
|
||||
out.extend_from_slice(&block_bytes.to_be_bytes());
|
||||
}
|
||||
let mut tmp = vec![0u8; (block * es).min(input.len())];
|
||||
let mut pos = 0;
|
||||
for n in blocks(size, block) {
|
||||
let bytes = n * es;
|
||||
let dst = &mut tmp[..bytes];
|
||||
bitshuffle_block(&input[pos..pos + bytes], dst, n, es);
|
||||
match codec {
|
||||
Codec::None => out.extend_from_slice(dst),
|
||||
Codec::Lz4 | Codec::Zstd => {
|
||||
let comp = if codec == Codec::Lz4 {
|
||||
lz4_flex::block::compress(dst)
|
||||
} else {
|
||||
zstd_encode(dst)
|
||||
};
|
||||
out.extend_from_slice(&(comp.len() as u32).to_be_bytes());
|
||||
out.extend_from_slice(&comp);
|
||||
}
|
||||
}
|
||||
pos += bytes;
|
||||
}
|
||||
out.extend_from_slice(&input[pos..]);
|
||||
Ok(out)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// The definition, one bit at a time.
|
||||
fn naive(input: &[u8], n: usize, es: usize) -> Vec<u8> {
|
||||
let mut out = vec![0u8; n * es];
|
||||
for i in 0..n {
|
||||
for j in 0..es {
|
||||
for k in 0..8 {
|
||||
if input[i * es + j] >> k & 1 == 1 {
|
||||
let p = (8 * j + k) * n + i;
|
||||
out[p / 8] |= 1 << (p % 8);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn transpose_matches_the_definition_and_inverts() {
|
||||
for (n, es) in [(8, 1), (16, 2), (24, 4), (128, 8), (64, 3), (8, 16)] {
|
||||
let input: Vec<u8> = (0..n * es)
|
||||
.map(|i| (i as u32).wrapping_mul(2_654_435_761).rotate_left(7) as u8)
|
||||
.collect();
|
||||
let mut out = vec![0u8; n * es];
|
||||
bitshuffle_block(&input, &mut out, n, es);
|
||||
assert_eq!(out, naive(&input, n, es), "n={n} es={es}");
|
||||
let mut back = vec![0u8; n * es];
|
||||
bitunshuffle_block(&out, &mut back, n, es);
|
||||
assert_eq!(back, input);
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(feature = "bitshuffle")]
|
||||
fn ctx_for(cd: Vec<u32>) -> crate::filter_pipeline::FilterDescription {
|
||||
crate::filter_pipeline::FilterDescription {
|
||||
filter_id: crate::filter_pipeline::FILTER_BITSHUFFLE,
|
||||
name: None,
|
||||
flags: 0,
|
||||
client_data: cd,
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(feature = "bitshuffle")]
|
||||
#[test]
|
||||
fn filter_round_trips_every_mode() {
|
||||
for es in [1usize, 2, 4, 8] {
|
||||
for n in [0usize, 1, 7, 8, 100, 1000, 5003] {
|
||||
let data: Vec<u8> = (0..n * es)
|
||||
.map(|i| (i % 97) as u8 ^ (i / 300) as u8)
|
||||
.collect();
|
||||
for (comp, block) in [(0, 0), (0, 16), (2, 0), (2, 64), (3, 0), (3, 1024)] {
|
||||
let f = ctx_for(vec![0, 4, es as u32, block, comp]);
|
||||
let ctx = FilterContext {
|
||||
filter: &f,
|
||||
element_size: es,
|
||||
max_output: data.len(),
|
||||
};
|
||||
let enc = bitshuffle_encode(&data, &ctx).unwrap();
|
||||
let dec = bitshuffle_decode(&enc, &ctx).unwrap();
|
||||
assert_eq!(dec, data, "es={es} n={n} comp={comp} block={block}");
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(feature = "bitshuffle")]
|
||||
#[test]
|
||||
fn rejects_oversized_and_truncated_chunks() {
|
||||
let data = vec![5u8; 4096];
|
||||
let f = ctx_for(vec![0, 4, 4, 0, 2]);
|
||||
let mut ctx = FilterContext {
|
||||
filter: &f,
|
||||
element_size: 4,
|
||||
max_output: data.len(),
|
||||
};
|
||||
let enc = bitshuffle_encode(&data, &ctx).unwrap();
|
||||
assert!(bitshuffle_decode(&enc[..enc.len() - 1], &ctx).is_err());
|
||||
ctx.max_output = 100;
|
||||
assert!(bitshuffle_decode(&enc, &ctx).is_err());
|
||||
}
|
||||
|
||||
/// Random and mutated chunks, in every mode, and hostile `cd_values`:
|
||||
/// errors are fine, panics are not.
|
||||
#[cfg(feature = "bitshuffle")]
|
||||
#[test]
|
||||
fn fuzzed_chunks_never_panic() {
|
||||
use crate::test_fuzz::{Rng, fuzz_decoder};
|
||||
let data: Vec<u8> = (0..3001u32)
|
||||
.flat_map(|i| ((i / 7) as u16).to_le_bytes())
|
||||
.collect();
|
||||
for (comp, block) in [(0, 0), (0, 16), (2, 0), (2, 64), (3, 0), (3, 1024)] {
|
||||
let f = ctx_for(vec![0, 4, 2, block, comp]);
|
||||
let ctx = FilterContext {
|
||||
filter: &f,
|
||||
element_size: 2,
|
||||
max_output: data.len(),
|
||||
};
|
||||
let seeds = vec![
|
||||
bitshuffle_encode(&data, &ctx).unwrap(),
|
||||
bitshuffle_encode(&data[..34], &ctx).unwrap(),
|
||||
bitshuffle_encode(&data[..512], &ctx).unwrap(),
|
||||
];
|
||||
fuzz_decoder(
|
||||
0xb5 + comp as u64 * 7 + block as u64,
|
||||
&seeds,
|
||||
4_000,
|
||||
data.len(),
|
||||
|s| bitshuffle_decode(s, &ctx),
|
||||
);
|
||||
}
|
||||
// Hostile filter parameters on a valid chunk.
|
||||
let mut rng = Rng::new(0xcd);
|
||||
let good = ctx_for(vec![0, 4, 2, 0, 2]);
|
||||
let enc = bitshuffle_encode(
|
||||
&data,
|
||||
&FilterContext {
|
||||
filter: &good,
|
||||
element_size: 2,
|
||||
max_output: data.len(),
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
for _ in 0..3_000 {
|
||||
let cd: Vec<u32> = (0..rng.below(7))
|
||||
.map(|_| match rng.below(4) {
|
||||
0 => rng.below(5) as u32,
|
||||
1 => u32::MAX - rng.below(4) as u32,
|
||||
2 => 1 << rng.below(32),
|
||||
_ => rng.next_u64() as u32,
|
||||
})
|
||||
.collect();
|
||||
let f = ctx_for(cd);
|
||||
let ctx = FilterContext {
|
||||
filter: &f,
|
||||
element_size: 2,
|
||||
max_output: data.len(),
|
||||
};
|
||||
let _ = bitshuffle_decode(&enc, &ctx);
|
||||
let _ = bitshuffle_decode(&data, &ctx);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,711 @@
|
||||
//! Blosc 1 (HDF5 filter 32001, `hdf5-blosc`, hdf5plugin's `Blosc`), in pure
|
||||
//! Rust: the Blosc 1 frame, its byte shuffle and bit shuffle, and the
|
||||
//! BloscLZ, LZ4/LZ4HC, Snappy, Zlib and Zstandard codecs inside it.
|
||||
//!
|
||||
//! **Frame** (c-blosc 1.x, format version 2). A 16-byte header — version
|
||||
//! (2), codec format version (1), flags, type size, then little-endian `u32`
|
||||
//! decoded size, block size and frame size. Flags: bit 0 byte shuffle, bit
|
||||
//! 1 stored raw ("memcpyed": the data follows the header), bit 2 bit
|
||||
//! shuffle, bit 4 "do not split", bits 5-7 the codec (0 BloscLZ, 1 LZ4 and
|
||||
//! LZ4HC, 2 Snappy, 3 Zlib, 4 Zstandard). Unless stored raw, a table of
|
||||
//! `u32` block offsets follows, one per block of `block size` bytes (the
|
||||
//! last one may be shorter). A block is one stream, or — when the "do not
|
||||
//! split" flag is clear, the type size is at most 16, the block holds at
|
||||
//! least 128 elements, and it is not the short last block — `type size`
|
||||
//! streams, one per byte plane. Each stream is a `u32` length and the
|
||||
//! codec's output; a length equal to the stream's decoded size means the
|
||||
//! bytes are stored raw. The decoded block is then unshuffled (byte shuffle
|
||||
//! for type size > 1; bit shuffle when the block holds a multiple of 8
|
||||
//! elements, the trailing partial element copied as is).
|
||||
//!
|
||||
//! **Filter** (`blosc_filter.c`) `cd_values`: `[0]` filter revision, `[1]`
|
||||
//! Blosc format version, `[2]` type size, `[3]` chunk size in bytes, `[4]`
|
||||
//! compression level, `[5]` shuffle (0 none, 1 byte, 2 bit), `[6]`
|
||||
//! compressor (0 blosclz, 1 lz4, 2 lz4hc, 3 snappy, 4 zlib, 5 zstd). The
|
||||
//! decoder needs only the frame.
|
||||
|
||||
use crate::error::FormatError;
|
||||
use crate::filter_registry::FilterContext;
|
||||
use crate::filters_bitshuffle::{bitshuffle_block, bitunshuffle_block};
|
||||
|
||||
const HEADER: usize = 16;
|
||||
const FLAG_SHUFFLE: u8 = 0x01;
|
||||
const FLAG_MEMCPYED: u8 = 0x02;
|
||||
const FLAG_BITSHUFFLE: u8 = 0x04;
|
||||
const FLAG_FUTURE: u8 = 0x08;
|
||||
const FLAG_DONT_SPLIT: u8 = 0x10;
|
||||
const MAX_SPLITS: usize = 16;
|
||||
const MIN_BUFFERSIZE: usize = 128;
|
||||
|
||||
fn err(msg: &str) -> FormatError {
|
||||
FormatError::DecompressionError(format!("blosc: {msg}"))
|
||||
}
|
||||
|
||||
fn le32(b: &[u8], at: usize) -> Result<usize, FormatError> {
|
||||
b.get(at..at + 4)
|
||||
.map(|s| u32::from_le_bytes(s.try_into().unwrap()) as usize)
|
||||
.ok_or_else(|| err("truncated frame"))
|
||||
}
|
||||
|
||||
/// The codec inside a Blosc frame (flags bits 5-7).
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
enum Codec {
|
||||
BloscLz,
|
||||
Lz4,
|
||||
Snappy,
|
||||
Zlib,
|
||||
Zstd,
|
||||
}
|
||||
|
||||
impl Codec {
|
||||
fn from_flags(flags: u8) -> Result<Codec, FormatError> {
|
||||
match flags >> 5 {
|
||||
0 => Ok(Codec::BloscLz),
|
||||
1 => Ok(Codec::Lz4),
|
||||
2 => Ok(Codec::Snappy),
|
||||
3 => Ok(Codec::Zlib),
|
||||
4 => Ok(Codec::Zstd),
|
||||
other => Err(err(&format!("unknown codec {other}"))),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Decode one codec stream into exactly `dst`.
|
||||
fn decode_stream(
|
||||
codec: Codec,
|
||||
src: &[u8],
|
||||
dst: &mut [u8],
|
||||
zstd: &mut Option<ruzstd::decoding::FrameDecoder>,
|
||||
) -> Result<(), FormatError> {
|
||||
let n = match codec {
|
||||
Codec::BloscLz => blosclz_decompress(src, dst),
|
||||
Codec::Lz4 => {
|
||||
lz4_flex::block::decompress_into(src, dst).map_err(|e| err(&format!("lz4: {e}")))?
|
||||
}
|
||||
Codec::Snappy => {
|
||||
let len = snap::raw::decompress_len(src).map_err(|e| err(&format!("snappy: {e}")))?;
|
||||
if len != dst.len() {
|
||||
return Err(err("snappy stream has the wrong size"));
|
||||
}
|
||||
snap::raw::Decoder::new()
|
||||
.decompress(src, dst)
|
||||
.map_err(|e| err(&format!("snappy: {e}")))?
|
||||
}
|
||||
Codec::Zlib => {
|
||||
let out = crate::filters::inflate_bounded(src, dst.len(), dst.len())
|
||||
.map_err(|e| err(&format!("zlib: {e}")))?;
|
||||
let n = out.len();
|
||||
if n == dst.len() {
|
||||
dst.copy_from_slice(&out);
|
||||
}
|
||||
n
|
||||
}
|
||||
Codec::Zstd => crate::filters_bitshuffle::zstd_decode_into(
|
||||
zstd.get_or_insert_with(ruzstd::decoding::FrameDecoder::new),
|
||||
src,
|
||||
dst,
|
||||
)?,
|
||||
};
|
||||
if n != dst.len() {
|
||||
return Err(err("stream decoded to the wrong size"));
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Decode a Blosc-filtered chunk: one Blosc 1 frame.
|
||||
///
|
||||
/// An HDF5 chunk is never empty, so a frame that decodes to nothing where
|
||||
/// the chunk size is known is corrupt (libhdf5's filter fails it too).
|
||||
pub(crate) fn blosc_decode(input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||
let out = blosc_decompress(input, ctx.output_limit())?;
|
||||
if out.is_empty() && ctx.max_output != 0 {
|
||||
return Err(err("empty frame for a non-empty chunk"));
|
||||
}
|
||||
Ok(out)
|
||||
}
|
||||
|
||||
/// Decompress a Blosc 1 frame, refusing more than `limit` bytes of output.
|
||||
pub fn blosc_decompress(input: &[u8], limit: usize) -> Result<Vec<u8>, FormatError> {
|
||||
if input.len() < HEADER {
|
||||
return Err(err("truncated header"));
|
||||
}
|
||||
let version = input[0];
|
||||
let codec_version = input[1];
|
||||
let flags = input[2];
|
||||
let typesize = input[3] as usize;
|
||||
let nbytes = le32(input, 4)?;
|
||||
let blocksize = le32(input, 8)?;
|
||||
let cbytes = le32(input, 12)?;
|
||||
if version != 1 && version != 2 {
|
||||
return Err(err(&format!(
|
||||
"frame format version {version} is not Blosc 1 (a Blosc 2 chunk?)"
|
||||
)));
|
||||
}
|
||||
if flags & FLAG_FUTURE != 0 {
|
||||
return Err(err("unknown header flags"));
|
||||
}
|
||||
if nbytes > limit {
|
||||
return Err(err("decoded size exceeds the chunk size"));
|
||||
}
|
||||
if cbytes > input.len() {
|
||||
return Err(err("frame is longer than the chunk"));
|
||||
}
|
||||
if cbytes < HEADER {
|
||||
return Err(err("truncated frame"));
|
||||
}
|
||||
let src = &input[..cbytes];
|
||||
if nbytes == 0 {
|
||||
return Ok(Vec::new());
|
||||
}
|
||||
if blocksize == 0 || typesize == 0 {
|
||||
return Err(err("bad block or type size"));
|
||||
}
|
||||
let mut out = vec![0u8; nbytes];
|
||||
if flags & FLAG_MEMCPYED != 0 {
|
||||
if cbytes != nbytes + HEADER {
|
||||
return Err(err("stored frame has the wrong size"));
|
||||
}
|
||||
out.copy_from_slice(&src[HEADER..]);
|
||||
return Ok(out);
|
||||
}
|
||||
let codec = Codec::from_flags(flags)?;
|
||||
if codec_version != 1 {
|
||||
return Err(err(&format!(
|
||||
"unsupported {codec:?} format version {codec_version}"
|
||||
)));
|
||||
}
|
||||
let nblocks = nbytes.div_ceil(blocksize);
|
||||
let leftover = nbytes % blocksize;
|
||||
if nblocks > (cbytes - HEADER) / 4 {
|
||||
return Err(err("block table is truncated"));
|
||||
}
|
||||
let block_len = blocksize.min(nbytes);
|
||||
let mut tmp = vec![0u8; block_len];
|
||||
let mut zstd = None;
|
||||
let dont_split = flags & FLAG_DONT_SPLIT != 0;
|
||||
for j in 0..nblocks {
|
||||
let is_leftover = j == nblocks - 1 && leftover > 0;
|
||||
let bsize = if is_leftover { leftover } else { blocksize };
|
||||
let nsplits = if !dont_split
|
||||
&& typesize <= MAX_SPLITS
|
||||
&& bsize / typesize >= MIN_BUFFERSIZE
|
||||
&& !is_leftover
|
||||
{
|
||||
typesize
|
||||
} else {
|
||||
1
|
||||
};
|
||||
let neblock = bsize / nsplits;
|
||||
let mut pos = le32(src, HEADER + 4 * j)?;
|
||||
let tmp = &mut tmp[..bsize];
|
||||
for s in 0..nsplits {
|
||||
let clen = src
|
||||
.get(pos..)
|
||||
.and_then(|rest| rest.get(..4))
|
||||
.map(|b| u32::from_le_bytes(b.try_into().unwrap()) as usize)
|
||||
.ok_or_else(|| err("block offset out of range"))?;
|
||||
pos += 4;
|
||||
let stream = src
|
||||
.get(pos..pos.saturating_add(clen))
|
||||
.ok_or_else(|| err("stream runs past the frame"))?;
|
||||
let dst = &mut tmp[s * neblock..(s + 1) * neblock];
|
||||
if clen == neblock {
|
||||
dst.copy_from_slice(stream);
|
||||
} else {
|
||||
decode_stream(codec, stream, dst, &mut zstd)?;
|
||||
}
|
||||
pos += clen;
|
||||
}
|
||||
// `bsize` is a whole number of splits by construction (`nsplits` > 1
|
||||
// only for full blocks, and c-blosc sizes those in whole elements).
|
||||
if nsplits * neblock != bsize {
|
||||
return Err(err("block is not a whole number of streams"));
|
||||
}
|
||||
let dest = &mut out[j * blocksize..j * blocksize + bsize];
|
||||
unshuffle_block(flags, typesize, tmp, dest);
|
||||
}
|
||||
Ok(out)
|
||||
}
|
||||
|
||||
/// Undo the frame's shuffle on one decoded block.
|
||||
fn unshuffle_block(flags: u8, typesize: usize, src: &[u8], dest: &mut [u8]) {
|
||||
let bsize = src.len();
|
||||
if flags & FLAG_SHUFFLE != 0 && typesize > 1 {
|
||||
let n = bsize / typesize;
|
||||
for i in 0..n {
|
||||
for b in 0..typesize {
|
||||
dest[i * typesize + b] = src[b * n + i];
|
||||
}
|
||||
}
|
||||
dest[n * typesize..].copy_from_slice(&src[n * typesize..]);
|
||||
} else if flags & FLAG_BITSHUFFLE != 0 && bsize >= typesize {
|
||||
let n = bsize / typesize;
|
||||
if n.is_multiple_of(8) {
|
||||
let body = n * typesize;
|
||||
bitunshuffle_block(&src[..body], &mut dest[..body], n, typesize);
|
||||
dest[body..].copy_from_slice(&src[body..]);
|
||||
} else {
|
||||
dest.copy_from_slice(src);
|
||||
}
|
||||
} else {
|
||||
dest.copy_from_slice(src);
|
||||
}
|
||||
}
|
||||
|
||||
/// BloscLZ decompression (c-blosc 1.21 `blosclz_decompress`): returns the
|
||||
/// number of bytes written, or 0 on malformed input — exactly as the C
|
||||
/// decoder, including stopping before a match that ends the stream, so a
|
||||
/// stream libblosc rejects is rejected here too.
|
||||
///
|
||||
/// Instructions: a control byte `ctrl`. Below 32, a literal run of
|
||||
/// `ctrl + 1` bytes. Otherwise a match: length `(ctrl >> 5) + 2`, extended
|
||||
/// by following bytes while they are 255 when the top three bits are all
|
||||
/// set; distance `((ctrl & 31) << 8) + next byte + 1`, or — when that byte
|
||||
/// is 255 and the high bits are 31 — a 16-bit big-endian distance plus 8192.
|
||||
/// The first instruction is always a literal.
|
||||
pub(crate) fn blosclz_decompress(input: &[u8], out: &mut [u8]) -> usize {
|
||||
const MAX_DISTANCE: usize = 8191;
|
||||
let limit = input.len();
|
||||
if limit == 0 {
|
||||
return 0;
|
||||
}
|
||||
let mut ip = 1usize;
|
||||
let mut op = 0usize;
|
||||
let mut ctrl = (input[0] & 31) as usize;
|
||||
loop {
|
||||
if ctrl >= 32 {
|
||||
let mut len = (ctrl >> 5) - 1;
|
||||
let ofs = (ctrl & 31) << 8;
|
||||
if len == 6 {
|
||||
loop {
|
||||
if ip + 1 >= limit {
|
||||
return 0;
|
||||
}
|
||||
let code = input[ip] as usize;
|
||||
ip += 1;
|
||||
len += code;
|
||||
if code != 255 {
|
||||
break;
|
||||
}
|
||||
}
|
||||
} else if ip + 1 >= limit {
|
||||
return 0;
|
||||
}
|
||||
let code = input[ip] as usize;
|
||||
ip += 1;
|
||||
len += 3;
|
||||
// The copy source is `distance` bytes back.
|
||||
let mut distance = ofs + code + 1;
|
||||
if code == 255 && ofs == 31 << 8 {
|
||||
if ip + 1 >= limit {
|
||||
return 0;
|
||||
}
|
||||
let far = ((input[ip] as usize) << 8) + input[ip + 1] as usize;
|
||||
ip += 2;
|
||||
distance = far + MAX_DISTANCE + 1;
|
||||
}
|
||||
if op + len > out.len() {
|
||||
return 0;
|
||||
}
|
||||
if distance > op {
|
||||
return 0;
|
||||
}
|
||||
if ip >= limit {
|
||||
break;
|
||||
}
|
||||
ctrl = input[ip] as usize;
|
||||
ip += 1;
|
||||
let start = op - distance;
|
||||
if distance >= len {
|
||||
out.copy_within(start..start + len, op);
|
||||
} else {
|
||||
for k in 0..len {
|
||||
out[op + k] = out[start + k];
|
||||
}
|
||||
}
|
||||
op += len;
|
||||
} else {
|
||||
let run = ctrl + 1;
|
||||
if op + run > out.len() || ip + run > limit {
|
||||
return 0;
|
||||
}
|
||||
out[op..op + run].copy_from_slice(&input[ip..ip + run]);
|
||||
op += run;
|
||||
ip += run;
|
||||
if ip >= limit {
|
||||
break;
|
||||
}
|
||||
ctrl = input[ip] as usize;
|
||||
ip += 1;
|
||||
}
|
||||
}
|
||||
op
|
||||
}
|
||||
|
||||
/// The codec our encoder puts inside the frame.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub(crate) enum EncodeCodec {
|
||||
Lz4,
|
||||
Snappy,
|
||||
Zlib,
|
||||
Zstd,
|
||||
}
|
||||
|
||||
impl EncodeCodec {
|
||||
/// From the filter's `cd_values[6]` compressor code.
|
||||
fn from_cd(code: u32) -> Result<EncodeCodec, FormatError> {
|
||||
match code {
|
||||
1 | 2 => Ok(EncodeCodec::Lz4),
|
||||
3 => Ok(EncodeCodec::Snappy),
|
||||
4 => Ok(EncodeCodec::Zlib),
|
||||
5 => Ok(EncodeCodec::Zstd),
|
||||
0 => Err(FormatError::CompressionError(
|
||||
"blosc: clawhdf5 cannot write BloscLZ; choose lz4, snappy, zlib or zstd".into(),
|
||||
)),
|
||||
other => Err(FormatError::CompressionError(format!(
|
||||
"blosc: unknown compressor {other}"
|
||||
))),
|
||||
}
|
||||
}
|
||||
|
||||
fn flags(self) -> u8 {
|
||||
(match self {
|
||||
EncodeCodec::Lz4 => 1,
|
||||
EncodeCodec::Snappy => 2,
|
||||
EncodeCodec::Zlib => 3,
|
||||
EncodeCodec::Zstd => 4,
|
||||
}) << 5
|
||||
}
|
||||
|
||||
fn encode(self, data: &[u8], level: u32) -> Result<Vec<u8>, FormatError> {
|
||||
match self {
|
||||
EncodeCodec::Lz4 => Ok(lz4_flex::block::compress(data)),
|
||||
EncodeCodec::Snappy => snap::raw::Encoder::new()
|
||||
.compress_vec(data)
|
||||
.map_err(|e| FormatError::CompressionError(format!("blosc: snappy: {e}"))),
|
||||
EncodeCodec::Zlib => crate::filters::deflate_bounded(data, level.min(9))
|
||||
.map_err(|e| FormatError::CompressionError(format!("blosc: zlib: {e}"))),
|
||||
EncodeCodec::Zstd => Ok(crate::filters_bitshuffle::zstd_encode(data)),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Block size our encoder uses: at most 256 KiB, a whole number of
|
||||
/// elements (and, for bit shuffle, of 8-element groups).
|
||||
fn encode_block_size(nbytes: usize, typesize: usize, bitshuffle: bool) -> usize {
|
||||
let unit = if bitshuffle { 8 * typesize } else { typesize };
|
||||
let target = (256 * 1024).min(nbytes);
|
||||
if target < unit {
|
||||
return nbytes.max(1);
|
||||
}
|
||||
target / unit * unit
|
||||
}
|
||||
|
||||
/// Encode a chunk as one Blosc 1 frame. `cd_values` as hdf5-blosc:
|
||||
/// `[2]` type size, `[4]` level (0 = store), `[5]` shuffle, `[6]` codec.
|
||||
pub(crate) fn blosc_encode(input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||
let cd = ctx.client_data();
|
||||
let cerr = |m: &str| FormatError::CompressionError(format!("blosc: {m}"));
|
||||
let typesize = match cd.get(2) {
|
||||
Some(&t) if t != 0 => t as usize,
|
||||
_ => ctx.element_size.max(1),
|
||||
};
|
||||
// Blosc records the type size in one byte; c-blosc treats larger types
|
||||
// as bytes.
|
||||
let typesize = if typesize > 255 { 1 } else { typesize };
|
||||
let level = cd.get(4).copied().unwrap_or(5);
|
||||
let shuffle = cd.get(5).copied().unwrap_or(1);
|
||||
let codec = EncodeCodec::from_cd(cd.get(6).copied().unwrap_or(1))?;
|
||||
let nbytes = input.len();
|
||||
if nbytes > i32::MAX as usize - HEADER {
|
||||
return Err(cerr("chunk too large for a Blosc frame"));
|
||||
}
|
||||
let mut flags = codec.flags();
|
||||
match shuffle {
|
||||
0 => {}
|
||||
1 => flags |= FLAG_SHUFFLE,
|
||||
2 => flags |= FLAG_BITSHUFFLE,
|
||||
other => return Err(cerr(&format!("unknown shuffle mode {other}"))),
|
||||
}
|
||||
let blocksize = encode_block_size(nbytes, typesize, shuffle == 2);
|
||||
let header = |flags: u8, blocksize: usize, cbytes: usize| {
|
||||
let mut h = Vec::with_capacity(HEADER);
|
||||
h.extend_from_slice(&[2, 1, flags, typesize as u8]);
|
||||
h.extend_from_slice(&(nbytes as u32).to_le_bytes());
|
||||
h.extend_from_slice(&(blocksize as u32).to_le_bytes());
|
||||
h.extend_from_slice(&(cbytes as u32).to_le_bytes());
|
||||
h
|
||||
};
|
||||
let stored = || {
|
||||
let mut out = header(
|
||||
FLAG_MEMCPYED | (flags & !(FLAG_SHUFFLE | FLAG_BITSHUFFLE)),
|
||||
blocksize,
|
||||
nbytes + HEADER,
|
||||
);
|
||||
out.extend_from_slice(input);
|
||||
out
|
||||
};
|
||||
if level == 0 || nbytes == 0 {
|
||||
return Ok(stored());
|
||||
}
|
||||
|
||||
let nblocks = nbytes.div_ceil(blocksize);
|
||||
let leftover = nbytes % blocksize;
|
||||
let mut body = Vec::with_capacity(nbytes / 2);
|
||||
let mut starts = Vec::with_capacity(nblocks);
|
||||
let table_end = HEADER + 4 * nblocks;
|
||||
let mut shuffled = vec![0u8; blocksize];
|
||||
for j in 0..nblocks {
|
||||
let is_leftover = j == nblocks - 1 && leftover > 0;
|
||||
let bsize = if is_leftover { leftover } else { blocksize };
|
||||
let block = &input[j * blocksize..j * blocksize + bsize];
|
||||
let sh = &mut shuffled[..bsize];
|
||||
shuffle_block(flags, typesize, block, sh);
|
||||
starts.push(table_end + body.len());
|
||||
let nsplits = if typesize <= MAX_SPLITS
|
||||
&& bsize / typesize >= MIN_BUFFERSIZE
|
||||
&& !is_leftover
|
||||
&& bsize.is_multiple_of(typesize)
|
||||
{
|
||||
typesize
|
||||
} else {
|
||||
1
|
||||
};
|
||||
let neblock = bsize / nsplits;
|
||||
for s in 0..nsplits {
|
||||
let part = &sh[s * neblock..(s + 1) * neblock];
|
||||
let comp = codec.encode(part, level)?;
|
||||
if comp.len() < neblock {
|
||||
body.extend_from_slice(&(comp.len() as u32).to_le_bytes());
|
||||
body.extend_from_slice(&comp);
|
||||
} else {
|
||||
body.extend_from_slice(&(neblock as u32).to_le_bytes());
|
||||
body.extend_from_slice(part);
|
||||
}
|
||||
}
|
||||
if table_end + body.len() >= nbytes + HEADER {
|
||||
// Incompressible: store instead, as c-blosc does.
|
||||
return Ok(stored());
|
||||
}
|
||||
}
|
||||
// A split block must decode as split: the decoder infers splitting from
|
||||
// the same rule, which requires a whole number of elements per block.
|
||||
let cbytes = table_end + body.len();
|
||||
let mut out = header(flags, blocksize, cbytes);
|
||||
for s in starts {
|
||||
out.extend_from_slice(&(s as u32).to_le_bytes());
|
||||
}
|
||||
out.extend_from_slice(&body);
|
||||
Ok(out)
|
||||
}
|
||||
|
||||
/// Apply the frame's shuffle to one block (the inverse of
|
||||
/// [`unshuffle_block`]).
|
||||
fn shuffle_block(flags: u8, typesize: usize, src: &[u8], dest: &mut [u8]) {
|
||||
let bsize = src.len();
|
||||
if flags & FLAG_SHUFFLE != 0 && typesize > 1 {
|
||||
let n = bsize / typesize;
|
||||
for i in 0..n {
|
||||
for b in 0..typesize {
|
||||
dest[b * n + i] = src[i * typesize + b];
|
||||
}
|
||||
}
|
||||
dest[n * typesize..].copy_from_slice(&src[n * typesize..]);
|
||||
} else if flags & FLAG_BITSHUFFLE != 0 && bsize >= typesize {
|
||||
let n = bsize / typesize;
|
||||
if n.is_multiple_of(8) {
|
||||
let body = n * typesize;
|
||||
bitshuffle_block(&src[..body], &mut dest[..body], n, typesize);
|
||||
dest[body..].copy_from_slice(&src[body..]);
|
||||
} else {
|
||||
dest.copy_from_slice(src);
|
||||
}
|
||||
} else {
|
||||
dest.copy_from_slice(src);
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::filter_pipeline::{FILTER_BLOSC, FilterDescription};
|
||||
|
||||
/// A blosclz stream: literal "abc", then a 9-byte match 3 back (a run
|
||||
/// of "abc"), then literal "Z".
|
||||
#[test]
|
||||
fn blosclz_decodes_literals_and_overlapping_matches() {
|
||||
// Match: length (ctrl >> 5) + 2 = 8, distance ofs + code + 1 = 3.
|
||||
let stream = [2, b'a', b'b', b'c', (6 << 5), 2, 0, b'Z'];
|
||||
let mut out = [0u8; 12];
|
||||
assert_eq!(blosclz_decompress(&stream, &mut out), 12);
|
||||
assert_eq!(&out, b"abcabcabcabZ");
|
||||
// A stream cut inside a match is malformed.
|
||||
let mut out = [0u8; 11];
|
||||
assert_eq!(blosclz_decompress(&stream[..6], &mut out), 0);
|
||||
// A match before the start of the output is malformed.
|
||||
assert_eq!(blosclz_decompress(&[0, b'a', 32, 5, 0, b'x'], &mut out), 0);
|
||||
}
|
||||
|
||||
fn desc(cd: Vec<u32>) -> FilterDescription {
|
||||
FilterDescription {
|
||||
filter_id: FILTER_BLOSC,
|
||||
name: None,
|
||||
flags: 0,
|
||||
client_data: cd,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn frame_round_trips_every_codec_and_shuffle() {
|
||||
for ts in [1usize, 2, 4, 8, 3, 32] {
|
||||
for n in [0usize, 5, 100, 1000, 70_000, 300_001] {
|
||||
if n * ts > 1 << 20 && ts > 1 {
|
||||
continue;
|
||||
}
|
||||
let data: Vec<u8> = (0..n * ts)
|
||||
.map(|i| ((i / ts) % 200) as u8 ^ (i % ts) as u8)
|
||||
.collect();
|
||||
for codec in [1u32, 3, 4, 5] {
|
||||
for shuffle in [0u32, 1, 2] {
|
||||
for level in [0u32, 5] {
|
||||
let f = desc(vec![2, 2, ts as u32, 0, level, shuffle, codec]);
|
||||
let ctx = FilterContext {
|
||||
filter: &f,
|
||||
element_size: ts,
|
||||
max_output: data.len(),
|
||||
};
|
||||
let enc = blosc_encode(&data, &ctx).unwrap();
|
||||
let dec = blosc_decode(&enc, &ctx).unwrap_or_else(|e| {
|
||||
panic!("ts={ts} n={n} codec={codec} shuffle={shuffle}: {e}")
|
||||
});
|
||||
assert!(
|
||||
dec == data,
|
||||
"ts={ts} n={n} codec={codec} shuffle={shuffle} level={level}"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rejects_bad_frames() {
|
||||
let data = vec![9u8; 50_000];
|
||||
let f = desc(vec![2, 2, 4, 0, 5, 1, 1]);
|
||||
let ctx = FilterContext {
|
||||
filter: &f,
|
||||
element_size: 4,
|
||||
max_output: data.len(),
|
||||
};
|
||||
let enc = blosc_encode(&data, &ctx).unwrap();
|
||||
assert!(blosc_decode(&enc[..enc.len() - 3], &ctx).is_err());
|
||||
let small = FilterContext {
|
||||
max_output: 49_999,
|
||||
..ctx
|
||||
};
|
||||
assert!(blosc_decode(&enc, &small).is_err());
|
||||
let mut v3 = enc.clone();
|
||||
v3[0] = 3;
|
||||
assert!(blosc_decode(&v3, &ctx).is_err());
|
||||
let f0 = desc(vec![2, 2, 4, 0, 5, 1, 0]);
|
||||
let ctx0 = FilterContext { filter: &f0, ..ctx };
|
||||
assert!(blosc_encode(&data, &ctx0).is_err());
|
||||
}
|
||||
|
||||
/// A frame that declares no data, for a chunk that has some.
|
||||
#[test]
|
||||
fn empty_frame_for_a_non_empty_chunk_is_an_error() {
|
||||
let mut frame = vec![2u8, 1, 0x20, 4];
|
||||
for v in [0u32, 64, 16] {
|
||||
frame.extend_from_slice(&v.to_le_bytes());
|
||||
}
|
||||
assert_eq!(blosc_decompress(&frame, 64).unwrap(), b"");
|
||||
let f = desc(vec![2, 2, 4, 64, 5, 1, 1]);
|
||||
let ctx = FilterContext {
|
||||
filter: &f,
|
||||
element_size: 4,
|
||||
max_output: 64,
|
||||
};
|
||||
assert!(blosc_decode(&frame, &ctx).is_err());
|
||||
}
|
||||
|
||||
/// A frame whose header claims a compressed size smaller than the
|
||||
/// header itself, not stored raw: an error, not an arithmetic overflow
|
||||
/// (it panicked in debug builds).
|
||||
#[test]
|
||||
fn frame_size_below_the_header_is_an_error() {
|
||||
let mut frame = vec![2u8, 1, 1 << 5, 4];
|
||||
for v in [64u32, 64, 8] {
|
||||
frame.extend_from_slice(&v.to_le_bytes());
|
||||
}
|
||||
frame.extend_from_slice(&[0; 40]);
|
||||
assert!(blosc_decompress(&frame, 1000).is_err());
|
||||
for cbytes in 0..16u32 {
|
||||
frame[12..16].copy_from_slice(&cbytes.to_le_bytes());
|
||||
assert!(blosc_decompress(&frame, 1000).is_err(), "cbytes={cbytes}");
|
||||
}
|
||||
}
|
||||
|
||||
/// A BloscLZ frame (our encoder cannot write one): a single block,
|
||||
/// one stream, no shuffle.
|
||||
fn blosclz_frame() -> Vec<u8> {
|
||||
let stream = [2, b'a', b'b', b'c', (6 << 5), 2, 0, b'Z'];
|
||||
let mut f = vec![2u8, 1, 0, 1];
|
||||
for v in [12u32, 12, (HEADER + 4 + 4 + stream.len()) as u32] {
|
||||
f.extend_from_slice(&v.to_le_bytes());
|
||||
}
|
||||
f.extend_from_slice(&((HEADER + 4) as u32).to_le_bytes());
|
||||
f.extend_from_slice(&(stream.len() as u32).to_le_bytes());
|
||||
f.extend_from_slice(&stream);
|
||||
f
|
||||
}
|
||||
|
||||
/// Random and mutated frames, every codec and shuffle: errors are fine,
|
||||
/// panics are not.
|
||||
#[test]
|
||||
fn fuzzed_frames_never_panic() {
|
||||
let limit = 6000;
|
||||
let data: Vec<u8> = (0..1500u32).flat_map(|i| (i / 5).to_le_bytes()).collect();
|
||||
let mut seeds = vec![blosclz_frame()];
|
||||
for codec in [1u32, 3, 4, 5] {
|
||||
for shuffle in [0u32, 1, 2] {
|
||||
for (ts, n) in [(4usize, data.len()), (4, 520), (1, 300), (2, 4)] {
|
||||
let f = desc(vec![2, 2, ts as u32, 0, 5, shuffle, codec]);
|
||||
let ctx = FilterContext {
|
||||
filter: &f,
|
||||
element_size: ts,
|
||||
max_output: n,
|
||||
};
|
||||
seeds.push(blosc_encode(&data[..n], &ctx).unwrap());
|
||||
}
|
||||
}
|
||||
}
|
||||
// Stored raw.
|
||||
let f = desc(vec![2, 2, 4, 0, 0, 1, 1]);
|
||||
let ctx = FilterContext {
|
||||
filter: &f,
|
||||
element_size: 4,
|
||||
max_output: 64,
|
||||
};
|
||||
seeds.push(blosc_encode(&data[..64], &ctx).unwrap());
|
||||
crate::test_fuzz::fuzz_decoder(0xb10, &seeds, 30_000, limit, |s| {
|
||||
blosc_decompress(s, limit)
|
||||
});
|
||||
}
|
||||
|
||||
/// BloscLZ streams on their own, random and mutated.
|
||||
#[test]
|
||||
fn fuzzed_blosclz_streams_never_panic() {
|
||||
let seed = blosclz_frame()[HEADER + 8..].to_vec();
|
||||
let mut out = [0u8; 64];
|
||||
crate::test_fuzz::fuzz_decoder(0xb11, &[seed], 30_000, 64, |s| {
|
||||
let n = blosclz_decompress(s, &mut out);
|
||||
if n == 0 {
|
||||
Err(err("malformed"))
|
||||
} else {
|
||||
Ok(out[..n].to_vec())
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,132 @@
|
||||
//! bzip2 (HDF5 filter 307, PyTables' `H5Zbzip2.c`, hdf5plugin's `BZip2`).
|
||||
//!
|
||||
//! The chunk is one bzip2 stream; `cd_values[0]` is the block size (1-9,
|
||||
//! the compression level). Decoded with the `bzip2` crate's default backend,
|
||||
//! `libbz2-rs-sys`, a pure-Rust port of libbzip2.
|
||||
|
||||
use crate::error::FormatError;
|
||||
use crate::filter_registry::FilterContext;
|
||||
|
||||
fn err(msg: &str) -> FormatError {
|
||||
FormatError::DecompressionError(format!("bzip2: {msg}"))
|
||||
}
|
||||
|
||||
/// Decode a bzip2-filtered chunk, refusing output beyond the chunk size.
|
||||
pub(crate) fn bzip2_decode(input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||
use bzip2::{Decompress, Status};
|
||||
let limit = ctx.output_limit();
|
||||
let max_capacity = limit.saturating_add(1);
|
||||
let hint = if ctx.max_output != 0 {
|
||||
ctx.max_output
|
||||
} else {
|
||||
input.len().saturating_mul(4)
|
||||
};
|
||||
let mut out = Vec::new();
|
||||
out.try_reserve_exact(hint.clamp(1, max_capacity))
|
||||
.map_err(|_| err("cannot allocate the output buffer"))?;
|
||||
let mut dec = Decompress::new(false);
|
||||
loop {
|
||||
let (in_before, out_before) = (dec.total_in(), dec.total_out());
|
||||
let status = dec
|
||||
.decompress_vec(&input[in_before as usize..], &mut out)
|
||||
.map_err(|e| err(&e.to_string()))?;
|
||||
if out.len() > limit {
|
||||
return Err(err("output exceeds the chunk size"));
|
||||
}
|
||||
if status == Status::StreamEnd {
|
||||
return Ok(out);
|
||||
}
|
||||
if out.len() == out.capacity() {
|
||||
let grow = out
|
||||
.capacity()
|
||||
.min(max_capacity.saturating_sub(out.capacity()))
|
||||
.max(1);
|
||||
out.try_reserve_exact(grow)
|
||||
.map_err(|_| err("cannot allocate the output buffer"))?;
|
||||
} else if dec.total_in() as usize >= input.len()
|
||||
|| (dec.total_in(), dec.total_out()) == (in_before, out_before)
|
||||
{
|
||||
return Err(err("truncated stream"));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Encode a chunk as one bzip2 stream at block size `cd_values[0]`
|
||||
/// (default 9, as hdf5plugin).
|
||||
pub(crate) fn bzip2_encode(input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||
use bzip2::{Action, Compress, Compression, Status};
|
||||
let level = ctx.client_data().first().copied().unwrap_or(9).clamp(1, 9);
|
||||
let cerr = |m: String| FormatError::CompressionError(format!("bzip2: {m}"));
|
||||
let mut enc = Compress::new(Compression::new(level), 0);
|
||||
// bzip2's worst case is about 1% + 600 bytes over the input.
|
||||
let mut out = Vec::with_capacity(input.len() + input.len() / 100 + 600);
|
||||
loop {
|
||||
let consumed = enc.total_in() as usize;
|
||||
let status = enc
|
||||
.compress_vec(&input[consumed..], &mut out, Action::Finish)
|
||||
.map_err(|e| cerr(e.to_string()))?;
|
||||
if status == Status::StreamEnd {
|
||||
return Ok(out);
|
||||
}
|
||||
if out.len() == out.capacity() {
|
||||
out.reserve(out.capacity().max(4096));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::filter_pipeline::{FILTER_BZIP2, FilterDescription};
|
||||
|
||||
fn desc(level: u32) -> FilterDescription {
|
||||
FilterDescription {
|
||||
filter_id: FILTER_BZIP2,
|
||||
name: None,
|
||||
flags: 0,
|
||||
client_data: vec![level],
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn round_trips_and_bounds() {
|
||||
let data: Vec<u8> = (0..100_000u32)
|
||||
.flat_map(|i| (i % 777).to_le_bytes())
|
||||
.collect();
|
||||
for level in [1, 5, 9] {
|
||||
let f = desc(level);
|
||||
let ctx = FilterContext {
|
||||
filter: &f,
|
||||
element_size: 4,
|
||||
max_output: data.len(),
|
||||
};
|
||||
let enc = bzip2_encode(&data, &ctx).unwrap();
|
||||
assert!(enc.len() < data.len() / 4);
|
||||
assert_eq!(bzip2_decode(&enc, &ctx).unwrap(), data);
|
||||
// Truncated, and larger than the chunk: errors, not data.
|
||||
assert!(bzip2_decode(&enc[..enc.len() / 2], &ctx).is_err());
|
||||
let small = FilterContext {
|
||||
max_output: data.len() - 1,
|
||||
..ctx
|
||||
};
|
||||
assert!(bzip2_decode(&enc, &small).is_err());
|
||||
}
|
||||
}
|
||||
|
||||
/// Random and mutated streams: errors are fine, panics are not.
|
||||
#[test]
|
||||
fn fuzzed_streams_never_panic() {
|
||||
let f = desc(9);
|
||||
let data: Vec<u8> = (0..4000u32).flat_map(|i| (i % 91).to_le_bytes()).collect();
|
||||
let ctx = FilterContext {
|
||||
filter: &f,
|
||||
element_size: 4,
|
||||
max_output: data.len(),
|
||||
};
|
||||
let seeds = vec![
|
||||
bzip2_encode(&data, &ctx).unwrap(),
|
||||
bzip2_encode(&data[..40], &ctx).unwrap(),
|
||||
];
|
||||
crate::test_fuzz::fuzz_decoder(0xb2, &seeds, 3_000, data.len(), |s| bzip2_decode(s, &ctx));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,260 @@
|
||||
//! LZF (HDF5 filter 32000) — h5py's built-in compression filter
|
||||
//! (`compression="lzf"`), in pure Rust.
|
||||
//!
|
||||
//! The chunk is one raw LZF stream (liblzf 3.x format, no header). The
|
||||
//! stream is a sequence of instructions, each starting with a control byte:
|
||||
//!
|
||||
//! * `000LLLLL` — a literal run: the next `L + 1` bytes (1..=32) are copied.
|
||||
//! * `LLLOOOOO [E] OOOOOOOO` — a back reference: copy `len + 2` bytes from
|
||||
//! `distance` bytes back, where `len` is the top three bits (1..=6), or
|
||||
//! `7 + E` when they are all ones, and `distance` is the 13-bit offset
|
||||
//! (high five bits in the control byte, low eight in the last byte) plus 1.
|
||||
//!
|
||||
//! h5py's filter (`lzf_filter.c`) records the chunk's size in bytes in
|
||||
//! `cd_values[2]` (slots 0 and 1 hold the filter and liblzf versions) and
|
||||
//! sizes its output buffer from it.
|
||||
|
||||
#[cfg(not(feature = "std"))]
|
||||
extern crate alloc;
|
||||
#[cfg(not(feature = "std"))]
|
||||
use alloc::{format, vec, vec::Vec};
|
||||
|
||||
use crate::error::FormatError;
|
||||
use crate::filter_registry::FilterContext;
|
||||
|
||||
/// `H5PY_FILTER_LZF_VERSION`, written to `cd_values[0]`.
|
||||
pub const LZF_FILTER_VERSION: u32 = 4;
|
||||
/// `LZF_VERSION` (liblzf 1.5), written to `cd_values[1]`.
|
||||
pub const LZF_API_VERSION: u32 = 0x0105;
|
||||
|
||||
const MAX_LITERAL: usize = 32;
|
||||
const MAX_OFFSET: usize = 1 << 13;
|
||||
const MAX_REF: usize = (1 << 8) + (1 << 3);
|
||||
const HASH_LOG: u32 = 14;
|
||||
|
||||
fn err(msg: &str) -> FormatError {
|
||||
FormatError::DecompressionError(format!("lzf: {msg}"))
|
||||
}
|
||||
|
||||
/// Decode an LZF-filtered chunk.
|
||||
pub(crate) fn lzf_decode(input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||
let limit = ctx.output_limit();
|
||||
let hint = match ctx.client_data().get(2) {
|
||||
Some(&n) if n != 0 => n as usize,
|
||||
_ => input.len().saturating_mul(2),
|
||||
};
|
||||
lzf_decompress(input, hint.min(limit), limit)
|
||||
}
|
||||
|
||||
/// Decompress a raw LZF stream, refusing to produce more than `limit` bytes.
|
||||
pub fn lzf_decompress(
|
||||
input: &[u8],
|
||||
size_hint: usize,
|
||||
limit: usize,
|
||||
) -> Result<Vec<u8>, FormatError> {
|
||||
let mut out: Vec<u8> = Vec::new();
|
||||
out.try_reserve(size_hint)
|
||||
.map_err(|_| err("cannot allocate the output buffer"))?;
|
||||
let mut ip = 0usize;
|
||||
while ip < input.len() {
|
||||
let ctrl = input[ip] as usize;
|
||||
ip += 1;
|
||||
if ctrl < 32 {
|
||||
let run = ctrl + 1;
|
||||
let lit = input
|
||||
.get(ip..ip + run)
|
||||
.ok_or_else(|| err("literal run past the end of the input"))?;
|
||||
if out.len() + run > limit {
|
||||
return Err(err("output exceeds the chunk size"));
|
||||
}
|
||||
out.extend_from_slice(lit);
|
||||
ip += run;
|
||||
} else {
|
||||
let mut len = ctrl >> 5;
|
||||
if len == 7 {
|
||||
len += *input
|
||||
.get(ip)
|
||||
.ok_or_else(|| err("truncated back reference"))?
|
||||
as usize;
|
||||
ip += 1;
|
||||
}
|
||||
let low = *input
|
||||
.get(ip)
|
||||
.ok_or_else(|| err("truncated back reference"))? as usize;
|
||||
ip += 1;
|
||||
let distance = ((ctrl & 0x1f) << 8) + low + 1;
|
||||
let len = len + 2;
|
||||
if distance > out.len() {
|
||||
return Err(err("back reference before the start of the output"));
|
||||
}
|
||||
if out.len() + len > limit {
|
||||
return Err(err("output exceeds the chunk size"));
|
||||
}
|
||||
let start = out.len() - distance;
|
||||
if distance >= len {
|
||||
out.extend_from_within(start..start + len);
|
||||
} else {
|
||||
// Overlapping copy: repeats the last `distance` bytes.
|
||||
for k in 0..len {
|
||||
let b = out[start + k];
|
||||
out.push(b);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
Ok(out)
|
||||
}
|
||||
|
||||
/// Encode a chunk with the LZF filter.
|
||||
pub(crate) fn lzf_encode(input: &[u8], _ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||
Ok(lzf_compress(input))
|
||||
}
|
||||
|
||||
fn hash3(b: &[u8]) -> usize {
|
||||
let v = (u32::from(b[0]) << 16) | (u32::from(b[1]) << 8) | u32::from(b[2]);
|
||||
(v.wrapping_mul(2_654_435_761) >> (32 - HASH_LOG)) as usize
|
||||
}
|
||||
|
||||
fn flush_literals(out: &mut Vec<u8>, lit: &[u8]) {
|
||||
for run in lit.chunks(MAX_LITERAL) {
|
||||
out.push((run.len() - 1) as u8);
|
||||
out.extend_from_slice(run);
|
||||
}
|
||||
}
|
||||
|
||||
/// Compress `input` into a raw LZF stream any liblzf decoder reads.
|
||||
///
|
||||
/// Incompressible input grows by one byte per 32. (h5py's own filter gives
|
||||
/// up on such a chunk and stores it unfiltered; storing the slightly larger
|
||||
/// stream is equally readable.)
|
||||
pub fn lzf_compress(input: &[u8]) -> Vec<u8> {
|
||||
let n = input.len();
|
||||
let mut out = Vec::with_capacity(n + n / MAX_LITERAL + 1);
|
||||
let mut table = vec![0u32; 1 << HASH_LOG];
|
||||
let mut lit_start = 0usize;
|
||||
let mut i = 0usize;
|
||||
while i + 2 < n {
|
||||
let h = hash3(&input[i..]);
|
||||
let cand = table[h] as usize;
|
||||
table[h] = (i + 1) as u32;
|
||||
if cand != 0 {
|
||||
let r = cand - 1;
|
||||
let distance = i - r;
|
||||
if distance <= MAX_OFFSET && input[r..r + 3] == input[i..i + 3] {
|
||||
let max_len = (n - i).min(MAX_REF);
|
||||
let mut len = 3;
|
||||
while len < max_len && input[r + len] == input[i + len] {
|
||||
len += 1;
|
||||
}
|
||||
flush_literals(&mut out, &input[lit_start..i]);
|
||||
let code = len - 2;
|
||||
let off = distance - 1;
|
||||
if code < 7 {
|
||||
out.push(((code << 5) | (off >> 8)) as u8);
|
||||
} else {
|
||||
out.push(((7 << 5) | (off >> 8)) as u8);
|
||||
out.push((code - 7) as u8);
|
||||
}
|
||||
out.push((off & 0xff) as u8);
|
||||
// Index the positions the match covered so later data can
|
||||
// refer back into it.
|
||||
let end = i + len;
|
||||
let mut j = i + 1;
|
||||
while j < end && j + 2 < n {
|
||||
table[hash3(&input[j..])] = (j + 1) as u32;
|
||||
j += 1;
|
||||
}
|
||||
i = end;
|
||||
lit_start = i;
|
||||
continue;
|
||||
}
|
||||
}
|
||||
i += 1;
|
||||
}
|
||||
flush_literals(&mut out, &input[lit_start..]);
|
||||
out
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn round_trip(data: &[u8]) {
|
||||
let c = lzf_compress(data);
|
||||
assert_eq!(lzf_decompress(&c, data.len(), data.len()).unwrap(), data);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn round_trips() {
|
||||
round_trip(b"");
|
||||
round_trip(b"a");
|
||||
round_trip(b"abcabcabcabcabcabcabcabcabcabcabcabc");
|
||||
round_trip(&[7u8; 10_000]);
|
||||
let noise: Vec<u8> = (0..70_000u32)
|
||||
.map(|i| (i.wrapping_mul(2_654_435_761) >> 13) as u8)
|
||||
.collect();
|
||||
round_trip(&noise);
|
||||
let ramp: Vec<u8> = (0..100_000u32)
|
||||
.flat_map(|i| (i % 1000).to_le_bytes())
|
||||
.collect();
|
||||
round_trip(&ramp);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn compresses_repetitive_data() {
|
||||
let data = [42u8; 4096];
|
||||
assert!(lzf_compress(&data).len() < 100);
|
||||
}
|
||||
|
||||
/// The chunk h5py 3.16's bundled liblzf writes for
|
||||
/// `b"hello hello hello hello"` (read back with `read_direct_chunk`): a
|
||||
/// 7-byte literal, a 14-byte back reference 6 bytes back (extended
|
||||
/// length), and a 2-byte literal.
|
||||
#[test]
|
||||
fn decodes_liblzf_output() {
|
||||
let stream = b"\x06hello h\xe0\x05\x05\x01lo";
|
||||
assert_eq!(
|
||||
lzf_decompress(stream, 23, 23).unwrap(),
|
||||
b"hello hello hello hello"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rejects_corrupt_streams() {
|
||||
// Back reference before the start.
|
||||
assert!(lzf_decompress(&[0x20, 0x00], 10, 10).is_err());
|
||||
// Literal run past the end.
|
||||
assert!(lzf_decompress(&[0x05, 1, 2], 10, 10).is_err());
|
||||
// Output over the limit.
|
||||
let c = lzf_compress(&[1u8; 100]);
|
||||
assert!(lzf_decompress(&c, 10, 99).is_err());
|
||||
}
|
||||
|
||||
/// Random and mutated streams: errors are fine, panics are not.
|
||||
#[test]
|
||||
fn fuzzed_streams_never_panic() {
|
||||
let seeds: Vec<Vec<u8>> = [
|
||||
b"hello hello hello hello".to_vec(),
|
||||
vec![7u8; 3000],
|
||||
(0..2000u32).flat_map(|i| (i % 37).to_le_bytes()).collect(),
|
||||
(0..500u32)
|
||||
.map(|i| (i.wrapping_mul(2_654_435_761) >> 13) as u8)
|
||||
.collect(),
|
||||
]
|
||||
.iter()
|
||||
.map(|d| lzf_compress(d))
|
||||
.collect();
|
||||
for limit in [0usize, 23, 4096, 8000] {
|
||||
crate::test_fuzz::fuzz_decoder(
|
||||
0x1f2 + limit as u64,
|
||||
&seeds[..1],
|
||||
5_000,
|
||||
limit.max(23),
|
||||
|s| lzf_decompress(s, limit, limit.max(23)),
|
||||
);
|
||||
}
|
||||
crate::test_fuzz::fuzz_decoder(0x1f3, &seeds, 20_000, 8000, |s| {
|
||||
lzf_decompress(s, 8000, 8000)
|
||||
});
|
||||
}
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user