Compare commits
399
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
ef428d756c | ||
|
|
4313917b4d | ||
|
|
011e0dbb96 | ||
|
|
f37e7ae326 | ||
|
|
7447dce121 | ||
|
|
93e2d5f365 | ||
|
|
2893b6c974 | ||
|
|
ea0508aaa5 | ||
|
|
75444950f3 | ||
|
|
8236b0e30a | ||
|
|
c2ae7846c9 | ||
|
|
a69c5be8b2 | ||
|
|
930921e8cb | ||
|
|
159e588550 | ||
|
|
6185874f9c | ||
|
|
89e7977943 | ||
|
|
67e72b30d7 | ||
|
|
0e98ffc498 | ||
|
|
efb88f94e3 | ||
|
|
ef480746da | ||
|
|
c5b2afbc35 | ||
|
|
b086dc3c2b | ||
|
|
30a1ed6b9c | ||
|
|
61e34927dc | ||
|
|
680c90b3a8 | ||
|
|
7d629f49e3 | ||
|
|
c04e34620e | ||
|
|
8df5b209a7 | ||
|
|
e8aaf050be | ||
|
|
5062b907bd | ||
|
|
4f5697fdd9 | ||
|
|
955dd1c691 | ||
|
|
ebe51f8e97 | ||
|
|
4e8109770d | ||
|
|
c513f7e6d7 | ||
|
|
a4f586e657 | ||
|
|
c54c64cc9b | ||
|
|
4ff3e40fea | ||
|
|
0aca0eb724 | ||
|
|
db2554dd81 | ||
|
|
955fdb660d | ||
|
|
304aed5813 | ||
|
|
1ffd013de9 | ||
|
|
dc9cfba6bb | ||
|
|
773f427f16 | ||
|
|
7e5e920c72 | ||
|
|
f191dc09d5 | ||
|
|
1c3ef98828 | ||
|
|
e9c71e5d2e | ||
|
|
17201e279d | ||
|
|
3fa5ed1dda | ||
|
|
42894bf93b | ||
|
|
8f59b2e1c2 | ||
|
|
0645dcf173 | ||
|
|
cadd27df5b | ||
|
|
b49ec39aff | ||
|
|
8fadb9f424 | ||
|
|
c233fbca6e | ||
|
|
437e81cfff | ||
|
|
234dd3e36c | ||
|
|
0e8522cfad | ||
|
|
895c79a2fe | ||
|
|
76c97f6c94 | ||
|
|
e5359354b7 | ||
|
|
052098bf36 | ||
|
|
fe377266e1 | ||
|
|
b668878129 | ||
|
|
485bea0f4f | ||
|
|
1ea9132e10 | ||
|
|
04a7f6f6c7 | ||
|
|
5b3d32b37d | ||
|
|
f7e2ab12f2 | ||
|
|
b6cbd2319f | ||
|
|
92c8285549 | ||
|
|
d2b25f154f | ||
|
|
85efde0b4a | ||
|
|
677dc5ec7c | ||
|
|
3c89a31df0 | ||
|
|
1b4a93f65a | ||
|
|
b41583113a | ||
|
|
02e89c1d2d | ||
|
|
5d17712adb | ||
|
|
476960f4b8 | ||
|
|
24f0c71939 | ||
|
|
5705866d40 | ||
|
|
2b7065998a | ||
|
|
6a535e2651 | ||
|
|
a65a2b7f18 | ||
|
|
2c292404d2 | ||
|
|
ba3f476be6 | ||
|
|
ff6d644391 | ||
|
|
cf2b408a63 | ||
|
|
d97b3d703a | ||
|
|
23a4784e72 | ||
|
|
a0160730f2 | ||
|
|
24cbf12f16 | ||
|
|
bcf3ae4856 | ||
|
|
06625b7470 | ||
|
|
aab7ea9e8f | ||
|
|
6a9bb02f37 | ||
|
|
0d908facd3 | ||
|
|
cd828725c7 | ||
|
|
512a6a753f | ||
|
|
6a4707d791 | ||
|
|
6248b411f0 | ||
|
|
479d8b47e0 | ||
|
|
f2ff2c424f | ||
|
|
c5334b1c97 | ||
|
|
d0e3beb3aa | ||
|
|
9a73299594 | ||
|
|
4b02e7d068 | ||
|
|
d7f07fa5c1 | ||
|
|
bdb2c0e36b | ||
|
|
55e0e7e9cf | ||
|
|
00f94d57ed | ||
|
|
4c01267b76 | ||
|
|
d493d4792e | ||
|
|
6559a91495 | ||
|
|
60502593b7 | ||
|
|
a6ed3a5c7d | ||
|
|
3da118d2ee | ||
|
|
7515e5dcbd | ||
|
|
742ed4dfb8 | ||
|
|
2ba4bc97d8 | ||
|
|
d9e4dfb6e6 | ||
|
|
22dc87b07c | ||
|
|
e05530a805 | ||
|
|
989335b67b | ||
|
|
bf4aefcd00 | ||
|
|
378afa1584 | ||
|
|
67958b08d9 | ||
|
|
f512bf3d09 | ||
|
|
8295d01614 | ||
|
|
94b6df986c | ||
|
|
6b3d003950 | ||
|
|
d110b1d945 | ||
|
|
b3058ca46e | ||
|
|
d3d73676c0 | ||
|
|
56abaec75e | ||
|
|
7334e21c93 | ||
|
|
b22b15f00a | ||
|
|
9e608b975c | ||
|
|
9e9b849dd7 | ||
|
|
193a5f8a82 | ||
|
|
3aab433edb | ||
|
|
1c1af460b6 | ||
|
|
c5cd14c2b2 | ||
|
|
16b7359485 | ||
|
|
1207df5189 | ||
|
|
f0db817678 | ||
|
|
9e59499c56 | ||
|
|
d63c76e7ab | ||
|
|
cc1c872a93 | ||
|
|
7acfb79584 | ||
|
|
de2a53f613 | ||
|
|
dda28d6c72 | ||
|
|
408f69ec1d | ||
|
|
73a01f1256 | ||
|
|
956e55c76a | ||
|
|
846c35455d | ||
|
|
ca779b2864 | ||
|
|
20bd381c87 | ||
|
|
5a202f3791 | ||
|
|
8dcce084ca | ||
|
|
45d617c39e | ||
|
|
d345ffbf80 | ||
|
|
17edfe2cf0 | ||
|
|
546fdb84fa | ||
|
|
05b0192a60 | ||
|
|
8bcae3c78e | ||
|
|
f0ecae38b6 | ||
|
|
400e3a9fec | ||
|
|
bd1d8f1a59 | ||
|
|
751edeb7e6 | ||
|
|
b43bd2e67f | ||
|
|
81a0e8685d | ||
|
|
41b7837d0a | ||
|
|
8c51b05b9c | ||
|
|
24412a0e59 | ||
|
|
3bcd443e63 | ||
|
|
37770f594a | ||
|
|
a5e41c1a53 | ||
|
|
b0a1e4f9a6 | ||
|
|
bd36fe883b | ||
|
|
d102c06306 | ||
|
|
6e8421a81e | ||
|
|
8ce6eca34d | ||
|
|
c3850a0b66 | ||
|
|
2bc4cb46a6 | ||
|
|
f7d88bb4fb | ||
|
|
f99587c27d | ||
|
|
2d4b211523 | ||
|
|
10da8f0d09 | ||
|
|
83cda847dd | ||
|
|
78c769f179 | ||
|
|
006bf3b131 | ||
|
|
8cbbef3fae | ||
|
|
55309dd242 | ||
|
|
63648c7000 | ||
|
|
91644d8aaf | ||
|
|
72306c6013 | ||
|
|
e60bde3579 | ||
|
|
c85a8222cc | ||
|
|
2b68791f6a | ||
|
|
f7c362cef5 | ||
|
|
13c095a3da | ||
|
|
591aa71d12 | ||
|
|
b9a2ce3077 | ||
|
|
743c32b512 | ||
|
|
993214723e | ||
|
|
3938f7f8a2 | ||
|
|
dd40bea467 | ||
|
|
afae86f3ea | ||
|
|
f713847e65 | ||
|
|
17fc8b1964 | ||
|
|
9238605661 | ||
|
|
738b9491b2 | ||
|
|
e10df68ed8 | ||
|
|
c4d96c1390 | ||
|
|
b8492bd28d | ||
|
|
a5bd70216c | ||
|
|
17f09375ad | ||
|
|
386bd1d41e | ||
|
|
b5e43bacd7 | ||
|
|
a14ccc36bf | ||
|
|
7f52a6f3ba | ||
|
|
f325d111f3 | ||
|
|
699ee9c447 | ||
|
|
9416c58723 | ||
|
|
6a8ee3ec7f | ||
|
|
a59d83d47d | ||
|
|
bb39be7f24 | ||
|
|
845a9d0125 | ||
|
|
7d7a7e75d4 | ||
|
|
0685037593 | ||
|
|
e92faa23a6 | ||
|
|
40968b3578 | ||
|
|
310448bfcb | ||
|
|
e73ac2af09 | ||
|
|
e01160299a | ||
|
|
5461a13984 | ||
|
|
056092b082 | ||
|
|
a5ca970015 | ||
|
|
e7a7951f1e | ||
|
|
1f71f3bcbc | ||
|
|
3cf8cd86f2 | ||
|
|
34987ec194 | ||
|
|
b58d61cfb7 | ||
|
|
1abd93e0f8 | ||
|
|
6dfd239011 | ||
|
|
07094e34a9 | ||
|
|
f4dee1cd08 | ||
|
|
e38f9123db | ||
|
|
a42b646689 | ||
|
|
74f9f50086 | ||
|
|
d16544b928 | ||
|
|
3b24e6753b | ||
|
|
e815eb922f | ||
|
|
c9c5337a62 | ||
|
|
bb78d70b99 | ||
|
|
a7de15534c | ||
|
|
10d1029ead | ||
|
|
883980f2bd | ||
|
|
d6e426e6d5 | ||
|
|
256e7b89e4 | ||
|
|
f2e704abf3 | ||
|
|
61f36516d7 | ||
|
|
45720fe5a6 | ||
|
|
adf961c883 | ||
|
|
b4a44a2e66 | ||
|
|
17fa783dce | ||
|
|
90e050944f | ||
|
|
a6e90f3ee3 | ||
|
|
0555794850 | ||
|
|
efc2dc53c9 | ||
|
|
5c2f656fe7 | ||
|
|
945b13a1f1 | ||
|
|
9179aa356e | ||
|
|
e94a52a88b | ||
|
|
d54a0f4737 | ||
|
|
aadfd18d4c | ||
|
|
2c6c6c176e | ||
|
|
190918a478 | ||
|
|
38d0d4de02 | ||
|
|
1c85986079 | ||
|
|
c7092722aa | ||
|
|
36356ba8a1 | ||
|
|
85eb7f5ce2 | ||
|
|
8196fab72a | ||
|
|
8ebd488d9e | ||
|
|
42b81d9f1c | ||
|
|
72b9cfb1e1 | ||
|
|
650f355219 | ||
|
|
7f5cfee281 | ||
|
|
c5302e587e | ||
|
|
e1115bc92a | ||
|
|
36d7a6f234 | ||
|
|
4b23ad697c | ||
|
|
e7f2d8575d | ||
|
|
7c1968a34a | ||
|
|
6db13c60b8 | ||
|
|
d99426be94 | ||
|
|
f5505fb03d | ||
|
|
95dcb04454 | ||
|
|
1dba7b465a | ||
|
|
57e938c438 | ||
|
|
3000b40cf3 | ||
|
|
bc820fbd8c | ||
|
|
9066d34eaa | ||
|
|
540fa08907 | ||
|
|
14876b8ae5 | ||
|
|
5935e13866 | ||
|
|
8c3ef996ea | ||
|
|
74fdf0582b | ||
|
|
44f5f8b5c5 | ||
|
|
2f252df084 | ||
|
|
c8c2930fc0 | ||
|
|
417c9516ca | ||
|
|
53dbddb07b | ||
|
|
081341b433 | ||
|
|
d074385944 | ||
|
|
4a1876faf2 | ||
|
|
be88e3fec7 | ||
|
|
e162c013fd | ||
|
|
06dda26d85 | ||
|
|
585e14d5e2 | ||
|
|
183d96ee26 | ||
|
|
bba1560416 | ||
|
|
b36998ef01 | ||
|
|
aef8e766ae | ||
|
|
9ea44d473d | ||
|
|
46203ea761 | ||
|
|
75bdb53342 | ||
|
|
dd5b3f6633 | ||
|
|
79dfa78e8f | ||
|
|
87d64588e5 | ||
|
|
bdadf3447c | ||
|
|
0c65a27b00 | ||
|
|
db9af7972c | ||
|
|
7706697feb | ||
|
|
4ecac65f22 | ||
|
|
c0f704c381 | ||
|
|
00b0cb0035 | ||
|
|
a7920bd4b3 | ||
|
|
7e43b5366c | ||
|
|
dce5559ff2 | ||
|
|
1b3bbb054a | ||
|
|
a8fb758489 | ||
|
|
73bb068264 | ||
|
|
5c8323cb1e | ||
|
|
dbaf3f505d | ||
|
|
c470244a6f | ||
|
|
d0db83812b | ||
|
|
5e4aa1c6bf | ||
|
|
1cceb930b2 | ||
|
|
fc7ae6549a | ||
|
|
735db117a7 | ||
|
|
e9b37a9602 | ||
|
|
4bed8b3765 | ||
|
|
36d689bc2c | ||
|
|
e7c08e06b4 | ||
|
|
c5049eb734 | ||
|
|
6f6bc97850 | ||
|
|
0cb72e8a60 | ||
|
|
e338d58ad5 | ||
|
|
8b85d9364b | ||
|
|
6598a7d02f | ||
|
|
114a2dfcba | ||
|
|
56a8c2f3d0 | ||
|
|
7b16dc90d6 | ||
|
|
4a5544da1d | ||
|
|
f4c6d43a3f | ||
|
|
e5e087f9ab | ||
|
|
16c9ee0554 | ||
|
|
1d767e3b93 | ||
|
|
a91df3f1c3 | ||
|
|
b41272487a | ||
|
|
0bc7a293ae | ||
|
|
dea02f5214 | ||
|
|
97e65f2adf | ||
|
|
fb58300b3f | ||
|
|
eb196e824f | ||
|
|
367faad7f7 | ||
|
|
0901fb1499 | ||
|
|
e9aeb110b7 | ||
|
|
5889b378e9 | ||
|
|
18dc35f7e5 | ||
|
|
e17ab0ceef | ||
|
|
105cf13347 | ||
|
|
0529f72a2c | ||
|
|
a29c1b224b | ||
|
|
c0a9206703 | ||
|
|
57756e69ec | ||
|
|
6ad8ceb426 | ||
|
|
2e7e0456c1 | ||
|
|
dc0113d015 | ||
|
|
8ea455bbcb | ||
|
|
1e18ff5a86 | ||
|
|
306a35347c |
+69
-12
@@ -9,32 +9,89 @@ jobs:
|
|||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
container: rust:latest
|
container: rust:latest
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v4
|
# Plain git rather than actions/checkout: that is a JavaScript action,
|
||||||
- name: Cache cargo registry/target
|
# and rust:latest has no `node`, so it failed with exit 127 before any
|
||||||
uses: actions/cache@v4
|
# code was built — on every push. actions/cache went for the same reason.
|
||||||
with:
|
- name: Check out
|
||||||
path: |
|
run: |
|
||||||
~/.cargo/registry
|
git init -q .
|
||||||
~/.cargo/git
|
git remote add origin "${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}.git"
|
||||||
target
|
for i in 1 2 3; do git fetch -q --depth 1 origin "${GITHUB_SHA}" && break; sleep 5; done
|
||||||
key: ${{ runner.os }}-cargo-${{ hashFiles('**/Cargo.lock') }}
|
git checkout -q FETCH_HEAD
|
||||||
- name: Install rustfmt & clippy components
|
- name: Install rustfmt & clippy components
|
||||||
run: rustup component add rustfmt clippy
|
run: rustup component add rustfmt clippy
|
||||||
- name: Install thumbv7em-none-eabihf target
|
- name: Install thumbv7em-none-eabihf target
|
||||||
run: rustup target add thumbv7em-none-eabihf
|
run: rustup target add thumbv7em-none-eabihf
|
||||||
|
- name: Install wasm32-unknown-unknown target
|
||||||
|
# ci-test.sh builds the reader and clawhdf5-wasm for the browser.
|
||||||
|
run: rustup target add wasm32-unknown-unknown
|
||||||
- name: Install Python interop dependencies
|
- name: Install Python interop dependencies
|
||||||
# The interop suites used to skip silently when python3/h5py were
|
# The interop suites used to skip silently when python3/h5py were
|
||||||
# missing, so they never ran in CI. Install them and make a missing
|
# missing, so they never ran in CI. Install them and make a missing
|
||||||
# dependency a failure (CLAWHDF5_REQUIRE_INTEROP below).
|
# dependency a failure (CLAWHDF5_REQUIRE_INTEROP below).
|
||||||
run: |
|
run: |
|
||||||
apt-get update
|
apt-get update
|
||||||
apt-get install -y --no-install-recommends python3 python3-venv
|
# cmake builds libz-ng-sys for the opt-in `fast-deflate` (zlib-ng)
|
||||||
|
# steps in ci-test.sh; rust:latest does not ship it. The default
|
||||||
|
# build (pure-Rust zlib-rs) does not need it.
|
||||||
|
# hdf5-tools: h5ls/h5stat/h5dump/h5diff, which the h5rs
|
||||||
|
# (clawhdf5-tools) interop tests compare against.
|
||||||
|
apt-get install -y --no-install-recommends python3 python3-venv cmake hdf5-tools
|
||||||
python3 -m venv /opt/interop
|
python3 -m venv /opt/interop
|
||||||
/opt/interop/bin/pip install --no-cache-dir h5py numpy netCDF4 xarray
|
# maturin + pytest: ci-test.sh builds the Python package
|
||||||
|
# (crates/clawhdf5-py) and runs its tests against h5py.
|
||||||
|
/opt/interop/bin/pip install --no-cache-dir h5py numpy netCDF4 xarray hdf5plugin maturin pytest
|
||||||
echo "/opt/interop/bin" >> "$GITHUB_PATH"
|
echo "/opt/interop/bin" >> "$GITHUB_PATH"
|
||||||
- name: Show interop library versions
|
- name: Show interop library versions
|
||||||
run: python3 -c "import h5py, netCDF4; print('h5py', h5py.__version__, 'HDF5', h5py.version.hdf5_version, 'netCDF4', netCDF4.__version__)"
|
# h5dump's version too: the h5rs dump test requires its exact output
|
||||||
|
# (checked against Debian's 1.14.5 in rust:latest and 1.14.6).
|
||||||
|
run: |
|
||||||
|
/opt/interop/bin/python -c "import h5py, netCDF4, hdf5plugin; print('h5py', h5py.__version__, 'HDF5', h5py.version.hdf5_version, 'netCDF4', netCDF4.__version__, 'hdf5plugin', hdf5plugin.version)"
|
||||||
|
h5dump --version
|
||||||
- name: Run CI script
|
- name: Run CI script
|
||||||
env:
|
env:
|
||||||
|
# Name the interpreter outright rather than relying on $GITHUB_PATH
|
||||||
|
# reaching the test processes: if `python3` resolved to the system
|
||||||
|
# one instead of the venv, every interop suite would skip.
|
||||||
|
# CLAWHDF5_REQUIRE_INTEROP turns that skip into a failure, so the
|
||||||
|
# two together mean the suites either run or the build goes red.
|
||||||
|
CLAWHDF5_PYTHON: /opt/interop/bin/python
|
||||||
CLAWHDF5_REQUIRE_INTEROP: "1"
|
CLAWHDF5_REQUIRE_INTEROP: "1"
|
||||||
run: bash scripts/ci-test.sh
|
run: bash scripts/ci-test.sh
|
||||||
|
|
||||||
|
test-arm64:
|
||||||
|
# The aarch64 kernels in clawhdf5-accel — NEON `dot_i8`, including the
|
||||||
|
# SDOT path, and the f32 NEON kernels — are cfg'd out on x86, so the job
|
||||||
|
# above never compiles, lints or tests them.
|
||||||
|
#
|
||||||
|
# `linux_arm64` is served by two runners that execute differently:
|
||||||
|
# vision-01 runs steps on the host (Rust already installed) and vision-02
|
||||||
|
# runs them in docker.gitea.com/runner-images. So the steps work in both:
|
||||||
|
# no `container:`, no JavaScript actions (they are fetched from GitHub,
|
||||||
|
# which not every runner reliably reaches), and an explicit `+stable`
|
||||||
|
# toolchain rather than whatever a host happens to default to.
|
||||||
|
runs-on: linux_arm64
|
||||||
|
env:
|
||||||
|
CARGO_NET_RETRY: "10"
|
||||||
|
CARGO_TERM_COLOR: always
|
||||||
|
steps:
|
||||||
|
- name: Check out
|
||||||
|
run: |
|
||||||
|
git init -q .
|
||||||
|
git remote add origin "${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}.git"
|
||||||
|
for i in 1 2 3; do git fetch -q --depth 1 origin "${GITHUB_SHA}" && break; sleep 5; done
|
||||||
|
git checkout -q FETCH_HEAD
|
||||||
|
- name: Rust stable
|
||||||
|
run: |
|
||||||
|
export PATH="$HOME/.cargo/bin:$PATH"
|
||||||
|
command -v rustup >/dev/null || curl -sSf --retry 5 https://sh.rustup.rs | sh -s -- -y --profile minimal --default-toolchain none
|
||||||
|
rustup toolchain install stable --profile minimal --component clippy
|
||||||
|
echo "$HOME/.cargo/bin" >> "$GITHUB_PATH"
|
||||||
|
- name: Confirm aarch64
|
||||||
|
run: |
|
||||||
|
test "$(uname -m)" = aarch64
|
||||||
|
if grep -q asimddp /proc/cpuinfo; then echo "dot-product extension present: SDOT kernel runs"; else echo "no dot-product extension: plain NEON kernel runs"; fi
|
||||||
|
- name: Clippy (aarch64 kernels)
|
||||||
|
run: cargo +stable clippy -p clawhdf5-accel --all-targets -- -D warnings
|
||||||
|
- name: Test
|
||||||
|
run: cargo +stable test -p clawhdf5-accel -p clawhdf5-ann -p clawhdf5-format
|
||||||
|
|||||||
@@ -0,0 +1,56 @@
|
|||||||
|
name: Conformance
|
||||||
|
# Nightly: read every file of the pinned public HDF5 corpora with clawhdf5 and
|
||||||
|
# with h5py/libhdf5 and compare (conformance/run.sh; CONFORMANCE.md explains
|
||||||
|
# the method). Fails on any panic, hang, crash or out-of-memory in clawhdf5,
|
||||||
|
# and when the ok count drops below conformance/baseline.json or a file the
|
||||||
|
# baseline lists as ok stops being ok. The report is printed into the job log;
|
||||||
|
# nothing is uploaded (artifact actions are JavaScript, which rust:latest
|
||||||
|
# cannot run — see CLAUDE.md).
|
||||||
|
on:
|
||||||
|
schedule:
|
||||||
|
- cron: "17 3 * * *"
|
||||||
|
workflow_dispatch:
|
||||||
|
jobs:
|
||||||
|
conformance:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
container: rust:latest
|
||||||
|
timeout-minutes: 60
|
||||||
|
env:
|
||||||
|
CARGO_NET_RETRY: "10"
|
||||||
|
steps:
|
||||||
|
# Plain git, not actions/checkout (a JavaScript action; see ci.yml).
|
||||||
|
- name: Check out
|
||||||
|
run: |
|
||||||
|
git init -q .
|
||||||
|
git remote add origin "${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}.git"
|
||||||
|
for i in 1 2 3; do git fetch -q --depth 1 origin "${GITHUB_SHA}" && break; sleep 5; done
|
||||||
|
git checkout -q FETCH_HEAD
|
||||||
|
- name: Install h5py, h5dump and the probe's codec libraries
|
||||||
|
# hdf5-tools: h5dump for the CVE-corpus comparison. libaec-dev and
|
||||||
|
# pkg-config: the probe builds clawhdf5-format with `szip` (the core
|
||||||
|
# crates' default build needs neither).
|
||||||
|
run: |
|
||||||
|
apt-get update
|
||||||
|
apt-get install -y --no-install-recommends python3 python3-venv hdf5-tools libaec-dev pkg-config
|
||||||
|
python3 -m venv /opt/conformance
|
||||||
|
/opt/conformance/bin/pip install --no-cache-dir -r conformance/requirements.txt
|
||||||
|
/opt/conformance/bin/python -c "import h5py, hdf5plugin; print('h5py', h5py.__version__, 'HDF5', h5py.version.hdf5_version, 'hdf5plugin', hdf5plugin.version)"
|
||||||
|
h5dump --version
|
||||||
|
- name: Probe unit tests
|
||||||
|
run: cargo test --release --manifest-path conformance/probe/Cargo.toml
|
||||||
|
env:
|
||||||
|
CARGO_TARGET_DIR: conformance/.cache/target
|
||||||
|
- name: Sweep
|
||||||
|
# The corpora come from GitHub (pinned commits, conformance/corpus.txt),
|
||||||
|
# so this job needs a runner that reaches github.com.
|
||||||
|
env:
|
||||||
|
CLAWHDF5_PYTHON: /opt/conformance/bin/python
|
||||||
|
run: bash conformance/run.sh
|
||||||
|
- name: Report
|
||||||
|
if: always()
|
||||||
|
run: |
|
||||||
|
if [ -f CONFORMANCE.md ]; then cat CONFORMANCE.md; else echo "no report was generated"; fi
|
||||||
|
if [ -f conformance/.cache/results/summary.md ]; then
|
||||||
|
echo; echo "---- per-file detail (conformance/.cache/results/summary.md) ----"
|
||||||
|
cat conformance/.cache/results/summary.md
|
||||||
|
fi
|
||||||
@@ -4,3 +4,6 @@ benchmarks/longmemeval/*.json
|
|||||||
|
|
||||||
# Local model weights (MiniLM etc.) — large, not committed
|
# Local model weights (MiniLM etc.) — large, not committed
|
||||||
weights/
|
weights/
|
||||||
|
.venv
|
||||||
|
__pycache__/
|
||||||
|
.pytest_cache/
|
||||||
|
|||||||
+1232
-139
File diff suppressed because it is too large
Load Diff
+2176
File diff suppressed because it is too large
Load Diff
@@ -1,33 +1,41 @@
|
|||||||
# clawhdf5
|
# clawhdf5
|
||||||
|
|
||||||
## Purpose
|
## Purpose
|
||||||
Pure-Rust HDF5 format implementation with HNSW vector search, WAL-backed persistence, agent memory storage, and GPU-accelerated I/O. Used by ZeroClaw as its persistent memory and knowledge graph backend.
|
Pure-Rust HDF5 format implementation with HNSW vector search, WAL-backed persistence, agent memory storage, and GPU-accelerated vector search. A standalone library. Its one verified consumer is ClawBrainHub (`.brain` files); no agent framework integrates it (OpenClaw and ZeroClaw claims were withdrawn on 2026-09-25 — neither was ever true).
|
||||||
|
|
||||||
## Architecture
|
## Architecture
|
||||||
|
|
||||||
Cargo workspace with 16 crates under `crates/` (plus `libaec-sys`, an internal FFI bindings crate for the optional `szip` feature):
|
Cargo workspace with 19 crates under `crates/` (plus `libaec-sys`, an internal FFI bindings crate for the optional `szip` feature):
|
||||||
|
|
||||||
| Crate | Role |
|
| Crate | Role |
|
||||||
|-------|------|
|
|-------|------|
|
||||||
| `clawhdf5-format` | HDF5 binary spec parser (superblock, B-tree, heap) — also holds shared type definitions and physical constants |
|
| `clawhdf5-format` | HDF5 binary spec parser (superblock, B-tree, heap) — also holds shared type definitions and physical constants |
|
||||||
| `clawhdf5-io` | Read/write implementation |
|
| `clawhdf5-io` | Read/write implementation |
|
||||||
| `clawhdf5-filters` | Compression filters (gzip, LZ4, Zstd, Blosc) |
|
| `clawhdf5-filters` | Deflate backends (zlib-rs, zlib-ng, Apple Compression); the HDF5 filter pipeline, the filter registry (`clawhdf5_format::filter_registry`) and the other codecs (LZ4, Zstd, SZIP, N-Bit, scale-offset, pcodec, and the pure-Rust plugin filters LZF, bitshuffle, bzip2, Blosc 1, and Blosc2 and ZFP read-only) live in `clawhdf5-format`. |
|
||||||
| `clawhdf5-derive` | Proc-macro derive for HDF5-serializable structs |
|
| `clawhdf5-derive` | Proc-macro derive for HDF5-serializable structs |
|
||||||
| `clawhdf5` | Main facade crate |
|
| `clawhdf5` | Main facade crate |
|
||||||
| `clawhdf5-netcdf4` | NetCDF-4 compatibility layer |
|
| `clawhdf5-netcdf4` | NetCDF-4 compatibility layer |
|
||||||
| `clawhdf5-ann` | HNSW approximate nearest-neighbor vector index |
|
| `clawhdf5-ann` | HNSW approximate nearest-neighbor vector index |
|
||||||
| `clawhdf5-agent` | Agent memory, session history, knowledge graph storage |
|
| `clawhdf5-agent` | Agent memory, session history, knowledge graph storage |
|
||||||
| `clawhdf5-gpu` | GPU-accelerated I/O via wgpu (hand-written WGSL compute shaders) |
|
| `clawhdf5-gpu` | GPU vector distance computation via wgpu (hand-written WGSL compute shaders) — not dataset I/O |
|
||||||
| `clawhdf5-accel` | CPU SIMD acceleration path |
|
| `clawhdf5-accel` | CPU SIMD acceleration path |
|
||||||
| `clawhdf5-migrate` | Schema migration engine |
|
| `clawhdf5-migrate` | SQLite → HDF5 agent-memory migration |
|
||||||
| `clawhdf5-android` | Android JNI bindings |
|
| `clawhdf5-android` | Android JNI bindings |
|
||||||
| `clawhdf5-cli` | Command-line interface |
|
| `clawhdf5-cli` | Command-line interface (agent memory) |
|
||||||
|
| `clawhdf5-tools` | `h5rs`: pure-Rust HDF5 tools — `ls`, `dump` (DDL / hdf5-json), `stat`, `diff`, `check` (structural + checksum validator) |
|
||||||
| `clawhdf5-napi` | Node.js native addon bindings |
|
| `clawhdf5-napi` | Node.js native addon bindings |
|
||||||
| `clawhdf5-py` | PyO3 Python bindings |
|
| `clawhdf5-py` | PyO3 Python bindings |
|
||||||
|
| `clawhdf5-wasm` | WebAssembly (wasm-bindgen) reader for the browser; demo in `examples/wasm-viewer/` |
|
||||||
|
| `clawhdf5-remote` | Remote files: `open_url` over HTTP(S) range requests and object stores (`object_store`: S3, GCS, Azure) through a mandatory block cache (`BlockCache`) |
|
||||||
| `clawhdf5-bench` | Benchmark suite |
|
| `clawhdf5-bench` | Benchmark suite |
|
||||||
|
|
||||||
## Key Features
|
## Key Features
|
||||||
- Zero-dependency HDF5 read/write (no libhdf5 C library required)
|
- Zero-C-dependency HDF5 read/write: no libhdf5, and deflate defaults to
|
||||||
|
pure-Rust zlib-rs (`fast-deflate` opts into zlib-ng, which needs cmake).
|
||||||
|
`ci-test.sh` fails if a C-building crate enters the core crates' default
|
||||||
|
tree. flate2 must keep `runtime_detection` with zlib-rs — without it zlib-rs
|
||||||
|
loses SIMD and inflates 3.5x slower. MSRV is 1.92 (`rust-version`, checked
|
||||||
|
in CI).
|
||||||
- HNSW vector index for semantic similarity search over agent memories — the
|
- HNSW vector index for semantic similarity search over agent memories — the
|
||||||
`clawhdf5-agent` `hnsw` feature is **on by default**, so `hybrid_search` uses
|
`clawhdf5-agent` `hnsw` feature is **on by default**, so `hybrid_search` uses
|
||||||
the approximate `clawhdf5-ann` index for the vector stage (the index mirrors
|
the approximate `clawhdf5-ann` index for the vector stage (the index mirrors
|
||||||
@@ -39,7 +47,20 @@ Cargo workspace with 16 crates under `crates/` (plus `libaec-sys`, an internal F
|
|||||||
(plain closest-M capped recall on clustered data: 0.31 recall@10 at 100K). Its
|
(plain closest-M capped recall on clustered data: 0.31 recall@10 at 100K). Its
|
||||||
graph is saved to `<store>.h5.ann` at each checkpoint and reloaded by `open()`
|
graph is saved to `<store>.h5.ann` at each checkpoint and reloaded by `open()`
|
||||||
(tied to the checkpoint by a generation id; stale/damaged sidecars are
|
(tied to the checkpoint by a generation id; stale/damaged sidecars are
|
||||||
ignored and the index rebuilt). `hybrid_search` keeps one incremental BM25
|
ignored and the index rebuilt). `MemoryConfig::quantized_index` (**on by
|
||||||
|
default** for new stores, persisted; stores predating the setting load as
|
||||||
|
`false` and keep their f32 index — guarded by
|
||||||
|
`tests/fixtures/store_v2_5_0.h5`; CLI opt-out is `create --f32-index`)
|
||||||
|
stores the index's own copy of the embeddings as `i8`,
|
||||||
|
which roughly halves a loaded store's memory (2.72x -> 1.74x the raw vectors
|
||||||
|
at 100K); because quantised distances are approximate and `ef` cannot
|
||||||
|
compensate, the query path then re-scores the candidate pool against the
|
||||||
|
exact embeddings, which holds recall at the f32 index's level. It is also
|
||||||
|
faster at equal recall: 1.63x the QPS on x86-64 (AVX2) and 1.18x on a
|
||||||
|
Raspberry Pi 5 (`clawhdf5_accel::dot_i8`, NEON `SDOT` via inline asm since
|
||||||
|
the intrinsic is unstable; plain NEON on pre-dotprod cores). The aarch64
|
||||||
|
code is `cfg`'d out on x86, so x86 CI never compiles or lints it — test it
|
||||||
|
on real ARM (`rpivision02`, 10.0.2.3, is a Pi 5). `hybrid_search` keeps one incremental BM25
|
||||||
index for the life of the store and never writes the store: Hebbian
|
index for the life of the store and never writes the store: Hebbian
|
||||||
activation boosts are persisted by the next checkpoint (or on drop), not per
|
activation boosts are persisted by the next checkpoint (or on drop), not per
|
||||||
query. Measure any search-path change with
|
query. Measure any search-path change with
|
||||||
@@ -69,8 +90,51 @@ Cargo workspace with 16 crates under `crates/` (plus `libaec-sys`, an internal F
|
|||||||
`export` do). An unreadable WAL (torn header, bad magic) is quarantined to
|
`export` do). An unreadable WAL (torn header, bad magic) is quarantined to
|
||||||
`<store>.h5.wal.corrupt-<ts>` rather than blocking `open()`; a WAL with an
|
`<store>.h5.wal.corrupt-<ts>` rather than blocking `open()`; a WAL with an
|
||||||
unknown *newer* version still fails and is left untouched.
|
unknown *newer* version still fails and is left untouched.
|
||||||
- `MemoryConfig::compression` uses deflate by default; enable the agent's
|
- `MemoryConfig::float16` (**on by default** for new stores, persisted;
|
||||||
`zstd` feature to compress embeddings with Zstd instead (links libzstd).
|
existing stores keep their recorded `false` — guarded by the v2.5.0
|
||||||
|
fixture in `tests/float16_store.rs`; CLI opt-out is `create --f32`) writes
|
||||||
|
`/memory/embeddings` as IEEE half precision (48% smaller file at 100K;
|
||||||
|
LongMemEval with real MiniLM embeddings identical to f32).
|
||||||
|
`MemoryCache::half_precision` rounds each embedding as it enters the cache (push, update, WAL replay, and on load of a store still
|
||||||
|
`f32` on disk), so memory and file agree bit for bit; the conversions live
|
||||||
|
in `clawhdf5_format::float16` and must stay the single implementation.
|
||||||
|
Values beyond ±65504 are `MemoryError::InvalidEntry`. Interop: every file
|
||||||
|
must open in h5py — `f32` datasets and empty datasets did not until
|
||||||
|
2026-09-23 (see `docs/known-issues.md`); the agent's `h5py_interop` test
|
||||||
|
guards a whole store.
|
||||||
|
- `HDF5Memory::search(query_emb, text, &SearchOptions)` is the full search
|
||||||
|
path: optional source-channel filter (applied before ranking; exact scan of
|
||||||
|
the allowed records whenever cheaper than `pool × M` index distance
|
||||||
|
evaluations, and as the fallback when the pool comes back short), fusion,
|
||||||
|
activation scaling, optional re-ranking and confidence rejection.
|
||||||
|
`hybrid_search`/`hybrid_search_with` are thin wrappers; `ClawhdfBackend`
|
||||||
|
(the `openclaw` module) is `search` with re-rank + confidence on.
|
||||||
|
- **OpenClaw is not supported** (decided 2026-09-25): clawhdf5 is not an
|
||||||
|
OpenClaw memory plugin and never was — the old `memory.backend = "clawhdf5"`
|
||||||
|
config was never valid. Don't reintroduce OpenClaw claims; `docs/openclaw.md`
|
||||||
|
records what a real plugin would need.
|
||||||
|
- **ZeroClaw does not use clawhdf5** (checked 2026-09-25 against upstream
|
||||||
|
v0.8.5 and the `osobh/zeroclaw` fork, and their full history): no
|
||||||
|
`clawhdf5` feature or backend exists; ZeroClaw's memory backends are
|
||||||
|
sqlite/lucid/postgres/qdrant/markdown/none behind its own `Memory` trait.
|
||||||
|
`clawhdf5-migrate`'s default SQLite layout (`memory_chunks`, `sessions`,
|
||||||
|
`entities`, `relations`) is not ZeroClaw's schema either (ZeroClaw's is a
|
||||||
|
`memories` table). Don't reintroduce integration claims without an
|
||||||
|
integration and a test against the real consumer. Measure changes with
|
||||||
|
`search_harness --options-study`.
|
||||||
|
- `MemoryConfig::compression` is off by default; when on, embeddings are
|
||||||
|
deflate-compressed, or Zstd with the agent's `zstd` feature (links libzstd).
|
||||||
|
- Signed checkpoints (`clawhdf5-agent` `signing` module): with
|
||||||
|
`HDF5Memory::set_signing_key` every checkpoint stores an Ed25519-signed
|
||||||
|
manifest (SHA-256 per record in a Merkle tree + settings/sessions/graph
|
||||||
|
hashes; per-record hashes in `/integrity/record_hashes`);
|
||||||
|
`HDF5Memory::verify(path, &pk)` locates edits. The hashes must cover exactly
|
||||||
|
what the file persists in the form the loader returns it (strings lose
|
||||||
|
trailing NULs; an empty WAL mark is not written) or untouched stores stop
|
||||||
|
verifying — `tests/signed_store.rs` round-trips awkward strings. The key is
|
||||||
|
never persisted; a signed store refuses to checkpoint without it
|
||||||
|
(`MemoryError::SigningKeyRequired`, and `MemoryError` is `#[non_exhaustive]`).
|
||||||
|
WAL entries after the checkpoint are not covered.
|
||||||
- `Dataset::verify_provenance()` (clawhdf5 facade, `provenance` feature, on by
|
- `Dataset::verify_provenance()` (clawhdf5 facade, `provenance` feature, on by
|
||||||
default) recomputes a dataset's SHA-256 and compares it against the
|
default) recomputes a dataset's SHA-256 and compares it against the
|
||||||
`_provenance_sha256` attribute written automatically on save when
|
`_provenance_sha256` attribute written automatically on save when
|
||||||
@@ -87,7 +151,44 @@ Cargo workspace with 16 crates under `crates/` (plus `libaec-sys`, an internal F
|
|||||||
Alerts never block a save — drain them with `HDF5Memory::take_anomaly_alerts`.
|
Alerts never block a save — drain them with `HDF5Memory::take_anomaly_alerts`.
|
||||||
`MemorySource` for this bookkeeping is inferred from the caller-supplied
|
`MemorySource` for this bookkeeping is inferred from the caller-supplied
|
||||||
`source_channel` string (a heuristic, not an authenticated trust boundary).
|
`source_channel` string (a heuristic, not an authenticated trust boundary).
|
||||||
- GPU-accelerated batch I/O for large dataset processing
|
- In-place modification: `clawhdf5::FileEditor` (`crates/clawhdf5/src/edit/`)
|
||||||
|
overwrites values, grows and shrinks chunked datasets (every chunk index,
|
||||||
|
version-2 B-trees included) and sets attributes (compact and dense
|
||||||
|
storage) in existing files (h5py- or clawhdf5-written) without rewriting
|
||||||
|
them, changing indexes and heaps as libhdf5 does (index shapes and heap
|
||||||
|
bookkeeping are compared with libhdf5's in the tests); space an edit
|
||||||
|
frees is reused by later edits of the same editor. Anything it cannot do
|
||||||
|
safely is `Error::Unsupported` before any write (limits in
|
||||||
|
`docs/known-issues.md`). Test changes with
|
||||||
|
`cargo test -p clawhdf5-tools --test edit_interop --test
|
||||||
|
edit_coverage_interop` (h5py, h5dump, `h5rs check`, structure comparisons
|
||||||
|
with libhdf5; libhdf5 sources for the algorithms are at
|
||||||
|
github.com/HDFGroup/hdf5, tag `hdf5_1_14_6`).
|
||||||
|
- Remote files (`clawhdf5-remote`, range-read milestone M3 of
|
||||||
|
`docs/design/range-reads.md`): `open_url("http://…")` gives a
|
||||||
|
`clawhdf5::File` over `File::open_storage`, read through `BlockCache`
|
||||||
|
(1 MiB blocks, LRU byte budget, per-block in-flight dedup across threads,
|
||||||
|
runs coalesced into parallel requests). `HttpStorage` pins the file by
|
||||||
|
ETag/Last-Modified and length (a change is `RemoteError::FileChanged`),
|
||||||
|
refuses servers that ignore `Range` unless a full download is allowed,
|
||||||
|
and retries transient failures. `ObjectStoreStorage` (feature
|
||||||
|
`object-store`, pure Rust) runs each read on a small owned tokio
|
||||||
|
runtime and waits on a channel, so it works from any thread, including
|
||||||
|
inside `spawn_blocking` or another runtime. Default build is plain HTTP with
|
||||||
|
no C; `https` (rustls + ring) and `s3`/`gcs`/`azure` (aws-lc-rs) are
|
||||||
|
opt-in. Tests run a std-only HTTP server
|
||||||
|
(`tests/common/server.rs`, also the `range_server` example);
|
||||||
|
`CLAWHDF5_REMOTE_CORPUS=conformance/.cache/corpus` compares every corpus
|
||||||
|
file over HTTP with `File::open`.
|
||||||
|
- GPU-accelerated vector distance computation (`clawhdf5-gpu`, wgpu); HDF5 I/O itself is CPU-only
|
||||||
|
- Browser: `clawhdf5-wasm` (wasm-bindgen, read-only, file held in memory;
|
||||||
|
no Zstd/SZIP since they link C) and the `examples/wasm-viewer/` page.
|
||||||
|
`examples/wasm-viewer/test/run.sh` builds the package (needs the
|
||||||
|
`wasm-bindgen` CLI at the crate's exact version) and tests it under Node
|
||||||
|
and headless Chromium (a Playwright download in `~/.cache/ms-playwright`
|
||||||
|
on tank); the CI container has neither, so CI runs the native
|
||||||
|
`clawhdf5-wasm` `h5py_interop` test on the same fixture. Size numbers are
|
||||||
|
in the example's README.
|
||||||
- Python and Node.js bindings for cross-language use
|
- Python and Node.js bindings for cross-language use
|
||||||
- NetCDF-4 compatibility for scientific data interop
|
- NetCDF-4 compatibility for scientific data interop
|
||||||
|
|
||||||
@@ -103,12 +204,40 @@ cargo build --release
|
|||||||
cargo test --workspace
|
cargo test --workspace
|
||||||
```
|
```
|
||||||
|
|
||||||
|
### CI
|
||||||
|
`.gitea/workflows/ci.yml` has two jobs, both green as of 2026-09-22:
|
||||||
|
- **`test`** (`ubuntu-latest`, in `rust:latest`) runs `scripts/ci-test.sh` with
|
||||||
|
the h5py/netCDF4 interop suites required (`CLAWHDF5_REQUIRE_INTEROP=1`).
|
||||||
|
Served by the `tank` and `architect` runners.
|
||||||
|
- **`test-arm64`** (`linux_arm64`) lints and tests the aarch64 code — the NEON
|
||||||
|
kernels are `cfg`'d out on x86, so this is the only place they are built.
|
||||||
|
Served by `vision-01` (host mode) and `vision-02` (Docker), so steps must
|
||||||
|
work in both.
|
||||||
|
|
||||||
|
Keep workflows free of JavaScript actions (`actions/checkout`, `actions/cache`,
|
||||||
|
…): `rust:latest` has no `node`, and not every runner reaches GitHub, where
|
||||||
|
they are fetched from. Check out with plain `git` instead. The `test` job
|
||||||
|
installs `cmake` for the opt-in `fast-deflate` (zlib-ng) steps; the default
|
||||||
|
build needs no C toolchain, so `test-arm64` does not.
|
||||||
|
All runners are on `gitea-runner` 3.5.0, from `docker.gitea.com/act_runner`
|
||||||
|
— `gitea/act_runner:latest` on Docker Hub is frozen at 0.6.1.
|
||||||
|
|
||||||
### CLI
|
### CLI
|
||||||
```bash
|
```bash
|
||||||
cargo run -p clawhdf5-cli -- --help
|
cargo run -p clawhdf5-cli -- --help
|
||||||
# create, save, search, recall, stats, flush-wal, agents-md, export, snapshot subcommands
|
# create, save, search, recall, stats, flush-wal, agents-md, export, snapshot subcommands
|
||||||
```
|
```
|
||||||
|
|
||||||
|
### HDF5 tools (`h5rs`, crate `clawhdf5-tools`)
|
||||||
|
```bash
|
||||||
|
cargo run -p clawhdf5-tools -- ls -r file.h5 # also dump [--json], stat, diff, check
|
||||||
|
bash scripts/h5rs-fuzz.sh # every subcommand over the CVE corpus: no panic/crash/hang
|
||||||
|
bash scripts/h5rs-check-ok-files.sh --data # check passes every fully-read conformance file
|
||||||
|
```
|
||||||
|
Its interop tests compare against h5ls/h5stat/h5dump/h5diff (Debian
|
||||||
|
`hdf5-tools`, installed in CI); `dump` must stay byte-identical to h5dump on
|
||||||
|
the test files.
|
||||||
|
|
||||||
### Python bindings
|
### Python bindings
|
||||||
```bash
|
```bash
|
||||||
cd crates/clawhdf5-py
|
cd crates/clawhdf5-py
|
||||||
@@ -117,4 +246,12 @@ python -c "import clawhdf5; print(clawhdf5.__version__)"
|
|||||||
```
|
```
|
||||||
|
|
||||||
## Integration
|
## Integration
|
||||||
ZeroClaw imports this as a Cargo feature (`clawhdf5` feature flag) to persist agent memory with HNSW vector search for context retrieval.
|
- **ClawBrainHub** (`clawverse/clawbrainhub` on git.redclaw.dev) is the one
|
||||||
|
verified consumer: `cbh-core` reads and writes `.brain` files through the
|
||||||
|
facade (`File`, `FileBuilder`, `AttrValue`, `Selection`), `cbh-scanner`
|
||||||
|
uses the facade, and `cbh-cli` uses `clawhdf5_agent::bm25::BM25Index`. It
|
||||||
|
depends on this repo by path (`../clawhdf5`), so it builds against whatever
|
||||||
|
is checked out — changes to those APIs reach it directly. Verified
|
||||||
|
2026-09-25 against main: builds, and its 204 tests pass.
|
||||||
|
- OpenClaw and ZeroClaw were both described as consumers; neither integrates
|
||||||
|
clawhdf5 (see Key Features and `docs/openclaw.md`).
|
||||||
|
|||||||
+291
@@ -0,0 +1,291 @@
|
|||||||
|
# clawhdf5 conformance report
|
||||||
|
|
||||||
|
Every HDF5 file of eight public corpora (pinned by commit) is read twice — by
|
||||||
|
clawhdf5 (`conformance/probe`, the same `clawhdf5-format` calls the facade
|
||||||
|
makes) and by h5py/libhdf5 (`conformance/ref.py`) — and the two readings are
|
||||||
|
compared object by object: the set of hard-linked objects, each dataset's and
|
||||||
|
attribute's shape, and a SHA-256 of its values in a canonical encoding. The
|
||||||
|
CVE corpus is also run through `h5dump`. Each side runs under a timeout and an
|
||||||
|
address-space limit, so a hang, crash or runaway allocation is recorded, not
|
||||||
|
fatal. This file is generated by `conformance/run.sh`; do not edit it by hand.
|
||||||
|
|
||||||
|
## Run
|
||||||
|
|
||||||
|
| | |
|
||||||
|
|---|---|
|
||||||
|
| date | 2026-09-27 00:34 UTC |
|
||||||
|
| clawhdf5 commit | `f37e7ae3263277319dba4bc39be5397194eb00c3` |
|
||||||
|
| machine | `tank`: AMD Ryzen 7 7800X3D 8-Core Processor, 16 CPUs, 61 GiB, Linux 7.0.0-34-generic x86_64 |
|
||||||
|
| command | `conformance/run.sh --no-fetch --update-baseline` |
|
||||||
|
| rustc | rustc 1.98.1 (48a229cea 2026-09-01) |
|
||||||
|
| reference | h5py 3.16.0, HDF5 2.0.0, numpy 2.5.3, hdf5plugin 7.1.0, Python 3.14.4 |
|
||||||
|
| h5dump | Version 1.14.6 (CVE corpus only) |
|
||||||
|
| limits | 20 s timeout (SIGKILL), 4096 MiB address space, per process; 16 files in parallel |
|
||||||
|
| runtime | 21 s probing + comparing (0 s fetch/build before it) |
|
||||||
|
|
||||||
|
## Results
|
||||||
|
|
||||||
|
A file's class is the first that applies:
|
||||||
|
|
||||||
|
- **panic / hang / crash / oom** — clawhdf5 panicked (caught per object or not), hit the timeout, died on a signal, or failed an allocation. The CI gate fails on any of these.
|
||||||
|
- **h5py-cannot-read** — libhdf5 could not open the file (or itself crashed or hung). Nothing to compare against; most are the deliberately malformed CVE reproducers.
|
||||||
|
- **our-error** — clawhdf5 returned an error for something h5py reads.
|
||||||
|
- **mismatch** — both read it, but the shapes, values, object set or attribute set differ.
|
||||||
|
- **ok** — every object h5py reads, clawhdf5 reads identically.
|
||||||
|
|
||||||
|
| corpus | files | ok | our-error | mismatch | h5py-cannot-read | panic | hang | crash | oom |
|
||||||
|
|---|---|---|---|---|---|---|---|---|---|
|
||||||
|
| NCAS-CMS_pyfive | 33 | 32 | 0 | 1 | 0 | 0 | 0 | 0 | 0 |
|
||||||
|
| cve_hdf5 | 147 | 113 | 2 | 0 | 32 | 0 | 0 | 0 | 0 |
|
||||||
|
| h5py_data | 4 | 4 | 0 | 0 | 0 | 0 | 0 | 0 | 0 |
|
||||||
|
| hdf5 | 466 | 404 | 1 | 1 | 60 | 0 | 0 | 0 | 0 |
|
||||||
|
| netcdf-c | 20 | 20 | 0 | 0 | 0 | 0 | 0 | 0 | 0 |
|
||||||
|
| netcdf4-python | 18 | 18 | 0 | 0 | 0 | 0 | 0 | 0 | 0 |
|
||||||
|
| usnistgov_h5wasm | 5 | 5 | 0 | 0 | 0 | 0 | 0 | 0 | 0 |
|
||||||
|
| xarray-data | 4 | 4 | 0 | 0 | 0 | 0 | 0 | 0 | 0 |
|
||||||
|
| **all** | **697** | **600** | **3** | **2** | **92** | **0** | **0** | **0** | **0** |
|
||||||
|
|
||||||
|
2 of the 2 mismatches are a known h5py bug, not ours (see *Known not-our-bug*).
|
||||||
|
|
||||||
|
3 of the 3 our-errors are corrupt data that HDF5 2.0 reads only through a bug and clawhdf5 refuses (see *Known not-our-bug*).
|
||||||
|
|
||||||
|
Corpora (fetched by `conformance/fetch-corpus.sh` into the gitignored `conformance/.cache/`):
|
||||||
|
|
||||||
|
| corpus | source | commit |
|
||||||
|
|---|---|---|
|
||||||
|
| hdf5 | https://github.com/HDFGroup/hdf5 | `a3cf1ea82cc7` |
|
||||||
|
| cve_hdf5 | https://github.com/HDFGroup/cve_hdf5 | `3fd1f5ae3869` |
|
||||||
|
| netcdf-c | https://github.com/Unidata/netcdf-c | `beb7b9585273` |
|
||||||
|
| NCAS-CMS_pyfive | https://github.com/NCAS-CMS/pyfive | `8cf07b874913` |
|
||||||
|
| usnistgov_h5wasm | https://github.com/usnistgov/h5wasm | `02f6336527d2` |
|
||||||
|
| netcdf4-python | https://github.com/Unidata/netcdf4-python | `6e67576d39ae` |
|
||||||
|
| xarray-data | https://github.com/pydata/xarray-data | `a35297e9da2c` |
|
||||||
|
| h5py_data | https://github.com/h5py/h5py (`h5py/tests/data_files`) | `b2f0347c4200` |
|
||||||
|
|
||||||
|
## Panics, hangs, crashes, out-of-memory
|
||||||
|
|
||||||
|
None.
|
||||||
|
|
||||||
|
## Our-error root causes
|
||||||
|
|
||||||
|
Grouped by normalised error message. *files* counts files whose class this cause affects.
|
||||||
|
|
||||||
|
| files | objects | error | examples |
|
||||||
|
|---:|---:|---|---|
|
||||||
|
| 3 | 3 | `ChunkedReadError("…")` | `cve_hdf5/cvefiles/cve-2025-2308.h5`, `cve_hdf5/cvefiles/cve-2025-44904.h5`, `hdf5/test/testfiles/bad_nbit_parms_walk.h5` |
|
||||||
|
|
||||||
|
## Mismatch root causes
|
||||||
|
|
||||||
|
| files | objects | cause | examples |
|
||||||
|
|---:|---:|---|---|
|
||||||
|
| 1 | 1 | `attr-values: ours=vlen(>u8) h5py=object layout=- filters=-` | `NCAS-CMS_pyfive/tests/data/attr_datatypes.hdf5` |
|
||||||
|
| 1 | 1 | `values: ours=vlen({r:>f4,i:>f4}8) h5py=object layout=contiguous filters=-` | `hdf5/tools/test/testfiles/tcomplex_be.h5` |
|
||||||
|
|
||||||
|
## CVE corpus: clawhdf5 vs h5dump vs h5py
|
||||||
|
|
||||||
|
The 147 files of [HDFGroup/cve_hdf5](https://github.com/HDFGroup/cve_hdf5) — reproducers for
|
||||||
|
published libhdf5 CVEs and fuzzer finds. *read* = produced output (possibly with per-object
|
||||||
|
errors), *error* = refused cleanly. h5dump exits non-zero on any error anywhere in a file, so
|
||||||
|
its read/error split is not comparable with the other two rows; the panic, crash, hang and oom
|
||||||
|
columns are.
|
||||||
|
|
||||||
|
| tool | read | error | panic | crash | hang | oom |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
| clawhdf5 | 121 | 26 | 0 | 0 | 0 | 0 |
|
||||||
|
| h5dump 1.14.6 | 16 | 129 | 0 | 2 | 0 | 0 |
|
||||||
|
| h5py 3.16.0 / HDF5 2.0.0 | 115 | 31 | 0 | 1 | 0 | 0 |
|
||||||
|
|
||||||
|
<details><summary>Per-file outcomes</summary>
|
||||||
|
|
||||||
|
| file | h5dump | h5py | clawhdf5 | class |
|
||||||
|
|---|---|---|---|---|
|
||||||
|
| cvefiles/cve-2016-4330.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2016-4331.h5 | error exit | read 25 obj, 1 errors | read 25 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2016-4332-mtime-new.h5 | error exit | read 25 obj, 1 errors | read 25 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2016-4332-mtime.h5 | error exit | read 4 obj, 3 errors | read 4 obj, 3 errors | ok |
|
||||||
|
| cvefiles/cve-2016-4332-stab.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2016-4333.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2017-17505.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2017-17506.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2017-17507.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2017-17508.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2017-17509.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-11202.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-11203.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-11204.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-11205.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-11206-new.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-11206-old.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-11207.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-13866.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2018-13867.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-13868.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-13869.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-13870.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-13871.h5 | error exit | read 2 obj | read 2 obj | ok |
|
||||||
|
| cvefiles/cve-2018-13872.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-13873.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-13874.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2018-13875.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-13876.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2018-14031.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-14033.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-14034.h5 | error exit | read 1 obj, 2 errors | read 1 obj | ok |
|
||||||
|
| cvefiles/cve-2018-14035.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-14460.h5 | error exit | read 3 obj, 2 errors | read 3 obj, 2 errors | ok |
|
||||||
|
| cvefiles/cve-2018-15671.h5 | ok | read 1 obj | read 1 obj | ok |
|
||||||
|
| cvefiles/cve-2018-15672.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-16438.h5 | error exit | read 1 obj, 1 errors | read 1 obj | ok |
|
||||||
|
| cvefiles/cve-2018-17233.h5 | error exit | read 6 obj, 1 errors | read 6 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-17234.h5 | error exit | read 6 obj, 1 errors | read 6 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-17237.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-17432.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-17433 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2018-17434.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-17435.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-17436 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2018-17437.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-17438 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-17439 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2019-8396.h5 | error exit | read 3 obj, 2 errors | read 3 obj, 2 errors | ok |
|
||||||
|
| cvefiles/cve-2019-8397.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2019-8398.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2019-9151.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 2 errors | ok |
|
||||||
|
| cvefiles/cve-2019-9152.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2020-10809 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2020-10810.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2020-10811.h5 | error exit | read 25 obj, 1 errors | read 25 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2020-10812.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2020-18232.h5 | error exit | read 3 obj, 2 errors | read 3 obj, 2 errors | ok |
|
||||||
|
| cvefiles/cve-2020-18494.h5 | ok | read 2 obj | read 2 obj | ok |
|
||||||
|
| cvefiles/cve-2021-36977.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2021-37501.h5 | error exit | read 18 obj, 1 errors | read 18 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2021-45829.h5 | error exit | read 1 obj, 2 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2021-45830.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2021-45833.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2021-46242.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2021-46243.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2021-46244.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29157.h5 | error exit | read 4 obj, 7 errors | read 4 obj, 7 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29158.h5 | ok | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29159.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29160.h5 | error exit | read 4 obj, 1 errors | read 4 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29161.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29162.h5 | error exit | read 17 obj, 4 errors | read 17 obj, 4 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29163.h5 | error exit | read 7 obj, 1 errors | read 7 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29164.h5 | ok | read 3 obj | read 3 obj | ok |
|
||||||
|
| cvefiles/cve-2024-29165.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29166.h5 | error exit | read 17 obj, 2 errors | read 17 obj | ok |
|
||||||
|
| cvefiles/cve-2024-32605.h5 | ok | read 6 obj, 1 errors | read 6 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32606.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32607-1.h5 | ok | read 10 obj | read 10 obj | ok |
|
||||||
|
| cvefiles/cve-2024-32607-2.h5 | error exit | read 9 obj, 1 errors | read 9 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32608.h5 | error exit | read 6 obj, 1 errors | read 6 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32609.h5 | error exit | SIGSEGV | read 3 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2024-32610.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32611.h5 | ok | read 6 obj | read 6 obj | ok |
|
||||||
|
| cvefiles/cve-2024-32612.h5 | ok | read 3 obj | read 3 obj | ok |
|
||||||
|
| cvefiles/cve-2024-32613.h5 | error exit | read 7 obj, 1 errors | read 7 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32614.h5 | error exit | read 25 obj, 2 errors | read 25 obj, 2 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32615.h5 | error exit | read 4 obj, 1 errors | read 4 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32616.h5 | error exit | read 10 obj, 7 errors | read 10 obj, 6 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32617.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32618.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32619.h5 | error exit | read 3 obj, 2 errors | read 3 obj, 2 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32620.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32621.h5 | ok | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32622.h5 | ok | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32623.h5 | ok | read 6 obj | read 6 obj | ok |
|
||||||
|
| cvefiles/cve-2024-32624.h5 | error exit | read 6 obj, 1 errors | read 6 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-33873.h5 | error exit | read 4 obj, 1 errors | read 4 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-33874.h5 | ok | read 6 obj, 1 errors | read 6 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-33875.h5 | ok | read 2 obj | read 2 obj | ok |
|
||||||
|
| cvefiles/cve-2024-33876.h5 | ok | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-33877.h5 | error exit | read 8 obj, 1 errors | read 8 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-2153.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-2308.h5 | error exit | read 25 obj, 1 errors | read 25 obj, 2 errors | our-error |
|
||||||
|
| cvefiles/cve-2025-2309.h5 | ok | read 6 obj, 1 errors | read 6 obj | ok |
|
||||||
|
| cvefiles/cve-2025-2310.h5 | error exit | read 24 obj, 8 errors | read 24 obj, 8 errors | ok |
|
||||||
|
| cvefiles/cve-2025-2912.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-2913.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-2914.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-2915.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-2923.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-2924.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-2925.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-2926.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-44904.h5 | error exit | read 25 obj, 1 errors | read 25 obj, 2 errors | our-error |
|
||||||
|
| cvefiles/cve-2025-44905.h5 | error exit | read 25 obj, 3 errors | read 25 obj, 3 errors | ok |
|
||||||
|
| cvefiles/cve-2025-6269-1.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-6269-2.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-6269-3.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-6269-4.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-6270-1.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-6270-2.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-6270-3.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-6516.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-6750.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-6816.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-6817.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-6818.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-6856.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-6857.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-6858.h5 | SIGSEGV | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-7067.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-7068.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-7069.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2026-26200.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2026-34734.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2026-92627.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/unknown-1.h5 | error exit | read 11 obj, 1 errors | read 11 obj, 1 errors | ok |
|
||||||
|
| fuzzerfiles/gh-4431-poc-03.h5 | error exit | read 1 obj | read 1 obj | ok |
|
||||||
|
| fuzzerfiles/gh-4432-poc-05.h5 | SIGSEGV | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| fuzzerfiles/gh-4433-poc-08.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| fuzzerfiles/gh-4434-poc-09.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| fuzzerfiles/gh-4435-poc-10.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| fuzzerfiles/gh-4585.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| fuzzerfiles/gh_2649_flawed.h5 | error exit | read 9 obj, 1 errors | read 9 obj, 1 errors | ok |
|
||||||
|
| fuzzerfiles/gh_2649_plain_model.h5 | ok | read 10 obj | read 10 obj | ok |
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
## Known not-our-bug
|
||||||
|
|
||||||
|
- **h5py big-endian variable-length sequences.** h5py returns the elements of a VL sequence
|
||||||
|
whose base type is big-endian with the file's big-endian bytes but a native (little-endian)
|
||||||
|
numpy dtype, so the values it reports are byte-swapped garbage; `h5dump` prints the values
|
||||||
|
clawhdf5 reads. Reproducer: `h5py.vlen_dtype(np.dtype('>f4'))` dataset holding `[1.0, 2.0]`
|
||||||
|
reads back in h5py as `[4.6e-41, 9.0e-44]`. Affected here: `NCAS-CMS_pyfive/tests/data/attr_datatypes.hdf5`, `hdf5/tools/test/testfiles/tcomplex_be.h5`.
|
||||||
|
- **Non-IEEE floats and partial-precision integers (N-Bit).** libhdf5 converts a float whose
|
||||||
|
bit layout is not IEEE (e.g. `H5Tset_precision` for the N-Bit filter) or an integer with a
|
||||||
|
bit offset / reduced precision into the plain numpy type of the same size. The probe
|
||||||
|
compares such values as converted numbers, not raw file bytes (before 2026-09-25 it compared
|
||||||
|
raw bytes, which reported every N-Bit float dataset as a mismatch).
|
||||||
|
- **Types h5py widens.** Where h5py reads a type into a numpy type of a different size
|
||||||
|
(FP8 -> float16, bfloat16 -> float32, x87 long double -> float128) the values are not
|
||||||
|
compared (shape and presence still are): dataset file type size 1 -> numpy float16 (2) (15x), attr file type size 1 -> numpy float16 (2) (15x), dataset file type size 2 -> numpy float32 (4) (2x), dataset file type size 8 -> numpy float128 (16) (1x), dataset file type size 12 -> numpy float128 (16) (1x), attr file type size 2 -> numpy float32 (4) (1x), dataset file type size 2 -> numpy >f4 (4) (1x), attr file type size 2 -> numpy >f4 (4) (1x).
|
||||||
|
- **Corrupt data HDF5 2.0 reads through a bug.** clawhdf5 refuses these objects; h5py 3.16 /
|
||||||
|
HDF5 2.0 returns values for them that the file does not hold:
|
||||||
|
- `cve_hdf5/cvefiles/cve-2025-2308.h5` `/Scale_offset_long_long_data_le`: scale-offset codes run past the end of the chunk: HDF5 2.0 reads past its buffer; libhdf5's develop branch refuses the chunk ("Buffer too short").
|
||||||
|
- `cve_hdf5/cvefiles/cve-2025-44904.h5` `/Scale_offset_float_data_le`: unfiltered chunks of 38 and 37 bytes for 48-byte chunks: HDF5 2.0 fills the rest with whatever its buffer held; libhdf5's develop branch refuses them ("incorrect chunk size returned from index for unfiltered chunk").
|
||||||
|
- `hdf5/test/testfiles/bad_nbit_parms_walk.h5` `/Nbit_int_data_le`: an N-Bit parameter list one value short: HDF5 2.0 reads past the list; libhdf5's own test (`test_filter_bad_params`, test/dsets.c) now requires the read to fail.
|
||||||
|
- **References** are compared by presence only (`R`), not by target.
|
||||||
|
|
||||||
|
## Objects h5py fails on but clawhdf5 reads
|
||||||
|
|
||||||
|
- 19 x `OSError: Can't synchronously read data (no appropriate function for conversion path)`
|
||||||
|
- 1 x `TypeError: unhandled dtype kind M (dtype('…'))`
|
||||||
|
- 1 x `TypeError: No NumPy equivalent for TypeTimeID exists`
|
||||||
|
- 1 x `ValueError: Insufficient precision in available types to represent (N, N, N, N, N)`
|
||||||
|
|
||||||
|
## Reproduce
|
||||||
|
|
||||||
|
```sh
|
||||||
|
# needs: Rust, python3 with h5py numpy hdf5plugin (conformance/requirements.txt), h5dump (hdf5-tools), git
|
||||||
|
CLAWHDF5_PYTHON=/path/to/venv/bin/python conformance/run.sh
|
||||||
|
```
|
||||||
|
|
||||||
|
The corpus (about 450 MB of sparse checkouts) is cached in `conformance/.cache/`; results for
|
||||||
|
every file, both sides' raw JSON and stderr, are in `conformance/.cache/results/`.
|
||||||
|
`conformance/baseline.json` holds the ok files the nightly CI job (`.gitea/workflows/conformance.yml`)
|
||||||
|
must keep; `conformance/run.sh --update-baseline` rewrites it.
|
||||||
+16
-1
@@ -16,13 +16,19 @@ members = [
|
|||||||
"crates/clawhdf5-cli",
|
"crates/clawhdf5-cli",
|
||||||
"crates/clawhdf5-napi",
|
"crates/clawhdf5-napi",
|
||||||
"crates/clawhdf5-bench",
|
"crates/clawhdf5-bench",
|
||||||
|
"crates/clawhdf5-tools",
|
||||||
|
"crates/clawhdf5-wasm",
|
||||||
|
"crates/clawhdf5-remote",
|
||||||
"crates/libaec-sys",
|
"crates/libaec-sys",
|
||||||
]
|
]
|
||||||
resolver = "2"
|
resolver = "2"
|
||||||
|
|
||||||
[workspace.package]
|
[workspace.package]
|
||||||
version = "2.5.0"
|
version = "2.7.0"
|
||||||
edition = "2024"
|
edition = "2024"
|
||||||
|
# Oldest toolchain that builds the whole workspace; CI checks it. wgpu (in
|
||||||
|
# clawhdf5-gpu) requires 1.92.
|
||||||
|
rust-version = "1.92"
|
||||||
license = "MIT"
|
license = "MIT"
|
||||||
repository = "https://git.redclaw.dev/quantumclaw/clawhdf5"
|
repository = "https://git.redclaw.dev/quantumclaw/clawhdf5"
|
||||||
|
|
||||||
@@ -31,3 +37,12 @@ tempfile = "3"
|
|||||||
criterion = { version = "0.5", features = ["html_reports"] }
|
criterion = { version = "0.5", features = ["html_reports"] }
|
||||||
half = "2.7"
|
half = "2.7"
|
||||||
serde = { version = "1", features = ["derive"] }
|
serde = { version = "1", features = ["derive"] }
|
||||||
|
|
||||||
|
# The browser build of clawhdf5-wasm (examples/wasm-viewer/build.sh): size
|
||||||
|
# over speed, whole-program optimisation. Native profiles are unaffected.
|
||||||
|
[profile.wasm-release]
|
||||||
|
inherits = "release"
|
||||||
|
opt-level = "s"
|
||||||
|
lto = true
|
||||||
|
codegen-units = 1
|
||||||
|
panic = "abort"
|
||||||
|
|||||||
+14
-8
@@ -105,24 +105,30 @@
|
|||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
## Track 7: OpenClaw Integration
|
## Track 7: OpenClaw Integration — withdrawn (2026-09-25)
|
||||||
**Status:** 🟢 Complete
|
**Status:** ⚪ Withdrawn (the items below were library work; no OpenClaw integration shipped)
|
||||||
**Priority:** Critical (for adoption)
|
**Priority:** Critical (for adoption)
|
||||||
**Crates:** `clawhdf5-agent`, `clawhdf5-napi`
|
**Crates:** `clawhdf5-agent`, `clawhdf5-napi`
|
||||||
|
|
||||||
- [x] **7.1** Memory backend trait — MemoryBackend with search/get/write/ingest/export/stats
|
- [x] **7.1** Memory backend trait — MemoryBackend with search/get/write/ingest/export/stats
|
||||||
- [x] **7.2** Hybrid retrieval pipeline — ClawhdfBackend wires RRF → reranker → confidence rejection
|
- [x] **7.2** Hybrid retrieval pipeline — ClawhdfBackend wires RRF → reranker → confidence rejection
|
||||||
- [x] **7.3** Markdown import/export — MarkdownParser + MarkdownExporter with line tracking + metadata
|
- [x] **7.3** Markdown import/export — MarkdownParser + MarkdownExporter with line tracking + metadata
|
||||||
- [x] **7.4** memory_search tool — backed by full hybrid retrieval pipeline
|
- [x] **7.4** `search()` — backed by the full hybrid retrieval pipeline (a Rust method; no OpenClaw tool was ever registered)
|
||||||
- [x] **7.5** memory_get tool — get() with path + line range support
|
- [x] **7.5** `get()` — read back by path, with a line slice (not an OpenClaw tool either)
|
||||||
- [x] **7.6** Compaction integration — run_compaction() (decay + compact + WAL flush), run_consolidation() (hippocampal engine), tick_session(), flush_wal()
|
- [x] **7.6** Compaction integration — run_compaction() (decay + compact + WAL flush), run_consolidation() (hippocampal engine), tick_session(), flush_wal()
|
||||||
- [x] **7.7** Config surface — `memory.backend = "clawhdf5"` schema documented in docs/openclaw-config.md
|
- [ ] **7.7** ~~Config surface — `memory.backend = "clawhdf5"`~~ — never valid OpenClaw config; docs removed
|
||||||
- [x] **7.8** Documentation + migration guide — docs/migration-guide.md, docs/openclaw-integration.md (architecture, full API reference, code patterns)
|
- [ ] **7.8** ~~Documentation + migration guide~~ — removed: they described an integration that never worked
|
||||||
|
|
||||||
**Node.js bridge:** `clawhdf5-napi` (napi-rs) → `@redclaw/clawhdf5` npm package with full TypeScript types.
|
**Node.js bridge:** `clawhdf5-napi` (napi-rs) and a TypeScript wrapper in `packages/clawhdf5-node` exist but are unpublished, untested in CI and known to be broken (docs/known-issues.md).
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
|
> **Withdrawn.** None of this track produced a working OpenClaw integration: no
|
||||||
|
> plugin was built, the documented `memory.backend = "clawhdf5"` config was never
|
||||||
|
> valid in any OpenClaw release, and the Node package was never published. The
|
||||||
|
> Rust `ClawhdfBackend` remains as a library API. Not pursued for now; see
|
||||||
|
> [docs/openclaw.md](docs/openclaw.md) for what a plugin would need today.
|
||||||
|
|
||||||
## Track 8: Benchmarking & Validation
|
## Track 8: Benchmarking & Validation
|
||||||
**Status:** 🟢 Complete
|
**Status:** 🟢 Complete
|
||||||
**Priority:** High
|
**Priority:** High
|
||||||
@@ -142,7 +148,7 @@
|
|||||||
|
|
||||||
**Phase 1:** ~~Tracks 1, 2, 3 — core memory intelligence~~ 🟢 Complete
|
**Phase 1:** ~~Tracks 1, 2, 3 — core memory intelligence~~ 🟢 Complete
|
||||||
**Phase 2:** ~~Track 4 (temporal) + Track 5 (security)~~ 🟢 Complete
|
**Phase 2:** ~~Track 4 (temporal) + Track 5 (security)~~ 🟢 Complete
|
||||||
**Phase 3:** ~~Track 6 (multi-modal) + Track 7 (OpenClaw integration)~~ 🟢 Complete
|
**Phase 3:** ~~Track 6 (multi-modal)~~ 🟢 Complete; Track 7 (OpenClaw integration) withdrawn
|
||||||
**Phase 4:** ~~Track 8 (benchmarking + validation)~~ 🟢 Complete
|
**Phase 4:** ~~Track 8 (benchmarking + validation)~~ 🟢 Complete
|
||||||
|
|
||||||
All 8 tracks delivered. 1,650+ tests passing, zero clippy warnings.
|
All 8 tracks delivered. 1,650+ tests passing, zero clippy warnings.
|
||||||
|
|||||||
@@ -0,0 +1,3 @@
|
|||||||
|
/.cache/
|
||||||
|
# pin the probe's dependencies (the workspace lock is not committed)
|
||||||
|
!/probe/Cargo.lock
|
||||||
@@ -0,0 +1,39 @@
|
|||||||
|
# Conformance sweep
|
||||||
|
|
||||||
|
Reads every HDF5 file of eight public corpora with clawhdf5 and with
|
||||||
|
h5py/libhdf5, compares the two readings object by object, and writes
|
||||||
|
[`CONFORMANCE.md`](../CONFORMANCE.md).
|
||||||
|
|
||||||
|
```sh
|
||||||
|
CLAWHDF5_PYTHON=/path/to/venv/bin/python conformance/run.sh # ~30 s once the corpus is cached
|
||||||
|
conformance/run.sh --update-baseline # after an intended change in results
|
||||||
|
```
|
||||||
|
|
||||||
|
Needs Rust, `git`, `h5dump` (Debian/Ubuntu `hdf5-tools`), `libaec` (for the
|
||||||
|
probe's `szip` feature; `libaec-dev`), and a Python with the packages in
|
||||||
|
`requirements.txt`. The first run downloads about 450 MB of sparse checkouts.
|
||||||
|
|
||||||
|
| file | role |
|
||||||
|
|---|---|
|
||||||
|
| `corpus.txt` | the corpora: git URL, pinned commit, swept root, sparse-checkout patterns |
|
||||||
|
| `fetch-corpus.sh` | shallow, sparse, blob-filtered checkout of each pinned commit into `.cache/src/` (gitignored); no-op when already there |
|
||||||
|
| `list_files.py` | which files are probed (HDF5/netCDF-4 extensions minus netCDF classic, plus the CVE reproducers) |
|
||||||
|
| `probe/` | the clawhdf5 side: a standalone crate (outside the workspace, so `cargo test --workspace` never builds it) that walks a file with `clawhdf5-format` and prints canonical JSON |
|
||||||
|
| `ref.py` | the h5py side: the same JSON from h5py |
|
||||||
|
| `run_one.sh` | runs both sides on one file (and `h5dump` on the CVE corpus) under a timeout and an address-space limit |
|
||||||
|
| `compare.py` | classifies each file (ok / our-error / mismatch / h5py-cannot-read / panic / hang / crash / oom) and groups root causes |
|
||||||
|
| `report.py` | writes `CONFORMANCE.md` |
|
||||||
|
| `check.py` | the gate: fails on any panic/hang/crash/oom, on an ok count below `baseline.json`, or on a baseline-ok file that is no longer ok |
|
||||||
|
| `baseline.json` | the ok files the gate holds the line on |
|
||||||
|
| `requirements.txt` | pinned h5py / numpy / hdf5plugin / netCDF4 |
|
||||||
|
|
||||||
|
Results for every file (both sides' JSON and stderr, `results.csv`,
|
||||||
|
`results.json`, `summary.md`) are left in `.cache/results/`.
|
||||||
|
|
||||||
|
The nightly job is `.gitea/workflows/conformance.yml`; it prints the report
|
||||||
|
into the job log.
|
||||||
|
|
||||||
|
The canonical value encoding both sides hash is documented at the top of
|
||||||
|
`probe/src/main.rs`. Values are compared as libhdf5 presents them: a float
|
||||||
|
with a non-IEEE bit layout (N-Bit) or an integer with a bit offset is compared
|
||||||
|
as the converted number, not as raw file bytes.
|
||||||
@@ -0,0 +1,648 @@
|
|||||||
|
{
|
||||||
|
"comment": "conformance/run.sh fails if the ok count drops below `ok` or a file in `ok_files` stops being ok. Regenerate with `conformance/run.sh --update-baseline` after an intended change.",
|
||||||
|
"commit": "f37e7ae3263277319dba4bc39be5397194eb00c3",
|
||||||
|
"date": "2026-09-27 00:34 UTC",
|
||||||
|
"reference": "h5py 3.16.0 / HDF5 2.0.0",
|
||||||
|
"files": 697,
|
||||||
|
"ok": 600,
|
||||||
|
"counts": {
|
||||||
|
"h5py-cannot-read": 92,
|
||||||
|
"mismatch": 2,
|
||||||
|
"ok": 600,
|
||||||
|
"our-error": 3
|
||||||
|
},
|
||||||
|
"per_corpus": {
|
||||||
|
"NCAS-CMS_pyfive": {
|
||||||
|
"mismatch": 1,
|
||||||
|
"ok": 32
|
||||||
|
},
|
||||||
|
"cve_hdf5": {
|
||||||
|
"h5py-cannot-read": 32,
|
||||||
|
"ok": 113,
|
||||||
|
"our-error": 2
|
||||||
|
},
|
||||||
|
"h5py_data": {
|
||||||
|
"ok": 4
|
||||||
|
},
|
||||||
|
"hdf5": {
|
||||||
|
"h5py-cannot-read": 60,
|
||||||
|
"mismatch": 1,
|
||||||
|
"ok": 404,
|
||||||
|
"our-error": 1
|
||||||
|
},
|
||||||
|
"netcdf-c": {
|
||||||
|
"ok": 20
|
||||||
|
},
|
||||||
|
"netcdf4-python": {
|
||||||
|
"ok": 18
|
||||||
|
},
|
||||||
|
"usnistgov_h5wasm": {
|
||||||
|
"ok": 5
|
||||||
|
},
|
||||||
|
"xarray-data": {
|
||||||
|
"ok": 4
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"ok_files": [
|
||||||
|
"NCAS-CMS_pyfive/tests/compact.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/btreev2.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/chunked.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/cmip_bad_eg.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/compressed.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/compressed_v1.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/dataset_datatypes.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/dataset_multidim.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/dim_scales.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/earliest.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/enum_h5variable.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/enum_variable.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/enum_variable.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/enums_from_netcdf.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/fillvalue_earliest.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/fillvalue_latest.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/filter_pipeline_v2.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/fletcher32.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/fractal_heap_no_mci_rlat.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/groups.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/h5netcdf_test.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/issue23_A.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/issue23_A_contiguous.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/issue23_B.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/latest.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/netcdf4_classic.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/new_style_groups.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/noy_AERmonZ_UKESM1-0-LL_piControl_r1i1p1f2_gnz_200001-200012.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/references.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/resizable.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/opaque_datetime.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/opaque_fixed.hdf5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2016-4330.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2016-4331.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2016-4332-mtime-new.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2016-4332-mtime.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2016-4333.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2017-17505.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2017-17506.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2017-17507.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2017-17508.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2017-17509.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-11202.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-11203.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-11204.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-11205.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-11206-new.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-11206-old.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-11207.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-13867.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-13868.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-13869.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-13870.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-13871.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-13872.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-13873.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-13875.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-14031.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-14033.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-14034.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-14035.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-14460.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-15671.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-15672.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-16438.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-17233.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-17234.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-17237.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-17432.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-17434.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-17435.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-17437.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-17438",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-17439",
|
||||||
|
"cve_hdf5/cvefiles/cve-2019-8396.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2019-8397.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2019-8398.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2019-9151.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2019-9152.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2020-10811.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2020-18232.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2020-18494.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2021-36977.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2021-37501.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2021-45829.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2021-45833.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2021-46243.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2021-46244.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29157.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29158.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29159.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29160.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29161.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29162.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29163.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29164.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29165.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29166.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32605.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32606.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32607-1.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32607-2.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32608.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32610.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32611.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32612.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32613.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32614.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32615.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32616.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32617.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32618.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32619.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32620.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32621.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32622.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32623.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32624.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-33873.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-33874.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-33875.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-33876.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-33877.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-2309.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-2310.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-2924.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-2925.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-44905.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-6269-1.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-6269-2.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-6269-3.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-6269-4.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-6516.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-6857.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-7067.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2026-26200.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2026-34734.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2026-92627.h5",
|
||||||
|
"cve_hdf5/cvefiles/unknown-1.h5",
|
||||||
|
"cve_hdf5/fuzzerfiles/gh-4431-poc-03.h5",
|
||||||
|
"cve_hdf5/fuzzerfiles/gh-4432-poc-05.h5",
|
||||||
|
"cve_hdf5/fuzzerfiles/gh-4433-poc-08.h5",
|
||||||
|
"cve_hdf5/fuzzerfiles/gh-4435-poc-10.h5",
|
||||||
|
"cve_hdf5/fuzzerfiles/gh_2649_flawed.h5",
|
||||||
|
"cve_hdf5/fuzzerfiles/gh_2649_plain_model.h5",
|
||||||
|
"h5py_data/compound-dtype-complex.h5",
|
||||||
|
"h5py_data/vlen_string_dset.h5",
|
||||||
|
"h5py_data/vlen_string_dset_utc.h5",
|
||||||
|
"h5py_data/vlen_string_s390x.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_bitgroom.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_blosc.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_blosc2.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_bshuf.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_bzip2.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_granularbr.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_jpeg.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_lz4.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_lzf.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_zfp.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_zstd.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5G/16/h5ex_g_iterate.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5G/16/h5ex_g_traverse.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5G/h5ex_g_iterate.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5G/h5ex_g_traverse.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5G/h5ex_g_visit.h5",
|
||||||
|
"hdf5/HDF5Examples/FORTRAN/H5G/h5ex_g_iterate.h5",
|
||||||
|
"hdf5/HDF5Examples/FORTRAN/H5G/h5ex_g_traverse.h5",
|
||||||
|
"hdf5/HDF5Examples/FORTRAN/H5G/h5ex_g_visit.h5",
|
||||||
|
"hdf5/HDF5Examples/JAVA/H5G/h5ex_g_iterate.h5",
|
||||||
|
"hdf5/HDF5Examples/JAVA/H5G/h5ex_g_visit.h5",
|
||||||
|
"hdf5/HDF5Examples/JAVA/compat/H5G/110/h5ex_g_iterate.h5",
|
||||||
|
"hdf5/HDF5Examples/JAVA/compat/H5G/110/h5ex_g_visit.h5",
|
||||||
|
"hdf5/HDF5Examples/JAVA/compat/H5G/h5ex_g_iterate.h5",
|
||||||
|
"hdf5/HDF5Examples/JAVA/compat/H5G/h5ex_g_visit.h5",
|
||||||
|
"hdf5/c++/test/th5s.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_ds_be.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_ds_be_new_ref-32bit.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_ds_be_new_ref.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_ds_le.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_ds_le_new_ref.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_ld.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_table_be.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_table_cray.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_table_le.h5",
|
||||||
|
"hdf5/test/testfiles/aggr.h5",
|
||||||
|
"hdf5/test/testfiles/bad_chunk_ndims.h5",
|
||||||
|
"hdf5/test/testfiles/bad_compound.h5",
|
||||||
|
"hdf5/test/testfiles/bad_offset.h5",
|
||||||
|
"hdf5/test/testfiles/be_data.h5",
|
||||||
|
"hdf5/test/testfiles/be_extlink1.h5",
|
||||||
|
"hdf5/test/testfiles/be_extlink2.h5",
|
||||||
|
"hdf5/test/testfiles/btree_idx_1_6.h5",
|
||||||
|
"hdf5/test/testfiles/btree_idx_1_8.h5",
|
||||||
|
"hdf5/test/testfiles/charsets.h5",
|
||||||
|
"hdf5/test/testfiles/corrupt_stab_msg.h5",
|
||||||
|
"hdf5/test/testfiles/deflate.h5",
|
||||||
|
"hdf5/test/testfiles/file_image_core_test.h5",
|
||||||
|
"hdf5/test/testfiles/filespace_1_6.h5",
|
||||||
|
"hdf5/test/testfiles/filespace_1_8.h5",
|
||||||
|
"hdf5/test/testfiles/fill18.h5",
|
||||||
|
"hdf5/test/testfiles/fill_old.h5",
|
||||||
|
"hdf5/test/testfiles/filter_error.h5",
|
||||||
|
"hdf5/test/testfiles/fsm_aggr_nopersist.h5",
|
||||||
|
"hdf5/test/testfiles/fsm_aggr_persist.h5",
|
||||||
|
"hdf5/test/testfiles/group_old.h5",
|
||||||
|
"hdf5/test/testfiles/h5fc_ext1_f.h5",
|
||||||
|
"hdf5/test/testfiles/h5fc_ext1_i.h5",
|
||||||
|
"hdf5/test/testfiles/h5fc_ext2_if.h5",
|
||||||
|
"hdf5/test/testfiles/h5fc_ext2_sf.h5",
|
||||||
|
"hdf5/test/testfiles/h5fc_ext3_isf.h5",
|
||||||
|
"hdf5/test/testfiles/h5fc_ext_none.h5",
|
||||||
|
"hdf5/test/testfiles/le_data.h5",
|
||||||
|
"hdf5/test/testfiles/le_extlink1.h5",
|
||||||
|
"hdf5/test/testfiles/le_extlink2.h5",
|
||||||
|
"hdf5/test/testfiles/memleak_H5O_dtype_decode_helper_H5Odtype.h5",
|
||||||
|
"hdf5/test/testfiles/mergemsg.h5",
|
||||||
|
"hdf5/test/testfiles/noencoder.h5",
|
||||||
|
"hdf5/test/testfiles/none.h5",
|
||||||
|
"hdf5/test/testfiles/paged_nopersist.h5",
|
||||||
|
"hdf5/test/testfiles/paged_persist.h5",
|
||||||
|
"hdf5/test/testfiles/specmetaread.h5",
|
||||||
|
"hdf5/test/testfiles/tarrold.h5",
|
||||||
|
"hdf5/test/testfiles/tbad_msg_count.h5",
|
||||||
|
"hdf5/test/testfiles/tbogus.h5",
|
||||||
|
"hdf5/test/testfiles/test_filters_be.h5",
|
||||||
|
"hdf5/test/testfiles/test_filters_le.h5",
|
||||||
|
"hdf5/test/testfiles/th5s.h5",
|
||||||
|
"hdf5/test/testfiles/tlayouto.h5",
|
||||||
|
"hdf5/test/testfiles/tmisc38a.h5",
|
||||||
|
"hdf5/test/testfiles/tmisc38b.h5",
|
||||||
|
"hdf5/test/testfiles/tmtimen.h5",
|
||||||
|
"hdf5/test/testfiles/tmtimeo.h5",
|
||||||
|
"hdf5/test/testfiles/tnullspace.h5",
|
||||||
|
"hdf5/test/testfiles/tsizeslheap.h5",
|
||||||
|
"hdf5/tools/test/testfiles/bigendian/tall.h5",
|
||||||
|
"hdf5/tools/test/testfiles/bigendian/tdset2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/binfp64.h5",
|
||||||
|
"hdf5/tools/test/testfiles/binin16.h5",
|
||||||
|
"hdf5/tools/test/testfiles/binin32.h5",
|
||||||
|
"hdf5/tools/test/testfiles/binin8.h5",
|
||||||
|
"hdf5/tools/test/testfiles/binin8w.h5",
|
||||||
|
"hdf5/tools/test/testfiles/binuin16.h5",
|
||||||
|
"hdf5/tools/test/testfiles/binuin32.h5",
|
||||||
|
"hdf5/tools/test/testfiles/bounds_latest_latest.h5",
|
||||||
|
"hdf5/tools/test/testfiles/charsets.h5",
|
||||||
|
"hdf5/tools/test/testfiles/compounds_array_vlen1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/compounds_array_vlen2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/err_attr_dspace.h5",
|
||||||
|
"hdf5/tools/test/testfiles/file_space.h5",
|
||||||
|
"hdf5/tools/test/testfiles/filter_fail.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5clear_fsm_persist_equal.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5clear_fsm_persist_less.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5clear_fsm_persist_noclose.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5clear_fsm_persist_user_equal.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5clear_fsm_persist_user_less.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5clear_mdc_image.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5clear_sec2_v0.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5clear_sec2_v2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5copy_extlinks_src.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5copy_extlinks_trg.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5copy_ref.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5copytst.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5copytst_new.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_attr1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_attr2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_attr3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_attr_v_level1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_attr_v_level2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_basic1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_basic2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_comp_vl_strs.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_danglelinks1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_danglelinks2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_dset1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_dset2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_dset3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_dset_zero_dim_size1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_dset_zero_dim_size2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_dtypes.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_empty.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_enum_invalid_values.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_eps1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_eps2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_exclude1-1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_exclude1-2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_exclude2-1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_exclude2-2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_exclude3-1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_exclude3-2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_ext2softlink_src.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_ext2softlink_trg.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_extlink_src.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_extlink_trg.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_grp_recurse1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_grp_recurse2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_grp_recurse_ext1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_grp_recurse_ext2-1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_grp_recurse_ext2-2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_grp_recurse_ext2-3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_hyper1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_hyper2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_linked_softlink.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_links.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_onion_dset_1d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_onion_dset_ext.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_onion_objs.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_softlinks.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_strings1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_strings2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_types.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_edge_v3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_err_level.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_ext1_f.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_ext1_i.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_ext1_s.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_ext2_if.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_ext2_is.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_ext2_sf.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_ext3_isf.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_ext_none.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_non_v3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_CVE-2018-14460.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_CVE-2018-17432.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_aggr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_attr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_attr_refs.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_deflate.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_early.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_ext.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_f32le.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_f32le_ex.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_fill.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_filters.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_fletcher.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_fsm_aggr_nopersist.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_fsm_aggr_persist.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_hlink.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_int32le_1d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_int32le_1d_ex.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_int32le_2d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_int32le_2d_ex.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_int32le_3d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_int32le_3d_ex.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_layout.UD.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_layout.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_layout2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_layout3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_layouto.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_named_dtypes.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_nbit.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_nested_8bit_enum.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_nested_8bit_enum_deflated.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_none.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_objs.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_paged_nopersist.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_paged_persist.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_refs.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_shuffle.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_soffset.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_szip.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_uint8be.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_uint8be_ex.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5stat_err_old_fill.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5stat_err_old_layout.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5stat_err_refcount.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5stat_filters.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5stat_idx.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5stat_newgrat.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5stat_threshold.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5stat_tsohm.h5",
|
||||||
|
"hdf5/tools/test/testfiles/mod_h5clear_mdc_image.h5",
|
||||||
|
"hdf5/tools/test/testfiles/non_comparables1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/non_comparables2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/old_h5fc_ext1_f.h5",
|
||||||
|
"hdf5/tools/test/testfiles/old_h5fc_ext1_i.h5",
|
||||||
|
"hdf5/tools/test/testfiles/old_h5fc_ext1_s.h5",
|
||||||
|
"hdf5/tools/test/testfiles/old_h5fc_ext2_if.h5",
|
||||||
|
"hdf5/tools/test/testfiles/old_h5fc_ext2_is.h5",
|
||||||
|
"hdf5/tools/test/testfiles/old_h5fc_ext2_sf.h5",
|
||||||
|
"hdf5/tools/test/testfiles/old_h5fc_ext3_isf.h5",
|
||||||
|
"hdf5/tools/test/testfiles/old_h5fc_ext_none.h5",
|
||||||
|
"hdf5/tools/test/testfiles/packedbits.h5",
|
||||||
|
"hdf5/tools/test/testfiles/t128bit_float.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tCVE-2021-37501_attr_decode.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tCVE_2018_11206_fill_new.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tCVE_2018_11206_fill_old.h5",
|
||||||
|
"hdf5/tools/test/testfiles/taindices.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tall.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tarray1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tarray1_big.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tarray2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tarray3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tarray4.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tarray5.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tarray8.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tattr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tattr2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tattr4_be.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tattrintsize.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tattrreg.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tbfloat16.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tbfloat16_be.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tbigdims.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tbinary.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tbitnopaque.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tchar.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tcmpdattrintsize.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tcmpdintarray.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tcmpdints.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tcmpdintsize.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tcomplex.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tcompound.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tcompound_complex.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tcompound_complex2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tdatareg.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tdset.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tdset2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tdset_idx.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tempty.h5",
|
||||||
|
"hdf5/tools/test/testfiles/textlink.h5",
|
||||||
|
"hdf5/tools/test/testfiles/textlinkfar.h5",
|
||||||
|
"hdf5/tools/test/testfiles/textlinksrc.h5",
|
||||||
|
"hdf5/tools/test/testfiles/textlinktar.h5",
|
||||||
|
"hdf5/tools/test/testfiles/textpfe.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfcontents1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfcontents2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfilters.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfloat16.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfloat16_be.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfloat4.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfloat6.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfloat8.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfloatsattrs.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfpformat.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfvalues.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tgroup.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tgrp_comments.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tgrpnullspace.h5",
|
||||||
|
"hdf5/tools/test/testfiles/thlink.h5",
|
||||||
|
"hdf5/tools/test/testfiles/thyperslab.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tintascii.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tints4dims.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tintsattrs.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tintsnodata.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tlarge_objname.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tldouble.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tldouble_scalar.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tlonglinks.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tloop.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tnamed_dtype_attr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tnestedcmpddt.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tnestedcomp.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tno-subset.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tnullspace.h5",
|
||||||
|
"hdf5/tools/test/testfiles/torderattr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tordergr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_attr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_compat.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_ext1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_ext2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_grp.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_obj.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_obj_del.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_param.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_reg.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_reg_1d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tsaf.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tscalarattrintsize.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tscalarintattrsize.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tscalarintsize.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tscalarstring.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tslink.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tsoftlinks.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tst_onion_dset_1d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tst_onion_dset_ext.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tst_onion_objs.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tstr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tstr2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tstr3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tudfilter.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tudfilter2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tudlink.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tvldtypes1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tvldtypes2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tvldtypes3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tvldtypes4.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tvldtypes5.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tvlenstr_array.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tvlstr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tvms.h5",
|
||||||
|
"hdf5/tools/test/testfiles/twithub.h5",
|
||||||
|
"hdf5/tools/test/testfiles/twithub513.h5",
|
||||||
|
"hdf5/tools/test/testfiles/txtfp32.h5",
|
||||||
|
"hdf5/tools/test/testfiles/txtfp64.h5",
|
||||||
|
"hdf5/tools/test/testfiles/txtin16.h5",
|
||||||
|
"hdf5/tools/test/testfiles/txtin32.h5",
|
||||||
|
"hdf5/tools/test/testfiles/txtin8.h5",
|
||||||
|
"hdf5/tools/test/testfiles/txtstr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/txtuin16.h5",
|
||||||
|
"hdf5/tools/test/testfiles/txtuin32.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/1_a.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/1_b.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/1_c.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/1_d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/1_e.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/1_f.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/1_vds.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/2_a.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/2_b.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/2_c.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/2_d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/2_e.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/2_vds.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/3_1_vds.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/3_2_vds.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/4_0.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/4_1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/4_2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/4_vds.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/5_a.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/5_b.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/5_c.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/5_vds.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/a.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/b.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/c.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/f-0.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/f-3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/vds-eiger.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/vds-percival-unlim-maxmin.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tbitfields.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tcompound2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tdset2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tenum.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/test35.nc",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tloop2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tmany.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tname-amp.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tname-apos.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tname-gt.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tname-lt.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tname-quot.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tname-sp.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tnodata.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tobjref.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/topaque.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tref-escapes-at.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tref-escapes.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tref.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tstring-at.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tstring.h5",
|
||||||
|
"hdf5/tools/test/testfiles/zerodim.h5",
|
||||||
|
"netcdf-c/h5_test/ref_tst_h_compounds.h5",
|
||||||
|
"netcdf-c/h5_test/ref_tst_h_compounds2.h5",
|
||||||
|
"netcdf-c/nc_test4/ref_hdf5_compat1.nc",
|
||||||
|
"netcdf-c/nc_test4/ref_hdf5_compat2.nc",
|
||||||
|
"netcdf-c/nc_test4/ref_hdf5_compat3.nc",
|
||||||
|
"netcdf-c/nc_test4/ref_szip.h5",
|
||||||
|
"netcdf-c/nc_test4/ref_tst_compounds.nc",
|
||||||
|
"netcdf-c/nc_test4/ref_tst_dims.nc",
|
||||||
|
"netcdf-c/nc_test4/ref_tst_interops4.nc",
|
||||||
|
"netcdf-c/nc_test4/ref_tst_xplatform2_1.nc",
|
||||||
|
"netcdf-c/nc_test4/ref_tst_xplatform2_2.nc",
|
||||||
|
"netcdf-c/nc_test4/tdset.h5",
|
||||||
|
"netcdf-c/ncdump/ref_nc_test_netcdf4_4_0.nc",
|
||||||
|
"netcdf-c/ncdump/ref_no_ncproperty.nc",
|
||||||
|
"netcdf-c/ncdump/ref_provenance_v1.nc",
|
||||||
|
"netcdf-c/ncdump/ref_test_corrupt_magic.nc",
|
||||||
|
"netcdf-c/ncdump/ref_tst_compounds2.nc",
|
||||||
|
"netcdf-c/ncdump/ref_tst_compounds3.nc",
|
||||||
|
"netcdf-c/ncdump/ref_tst_compounds4.nc",
|
||||||
|
"netcdf-c/ncdump/ref_tst_irish_rover.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2000.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2001.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2002.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2003.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2004.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2005.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2006.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2007.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2008.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2009.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2010.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2011.nc",
|
||||||
|
"netcdf4-python/examples/data/rtofs_glo_3dz_f006_6hrly_reg3.nc",
|
||||||
|
"netcdf4-python/test/20171025_2056.Cloud_Top_Height.nc",
|
||||||
|
"netcdf4-python/test/issue1152.nc",
|
||||||
|
"netcdf4-python/test/issue671.nc",
|
||||||
|
"netcdf4-python/test/issue672.nc",
|
||||||
|
"netcdf4-python/test/test_gold.nc",
|
||||||
|
"usnistgov_h5wasm/test/array.h5",
|
||||||
|
"usnistgov_h5wasm/test/compressed.h5",
|
||||||
|
"usnistgov_h5wasm/test/empty.h5",
|
||||||
|
"usnistgov_h5wasm/test/float16.h5",
|
||||||
|
"usnistgov_h5wasm/test/vlen.h5",
|
||||||
|
"xarray-data/ROMS_example.nc",
|
||||||
|
"xarray-data/basin_mask.nc",
|
||||||
|
"xarray-data/imerghh_730.hdf5",
|
||||||
|
"xarray-data/precipitation.nc4"
|
||||||
|
]
|
||||||
|
}
|
||||||
Executable
+88
@@ -0,0 +1,88 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""check.py <results_dir> <baseline.json> [--update]
|
||||||
|
|
||||||
|
The conformance gate. Fails (exit 1) when
|
||||||
|
* clawhdf5 panicked, hung, crashed or ran out of memory on any file, or
|
||||||
|
* the ok count fell below the baseline's, or
|
||||||
|
* a file the baseline lists as ok is no longer ok (even if another file
|
||||||
|
became ok and the total held).
|
||||||
|
New ok files are reported so the baseline can be raised (--update rewrites it
|
||||||
|
from the results).
|
||||||
|
"""
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
|
||||||
|
FATAL = ("panic", "hang", "crash", "oom")
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
args = [a for a in sys.argv[1:] if not a.startswith("--")]
|
||||||
|
update = "--update" in sys.argv
|
||||||
|
res_dir, base_path = args
|
||||||
|
res = json.load(open(os.path.join(res_dir, "results.json")))
|
||||||
|
rows = res["rows"]
|
||||||
|
counts = {}
|
||||||
|
per_corpus = {}
|
||||||
|
for r in rows:
|
||||||
|
counts[r["class"]] = counts.get(r["class"], 0) + 1
|
||||||
|
pc = per_corpus.setdefault(r["corpus"], {})
|
||||||
|
pc[r["class"]] = pc.get(r["class"], 0) + 1
|
||||||
|
ok_files = sorted(r["file"] for r in rows if r["class"] == "ok")
|
||||||
|
|
||||||
|
if update:
|
||||||
|
meta = {}
|
||||||
|
mp = os.path.join(res_dir, "report-meta.json")
|
||||||
|
if os.path.exists(mp):
|
||||||
|
meta = json.load(open(mp))
|
||||||
|
base = {
|
||||||
|
"comment": "conformance/run.sh fails if the ok count drops below `ok` or a file in `ok_files` stops being ok. "
|
||||||
|
"Regenerate with `conformance/run.sh --update-baseline` after an intended change.",
|
||||||
|
"commit": meta.get("commit", ""),
|
||||||
|
"date": meta.get("date", ""),
|
||||||
|
"reference": meta.get("reference", ""),
|
||||||
|
"files": len(rows),
|
||||||
|
"ok": len(ok_files),
|
||||||
|
"counts": dict(sorted(counts.items())),
|
||||||
|
"per_corpus": {k: dict(sorted(v.items())) for k, v in sorted(per_corpus.items())},
|
||||||
|
"ok_files": ok_files,
|
||||||
|
}
|
||||||
|
with open(base_path, "w") as fh:
|
||||||
|
json.dump(base, fh, indent=1)
|
||||||
|
fh.write("\n")
|
||||||
|
print(f"baseline updated: {len(ok_files)} ok of {len(rows)} files -> {base_path}")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
base = json.load(open(base_path))
|
||||||
|
failures = []
|
||||||
|
fatal = [r for r in rows if r["class"] in FATAL]
|
||||||
|
for r in fatal:
|
||||||
|
failures.append(f"{r['class']}: {r['file']}: {r['ours_detail'][:200]}")
|
||||||
|
if len(ok_files) < base["ok"]:
|
||||||
|
failures.append(f"ok count dropped: {len(ok_files)} < baseline {base['ok']}")
|
||||||
|
now_ok = set(ok_files)
|
||||||
|
by_file = {r["file"]: r for r in rows}
|
||||||
|
for f in base["ok_files"]:
|
||||||
|
if f not in now_ok:
|
||||||
|
r = by_file.get(f)
|
||||||
|
why = f"now {r['class']}: {(r['ours_detail'] or r['first_issue'])[:200]}" if r else "no longer in the corpus"
|
||||||
|
failures.append(f"regressed: {f}: {why}")
|
||||||
|
gained = sorted(now_ok - set(base["ok_files"]))
|
||||||
|
|
||||||
|
print(f"conformance: {len(ok_files)} ok of {len(rows)} files (baseline {base['ok']} of {base['files']}); "
|
||||||
|
+ ", ".join(f"{k} {v}" for k, v in sorted(counts.items())))
|
||||||
|
if gained:
|
||||||
|
print(f"{len(gained)} file(s) newly ok — raise the baseline with `conformance/run.sh --update-baseline`:")
|
||||||
|
for f in gained:
|
||||||
|
print(f" + {f}")
|
||||||
|
if failures:
|
||||||
|
print(f"CONFORMANCE GATE FAILED ({len(failures)}):")
|
||||||
|
for f in failures:
|
||||||
|
print(f" - {f}")
|
||||||
|
return 1
|
||||||
|
print("conformance gate passed")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
sys.exit(main())
|
||||||
Executable
+295
@@ -0,0 +1,295 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""compare.py <results_dir>: classify each file and group failures by root cause.
|
||||||
|
|
||||||
|
Writes <results_dir>/results.csv, results.json and summary.md.
|
||||||
|
File classes (first match wins):
|
||||||
|
hang, oom, crash, panic ours: timeout / allocation failure / signal / any panic (caught or not)
|
||||||
|
h5py-cannot-read libhdf5/h5py failed to open the file (or crashed/hung)
|
||||||
|
our-error we fail to open, list, or read something h5py reads
|
||||||
|
mismatch we read something with different shape/values, or a different object set
|
||||||
|
ok
|
||||||
|
"""
|
||||||
|
import collections
|
||||||
|
import csv
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
|
||||||
|
R = sys.argv[1]
|
||||||
|
RUNS = os.path.join(R, "runs")
|
||||||
|
|
||||||
|
|
||||||
|
def load(d, name):
|
||||||
|
rc_p = os.path.join(d, name + ".rc")
|
||||||
|
if not os.path.exists(rc_p):
|
||||||
|
return None
|
||||||
|
rc = int(open(rc_p).read().strip() or -1)
|
||||||
|
err = open(os.path.join(d, name + ".err"), errors="replace").read()
|
||||||
|
js = None
|
||||||
|
try:
|
||||||
|
js = json.load(open(os.path.join(d, name + ".json")))
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
pass
|
||||||
|
return {"rc": rc, "err": err, "json": js}
|
||||||
|
|
||||||
|
|
||||||
|
def proc_status(p):
|
||||||
|
"""-> (status, detail)"""
|
||||||
|
if p is None:
|
||||||
|
return "missing", ""
|
||||||
|
rc, err = p["rc"], p["err"]
|
||||||
|
first_panic = next((ln for ln in err.splitlines() if ln.startswith("PANIC:") or "panicked at" in ln), "")
|
||||||
|
if rc == 0 and p["json"] is not None:
|
||||||
|
return "ok", ""
|
||||||
|
if rc == 137 or rc == 124:
|
||||||
|
return "hang", f"timeout ({os.environ.get('TMO', '20')} s)"
|
||||||
|
if "memory allocation of" in err or "MemoryError" in err or "std::bad_alloc" in err:
|
||||||
|
m = re.search(r"memory allocation of \d+ bytes failed", err)
|
||||||
|
return "oom", m.group(0) if m else "allocation failure"
|
||||||
|
if "overflowed its stack" in err:
|
||||||
|
return "crash", "stack overflow"
|
||||||
|
if rc == 101:
|
||||||
|
return "panic", first_panic or (err.strip().splitlines() or [""])[-1]
|
||||||
|
if rc in (134, 139, 136, 135, 132) or rc > 128:
|
||||||
|
sig = {134: "SIGABRT", 139: "SIGSEGV", 136: "SIGFPE", 135: "SIGBUS", 132: "SIGILL"}.get(rc, f"signal {rc - 128}")
|
||||||
|
tail = [ln for ln in err.strip().splitlines() if ln.strip()][-1:]
|
||||||
|
return "crash", f"{sig}: {tail[0][:200] if tail else ''}"
|
||||||
|
tail = [ln for ln in err.strip().splitlines() if ln.strip()][-1:]
|
||||||
|
return "crash", f"rc={rc}: {tail[0][:200] if tail else ''}"
|
||||||
|
|
||||||
|
|
||||||
|
def norm(msg):
|
||||||
|
m = msg.split("\n")[0]
|
||||||
|
m = re.sub(r"0x[0-9a-fA-F]+", "X", m)
|
||||||
|
m = re.sub(r'"[^"]*"', '"…"', m)
|
||||||
|
m = re.sub(r"'[^']*'", "'…'", m)
|
||||||
|
m = re.sub(r"\d+", "N", m)
|
||||||
|
return m[:160]
|
||||||
|
|
||||||
|
|
||||||
|
def panic_head(msg):
|
||||||
|
"""First line + first clawhdf5 frame of a PANIC record."""
|
||||||
|
lines = msg.split("\n")
|
||||||
|
frame = next((ln.strip() for ln in lines[1:] if "clawhdf5_format" in ln), "")
|
||||||
|
return lines[0][:300], frame[:300]
|
||||||
|
|
||||||
|
|
||||||
|
def eq_shape(a, b):
|
||||||
|
return a == b
|
||||||
|
|
||||||
|
|
||||||
|
rows = []
|
||||||
|
issues_by_file = {}
|
||||||
|
root_causes = collections.defaultdict(lambda: {"files": set(), "count": 0, "examples": []})
|
||||||
|
mismatch_causes = collections.defaultdict(lambda: {"files": set(), "count": 0, "examples": []})
|
||||||
|
panics = []
|
||||||
|
ref_only_errors = collections.Counter()
|
||||||
|
incomparable = collections.Counter()
|
||||||
|
|
||||||
|
|
||||||
|
def add(bucket, key, file, example):
|
||||||
|
b = bucket[key]
|
||||||
|
b["count"] += 1
|
||||||
|
if file not in b["files"] and len(b["examples"]) < 6:
|
||||||
|
b["examples"].append(example)
|
||||||
|
b["files"].add(file)
|
||||||
|
|
||||||
|
|
||||||
|
files = [ln.strip() for ln in open(os.path.join(R, "files.txt")) if ln.strip()]
|
||||||
|
for rel in files:
|
||||||
|
d = os.path.join(RUNS, rel.replace("/", "__"))
|
||||||
|
corpus = rel.split("/")[0]
|
||||||
|
ours, ref = load(d, "ours"), load(d, "ref")
|
||||||
|
h5dump = load(d, "h5dump")
|
||||||
|
os_, od = proc_status(ours)
|
||||||
|
rs, rd = proc_status(ref)
|
||||||
|
oj = ours["json"] if ours else None
|
||||||
|
rj = ref["json"] if ref else None
|
||||||
|
issues = [] # (kind, detail)
|
||||||
|
caught_panics = []
|
||||||
|
|
||||||
|
def scan_err(path, what, msg):
|
||||||
|
if msg.startswith("PANIC:"):
|
||||||
|
caught_panics.append((path, what, msg))
|
||||||
|
|
||||||
|
if oj:
|
||||||
|
for o in oj.get("objects", []):
|
||||||
|
for k in ("error", "attrs_error", "list_error"):
|
||||||
|
if k in o:
|
||||||
|
scan_err(o["path"], k, o[k])
|
||||||
|
for an, av in (o.get("attrs") or {}).items():
|
||||||
|
if "error" in av:
|
||||||
|
scan_err(o["path"], f"attr {an}", av["error"])
|
||||||
|
if oj.get("open_error", "").startswith("PANIC:"):
|
||||||
|
caught_panics.append(("<open>", "open", oj["open_error"]))
|
||||||
|
|
||||||
|
ref_open_fail = rs != "ok" or (rj is not None and "open_error" in rj)
|
||||||
|
ours_open_err = oj.get("open_error") if oj else None
|
||||||
|
n_obj = n_ok = 0
|
||||||
|
if os_ == "ok" and rj and not ref_open_fail and not ours_open_err:
|
||||||
|
ro = {x["path"]: x for x in rj.get("objects", [])}
|
||||||
|
oo = {x["path"]: x for x in oj.get("objects", [])}
|
||||||
|
our_list_errors = [x for x in oo.values() if "list_error" in x]
|
||||||
|
for p in sorted(set(ro) | set(oo)):
|
||||||
|
a, b = ro.get(p), oo.get(p)
|
||||||
|
n_obj += 1
|
||||||
|
if a is None:
|
||||||
|
issues.append(("mismatch", f"extra object {p} (kind={b.get('kind')})", "extra-object", b))
|
||||||
|
continue
|
||||||
|
if b is None:
|
||||||
|
if our_list_errors:
|
||||||
|
continue # accounted for by the list_error
|
||||||
|
issues.append(("mismatch", f"missing object {p} (kind={a.get('kind')})", "missing-object", a))
|
||||||
|
continue
|
||||||
|
ok = True
|
||||||
|
if a.get("kind") != b.get("kind") and "error" not in b and "error" not in a:
|
||||||
|
issues.append(("mismatch", f"{p}: kind {a.get('kind')} vs ours {b.get('kind')}", "kind", b))
|
||||||
|
ok = False
|
||||||
|
# h5py could not open the object at all: it read none of its
|
||||||
|
# attributes or links, so there is nothing to compare ours with
|
||||||
|
# (the object's own error is compared above and below).
|
||||||
|
ref_unopened = a.get("kind") == "unknown" and "error" in a
|
||||||
|
for k in ("error", "list_error", "attrs_error"):
|
||||||
|
if ref_unopened and k != "error":
|
||||||
|
continue
|
||||||
|
if k in b and k not in a:
|
||||||
|
issues.append(("our-error", f"{p}: {k}: {b[k]}", b[k], b))
|
||||||
|
ok = False
|
||||||
|
elif k in a and k not in b and k == "error":
|
||||||
|
ref_only_errors[norm(a[k])] += 1
|
||||||
|
if a.get("kind") == "dataset" and "error" not in a and "error" not in b:
|
||||||
|
if "skipped" in a or "skipped" in b:
|
||||||
|
pass
|
||||||
|
elif a.get("converted"):
|
||||||
|
incomparable[f"dataset {a['converted']}"] += 1
|
||||||
|
elif a.get("shape") != b.get("shape"):
|
||||||
|
issues.append(("mismatch", f"{p}: shape {a.get('shape')} vs ours {b.get('shape')}", "shape", b))
|
||||||
|
ok = False
|
||||||
|
elif a.get("hash") != b.get("hash"):
|
||||||
|
issues.append(("mismatch", f"{p}: values differ (h5py {a.get('dtype')} vs ours {b.get('dtype')})", "values", b | {"ref_head": a.get("head"), "ref_dtype": a.get("dtype")}))
|
||||||
|
ok = False
|
||||||
|
ra, oa = a.get("attrs") or {}, b.get("attrs") or {}
|
||||||
|
if "attrs_error" not in b and "attrs_error" not in a and not ref_unopened:
|
||||||
|
for an in sorted(set(ra) | set(oa)):
|
||||||
|
x, y = ra.get(an), oa.get(an)
|
||||||
|
if x is None:
|
||||||
|
issues.append(("mismatch", f"{p}@{an}: extra attribute", "extra-attr", y or {}))
|
||||||
|
elif y is None:
|
||||||
|
issues.append(("mismatch", f"{p}@{an}: missing attribute", "missing-attr", x))
|
||||||
|
elif "error" in y and "error" not in x:
|
||||||
|
issues.append(("our-error", f"{p}@{an}: {y['error']}", y["error"], y))
|
||||||
|
elif "error" in x:
|
||||||
|
continue
|
||||||
|
elif x.get("converted"):
|
||||||
|
incomparable[f"attr {x['converted']}"] += 1
|
||||||
|
elif x.get("shape") != y.get("shape"):
|
||||||
|
issues.append(("mismatch", f"{p}@{an}: attr shape {x.get('shape')} vs ours {y.get('shape')}", "attr-shape", y | {"ref_dtype": x.get("dtype")}))
|
||||||
|
elif x.get("hash") != y.get("hash"):
|
||||||
|
issues.append(("mismatch", f"{p}@{an}: attr values differ (h5py {x.get('dtype')} vs ours {y.get('dtype')})", "attr-values", y | {"ref_head": x.get("head"), "ref_dtype": x.get("dtype")}))
|
||||||
|
if ok:
|
||||||
|
n_ok += 1
|
||||||
|
|
||||||
|
# classify
|
||||||
|
if os_ in ("hang", "oom", "crash", "panic"):
|
||||||
|
cls = os_
|
||||||
|
elif caught_panics:
|
||||||
|
cls = "panic"
|
||||||
|
elif ref_open_fail:
|
||||||
|
cls = "h5py-cannot-read"
|
||||||
|
elif ours_open_err:
|
||||||
|
cls = "our-error"
|
||||||
|
issues.append(("our-error", f"open: {ours_open_err}", ours_open_err, {}))
|
||||||
|
elif any(i[0] == "our-error" for i in issues):
|
||||||
|
cls = "our-error"
|
||||||
|
elif issues:
|
||||||
|
cls = "mismatch"
|
||||||
|
else:
|
||||||
|
cls = "ok"
|
||||||
|
|
||||||
|
if os_ in ("hang", "oom", "crash", "panic") or caught_panics:
|
||||||
|
panics.append({
|
||||||
|
"file": rel, "class": cls, "detail": od,
|
||||||
|
"stderr": (ours["err"] if ours else "")[:3000],
|
||||||
|
"caught": [(p, w, m[:2500]) for p, w, m in caught_panics[:3]],
|
||||||
|
"n_caught": len(caught_panics),
|
||||||
|
})
|
||||||
|
for kind, detail, key, rec in issues:
|
||||||
|
if kind == "our-error":
|
||||||
|
add(root_causes, norm(key), rel, detail[:300])
|
||||||
|
else:
|
||||||
|
if key in ("values", "attr-values", "shape", "attr-shape"):
|
||||||
|
mk = f"{key}: ours={rec.get('dtype')} h5py={rec.get('ref_dtype')} layout={rec.get('layout','-')} filters={rec.get('filters','-')}"
|
||||||
|
else:
|
||||||
|
mk = key
|
||||||
|
add(mismatch_causes, mk, rel, detail[:300] + (f" | ref_head={rec.get('ref_head')} our_head={rec.get('head')}" if rec.get("ref_head") else ""))
|
||||||
|
ref_detail = rd if rs != "ok" else ((rj or {}).get("open_error") or "")
|
||||||
|
h5d = ""
|
||||||
|
if h5dump:
|
||||||
|
rc = h5dump["rc"]
|
||||||
|
h5d = {0: "ok", 1: "error", 137: "hang", 124: "hang", 134: "SIGABRT", 139: "SIGSEGV", 136: "SIGFPE", 135: "SIGBUS"}.get(rc, f"rc={rc}")
|
||||||
|
if "memory allocation" in h5dump["err"] or "Cannot allocate" in h5dump["err"]:
|
||||||
|
h5d += "(oom)"
|
||||||
|
rows.append({
|
||||||
|
"file": rel, "corpus": corpus, "class": cls,
|
||||||
|
"ours": os_ if os_ != "ok" else ("open-error" if ours_open_err else ("panic" if caught_panics else "ok")),
|
||||||
|
"ours_detail": (od or ours_open_err or (caught_panics[0][2].split("\n")[0] if caught_panics else ""))[:300],
|
||||||
|
"ref": rs if rs != "ok" else ("open-error" if (rj or {}).get("open_error") else "ok"),
|
||||||
|
"ref_detail": ref_detail[:300],
|
||||||
|
"h5dump_1_14_6": h5d,
|
||||||
|
"h5dump_detail": ([ln for ln in h5dump["err"].splitlines() if ln.strip()][-1:] or [""])[0][:200] if h5dump else "",
|
||||||
|
"objects": n_obj, "objects_ok": n_ok,
|
||||||
|
"issues": len(issues), "first_issue": issues[0][1][:300] if issues else "",
|
||||||
|
"superblock": (oj or {}).get("superblock_version", ""),
|
||||||
|
})
|
||||||
|
# the first issues of each file, for report.py's known-cause matching
|
||||||
|
issues_by_file[rel] = [
|
||||||
|
{"kind": k, "key": key, "detail": det[:300], "ours_dtype": rec.get("dtype"), "ref_dtype": rec.get("ref_dtype")}
|
||||||
|
for k, det, key, rec in issues[:50]
|
||||||
|
]
|
||||||
|
|
||||||
|
with open(os.path.join(R, "results.csv"), "w", newline="") as fh:
|
||||||
|
w = csv.DictWriter(fh, fieldnames=list(rows[0].keys()))
|
||||||
|
w.writeheader()
|
||||||
|
w.writerows(rows)
|
||||||
|
|
||||||
|
|
||||||
|
def ser(b):
|
||||||
|
return {k: {"files": len(v["files"]), "count": v["count"], "examples": v["examples"], "file_list": sorted(v["files"])} for k, v in sorted(b.items(), key=lambda kv: -len(kv[1]["files"]))}
|
||||||
|
|
||||||
|
|
||||||
|
json.dump({"rows": rows, "issues": issues_by_file, "root_causes": ser(root_causes), "mismatch_causes": ser(mismatch_causes),
|
||||||
|
"panics": panics, "incomparable": incomparable.most_common(), "ref_only_errors": ref_only_errors.most_common()},
|
||||||
|
open(os.path.join(R, "results.json"), "w"), indent=1)
|
||||||
|
|
||||||
|
classes = ["ok", "our-error", "mismatch", "h5py-cannot-read", "hang", "panic", "crash", "oom"]
|
||||||
|
by_corpus = collections.defaultdict(collections.Counter)
|
||||||
|
for r in rows:
|
||||||
|
by_corpus[r["corpus"]][r["class"]] += 1
|
||||||
|
by_corpus["ALL"][r["class"]] += 1
|
||||||
|
lines = ["# Conformance sweep summary", "", "| corpus | files | " + " | ".join(classes) + " |", "|---" * (len(classes) + 2) + "|"]
|
||||||
|
for c in sorted(by_corpus, key=lambda k: (k == "ALL", k)):
|
||||||
|
cnt = by_corpus[c]
|
||||||
|
lines.append(f"| {c} | {sum(cnt.values())} | " + " | ".join(str(cnt.get(k, 0)) for k in classes) + " |")
|
||||||
|
lines += ["", "## Panics / hangs / crashes / OOM", ""]
|
||||||
|
for p in panics:
|
||||||
|
lines.append(f"- **{p['file']}** [{p['class']}] {p['detail']}")
|
||||||
|
for path, what, m in p["caught"][:1]:
|
||||||
|
lines.append(" ```\n " + f"{path} ({what}): " + m.replace("\n", "\n ")[:1500] + "\n ```")
|
||||||
|
if not p["caught"] and p["stderr"]:
|
||||||
|
lines.append(" ```\n " + p["stderr"].strip()[:1500].replace("\n", "\n ") + "\n ```")
|
||||||
|
lines += ["", "## Our-error root causes (files affected)", ""]
|
||||||
|
for k, v in ser(root_causes).items():
|
||||||
|
lines.append(f"- [{v['files']} files, {v['count']} objs] `{k}`")
|
||||||
|
for ex in v["examples"][:3]:
|
||||||
|
lines.append(f" - {ex}")
|
||||||
|
lines += ["", "## Mismatch root causes", ""]
|
||||||
|
for k, v in ser(mismatch_causes).items():
|
||||||
|
lines.append(f"- [{v['files']} files, {v['count']} objs] `{k}`")
|
||||||
|
for ex in v["examples"][:3]:
|
||||||
|
lines.append(f" - {ex}")
|
||||||
|
lines += ["", "## Objects h5py fails on but we read (top)", ""]
|
||||||
|
for k, n in ref_only_errors.most_common(15):
|
||||||
|
lines.append(f"- {n} x `{k}`")
|
||||||
|
open(os.path.join(R, "summary.md"), "w").write("\n".join(lines) + "\n")
|
||||||
|
print("\n".join(lines[:4 + len(by_corpus)]))
|
||||||
@@ -0,0 +1,19 @@
|
|||||||
|
# Conformance corpora, pinned by commit. fetch-corpus.sh reads this file.
|
||||||
|
#
|
||||||
|
# name git-url commit root [sparse-checkout patterns...]
|
||||||
|
#
|
||||||
|
# `root` is the directory inside the checkout that is swept ("." = all of it).
|
||||||
|
# Patterns are git non-cone sparse-checkout patterns; none = whole repository.
|
||||||
|
# Every file under <root> with an HDF5/netCDF-4 extension is probed; for
|
||||||
|
# cve_hdf5 the extension-less files in cvefiles/ and fuzzerfiles/ are too.
|
||||||
|
# Licences: each corpus keeps its upstream licence; nothing here is committed
|
||||||
|
# to this repository — the files are downloaded into the gitignored cache.
|
||||||
|
hdf5 https://github.com/HDFGroup/hdf5.git a3cf1ea82cc7a66e50029a688121e1b105a7ce88 . *.h5 *.he5 *.nc *.hdf5 *.h5f
|
||||||
|
cve_hdf5 https://github.com/HDFGroup/cve_hdf5.git 3fd1f5ae3869e01b8ae02b41d7108de7ffb1a374 .
|
||||||
|
netcdf-c https://github.com/Unidata/netcdf-c.git beb7b9585273c1548386231a59b809d906359033 . /nc_test4/*.nc /ncdump/*.nc /nc_test4/*.h5 /ncdump/*.h5 /h5_test/*.h5 /hdf5_test/*.h5
|
||||||
|
NCAS-CMS_pyfive https://github.com/NCAS-CMS/pyfive.git 8cf07b8749133f41c5e30b8a4c604486f687fe74 . *.h5 *.hdf5 *.hdf *.nc *.he5
|
||||||
|
usnistgov_h5wasm https://github.com/usnistgov/h5wasm.git 02f6336527d2812783fcedabfbf42127ec8d06d2 . *.h5 *.hdf5 *.hdf *.nc *.he5
|
||||||
|
netcdf4-python https://github.com/Unidata/netcdf4-python.git 6e67576d39aef8091fb20bd767b4f1a52ddc1bec . *.nc *.h5
|
||||||
|
xarray-data https://github.com/pydata/xarray-data.git a35297e9da2cc99c811014f0c8a4297345a5c28d . /basin_mask.nc /precipitation.nc4 /imerghh_730.hdf5 /eraint_uvz.nc /ROMS_example.nc /tiny.nc
|
||||||
|
# h5py 3.16.0 (tag 3.16.0), its test data files.
|
||||||
|
h5py_data https://github.com/h5py/h5py.git b2f0347c4200333acd89b43733f1caa0c115162f h5py/tests/data_files /h5py/tests/data_files/*
|
||||||
Executable
+39
@@ -0,0 +1,39 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# fetch-corpus.sh [cache_dir]
|
||||||
|
#
|
||||||
|
# Download the corpora pinned in conformance/corpus.txt into the (gitignored)
|
||||||
|
# cache: <cache>/src/<name> is a shallow, sparse, blob-filtered checkout of the
|
||||||
|
# pinned commit and <cache>/corpus/<name> links to the swept root inside it.
|
||||||
|
# A corpus already checked out at its pinned commit is left alone, so a second
|
||||||
|
# run costs nothing and needs no network.
|
||||||
|
set -euo pipefail
|
||||||
|
HERE="$(cd "$(dirname "$0")" && pwd)"
|
||||||
|
CACHE="${1:-${CONFORMANCE_CACHE:-$HERE/.cache}}"
|
||||||
|
mkdir -p "$CACHE/src" "$CACHE/corpus"
|
||||||
|
CACHE="$(cd "$CACHE" && pwd)"
|
||||||
|
|
||||||
|
retry() { local i; for i in 1 2 3 4; do "$@" && return 0; sleep $((i * 5)); done; return 1; }
|
||||||
|
|
||||||
|
grep -v '^[[:space:]]*\(#\|$\)' "$HERE/corpus.txt" | while read -r name url commit root patterns; do
|
||||||
|
src="$CACHE/src/$name"
|
||||||
|
if [ -d "$src/.git" ] && [ "$(git -C "$src" rev-parse HEAD 2>/dev/null)" = "$commit" ]; then
|
||||||
|
echo "cached $name @ ${commit:0:12}"
|
||||||
|
else
|
||||||
|
echo "fetching $name @ ${commit:0:12} from $url"
|
||||||
|
rm -rf "$src"
|
||||||
|
git init -q "$src"
|
||||||
|
git -C "$src" remote add origin "$url"
|
||||||
|
git -C "$src" config advice.detachedHead false
|
||||||
|
if [ -n "$patterns" ]; then
|
||||||
|
git -C "$src" config core.sparseCheckout true
|
||||||
|
# no-cone patterns (globs); `set -f` keeps the shell from expanding them
|
||||||
|
(set -f; printf '%s\n' $patterns) > "$src/.git/info/sparse-checkout"
|
||||||
|
fi
|
||||||
|
retry git -C "$src" fetch -q --depth 1 --filter=blob:none origin "$commit"
|
||||||
|
retry git -C "$src" checkout -q FETCH_HEAD
|
||||||
|
got="$(git -C "$src" rev-parse HEAD)"
|
||||||
|
[ "$got" = "$commit" ] || { echo "error: $name checked out $got, expected $commit" >&2; exit 1; }
|
||||||
|
fi
|
||||||
|
ln -sfn "$src/$root" "$CACHE/corpus/$name"
|
||||||
|
done
|
||||||
|
echo "corpus ready in $CACHE/corpus"
|
||||||
@@ -0,0 +1,48 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""list_files.py <corpus_dir>: print the files the sweep probes, one per line,
|
||||||
|
as <corpus>/<path> in byte order.
|
||||||
|
|
||||||
|
* every file named *.h5 *.hdf5 *.he5 *.nc *.nc4 *.hdf *.h5f in each corpus,
|
||||||
|
except netCDF classic / 64-bit-offset / CDF5 files (magic "CDF"): they are
|
||||||
|
not HDF5, so neither side can read them and they say nothing;
|
||||||
|
* plus, for cve_hdf5, every file in cvefiles/ and fuzzerfiles/ except
|
||||||
|
.md/.c sources — the reproducers are mostly extension-less, and they are
|
||||||
|
kept whatever their bytes look like (that is their point).
|
||||||
|
"""
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
|
||||||
|
EXTS = (".h5", ".hdf5", ".he5", ".nc", ".nc4", ".hdf", ".h5f")
|
||||||
|
|
||||||
|
|
||||||
|
def walk(top):
|
||||||
|
for dirpath, dirnames, filenames in os.walk(top):
|
||||||
|
dirnames[:] = [d for d in dirnames if d != ".git"]
|
||||||
|
for fn in filenames:
|
||||||
|
p = os.path.join(dirpath, fn)
|
||||||
|
if os.path.isfile(p) and not os.path.islink(p):
|
||||||
|
yield os.path.relpath(p, top)
|
||||||
|
|
||||||
|
|
||||||
|
def main(root):
|
||||||
|
out = set()
|
||||||
|
for corpus in sorted(os.listdir(root)):
|
||||||
|
top = os.path.join(root, corpus)
|
||||||
|
if not os.path.isdir(top):
|
||||||
|
continue
|
||||||
|
for rel in walk(top):
|
||||||
|
path = os.path.join(top, rel)
|
||||||
|
if rel.lower().endswith(EXTS):
|
||||||
|
with open(path, "rb") as fh:
|
||||||
|
if fh.read(3) == b"CDF":
|
||||||
|
continue
|
||||||
|
out.add(f"{corpus}/{rel}")
|
||||||
|
elif corpus == "cve_hdf5" and rel.split(os.sep)[0] in ("cvefiles", "fuzzerfiles") \
|
||||||
|
and not rel.endswith((".md", ".c")):
|
||||||
|
out.add(f"{corpus}/{rel}")
|
||||||
|
for f in sorted(out, key=lambda s: s.encode()):
|
||||||
|
print(f)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main(sys.argv[1])
|
||||||
Generated
+492
@@ -0,0 +1,492 @@
|
|||||||
|
# This file is automatically @generated by Cargo.
|
||||||
|
# It is not intended for manual editing.
|
||||||
|
version = 4
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "adler2"
|
||||||
|
version = "2.0.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "320119579fcad9c21884f5c4861d16174d0e06250625266f50fe6898340abefa"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "better_io"
|
||||||
|
version = "0.2.0"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "ef0a3155e943e341e557863e69a708999c94ede624e37865c8e2a91b94efa78f"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "block-buffer"
|
||||||
|
version = "0.10.4"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "3078c7629b62d3f0439517fa394996acacc5cbc91c5a20d8c658e77abd503a71"
|
||||||
|
dependencies = [
|
||||||
|
"generic-array",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "byteorder"
|
||||||
|
version = "1.5.0"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "1fd0f2584146f6f2ef48085050886acf353beff7305ebd1ae69500e27c67f64b"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "bzip2"
|
||||||
|
version = "0.6.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "f3a53fac24f34a81bc9954b5d6cfce0c21e18ec6959f44f56e8e90e4bb7c346c"
|
||||||
|
dependencies = [
|
||||||
|
"libbz2-rs-sys",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "cc"
|
||||||
|
version = "1.5.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "f360145194ee8e21db5ee7f3fcd4fe52210864c75c985dae33218202c8bbe040"
|
||||||
|
dependencies = [
|
||||||
|
"find-msvc-tools",
|
||||||
|
"jobserver",
|
||||||
|
"libc",
|
||||||
|
"shlex",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "cfg-if"
|
||||||
|
version = "1.0.5"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "4e7648175b45a9a48536d676f68d918270699102aa8dab5496df06904c914600"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "clawhdf5-format"
|
||||||
|
version = "2.7.0"
|
||||||
|
dependencies = [
|
||||||
|
"byteorder",
|
||||||
|
"bzip2",
|
||||||
|
"flate2",
|
||||||
|
"libaec-sys",
|
||||||
|
"libc",
|
||||||
|
"lz4_flex",
|
||||||
|
"pco",
|
||||||
|
"portable-atomic",
|
||||||
|
"ruzstd",
|
||||||
|
"sha2",
|
||||||
|
"snap",
|
||||||
|
"zstd",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "conformance-probe"
|
||||||
|
version = "0.1.0"
|
||||||
|
dependencies = [
|
||||||
|
"clawhdf5-format",
|
||||||
|
"serde_json",
|
||||||
|
"sha2",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "cpufeatures"
|
||||||
|
version = "0.2.17"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "59ed5838eebb26a2bb2e58f6d5b5316989ae9d08bab10e0e6d103e656d1b0280"
|
||||||
|
dependencies = [
|
||||||
|
"libc",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "crc32fast"
|
||||||
|
version = "1.5.2"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "01a7799fd6b852db0e61728dde9a204c423b44d689dbd432522543614b490e78"
|
||||||
|
dependencies = [
|
||||||
|
"cfg-if",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "crunchy"
|
||||||
|
version = "0.2.4"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "460fbee9c2c2f33933d720630a6a0bac33ba7053db5344fac858d4b8952d77d5"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "crypto-common"
|
||||||
|
version = "0.1.7"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "78c8292055d1c1df0cce5d180393dc8cce0abec0a7102adb6c7b1eef6016d60a"
|
||||||
|
dependencies = [
|
||||||
|
"generic-array",
|
||||||
|
"typenum",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "digest"
|
||||||
|
version = "0.10.7"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292"
|
||||||
|
dependencies = [
|
||||||
|
"block-buffer",
|
||||||
|
"crypto-common",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "dtype_dispatch"
|
||||||
|
version = "0.2.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "ab23e69df104e2fd85ee63a533a22d2132ef5975dc6b36f9f3e5a7305e4a8ed7"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "find-msvc-tools"
|
||||||
|
version = "0.1.14"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "aedcfb3409746eddb02b9e19ebda1c3394f759a152e48ee875a0844d1b955484"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "flate2"
|
||||||
|
version = "1.1.10"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "6e634e2e0ebac1ee034020da1ca582e17ffe4e0f5e985823721e168928136dcb"
|
||||||
|
dependencies = [
|
||||||
|
"crc32fast",
|
||||||
|
"miniz_oxide",
|
||||||
|
"zlib-rs",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "generic-array"
|
||||||
|
version = "0.14.7"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "85649ca51fd72272d7821adaf274ad91c288277713d9c18820d8499a7ff69e9a"
|
||||||
|
dependencies = [
|
||||||
|
"typenum",
|
||||||
|
"version_check",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "getrandom"
|
||||||
|
version = "0.4.3"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "300e883d756b2e4ec94e02791f39b04b522276138852cfc41d9fb7e904106099"
|
||||||
|
dependencies = [
|
||||||
|
"cfg-if",
|
||||||
|
"libc",
|
||||||
|
"r-efi",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "half"
|
||||||
|
version = "2.7.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "6ea2d84b969582b4b1864a92dc5d27cd2b77b622a8d79306834f1be5ba20d84b"
|
||||||
|
dependencies = [
|
||||||
|
"cfg-if",
|
||||||
|
"crunchy",
|
||||||
|
"zerocopy",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "itoa"
|
||||||
|
version = "1.0.18"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "jobserver"
|
||||||
|
version = "0.1.35"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "1c00acbd29eabad4a2392fa0e921c874934dbbf4194312ad20f04a0ed67a3cb3"
|
||||||
|
dependencies = [
|
||||||
|
"getrandom",
|
||||||
|
"libc",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "libaec-sys"
|
||||||
|
version = "0.1.0"
|
||||||
|
dependencies = [
|
||||||
|
"pkg-config",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "libbz2-rs-sys"
|
||||||
|
version = "0.2.5"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "34b357333733e8260735ba5894eb928c02ecc69c78715f01a8019e7fa7f2db4c"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "libc"
|
||||||
|
version = "0.2.189"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "lz4_flex"
|
||||||
|
version = "0.11.6"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "373f5eceeeab7925e0c1098212f2fbc4d416adec9d35051a6ab251e824c1854a"
|
||||||
|
dependencies = [
|
||||||
|
"twox-hash",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "memchr"
|
||||||
|
version = "2.8.3"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "miniz_oxide"
|
||||||
|
version = "0.9.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "b63fbc4a50860e98e7b2aa7804ded1db5cbc3aff9193adaff57a6931bf7c4b4c"
|
||||||
|
dependencies = [
|
||||||
|
"adler2",
|
||||||
|
"simd-adler32",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "pco"
|
||||||
|
version = "1.0.3"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "386342cad4c6e97f081568e5d910ea7d871314c843aa8fc564f2a6b64cab9456"
|
||||||
|
dependencies = [
|
||||||
|
"better_io",
|
||||||
|
"dtype_dispatch",
|
||||||
|
"half",
|
||||||
|
"rand_xoshiro",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "pkg-config"
|
||||||
|
version = "0.3.34"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "f6b464fbc74e149a392436b17d523f769e057cb6877f6a5c4618bc6f11800548"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "portable-atomic"
|
||||||
|
version = "1.15.0"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "05c8b63e8d9609db387f0324918f81d68fe27748f084ef092fb35954d0539a85"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "proc-macro2"
|
||||||
|
version = "1.0.107"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9"
|
||||||
|
dependencies = [
|
||||||
|
"unicode-ident",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "quote"
|
||||||
|
version = "1.0.47"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001"
|
||||||
|
dependencies = [
|
||||||
|
"proc-macro2",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "r-efi"
|
||||||
|
version = "6.0.0"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "f8dcc9c7d52a811697d2151c701e0d08956f92b0e24136cf4cf27b57a6a0d9bf"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "rand_core"
|
||||||
|
version = "0.6.4"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "ec0be4795e2f6a28069bec0b5ff3e2ac9bafc99e6a9a7dc3547996c5c816922c"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "rand_xoshiro"
|
||||||
|
version = "0.6.0"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "6f97cdb2a36ed4183de61b2f824cc45c9f1037f28afe0a322e9fff4c108b5aaa"
|
||||||
|
dependencies = [
|
||||||
|
"rand_core",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "ruzstd"
|
||||||
|
version = "0.9.0"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "a252f5e20f038fe7b4ea53e073e65398d652c864cc162fc77c56c2f13717b888"
|
||||||
|
dependencies = [
|
||||||
|
"twox-hash",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "serde"
|
||||||
|
version = "1.0.229"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba"
|
||||||
|
dependencies = [
|
||||||
|
"serde_core",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "serde_core"
|
||||||
|
version = "1.0.229"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48"
|
||||||
|
dependencies = [
|
||||||
|
"serde_derive",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "serde_derive"
|
||||||
|
version = "1.0.229"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348"
|
||||||
|
dependencies = [
|
||||||
|
"proc-macro2",
|
||||||
|
"quote",
|
||||||
|
"syn 3.0.6",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "serde_json"
|
||||||
|
version = "1.0.151"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14"
|
||||||
|
dependencies = [
|
||||||
|
"itoa",
|
||||||
|
"memchr",
|
||||||
|
"serde",
|
||||||
|
"serde_core",
|
||||||
|
"zmij",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "sha2"
|
||||||
|
version = "0.10.9"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "a7507d819769d01a365ab707794a4084392c824f54a7a6a7862f8c3d0892b283"
|
||||||
|
dependencies = [
|
||||||
|
"cfg-if",
|
||||||
|
"cpufeatures",
|
||||||
|
"digest",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "shlex"
|
||||||
|
version = "2.0.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "simd-adler32"
|
||||||
|
version = "0.3.10"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "3a219298ac11a56ea9a6d2120044824d6f01aeb034955e7af7bc16858527deea"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "snap"
|
||||||
|
version = "1.1.2"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "199905e6153d6405f9728fe44daace35f8f837bbf830bb6e85fbd5828709a886"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "syn"
|
||||||
|
version = "2.0.119"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "872831b642d1a07999a962a351ed35b955ea2cfc8f3862091e2a240a84f17297"
|
||||||
|
dependencies = [
|
||||||
|
"proc-macro2",
|
||||||
|
"quote",
|
||||||
|
"unicode-ident",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "syn"
|
||||||
|
version = "3.0.6"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "8593e8e72159ed2257d083c7a454a85cbf854f37a0966d8d483aff8c8a3ebcee"
|
||||||
|
dependencies = [
|
||||||
|
"proc-macro2",
|
||||||
|
"quote",
|
||||||
|
"unicode-ident",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "twox-hash"
|
||||||
|
version = "2.1.4"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "5283634e518fe9e82c7b20520bb4bc209009fd16c82077c802f8111ecbb0117a"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "typenum"
|
||||||
|
version = "1.20.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "b6f5e870be6c3b371b77fe0ee0bafb859fa4964b4404c27de1d380043c4dda20"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "unicode-ident"
|
||||||
|
version = "1.0.26"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "d245f478577f809a851594d02313b640fb437e0bb33866753cff937863096954"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "version_check"
|
||||||
|
version = "0.9.5"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "zerocopy"
|
||||||
|
version = "0.8.59"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "6df92bf3d9227be3d53173901ddbffac2babc27ae50f397776ffd6dc33f800cb"
|
||||||
|
dependencies = [
|
||||||
|
"zerocopy-derive",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "zerocopy-derive"
|
||||||
|
version = "0.8.59"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "ac4f328cf2f05d084e496c3e9c3f33ed0a183656a16e1fcec4d464d8373aec82"
|
||||||
|
dependencies = [
|
||||||
|
"proc-macro2",
|
||||||
|
"quote",
|
||||||
|
"syn 2.0.119",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "zlib-rs"
|
||||||
|
version = "0.6.8"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "b268e58e7c693d7c271f93ffc4ba3b380412554231c85bf61ca7af91042a4112"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "zmij"
|
||||||
|
version = "1.0.23"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "zstd"
|
||||||
|
version = "0.13.3"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "e91ee311a569c327171651566e07972200e76fcfe2242a4fa446149a3881c08a"
|
||||||
|
dependencies = [
|
||||||
|
"zstd-safe",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "zstd-safe"
|
||||||
|
version = "7.3.0"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "64d80649ab6db9d9f6f9c80a40becd948eda4714a0a5ac8c4d157a32231c7882"
|
||||||
|
dependencies = [
|
||||||
|
"zstd-sys",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "zstd-sys"
|
||||||
|
version = "2.1.0+zstd.1.5.7"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "0ef0a8027ec3ee71300ab3bcbcd0393f434aa72b91ca6d635a39941deae8eea0"
|
||||||
|
dependencies = [
|
||||||
|
"cc",
|
||||||
|
"pkg-config",
|
||||||
|
]
|
||||||
@@ -0,0 +1,25 @@
|
|||||||
|
[package]
|
||||||
|
name = "conformance-probe"
|
||||||
|
version = "0.1.0"
|
||||||
|
edition = "2024"
|
||||||
|
rust-version = "1.92"
|
||||||
|
publish = false
|
||||||
|
description = "Walks an HDF5 file with clawhdf5-format and prints a canonical JSON description (see conformance/README.md)"
|
||||||
|
|
||||||
|
# Deliberately outside the main workspace: `cargo test --workspace` never
|
||||||
|
# builds it, and it links the optional C codecs (zstd, libaec) that the core
|
||||||
|
# crates' default build must not.
|
||||||
|
[workspace]
|
||||||
|
|
||||||
|
[dependencies]
|
||||||
|
clawhdf5-format = { path = "../../crates/clawhdf5-format", features = ["lz4", "zstd", "szip", "pcodec", "plugin-filters"] }
|
||||||
|
serde_json = "1"
|
||||||
|
sha2 = "0.10"
|
||||||
|
|
||||||
|
[profile.release]
|
||||||
|
# Keep panics catchable (the probe records them per object) and turn integer
|
||||||
|
# overflow into a reported panic instead of silent wraparound.
|
||||||
|
debug = 1
|
||||||
|
overflow-checks = true
|
||||||
|
debug-assertions = true
|
||||||
|
panic = "unwind"
|
||||||
@@ -0,0 +1,960 @@
|
|||||||
|
//! Conformance probe: walks an HDF5 file with clawhdf5-format (the same calls
|
||||||
|
//! the `clawhdf5` facade makes) and prints a canonical JSON description:
|
||||||
|
//! every hard-linked object (sorted-name DFS, deduplicated by header address),
|
||||||
|
//! and for each dataset / attribute its shape plus the SHA-256 of its values
|
||||||
|
//! in a canonical encoding shared with `ref.py`.
|
||||||
|
//!
|
||||||
|
//! Canonical value encoding (per element, concatenated, row-major):
|
||||||
|
//! int / float / bitfield / enum / time : element bytes, little-endian
|
||||||
|
//! non-IEEE-layout float (e.g. N-Bit) : the IEEE float of the same size it converts to
|
||||||
|
//! int with bit offset / short precision: the full-width integer it converts to
|
||||||
|
//! opaque : raw bytes
|
||||||
|
//! compound : members in declaration order (padding dropped)
|
||||||
|
//! array : base elements row-major
|
||||||
|
//! string (fixed or VL) : b'S' + u32le len + bytes (cut at first NUL, trailing spaces stripped)
|
||||||
|
//! VL sequence : b'V' + u32le count + base elements
|
||||||
|
//! reference : b'R' (payload not compared)
|
||||||
|
//!
|
||||||
|
//! Every object is processed inside catch_unwind; a caught panic is recorded
|
||||||
|
//! with its message, location and the clawhdf5 frames of its backtrace.
|
||||||
|
|
||||||
|
use std::cell::RefCell;
|
||||||
|
use std::collections::HashSet;
|
||||||
|
use std::panic::{self, AssertUnwindSafe};
|
||||||
|
|
||||||
|
use clawhdf5_format::attribute::extract_attributes_full;
|
||||||
|
use clawhdf5_format::data_layout::DataLayout;
|
||||||
|
use clawhdf5_format::data_read;
|
||||||
|
use clawhdf5_format::dataspace::{Dataspace, DataspaceType};
|
||||||
|
use clawhdf5_format::datatype::{Datatype, DatatypeByteOrder};
|
||||||
|
use clawhdf5_format::filter_pipeline::FilterPipeline;
|
||||||
|
use clawhdf5_format::group_v1::{self, GroupEntry};
|
||||||
|
use clawhdf5_format::group_v2;
|
||||||
|
use clawhdf5_format::message_type::MessageType;
|
||||||
|
use clawhdf5_format::object_header::{ObjectClass, ObjectHeader};
|
||||||
|
use clawhdf5_format::signature;
|
||||||
|
use clawhdf5_format::superblock::Superblock;
|
||||||
|
use clawhdf5_format::symbol_table::SymbolTableMessage;
|
||||||
|
use clawhdf5_format::vl_data::{VlResolver, check_element_size};
|
||||||
|
use serde_json::{Map, Value, json};
|
||||||
|
use sha2::{Digest, Sha256};
|
||||||
|
|
||||||
|
const MAX_BYTES: u64 = 200 * 1024 * 1024;
|
||||||
|
const MAX_OBJECTS: usize = 200_000;
|
||||||
|
|
||||||
|
thread_local! {
|
||||||
|
static LAST_PANIC: RefCell<Option<String>> = const { RefCell::new(None) };
|
||||||
|
}
|
||||||
|
|
||||||
|
fn install_hook() {
|
||||||
|
panic::set_hook(Box::new(|info| {
|
||||||
|
let msg = if let Some(s) = info.payload().downcast_ref::<&str>() {
|
||||||
|
s.to_string()
|
||||||
|
} else if let Some(s) = info.payload().downcast_ref::<String>() {
|
||||||
|
s.clone()
|
||||||
|
} else {
|
||||||
|
"<non-string panic>".into()
|
||||||
|
};
|
||||||
|
let loc = info
|
||||||
|
.location()
|
||||||
|
.map(|l| format!("{}:{}", l.file(), l.line()))
|
||||||
|
.unwrap_or_default();
|
||||||
|
let bt = std::backtrace::Backtrace::force_capture().to_string();
|
||||||
|
// keep only frames from clawhdf5 code
|
||||||
|
let mut frames = Vec::new();
|
||||||
|
let lines: Vec<&str> = bt.lines().collect();
|
||||||
|
for (i, l) in lines.iter().enumerate() {
|
||||||
|
let t = l.trim();
|
||||||
|
if t.contains("clawhdf5_format::") || t.contains("conformance_probe::") {
|
||||||
|
let at = lines
|
||||||
|
.get(i + 1)
|
||||||
|
.map(|n| n.trim())
|
||||||
|
.filter(|n| n.starts_with("at "))
|
||||||
|
.map(|n| {
|
||||||
|
let n = n.trim_start_matches("at ");
|
||||||
|
match n.find("/crates/") {
|
||||||
|
Some(p) => n[p + 1..].to_string(),
|
||||||
|
None => n.to_string(),
|
||||||
|
}
|
||||||
|
})
|
||||||
|
.unwrap_or_default();
|
||||||
|
let name = t.split_once(": ").map(|x| x.1).unwrap_or(t);
|
||||||
|
frames.push(format!("{name} ({at})"));
|
||||||
|
if frames.len() >= 12 {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let full = format!("PANIC: {msg} @ {loc}\n {}", frames.join("\n "));
|
||||||
|
eprintln!("{full}");
|
||||||
|
LAST_PANIC.with(|p| *p.borrow_mut() = Some(full));
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Run `f`, turning a panic into Err("PANIC: ...").
|
||||||
|
fn guarded<T>(f: impl FnOnce() -> Result<T, String>) -> Result<T, String> {
|
||||||
|
match panic::catch_unwind(AssertUnwindSafe(f)) {
|
||||||
|
Ok(r) => r,
|
||||||
|
Err(_) => Err(LAST_PANIC
|
||||||
|
.with(|p| p.borrow_mut().take())
|
||||||
|
.unwrap_or_else(|| "PANIC: <unknown>".into())),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn e<E: std::fmt::Debug>(x: E) -> String {
|
||||||
|
format!("{x:?}")
|
||||||
|
}
|
||||||
|
|
||||||
|
struct Ctx<'a> {
|
||||||
|
data: &'a [u8],
|
||||||
|
os: u8,
|
||||||
|
ls: u8,
|
||||||
|
base_dir: std::path::PathBuf,
|
||||||
|
/// Resolves variable-length elements as the library does (null
|
||||||
|
/// elements, strings cut at a NUL, heap objects of the wrong size
|
||||||
|
/// refused), caching each heap collection.
|
||||||
|
vl: RefCell<VlResolver<'a>>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl<'a> Ctx<'a> {
|
||||||
|
fn header(&self, addr: u64) -> Result<ObjectHeader, String> {
|
||||||
|
ObjectHeader::parse(self.data, addr as usize, self.os, self.ls).map_err(e)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn payload(&self, h: &ObjectHeader, t: MessageType) -> Result<Option<Vec<u8>>, String> {
|
||||||
|
match h.messages.iter().find(|m| m.msg_type == t) {
|
||||||
|
None => Ok(None),
|
||||||
|
Some(m) => {
|
||||||
|
clawhdf5_format::shared_message::message_data(self.data, m, self.os, self.ls)
|
||||||
|
.map(|c| Some(c.into_owned()))
|
||||||
|
.map_err(e)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn canon(&self, dt: &Datatype, b: &[u8], out: &mut Vec<u8>) -> Result<(), String> {
|
||||||
|
let size = dt.type_size() as usize;
|
||||||
|
if b.len() < size {
|
||||||
|
return Err(format!(
|
||||||
|
"canon: element slice {} < type size {size}",
|
||||||
|
b.len()
|
||||||
|
));
|
||||||
|
}
|
||||||
|
match dt {
|
||||||
|
Datatype::FloatingPoint { .. } if !ieee_layout(dt) => {
|
||||||
|
canon_custom_float(dt, &b[..size], out)?
|
||||||
|
}
|
||||||
|
Datatype::FixedPoint { .. } if partial_int(dt) => {
|
||||||
|
canon_partial_int(dt, &b[..size], out)?
|
||||||
|
}
|
||||||
|
Datatype::FixedPoint { byte_order, .. }
|
||||||
|
| Datatype::BitField { byte_order, .. }
|
||||||
|
| Datatype::FloatingPoint { byte_order, .. } => match byte_order {
|
||||||
|
DatatypeByteOrder::LittleEndian => out.extend_from_slice(&b[..size]),
|
||||||
|
DatatypeByteOrder::BigEndian => out.extend(b[..size].iter().rev()),
|
||||||
|
DatatypeByteOrder::Vax => return Err("canon: VAX byte order".into()),
|
||||||
|
},
|
||||||
|
Datatype::Time { .. } | Datatype::Opaque { .. } => out.extend_from_slice(&b[..size]),
|
||||||
|
Datatype::String { .. } => canon_str(&b[..size], out),
|
||||||
|
Datatype::Compound { members, .. } => {
|
||||||
|
for m in members {
|
||||||
|
let off = m.byte_offset as usize;
|
||||||
|
let ms = m.datatype.type_size() as usize;
|
||||||
|
if off.checked_add(ms).is_none_or(|end| end > size) {
|
||||||
|
return Err(format!("canon: member {} out of bounds", m.name));
|
||||||
|
}
|
||||||
|
self.canon(&m.datatype, &b[off..off + ms], out)?;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Datatype::Reference { .. } => out.push(b'R'),
|
||||||
|
Datatype::Enumeration { base_type, .. } => self.canon(base_type, b, out)?,
|
||||||
|
Datatype::Array {
|
||||||
|
base_type,
|
||||||
|
dimensions,
|
||||||
|
} => {
|
||||||
|
let n: usize = dimensions.iter().map(|d| *d as usize).product();
|
||||||
|
let bs = base_type.type_size() as usize;
|
||||||
|
for i in 0..n {
|
||||||
|
self.canon(base_type, &b[i * bs..], out)?;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Datatype::VariableLength {
|
||||||
|
size: vl_size,
|
||||||
|
is_string,
|
||||||
|
base_type,
|
||||||
|
..
|
||||||
|
} => {
|
||||||
|
check_element_size(*vl_size, self.os).map_err(e)?;
|
||||||
|
let el = &b[..size];
|
||||||
|
if *is_string {
|
||||||
|
let s = self.vl.borrow_mut().string_bytes(el).map_err(e)?;
|
||||||
|
canon_str(&s[0], out);
|
||||||
|
} else {
|
||||||
|
let bs = base_type.type_size() as usize;
|
||||||
|
// The borrow ends here: the base type may itself be
|
||||||
|
// variable-length.
|
||||||
|
let seq = self.vl.borrow_mut().sequences(el, bs).map_err(e)?;
|
||||||
|
let seq = &seq[0];
|
||||||
|
let len = seq.len() / bs;
|
||||||
|
out.push(b'V');
|
||||||
|
out.extend_from_slice(&(len as u32).to_le_bytes());
|
||||||
|
for i in 0..len {
|
||||||
|
self.canon(base_type, &seq[i * bs..], out)?;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Returns (shape json, n_elements)
|
||||||
|
fn shape(ds: &Dataspace) -> (Value, u64) {
|
||||||
|
match ds.space_type {
|
||||||
|
DataspaceType::Null => (Value::String("null".into()), 0),
|
||||||
|
DataspaceType::Scalar => (json!([]), 1),
|
||||||
|
DataspaceType::Simple => {
|
||||||
|
let n = ds.dimensions.iter().fold(1u64, |a, d| a.saturating_mul(*d));
|
||||||
|
(json!(ds.dimensions), n)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn hash_values(
|
||||||
|
&self,
|
||||||
|
dt: &Datatype,
|
||||||
|
raw: &[u8],
|
||||||
|
n: u64,
|
||||||
|
rec: &mut Map<String, Value>,
|
||||||
|
) -> Result<(), String> {
|
||||||
|
let size = dt.type_size() as usize;
|
||||||
|
let need = (n as usize).checked_mul(size).ok_or("n*size overflow")?;
|
||||||
|
if raw.len() != need {
|
||||||
|
return Err(format!(
|
||||||
|
"raw length {} != n_elements {n} * type_size {size}",
|
||||||
|
raw.len()
|
||||||
|
));
|
||||||
|
}
|
||||||
|
let mut canon = Vec::with_capacity(need);
|
||||||
|
for i in 0..n as usize {
|
||||||
|
self.canon(dt, &raw[i * size..(i + 1) * size], &mut canon)?;
|
||||||
|
}
|
||||||
|
let h = Sha256::digest(&canon);
|
||||||
|
rec.insert("hash".into(), Value::String(hex(&h)));
|
||||||
|
rec.insert(
|
||||||
|
"head".into(),
|
||||||
|
Value::String(hex(&canon[..canon.len().min(48)])),
|
||||||
|
);
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// VDS source files resolve next to the virtual file; like the library,
|
||||||
|
/// refuse absolute paths and `..`.
|
||||||
|
fn vds_resolver(
|
||||||
|
&self,
|
||||||
|
) -> impl Fn(&str) -> Result<Option<Vec<u8>>, clawhdf5_format::error::FormatError> + use<> {
|
||||||
|
let base = self.base_dir.clone();
|
||||||
|
move |name: &str| {
|
||||||
|
use clawhdf5_format::error::FormatError;
|
||||||
|
let p = std::path::Path::new(name);
|
||||||
|
if p.is_absolute()
|
||||||
|
|| p.components()
|
||||||
|
.any(|c| matches!(c, std::path::Component::ParentDir))
|
||||||
|
{
|
||||||
|
return Err(FormatError::ChunkedReadError(format!("refused {name}")));
|
||||||
|
}
|
||||||
|
match std::fs::read(base.join(p)) {
|
||||||
|
Ok(b) => Ok(Some(b)),
|
||||||
|
Err(err) if err.kind() == std::io::ErrorKind::NotFound => Ok(None),
|
||||||
|
Err(err) => Err(FormatError::ChunkedReadError(err.to_string())),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn read_named_datatype(&self, h: &ObjectHeader) -> Result<(), String> {
|
||||||
|
let dtb = self
|
||||||
|
.payload(h, MessageType::Datatype)?
|
||||||
|
.ok_or("MissingMessage(Datatype)")?;
|
||||||
|
Datatype::parse_in_header(&dtb, h.version).map_err(e)?;
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn read_dataset(&self, h: &ObjectHeader, rec: &mut Map<String, Value>) -> Result<(), String> {
|
||||||
|
let dtb = self
|
||||||
|
.payload(h, MessageType::Datatype)?
|
||||||
|
.ok_or("MissingMessage(Datatype)")?;
|
||||||
|
let (dt, _) = Datatype::parse_in_header(&dtb, h.version).map_err(e)?;
|
||||||
|
rec.insert("dtype".into(), Value::String(dtype_str(&dt)));
|
||||||
|
let dsb = self
|
||||||
|
.payload(h, MessageType::Dataspace)?
|
||||||
|
.ok_or("MissingMessage(Dataspace)")?;
|
||||||
|
let mut ds = Dataspace::parse(&dsb, self.ls).map_err(e)?;
|
||||||
|
// A virtual dataset's extent can come from its sources (unlimited /
|
||||||
|
// printf mappings), as h5py reports it, rather than the stored one.
|
||||||
|
if let Some(lm) = h
|
||||||
|
.messages
|
||||||
|
.iter()
|
||||||
|
.find(|m| m.msg_type == MessageType::DataLayout)
|
||||||
|
&& let Ok(dl @ DataLayout::Virtual { .. }) =
|
||||||
|
DataLayout::parse(&lm.data, self.os, self.ls)
|
||||||
|
{
|
||||||
|
let resolver = self.vds_resolver();
|
||||||
|
ds.dimensions = clawhdf5_format::vds::virtual_dataset_extent(
|
||||||
|
self.data,
|
||||||
|
&dl,
|
||||||
|
&ds,
|
||||||
|
self.os,
|
||||||
|
self.ls,
|
||||||
|
Some(&resolver),
|
||||||
|
)
|
||||||
|
.map_err(e)?;
|
||||||
|
}
|
||||||
|
let lm = h
|
||||||
|
.messages
|
||||||
|
.iter()
|
||||||
|
.find(|m| m.msg_type == MessageType::DataLayout)
|
||||||
|
.ok_or("MissingMessage(DataLayout)")?;
|
||||||
|
let dl = DataLayout::parse(&lm.data, self.os, self.ls).map_err(e)?;
|
||||||
|
// What libhdf5 checks when it opens the dataset (as File::dataset).
|
||||||
|
data_read::check_dataset_storage(&dl, &ds, &dt, self.data.len() as u64).map_err(e)?;
|
||||||
|
let (shape, n) = Self::shape(&ds);
|
||||||
|
rec.insert("shape".into(), shape);
|
||||||
|
if n.saturating_mul(dt.type_size() as u64) > MAX_BYTES {
|
||||||
|
rec.insert("skipped".into(), Value::String("too large".into()));
|
||||||
|
return Ok(());
|
||||||
|
}
|
||||||
|
rec.insert(
|
||||||
|
"layout".into(),
|
||||||
|
Value::String(
|
||||||
|
match &dl {
|
||||||
|
DataLayout::Compact { .. } => "compact",
|
||||||
|
DataLayout::Contiguous { .. } => "contiguous",
|
||||||
|
DataLayout::Chunked { .. } => "chunked",
|
||||||
|
DataLayout::Virtual { .. } => "virtual",
|
||||||
|
}
|
||||||
|
.into(),
|
||||||
|
),
|
||||||
|
);
|
||||||
|
let pipeline = match self.payload(h, MessageType::FilterPipeline)? {
|
||||||
|
Some(p) => Some(FilterPipeline::parse(&p).map_err(e)?),
|
||||||
|
None => None,
|
||||||
|
};
|
||||||
|
if let Some(p) = &pipeline {
|
||||||
|
rec.insert(
|
||||||
|
"filters".into(),
|
||||||
|
json!(p.filters.iter().map(|f| f.filter_id).collect::<Vec<_>>()),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
let raw = if matches!(dl, DataLayout::Virtual { .. }) {
|
||||||
|
let resolver = self.vds_resolver();
|
||||||
|
let fill = clawhdf5_format::fill_value::dataset_fill_value_in(
|
||||||
|
self.data,
|
||||||
|
&h.messages,
|
||||||
|
self.os,
|
||||||
|
self.ls,
|
||||||
|
)
|
||||||
|
.map_err(e)?;
|
||||||
|
clawhdf5_format::vds::read_virtual_dataset(
|
||||||
|
self.data,
|
||||||
|
&dl,
|
||||||
|
&ds,
|
||||||
|
&dt,
|
||||||
|
fill.as_deref(),
|
||||||
|
self.os,
|
||||||
|
self.ls,
|
||||||
|
Some(&resolver),
|
||||||
|
)
|
||||||
|
.map_err(e)?
|
||||||
|
.data
|
||||||
|
} else {
|
||||||
|
let cache = clawhdf5_format::chunk_cache::ChunkCache::new();
|
||||||
|
clawhdf5_format::fill_value::read_full_with_fill::<clawhdf5_format::error::FormatError>(
|
||||||
|
&h.messages,
|
||||||
|
self.data,
|
||||||
|
&dl,
|
||||||
|
&ds,
|
||||||
|
dt.type_size() as usize,
|
||||||
|
self.os,
|
||||||
|
self.ls,
|
||||||
|
|| {
|
||||||
|
data_read::read_raw_data_cached(
|
||||||
|
self.data,
|
||||||
|
&dl,
|
||||||
|
&ds,
|
||||||
|
&dt,
|
||||||
|
pipeline.as_ref(),
|
||||||
|
self.os,
|
||||||
|
self.ls,
|
||||||
|
&cache,
|
||||||
|
)
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.map_err(e)?
|
||||||
|
};
|
||||||
|
self.hash_values(&dt, &raw, n, rec)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn attrs(&self, h: &ObjectHeader) -> Result<Map<String, Value>, String> {
|
||||||
|
let msgs = extract_attributes_full(self.data, h, self.os, self.ls).map_err(e)?;
|
||||||
|
let mut out = Map::new();
|
||||||
|
for a in &msgs {
|
||||||
|
let r = guarded(|| {
|
||||||
|
let mut rec = Map::new();
|
||||||
|
rec.insert("dtype".into(), Value::String(dtype_str(&a.datatype)));
|
||||||
|
let (shape, n) = Self::shape(&a.dataspace);
|
||||||
|
rec.insert("shape".into(), shape);
|
||||||
|
self.hash_values(&a.datatype, &a.raw_data, n, &mut rec)?;
|
||||||
|
Ok(rec)
|
||||||
|
});
|
||||||
|
let v = match r {
|
||||||
|
Ok(rec) => Value::Object(rec),
|
||||||
|
Err(msg) => json!({ "error": msg }),
|
||||||
|
};
|
||||||
|
out.insert(a.name.clone(), v);
|
||||||
|
}
|
||||||
|
Ok(out)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn entries(&self, h: &ObjectHeader) -> Result<Vec<GroupEntry>, String> {
|
||||||
|
let v1 = h
|
||||||
|
.messages
|
||||||
|
.iter()
|
||||||
|
.find(|m| m.msg_type == MessageType::SymbolTable);
|
||||||
|
if let Some(m) = v1 {
|
||||||
|
let stm = SymbolTableMessage::parse(&m.data, self.os).map_err(e)?;
|
||||||
|
group_v1::resolve_v1_group_entries(self.data, &stm, self.os, self.ls).map_err(e)
|
||||||
|
} else if h
|
||||||
|
.messages
|
||||||
|
.iter()
|
||||||
|
.any(|m| m.msg_type == MessageType::LinkInfo || m.msg_type == MessageType::Link)
|
||||||
|
{
|
||||||
|
group_v2::resolve_v2_group_entries(self.data, h, self.os, self.ls).map_err(e)
|
||||||
|
} else {
|
||||||
|
Ok(Vec::new())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Element bytes as an unsigned integer (at most 16 bytes), honouring byte order.
|
||||||
|
fn element_bits(b: &[u8], byte_order: &DatatypeByteOrder) -> Result<u128, String> {
|
||||||
|
if b.len() > 16 {
|
||||||
|
return Err(format!("canon: {}-byte numeric element", b.len()));
|
||||||
|
}
|
||||||
|
let mut v = 0u128;
|
||||||
|
match byte_order {
|
||||||
|
DatatypeByteOrder::LittleEndian => {
|
||||||
|
for (i, x) in b.iter().enumerate() {
|
||||||
|
v |= u128::from(*x) << (8 * i);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
DatatypeByteOrder::BigEndian => {
|
||||||
|
for x in b {
|
||||||
|
v = (v << 8) | u128::from(*x);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
DatatypeByteOrder::Vax => return Err("canon: VAX byte order".into()),
|
||||||
|
}
|
||||||
|
Ok(v)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn field(v: u128, pos: u32, len: u32) -> u128 {
|
||||||
|
if len == 0 || pos >= 128 {
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
let v = v >> pos;
|
||||||
|
if len >= 128 {
|
||||||
|
v
|
||||||
|
} else {
|
||||||
|
v & ((1u128 << len) - 1)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// True when a float's bit fields are exactly IEEE 754 binary16/32/64 for its
|
||||||
|
/// size. h5py hands back such a type's bytes untouched; any other layout (an
|
||||||
|
/// N-Bit `H5Tset_precision` float, say) is *converted* by libhdf5 into the
|
||||||
|
/// numpy float of the same size, so comparing raw bytes would be meaningless.
|
||||||
|
fn ieee_layout(dt: &Datatype) -> bool {
|
||||||
|
let Datatype::FloatingPoint {
|
||||||
|
size,
|
||||||
|
bit_offset,
|
||||||
|
bit_precision,
|
||||||
|
exponent_location,
|
||||||
|
exponent_size,
|
||||||
|
mantissa_location,
|
||||||
|
mantissa_size,
|
||||||
|
exponent_bias,
|
||||||
|
..
|
||||||
|
} = dt
|
||||||
|
else {
|
||||||
|
return true;
|
||||||
|
};
|
||||||
|
let std = match size {
|
||||||
|
2 => (16, 10, 5, 10, 15),
|
||||||
|
4 => (32, 23, 8, 23, 127),
|
||||||
|
8 => (64, 52, 11, 52, 1023),
|
||||||
|
_ => return true, // no same-size numpy float to convert to: compare raw
|
||||||
|
};
|
||||||
|
*bit_offset == 0
|
||||||
|
&& (
|
||||||
|
*bit_precision,
|
||||||
|
*exponent_location,
|
||||||
|
*exponent_size,
|
||||||
|
*mantissa_size,
|
||||||
|
*exponent_bias,
|
||||||
|
) == (std.0, std.1, std.2, std.3, std.4)
|
||||||
|
&& *mantissa_location == 0
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Canonicalise a non-IEEE-layout float the way libhdf5's float->float
|
||||||
|
/// conversion presents it to h5py: as the IEEE float of the same size.
|
||||||
|
/// Assumes the implied-leading-one normalisation and the sign bit at the top
|
||||||
|
/// of the precision (what `H5Tset_precision` produces; the parser does not
|
||||||
|
/// keep either field).
|
||||||
|
fn canon_custom_float(dt: &Datatype, b: &[u8], out: &mut Vec<u8>) -> Result<(), String> {
|
||||||
|
let Datatype::FloatingPoint {
|
||||||
|
size,
|
||||||
|
byte_order,
|
||||||
|
bit_offset,
|
||||||
|
bit_precision,
|
||||||
|
exponent_location,
|
||||||
|
exponent_size,
|
||||||
|
mantissa_location,
|
||||||
|
mantissa_size,
|
||||||
|
exponent_bias,
|
||||||
|
} = dt
|
||||||
|
else {
|
||||||
|
unreachable!()
|
||||||
|
};
|
||||||
|
let (esize, msize) = (u32::from(*exponent_size), u32::from(*mantissa_size));
|
||||||
|
if esize == 0 || esize > 30 || msize > 64 {
|
||||||
|
return Err(format!("canon: unsupported float layout e{esize} m{msize}"));
|
||||||
|
}
|
||||||
|
let v = element_bits(b, byte_order)?;
|
||||||
|
let sign_pos = (u32::from(*bit_offset) + u32::from(*bit_precision)).saturating_sub(1);
|
||||||
|
let neg = field(v, sign_pos, 1) == 1;
|
||||||
|
let e = field(v, u32::from(*exponent_location), esize) as i64;
|
||||||
|
let m = field(v, u32::from(*mantissa_location), msize);
|
||||||
|
let emax = (1i64 << esize) - 1;
|
||||||
|
let bias = i64::from(*exponent_bias);
|
||||||
|
let mag = if e == emax {
|
||||||
|
if m == 0 { f64::INFINITY } else { f64::NAN }
|
||||||
|
} else if e == 0 {
|
||||||
|
(m as f64) * 2f64.powi((1 - bias - msize as i64) as i32)
|
||||||
|
} else {
|
||||||
|
((1u128 << msize) as f64 + m as f64) * 2f64.powi((e - bias - msize as i64) as i32)
|
||||||
|
};
|
||||||
|
let x = if neg { -mag } else { mag };
|
||||||
|
match size {
|
||||||
|
2 => out
|
||||||
|
.extend_from_slice(&clawhdf5_format::float16::f32_to_f16_bits(x as f32).to_le_bytes()),
|
||||||
|
4 => out.extend_from_slice(&(x as f32).to_le_bytes()),
|
||||||
|
8 => out.extend_from_slice(&x.to_le_bytes()),
|
||||||
|
_ => unreachable!("ieee_layout keeps other sizes raw"),
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Integers stored with a bit offset or reduced precision (N-Bit): libhdf5
|
||||||
|
/// converts them to the full-width integer of the same size, shifting the
|
||||||
|
/// value down and sign-extending from the top precision bit.
|
||||||
|
fn canon_partial_int(dt: &Datatype, b: &[u8], out: &mut Vec<u8>) -> Result<(), String> {
|
||||||
|
let Datatype::FixedPoint {
|
||||||
|
size,
|
||||||
|
byte_order,
|
||||||
|
signed,
|
||||||
|
bit_offset,
|
||||||
|
bit_precision,
|
||||||
|
} = dt
|
||||||
|
else {
|
||||||
|
unreachable!()
|
||||||
|
};
|
||||||
|
let prec = u32::from(*bit_precision);
|
||||||
|
let v = element_bits(b, byte_order)?;
|
||||||
|
let mut x = field(v, u32::from(*bit_offset), prec);
|
||||||
|
if *signed && prec > 0 && prec < 128 && field(x, prec - 1, 1) == 1 {
|
||||||
|
x |= !0u128 << prec;
|
||||||
|
}
|
||||||
|
out.extend_from_slice(&x.to_le_bytes()[..*size as usize]);
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn partial_int(dt: &Datatype) -> bool {
|
||||||
|
matches!(dt, Datatype::FixedPoint { size, bit_offset, bit_precision, .. }
|
||||||
|
if *bit_offset != 0 || u32::from(*bit_precision) != size * 8)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn canon_str(b: &[u8], out: &mut Vec<u8>) {
|
||||||
|
let cut = b.iter().position(|&c| c == 0).unwrap_or(b.len());
|
||||||
|
let mut s = &b[..cut];
|
||||||
|
while let [rest @ .., b' '] = s {
|
||||||
|
s = rest;
|
||||||
|
}
|
||||||
|
out.push(b'S');
|
||||||
|
out.extend_from_slice(&(s.len() as u32).to_le_bytes());
|
||||||
|
out.extend_from_slice(s);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn hex(b: &[u8]) -> String {
|
||||||
|
b.iter().map(|x| format!("{x:02x}")).collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn dtype_str(dt: &Datatype) -> String {
|
||||||
|
match dt {
|
||||||
|
Datatype::FixedPoint {
|
||||||
|
size,
|
||||||
|
signed,
|
||||||
|
byte_order,
|
||||||
|
..
|
||||||
|
} => {
|
||||||
|
format!(
|
||||||
|
"{}{}{}",
|
||||||
|
bo(byte_order),
|
||||||
|
if *signed { "i" } else { "u" },
|
||||||
|
size
|
||||||
|
)
|
||||||
|
}
|
||||||
|
Datatype::FloatingPoint {
|
||||||
|
size, byte_order, ..
|
||||||
|
} => format!("{}f{}", bo(byte_order), size),
|
||||||
|
Datatype::BitField {
|
||||||
|
size, byte_order, ..
|
||||||
|
} => format!("{}b{}", bo(byte_order), size),
|
||||||
|
Datatype::Time { size, .. } => format!("time{size}"),
|
||||||
|
Datatype::String { size, .. } => format!("S{size}"),
|
||||||
|
Datatype::Opaque { size, .. } => format!("V{size}"),
|
||||||
|
Datatype::Compound { size, members } => format!(
|
||||||
|
"{{{}}}{size}",
|
||||||
|
members
|
||||||
|
.iter()
|
||||||
|
.map(|m| format!("{}:{}", m.name, dtype_str(&m.datatype)))
|
||||||
|
.collect::<Vec<_>>()
|
||||||
|
.join(",")
|
||||||
|
),
|
||||||
|
Datatype::Reference { ref_type, .. } => format!("ref({ref_type:?})"),
|
||||||
|
Datatype::Enumeration { base_type, .. } => format!("enum({})", dtype_str(base_type)),
|
||||||
|
Datatype::VariableLength {
|
||||||
|
is_string: true, ..
|
||||||
|
} => "vlstr".into(),
|
||||||
|
Datatype::VariableLength { base_type, .. } => format!("vlen({})", dtype_str(base_type)),
|
||||||
|
Datatype::Array {
|
||||||
|
base_type,
|
||||||
|
dimensions,
|
||||||
|
} => format!("({}){dimensions:?}", dtype_str(base_type)),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn bo(b: &DatatypeByteOrder) -> &'static str {
|
||||||
|
match b {
|
||||||
|
DatatypeByteOrder::LittleEndian => "<",
|
||||||
|
DatatypeByteOrder::BigEndian => ">",
|
||||||
|
DatatypeByteOrder::Vax => "vax",
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn is_group(h: &ObjectHeader) -> bool {
|
||||||
|
h.messages.iter().any(|m| {
|
||||||
|
matches!(
|
||||||
|
m.msg_type,
|
||||||
|
MessageType::LinkInfo | MessageType::Link | MessageType::SymbolTable
|
||||||
|
)
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The probe's kind for an object header: libhdf5's object class
|
||||||
|
/// ([`ObjectHeader::object_class`]: group, then dataset — a datatype *and* a
|
||||||
|
/// dataspace — then named datatype), which is what h5py opens the object as.
|
||||||
|
/// The root group, and a header with only link messages, count as groups.
|
||||||
|
fn kind_of(h: &ObjectHeader, is_root: bool) -> &'static str {
|
||||||
|
match h.object_class() {
|
||||||
|
Some(ObjectClass::Group) => "group",
|
||||||
|
Some(ObjectClass::Dataset) => "dataset",
|
||||||
|
_ if is_root || is_group(h) => "group",
|
||||||
|
Some(ObjectClass::NamedDatatype) => "datatype",
|
||||||
|
None => "unknown",
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn main() {
|
||||||
|
install_hook();
|
||||||
|
let path = std::env::args().nth(1).expect("usage: probe <file>");
|
||||||
|
let mut top = Map::new();
|
||||||
|
top.insert("file".into(), Value::String(path.clone()));
|
||||||
|
let data = match std::fs::read(&path) {
|
||||||
|
Ok(d) => d,
|
||||||
|
Err(err) => {
|
||||||
|
top.insert("open_error".into(), Value::String(format!("Io({err})")));
|
||||||
|
println!("{}", Value::Object(top));
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
// Every address is relative to the superblock: look at the file from
|
||||||
|
// there on (past any user block), as libhdf5 does.
|
||||||
|
let hdf5: &[u8] = match signature::find_signature(&data) {
|
||||||
|
Ok(off) => &data[off..],
|
||||||
|
Err(_) => &data,
|
||||||
|
};
|
||||||
|
let sb = guarded(|| Superblock::parse(hdf5, 0).map_err(e));
|
||||||
|
let sb = match sb {
|
||||||
|
Ok(sb) => sb,
|
||||||
|
Err(msg) => {
|
||||||
|
top.insert("open_error".into(), Value::String(msg));
|
||||||
|
println!("{}", Value::Object(top));
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
// libhdf5 refuses a truncated file and reads nothing past the recorded
|
||||||
|
// end of file.
|
||||||
|
let base = (data.len() - hdf5.len()) as u64;
|
||||||
|
let hdf5 = match sb.data_end(base, data.len() as u64) {
|
||||||
|
Ok(end) => &hdf5[..end as usize],
|
||||||
|
Err(err) => {
|
||||||
|
top.insert("open_error".into(), Value::String(e(err)));
|
||||||
|
println!("{}", Value::Object(top));
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
// libhdf5 decodes the superblock extension at open (an error refuses
|
||||||
|
// the file), and loads a metadata cache image over the file's own
|
||||||
|
// metadata. It loads the image only when it first reads metadata — the
|
||||||
|
// root group — so a file whose image it cannot load still opens and
|
||||||
|
// that read fails. The library decides all three cases with the same
|
||||||
|
// `cache_image_state`: `File` and `MmapFile` open such a file and fail
|
||||||
|
// every object lookup with the image's error, which is what the probe
|
||||||
|
// records here (on the root group, where libhdf5 reports it).
|
||||||
|
use clawhdf5_format::superblock_ext::{self, CacheImageState};
|
||||||
|
let state = match guarded(|| superblock_ext::cache_image_state(hdf5, &sb).map_err(e)) {
|
||||||
|
Ok(x) => x,
|
||||||
|
Err(msg) => {
|
||||||
|
top.insert("open_error".into(), Value::String(msg));
|
||||||
|
println!("{}", Value::Object(top));
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let mut image_error = None;
|
||||||
|
let view = match state {
|
||||||
|
CacheImageState::Absent => None,
|
||||||
|
CacheImageState::Unloadable(err) => {
|
||||||
|
image_error = Some(e(err));
|
||||||
|
None
|
||||||
|
}
|
||||||
|
CacheImageState::Loaded(image) => {
|
||||||
|
let mut v = hdf5.to_vec();
|
||||||
|
match image.block(hdf5).and_then(|b| image.apply(b, &mut v)) {
|
||||||
|
Ok(()) => Some(v),
|
||||||
|
Err(err) => {
|
||||||
|
image_error = Some(e(err));
|
||||||
|
None
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let hdf5: &[u8] = view.as_deref().unwrap_or(hdf5);
|
||||||
|
top.insert("superblock_version".into(), json!(sb.version));
|
||||||
|
let ctx = Ctx {
|
||||||
|
data: hdf5,
|
||||||
|
os: sb.offset_size,
|
||||||
|
ls: sb.length_size,
|
||||||
|
base_dir: std::path::Path::new(&path)
|
||||||
|
.parent()
|
||||||
|
.map(|p| p.to_path_buf())
|
||||||
|
.unwrap_or_default(),
|
||||||
|
vl: RefCell::new(VlResolver::new(hdf5, sb.offset_size, sb.length_size)),
|
||||||
|
};
|
||||||
|
let mut objects: Vec<Value> = Vec::new();
|
||||||
|
let mut visited = HashSet::new();
|
||||||
|
let mut soft_v1 = 0u64;
|
||||||
|
// explicit DFS stack: (address, path)
|
||||||
|
let mut stack: Vec<(u64, String)> = vec![(sb.root_group_address, "/".to_string())];
|
||||||
|
while let Some((addr, p)) = stack.pop() {
|
||||||
|
if objects.len() >= MAX_OBJECTS {
|
||||||
|
top.insert("truncated".into(), json!(true));
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
if !visited.insert(addr) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let mut rec = Map::new();
|
||||||
|
rec.insert("path".into(), Value::String(p.clone()));
|
||||||
|
let r = guarded(|| {
|
||||||
|
if let Some(msg) = &image_error {
|
||||||
|
return Err(msg.clone());
|
||||||
|
}
|
||||||
|
let h = ctx.header(addr)?;
|
||||||
|
Ok(h)
|
||||||
|
});
|
||||||
|
let h = match r {
|
||||||
|
Ok(h) => h,
|
||||||
|
Err(msg) => {
|
||||||
|
rec.insert("kind".into(), Value::String("unknown".into()));
|
||||||
|
rec.insert("error".into(), Value::String(msg));
|
||||||
|
objects.push(Value::Object(rec));
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let kind = kind_of(&h, addr == sb.root_group_address);
|
||||||
|
rec.insert("kind".into(), Value::String(kind.into()));
|
||||||
|
if kind == "dataset"
|
||||||
|
&& let Err(msg) = guarded(|| ctx.read_dataset(&h, &mut rec))
|
||||||
|
{
|
||||||
|
rec.insert("error".into(), Value::String(msg));
|
||||||
|
}
|
||||||
|
// Opening a committed datatype decodes it (h5py's `f[name]` fails on
|
||||||
|
// one libhdf5 cannot decode), so decode it here too.
|
||||||
|
if kind == "datatype"
|
||||||
|
&& let Err(msg) = guarded(|| ctx.read_named_datatype(&h))
|
||||||
|
{
|
||||||
|
rec.insert("error".into(), Value::String(msg));
|
||||||
|
}
|
||||||
|
if kind != "datatype" {
|
||||||
|
match guarded(|| ctx.attrs(&h)) {
|
||||||
|
Ok(m) => {
|
||||||
|
rec.insert("attrs".into(), Value::Object(m));
|
||||||
|
}
|
||||||
|
Err(msg) => {
|
||||||
|
rec.insert("attrs_error".into(), Value::String(msg));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if kind == "group" {
|
||||||
|
match guarded(|| ctx.entries(&h)) {
|
||||||
|
Ok(mut ents) => {
|
||||||
|
ents.retain(|en| {
|
||||||
|
if en.cache_type == 2 {
|
||||||
|
soft_v1 += 1;
|
||||||
|
false
|
||||||
|
} else {
|
||||||
|
true
|
||||||
|
}
|
||||||
|
});
|
||||||
|
ents.sort_by(|a, b| a.name.cmp(&b.name));
|
||||||
|
let base = if p == "/" { String::new() } else { p.clone() };
|
||||||
|
for en in ents.into_iter().rev() {
|
||||||
|
stack.push((en.object_header_address, format!("{base}/{}", en.name)));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Err(msg) => {
|
||||||
|
rec.insert("list_error".into(), Value::String(msg));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
objects.push(Value::Object(rec));
|
||||||
|
}
|
||||||
|
if soft_v1 > 0 {
|
||||||
|
top.insert("v1_soft_link_entries".into(), json!(soft_v1));
|
||||||
|
}
|
||||||
|
top.insert("objects".into(), Value::Array(objects));
|
||||||
|
println!("{}", Value::Object(top));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
/// The N-Bit float of libhdf5's `test/testfiles/le_data.h5`
|
||||||
|
/// (`Nbit_float_data_le`): offset 7, precision 20, sign bit 26, exponent
|
||||||
|
/// 20+6 (bias 31), mantissa 7+13.
|
||||||
|
fn nbit_f32(byte_order: DatatypeByteOrder) -> Datatype {
|
||||||
|
Datatype::FloatingPoint {
|
||||||
|
size: 4,
|
||||||
|
byte_order,
|
||||||
|
bit_offset: 7,
|
||||||
|
bit_precision: 20,
|
||||||
|
exponent_location: 20,
|
||||||
|
exponent_size: 6,
|
||||||
|
mantissa_location: 7,
|
||||||
|
mantissa_size: 13,
|
||||||
|
exponent_bias: 31,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn canon_one(dt: &Datatype, bytes: &[u8]) -> Vec<u8> {
|
||||||
|
let mut out = Vec::new();
|
||||||
|
canon_custom_float(dt, bytes, &mut out).unwrap();
|
||||||
|
out
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn nbit_float_canonicalises_to_the_value_libhdf5_returns() {
|
||||||
|
let le = nbit_f32(DatatypeByteOrder::LittleEndian);
|
||||||
|
let be = nbit_f32(DatatypeByteOrder::BigEndian);
|
||||||
|
assert!(!ieee_layout(&le));
|
||||||
|
// 1.0: exponent = bias, mantissa 0
|
||||||
|
let one: u32 = 31 << 20;
|
||||||
|
assert_eq!(canon_one(&le, &one.to_le_bytes()), 1.0f32.to_le_bytes());
|
||||||
|
assert_eq!(canon_one(&be, &one.to_be_bytes()), 1.0f32.to_le_bytes());
|
||||||
|
// -2.1999512 (h5py's reading of the file's -2.2): sign, e = 32, m = 819
|
||||||
|
let v: u32 = (1 << 26) | (32 << 20) | (819 << 7);
|
||||||
|
assert_eq!(
|
||||||
|
canon_one(&le, &v.to_le_bytes()),
|
||||||
|
(-2.199_951_2f32).to_le_bytes()
|
||||||
|
);
|
||||||
|
assert_eq!(canon_one(&le, &[0; 4]), 0.0f32.to_le_bytes());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn ieee_floats_keep_their_raw_bytes() {
|
||||||
|
let f32le = Datatype::FloatingPoint {
|
||||||
|
size: 4,
|
||||||
|
byte_order: DatatypeByteOrder::LittleEndian,
|
||||||
|
bit_offset: 0,
|
||||||
|
bit_precision: 32,
|
||||||
|
exponent_location: 23,
|
||||||
|
exponent_size: 8,
|
||||||
|
mantissa_location: 0,
|
||||||
|
mantissa_size: 23,
|
||||||
|
exponent_bias: 127,
|
||||||
|
};
|
||||||
|
assert!(ieee_layout(&f32le));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn kind_follows_libhdf5_object_class() {
|
||||||
|
use clawhdf5_format::object_header::HeaderMessage;
|
||||||
|
let header = |types: &[MessageType]| ObjectHeader {
|
||||||
|
version: 2,
|
||||||
|
messages: types
|
||||||
|
.iter()
|
||||||
|
.map(|&msg_type| HeaderMessage {
|
||||||
|
msg_type,
|
||||||
|
size: 0,
|
||||||
|
flags: 0,
|
||||||
|
creation_order: None,
|
||||||
|
data: Vec::new(),
|
||||||
|
})
|
||||||
|
.collect(),
|
||||||
|
reference_count: None,
|
||||||
|
flags: 0,
|
||||||
|
access_time: None,
|
||||||
|
modification_time: None,
|
||||||
|
change_time: None,
|
||||||
|
birth_time: None,
|
||||||
|
};
|
||||||
|
use MessageType::*;
|
||||||
|
// cve-2024-33874 `/Dset1`: a datatype and a layout but no dataspace
|
||||||
|
// is a named datatype to libhdf5 (h5py opens it as one).
|
||||||
|
assert_eq!(kind_of(&header(&[Datatype, DataLayout]), false), "datatype");
|
||||||
|
assert_eq!(
|
||||||
|
kind_of(&header(&[Datatype, Dataspace, DataLayout]), false),
|
||||||
|
"dataset"
|
||||||
|
);
|
||||||
|
assert_eq!(kind_of(&header(&[SymbolTable]), false), "group");
|
||||||
|
assert_eq!(kind_of(&header(&[Link]), false), "group");
|
||||||
|
assert_eq!(kind_of(&header(&[]), true), "group");
|
||||||
|
assert_eq!(kind_of(&header(&[]), false), "unknown");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn partial_precision_int_is_shifted_and_sign_extended() {
|
||||||
|
let dt = Datatype::FixedPoint {
|
||||||
|
size: 4,
|
||||||
|
byte_order: DatatypeByteOrder::BigEndian,
|
||||||
|
signed: true,
|
||||||
|
bit_offset: 4,
|
||||||
|
bit_precision: 17,
|
||||||
|
};
|
||||||
|
assert!(partial_int(&dt));
|
||||||
|
let stored = (((-5i32) as u32) & 0x1_FFFF) << 4;
|
||||||
|
let mut out = Vec::new();
|
||||||
|
canon_partial_int(&dt, &stored.to_be_bytes(), &mut out).unwrap();
|
||||||
|
assert_eq!(out, (-5i32).to_le_bytes());
|
||||||
|
}
|
||||||
|
}
|
||||||
Executable
+273
@@ -0,0 +1,273 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Reference probe: same JSON as the Rust `conformance-probe`, produced with h5py.
|
||||||
|
|
||||||
|
Walk: iterative DFS from '/', children in sorted (UTF-8 byte) name order, hard
|
||||||
|
links only, each object once (first path wins, deduplicated by object identity).
|
||||||
|
Canonical value encoding: see harness/src/main.rs.
|
||||||
|
"""
|
||||||
|
import hashlib
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import struct
|
||||||
|
import sys
|
||||||
|
|
||||||
|
import numpy as np
|
||||||
|
import h5py
|
||||||
|
|
||||||
|
try:
|
||||||
|
import hdf5plugin # noqa: F401 registers blosc/lz4/zstd/bzip2/... filters
|
||||||
|
except Exception: # pragma: no cover
|
||||||
|
pass
|
||||||
|
|
||||||
|
MAX_BYTES = 200 * 1024 * 1024
|
||||||
|
MAX_OBJECTS = 200_000
|
||||||
|
|
||||||
|
|
||||||
|
def canon_str(b, out):
|
||||||
|
if isinstance(b, str):
|
||||||
|
b = b.encode("utf-8", "surrogateescape")
|
||||||
|
b = bytes(b)
|
||||||
|
cut = b.find(b"\x00")
|
||||||
|
if cut >= 0:
|
||||||
|
b = b[:cut]
|
||||||
|
b = b.rstrip(b" ")
|
||||||
|
out += b"S" + struct.pack("<I", len(b)) + b
|
||||||
|
|
||||||
|
|
||||||
|
def simple(dt):
|
||||||
|
if dt.fields:
|
||||||
|
return all(simple(dt.fields[n][0]) for n in dt.names)
|
||||||
|
if dt.subdtype:
|
||||||
|
return simple(dt.subdtype[0])
|
||||||
|
return dt.kind in "iufcbV"
|
||||||
|
|
||||||
|
|
||||||
|
def packed(dt):
|
||||||
|
if dt.fields:
|
||||||
|
return np.dtype([(n, packed(dt.fields[n][0])) for n in dt.names])
|
||||||
|
if dt.subdtype:
|
||||||
|
base, shape = dt.subdtype
|
||||||
|
return np.dtype((packed(base), shape))
|
||||||
|
if dt.kind in "iufcb":
|
||||||
|
return dt.newbyteorder("<")
|
||||||
|
return dt
|
||||||
|
|
||||||
|
|
||||||
|
def canon_el(dt, val, out):
|
||||||
|
if dt.fields:
|
||||||
|
for n in dt.names:
|
||||||
|
canon_el(dt.fields[n][0], val[n], out)
|
||||||
|
return
|
||||||
|
if dt.subdtype:
|
||||||
|
base, _ = dt.subdtype
|
||||||
|
for x in np.asarray(val).reshape(-1):
|
||||||
|
canon_el(base, x, out)
|
||||||
|
return
|
||||||
|
k = dt.kind
|
||||||
|
if k in "iufcb":
|
||||||
|
out += np.asarray(val, dtype=dt).astype(dt.newbyteorder("<")).tobytes()
|
||||||
|
elif k == "V":
|
||||||
|
out += np.asarray(val, dtype=dt).tobytes()
|
||||||
|
elif k == "S":
|
||||||
|
canon_str(val, out)
|
||||||
|
elif k == "O":
|
||||||
|
if h5py.check_string_dtype(dt) is not None:
|
||||||
|
canon_str(val if val is not None else b"", out)
|
||||||
|
elif h5py.check_ref_dtype(dt) is not None:
|
||||||
|
out += b"R"
|
||||||
|
else:
|
||||||
|
base = h5py.check_vlen_dtype(dt)
|
||||||
|
if base is None:
|
||||||
|
raise TypeError(f"unhandled object dtype {dt!r}")
|
||||||
|
arr = np.asarray(val if val is not None else [], dtype=base).reshape(-1)
|
||||||
|
out += b"V" + struct.pack("<I", arr.shape[0])
|
||||||
|
if simple(base):
|
||||||
|
out += arr.astype(packed(base)).tobytes()
|
||||||
|
else:
|
||||||
|
for x in arr:
|
||||||
|
canon_el(base, x, out)
|
||||||
|
elif k == "U":
|
||||||
|
canon_str(str(val), out)
|
||||||
|
else:
|
||||||
|
raise TypeError(f"unhandled dtype kind {k} ({dt!r})")
|
||||||
|
|
||||||
|
|
||||||
|
def has_obj(dt):
|
||||||
|
if dt.fields:
|
||||||
|
return any(has_obj(dt.fields[n][0]) for n in dt.names)
|
||||||
|
if dt.subdtype:
|
||||||
|
return has_obj(dt.subdtype[0])
|
||||||
|
return dt.kind == "O"
|
||||||
|
|
||||||
|
|
||||||
|
def note_conversion(tid, dt, rec):
|
||||||
|
"""h5py converts some file types (FP8, bfloat16, x87 long double, ...) to a
|
||||||
|
different-sized numpy type; then value bytes are not comparable."""
|
||||||
|
try:
|
||||||
|
if not has_obj(dt) and tid.get_size() != dt.itemsize:
|
||||||
|
rec["converted"] = f"file type size {tid.get_size()} -> numpy {dt} ({dt.itemsize})"
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
def hash_values(arr, dt, rec):
|
||||||
|
# h5py expands an HDF5 array element type into trailing array dims, a
|
||||||
|
# nested array type (an array of arrays) into all of them. Converting the
|
||||||
|
# expanded array back to the inner subarray type would broadcast every
|
||||||
|
# element into a whole subarray, so strip every level.
|
||||||
|
while dt.subdtype is not None:
|
||||||
|
dt = dt.subdtype[0]
|
||||||
|
arr = np.asarray(arr, dtype=dt)
|
||||||
|
if simple(dt):
|
||||||
|
c = np.ascontiguousarray(arr).astype(packed(dt)).tobytes()
|
||||||
|
else:
|
||||||
|
out = bytearray()
|
||||||
|
for x in arr.reshape(-1):
|
||||||
|
canon_el(dt, x, out)
|
||||||
|
c = bytes(out)
|
||||||
|
rec["hash"] = hashlib.sha256(c).hexdigest()
|
||||||
|
rec["head"] = c[:48].hex()
|
||||||
|
|
||||||
|
|
||||||
|
def err(e):
|
||||||
|
s = f"{type(e).__name__}: {e}"
|
||||||
|
return s.splitlines()[0][:400] if s else type(e).__name__
|
||||||
|
|
||||||
|
|
||||||
|
def shape_of(s):
|
||||||
|
return "null" if s is None else list(s)
|
||||||
|
|
||||||
|
|
||||||
|
def n_bytes(shape, tid):
|
||||||
|
n = 1
|
||||||
|
for d in shape or ():
|
||||||
|
n *= d
|
||||||
|
return n * tid.get_size()
|
||||||
|
|
||||||
|
|
||||||
|
def read_attrs(obj):
|
||||||
|
out = {}
|
||||||
|
names = sorted(obj.attrs.keys(), key=lambda s: s.encode("utf-8", "surrogateescape"))
|
||||||
|
for name in names:
|
||||||
|
rec = {}
|
||||||
|
try:
|
||||||
|
aid = obj.attrs.get_id(name)
|
||||||
|
rec["dtype"] = str(aid.dtype)
|
||||||
|
rec["shape"] = shape_of(aid.shape)
|
||||||
|
note_conversion(aid.get_type(), aid.dtype, rec)
|
||||||
|
if aid.shape is None:
|
||||||
|
hash_values(np.empty((0,), dtype=aid.dtype), aid.dtype, rec)
|
||||||
|
else:
|
||||||
|
val = obj.attrs[name]
|
||||||
|
hash_values(val, aid.dtype, rec)
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
rec = {"error": err(e)}
|
||||||
|
out[name] = rec
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def main(path):
|
||||||
|
top = {"file": path}
|
||||||
|
try:
|
||||||
|
f = h5py.File(path, "r")
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
top["open_error"] = err(e)
|
||||||
|
print(json.dumps(top))
|
||||||
|
return
|
||||||
|
objects = []
|
||||||
|
seen = set()
|
||||||
|
# Objects h5py cannot open have no ObjectID to deduplicate by; they are
|
||||||
|
# deduplicated by the address their hard link points at instead, as the
|
||||||
|
# probe deduplicates every object by header address.
|
||||||
|
seen_unopenable = set()
|
||||||
|
stack = [("/", None, None)]
|
||||||
|
while stack:
|
||||||
|
p, obj, link_addr = stack.pop()
|
||||||
|
if len(objects) >= MAX_OBJECTS:
|
||||||
|
top["truncated"] = True
|
||||||
|
break
|
||||||
|
rec = {"path": p}
|
||||||
|
try:
|
||||||
|
if obj is None:
|
||||||
|
obj = f[p]
|
||||||
|
key = hash(obj.id) # h5py ObjectID hash = (fileno, object address/token)
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
if link_addr is not None:
|
||||||
|
if link_addr in seen_unopenable:
|
||||||
|
continue
|
||||||
|
seen_unopenable.add(link_addr)
|
||||||
|
rec["kind"] = "unknown"
|
||||||
|
rec["error"] = err(e)
|
||||||
|
objects.append(rec)
|
||||||
|
continue
|
||||||
|
if key in seen:
|
||||||
|
continue
|
||||||
|
seen.add(key)
|
||||||
|
if isinstance(obj, h5py.Dataset):
|
||||||
|
kind = "dataset"
|
||||||
|
elif isinstance(obj, h5py.Group):
|
||||||
|
kind = "group"
|
||||||
|
elif isinstance(obj, h5py.Datatype):
|
||||||
|
kind = "datatype"
|
||||||
|
else:
|
||||||
|
kind = "unknown"
|
||||||
|
rec["kind"] = kind
|
||||||
|
if kind == "dataset":
|
||||||
|
try:
|
||||||
|
dt = obj.dtype
|
||||||
|
rec["dtype"] = str(dt)
|
||||||
|
rec["shape"] = shape_of(obj.shape)
|
||||||
|
note_conversion(obj.id.get_type(), dt, rec)
|
||||||
|
if obj.shape is None:
|
||||||
|
hash_values(np.empty((0,), dtype=dt), dt, rec)
|
||||||
|
elif n_bytes(obj.shape, obj.id.get_type()) > MAX_BYTES:
|
||||||
|
rec["skipped"] = "too large"
|
||||||
|
else:
|
||||||
|
arr = np.empty(obj.shape, dtype=dt)
|
||||||
|
if arr.size:
|
||||||
|
try:
|
||||||
|
obj.read_direct(arr)
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
arr = obj[()]
|
||||||
|
hash_values(arr, dt, rec)
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
rec["error"] = err(e)
|
||||||
|
if kind != "datatype":
|
||||||
|
try:
|
||||||
|
rec["attrs"] = read_attrs(obj)
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
rec["attrs_error"] = err(e)
|
||||||
|
if kind == "group":
|
||||||
|
try:
|
||||||
|
names = sorted(obj.keys(), key=lambda s: s.encode("utf-8", "surrogateescape"))
|
||||||
|
base = "" if p == "/" else p
|
||||||
|
kids = []
|
||||||
|
for n in names:
|
||||||
|
# The link's own type: `obj.get(n, getlink=True)` reports
|
||||||
|
# a user-defined link (type 64-255) as a HardLink.
|
||||||
|
try:
|
||||||
|
info = obj.id.links.get_info(n.encode("utf-8", "surrogateescape"))
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
info = None
|
||||||
|
if info is not None and info.type != h5py.h5l.TYPE_HARD:
|
||||||
|
continue
|
||||||
|
addr = info.u if info is not None else None
|
||||||
|
kids.append((f"{base}/{n}", addr))
|
||||||
|
for k, addr in reversed(kids):
|
||||||
|
stack.append((k, None, addr))
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
rec["list_error"] = err(e)
|
||||||
|
objects.append(rec)
|
||||||
|
top["objects"] = objects
|
||||||
|
print(json.dumps(top), flush=True)
|
||||||
|
# Exit without tearing down the h5py objects: freeing them for some files
|
||||||
|
# that hold references (hdf5's h5repack_attr_refs.h5, cve-2024-32623.h5)
|
||||||
|
# makes libhdf5 2.0 abort with "free(): chunks in smallbin corrupted"
|
||||||
|
# about half the time. That happens after the reading is done, so it says
|
||||||
|
# nothing about what h5py read, but it flipped those files between ok and
|
||||||
|
# h5py-cannot-read from one run to the next.
|
||||||
|
os._exit(0)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main(sys.argv[1])
|
||||||
@@ -0,0 +1,368 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""report.py <results_dir> <CONFORMANCE.md> <corpus_dir>
|
||||||
|
|
||||||
|
Render the sweep's results (compare.py's results.json plus the raw per-side
|
||||||
|
runs) as CONFORMANCE.md, and write <results_dir>/report-meta.json (commit,
|
||||||
|
date, versions) for check.py --update.
|
||||||
|
"""
|
||||||
|
import collections
|
||||||
|
import datetime
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import platform
|
||||||
|
|
||||||
|
import subprocess
|
||||||
|
import sys
|
||||||
|
|
||||||
|
import h5py
|
||||||
|
import numpy
|
||||||
|
|
||||||
|
try:
|
||||||
|
import hdf5plugin
|
||||||
|
HDF5PLUGIN = hdf5plugin.version
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
HDF5PLUGIN = "not installed"
|
||||||
|
|
||||||
|
R, OUT_MD, CORPUS = sys.argv[1], sys.argv[2], sys.argv[3]
|
||||||
|
HERE = os.path.dirname(os.path.abspath(__file__))
|
||||||
|
ROOT = os.path.dirname(HERE)
|
||||||
|
CLASSES = ["ok", "our-error", "mismatch", "h5py-cannot-read", "panic", "hang", "crash", "oom"]
|
||||||
|
|
||||||
|
|
||||||
|
def sh(*cmd, cwd=ROOT):
|
||||||
|
try:
|
||||||
|
return subprocess.run(cmd, cwd=cwd, capture_output=True, text=True, timeout=30).stdout.strip()
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
return ""
|
||||||
|
|
||||||
|
|
||||||
|
def cpu_model():
|
||||||
|
try:
|
||||||
|
for ln in open("/proc/cpuinfo"):
|
||||||
|
if ln.startswith(("model name", "Model")):
|
||||||
|
return ln.split(":", 1)[1].strip()
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
return platform.processor() or "unknown"
|
||||||
|
|
||||||
|
|
||||||
|
def mem_gib():
|
||||||
|
try:
|
||||||
|
for ln in open("/proc/meminfo"):
|
||||||
|
if ln.startswith("MemTotal:"):
|
||||||
|
return f"{int(ln.split()[1]) / 1048576:.0f} GiB"
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
return "?"
|
||||||
|
|
||||||
|
|
||||||
|
res = json.load(open(os.path.join(R, "results.json")))
|
||||||
|
meta_run = json.load(open(os.path.join(R, "meta.json"))) if os.path.exists(os.path.join(R, "meta.json")) else {}
|
||||||
|
rows = res["rows"]
|
||||||
|
issues = res.get("issues", {})
|
||||||
|
|
||||||
|
# safe.directory: a checkout owned by another user (a container) is still ours to read
|
||||||
|
commit = sh("git", "-c", "safe.directory=*", "rev-parse", "HEAD") or os.environ.get("GITHUB_SHA", "unknown")
|
||||||
|
lib_dirty = sh("git", "-c", "safe.directory=*", "status", "--porcelain", "--", "crates", "Cargo.toml")
|
||||||
|
h5dump_v = sh("h5dump", "--version").replace("h5dump: ", "")
|
||||||
|
meta = {
|
||||||
|
"date": datetime.datetime.now(datetime.timezone.utc).strftime("%Y-%m-%d %H:%M UTC"),
|
||||||
|
"commit": commit + (" (library sources modified)" if lib_dirty else ""),
|
||||||
|
"reference": f"h5py {h5py.__version__} / HDF5 {h5py.version.hdf5_version}",
|
||||||
|
}
|
||||||
|
json.dump(meta, open(os.path.join(R, "report-meta.json"), "w"), indent=1)
|
||||||
|
|
||||||
|
pins = []
|
||||||
|
for ln in open(os.path.join(HERE, "corpus.txt")):
|
||||||
|
if ln.strip() and not ln.lstrip().startswith("#"):
|
||||||
|
name, url, rev, root, *_ = ln.split()
|
||||||
|
pins.append((name, url, rev, root))
|
||||||
|
|
||||||
|
by_corpus = collections.defaultdict(collections.Counter)
|
||||||
|
for r in rows:
|
||||||
|
by_corpus[r["corpus"]][r["class"]] += 1
|
||||||
|
total = collections.Counter(r["class"] for r in rows)
|
||||||
|
|
||||||
|
|
||||||
|
def ex_list(files, n=3):
|
||||||
|
s = ", ".join(f"`{f}`" for f in files[:n])
|
||||||
|
return s + (f" (+{len(files) - n} more)" if len(files) > n else "")
|
||||||
|
|
||||||
|
|
||||||
|
# --- known causes that are not clawhdf5 bugs --------------------------------
|
||||||
|
def is_h5py_be_vlen(i):
|
||||||
|
"""h5py returns the elements of a VL sequence of a big-endian base type
|
||||||
|
with their file (big-endian) bytes but a native-endian dtype."""
|
||||||
|
return (i["kind"] == "mismatch" and i["key"] in ("values", "attr-values")
|
||||||
|
and (i.get("ref_dtype") == "object") and (i.get("ours_dtype") or "").startswith("vlen(")
|
||||||
|
and ">" in (i.get("ours_dtype") or ""))
|
||||||
|
|
||||||
|
|
||||||
|
# Objects the reference (h5py 3.16 / HDF5 2.0) reads only because of an
|
||||||
|
# HDF5 2.0 bug, and that clawhdf5 refuses: each one reads past a buffer or
|
||||||
|
# returns bytes the file does not hold, and libhdf5's develop branch refuses all
|
||||||
|
# three. (file, object) -> why. Checked 2026-09-26 against HDF5 2.0.0
|
||||||
|
# and HDFGroup/hdf5 develop sources; see docs/known-issues.md.
|
||||||
|
LIBHDF5_BUGS = {
|
||||||
|
("cve_hdf5/cvefiles/cve-2025-2308.h5", "/Scale_offset_long_long_data_le"):
|
||||||
|
"scale-offset codes run past the end of the chunk: HDF5 2.0 reads past its buffer; "
|
||||||
|
"libhdf5's develop branch refuses the chunk (\"Buffer too short\")",
|
||||||
|
("cve_hdf5/cvefiles/cve-2025-44904.h5", "/Scale_offset_float_data_le"):
|
||||||
|
"unfiltered chunks of 38 and 37 bytes for 48-byte chunks: HDF5 2.0 fills the rest with "
|
||||||
|
"whatever its buffer held; libhdf5's develop branch refuses them (\"incorrect chunk size returned "
|
||||||
|
"from index for unfiltered chunk\")",
|
||||||
|
("hdf5/test/testfiles/bad_nbit_parms_walk.h5", "/Nbit_int_data_le"):
|
||||||
|
"an N-Bit parameter list one value short: HDF5 2.0 reads past the list; libhdf5's own "
|
||||||
|
"test (`test_filter_bad_params`, test/dsets.c) now requires the read to fail",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def is_libhdf5_bug(rel, i):
|
||||||
|
return i["kind"] == "our-error" and any(
|
||||||
|
f == rel and i["detail"].startswith(obj + ":") for (f, obj) in LIBHDF5_BUGS)
|
||||||
|
|
||||||
|
|
||||||
|
known = collections.defaultdict(list)
|
||||||
|
for r in rows:
|
||||||
|
iss = issues.get(r["file"], [])
|
||||||
|
if r["class"] == "mismatch" and iss and all(is_h5py_be_vlen(i) for i in iss):
|
||||||
|
known["h5py-be-vlen"].append(r["file"])
|
||||||
|
if r["class"] == "our-error" and iss and all(is_libhdf5_bug(r["file"], i) for i in iss):
|
||||||
|
known["libhdf5-2.0"].append(r["file"])
|
||||||
|
|
||||||
|
|
||||||
|
# --- the CVE corpus: clawhdf5 vs h5dump vs h5py ------------------------------
|
||||||
|
def side(run, name):
|
||||||
|
p = os.path.join(R, "runs", run, name)
|
||||||
|
if not os.path.exists(p + ".rc"):
|
||||||
|
return None
|
||||||
|
rc = int(open(p + ".rc").read().strip() or -1)
|
||||||
|
err = open(p + ".err", errors="replace").read()
|
||||||
|
try:
|
||||||
|
j = json.load(open(p + ".json"))
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
j = None
|
||||||
|
return rc, err, j
|
||||||
|
|
||||||
|
|
||||||
|
def outcome(s, rust=False):
|
||||||
|
"""-> (bucket, text). bucket in read / error / panic / crash / hang / oom."""
|
||||||
|
if s is None:
|
||||||
|
return "missing", "not run"
|
||||||
|
rc, err, j = s
|
||||||
|
if rc in (137, 124):
|
||||||
|
return "hang", "hang (killed at timeout)"
|
||||||
|
if "memory allocation of" in err or "MemoryError" in err or "bad_alloc" in err or "Cannot allocate" in err:
|
||||||
|
return "oom", "out of memory"
|
||||||
|
if rust and (rc == 101 or "PANIC:" in err):
|
||||||
|
return "panic", "panic"
|
||||||
|
if "overflowed its stack" in err:
|
||||||
|
return "crash", "stack overflow"
|
||||||
|
if rc == 139:
|
||||||
|
return "crash", "SIGSEGV"
|
||||||
|
if rc == 134:
|
||||||
|
return "crash", "SIGABRT" + (" (heap corruption)" if ("corrupted" in err or "free()" in err) else "")
|
||||||
|
if rc > 128:
|
||||||
|
return "crash", f"signal {rc - 128}"
|
||||||
|
if j is None:
|
||||||
|
return ("error", "error exit") if rc in (0, 1) else ("crash", f"exit {rc}")
|
||||||
|
if "open_error" in j:
|
||||||
|
return "error", "open error"
|
||||||
|
objs = j.get("objects", [])
|
||||||
|
ne = sum(1 for o in objs for k in ("error", "attrs_error", "list_error") if k in o)
|
||||||
|
ne += sum(1 for o in objs for a in (o.get("attrs") or {}).values() if "error" in a)
|
||||||
|
return "read", f"read {len(objs)} obj" + (f", {ne} errors" if ne else "")
|
||||||
|
|
||||||
|
|
||||||
|
def h5dump_outcome(s):
|
||||||
|
if s is None:
|
||||||
|
return "missing", "not run"
|
||||||
|
rc, err, _ = s
|
||||||
|
if rc in (137, 124):
|
||||||
|
return "hang", "hang (killed at timeout)"
|
||||||
|
if "memory allocation" in err or "Cannot allocate" in err:
|
||||||
|
return "oom", "out of memory"
|
||||||
|
if rc == 139:
|
||||||
|
return "crash", "SIGSEGV"
|
||||||
|
if rc == 134:
|
||||||
|
return "crash", "SIGABRT" + (" (heap corruption)" if ("corrupted" in err or "free()" in err) else "")
|
||||||
|
if rc > 128:
|
||||||
|
return "crash", f"signal {rc - 128}"
|
||||||
|
return ("read", "ok") if rc == 0 else ("error", "error exit")
|
||||||
|
|
||||||
|
|
||||||
|
cve_rows = []
|
||||||
|
buckets = {"clawhdf5": collections.Counter(), "h5dump": collections.Counter(), "h5py": collections.Counter()}
|
||||||
|
ours_panic = {r["file"] for r in rows if r["class"] == "panic"}
|
||||||
|
for r in rows:
|
||||||
|
if r["corpus"] != "cve_hdf5":
|
||||||
|
continue
|
||||||
|
run = r["file"].replace("/", "__")
|
||||||
|
o = outcome(side(run, "ours"), rust=True)
|
||||||
|
if o[0] == "read" and r["file"] in ours_panic:
|
||||||
|
o = ("panic", "caught panic")
|
||||||
|
p = outcome(side(run, "ref"))
|
||||||
|
d = h5dump_outcome(side(run, "h5dump"))
|
||||||
|
buckets["clawhdf5"][o[0]] += 1
|
||||||
|
buckets["h5py"][p[0]] += 1
|
||||||
|
buckets["h5dump"][d[0]] += 1
|
||||||
|
cve_rows.append((r["file"].split("/", 1)[1], d[1], p[1], o[1], r["class"]))
|
||||||
|
|
||||||
|
# --- render -----------------------------------------------------------------
|
||||||
|
L = []
|
||||||
|
w = L.append
|
||||||
|
w("# clawhdf5 conformance report")
|
||||||
|
w("")
|
||||||
|
w("Every HDF5 file of eight public corpora (pinned by commit) is read twice — by")
|
||||||
|
w("clawhdf5 (`conformance/probe`, the same `clawhdf5-format` calls the facade")
|
||||||
|
w("makes) and by h5py/libhdf5 (`conformance/ref.py`) — and the two readings are")
|
||||||
|
w("compared object by object: the set of hard-linked objects, each dataset's and")
|
||||||
|
w("attribute's shape, and a SHA-256 of its values in a canonical encoding. The")
|
||||||
|
w("CVE corpus is also run through `h5dump`. Each side runs under a timeout and an")
|
||||||
|
w("address-space limit, so a hang, crash or runaway allocation is recorded, not")
|
||||||
|
w("fatal. This file is generated by `conformance/run.sh`; do not edit it by hand.")
|
||||||
|
w("")
|
||||||
|
w("## Run")
|
||||||
|
w("")
|
||||||
|
w("| | |")
|
||||||
|
w("|---|---|")
|
||||||
|
w(f"| date | {meta['date']} |")
|
||||||
|
w(f"| clawhdf5 commit | `{meta['commit']}` |")
|
||||||
|
w(f"| machine | `{platform.node()}`: {cpu_model()}, {os.cpu_count()} CPUs, {mem_gib()}, {platform.system()} {platform.release()} {platform.machine()} |")
|
||||||
|
w(f"| command | `{os.environ.get('CONFORMANCE_CMD', 'conformance/run.sh')}` |")
|
||||||
|
w(f"| rustc | {sh('rustc', '-V')} |")
|
||||||
|
w(f"| reference | h5py {h5py.__version__}, HDF5 {h5py.version.hdf5_version}, numpy {numpy.__version__}, hdf5plugin {HDF5PLUGIN}, Python {platform.python_version()} |")
|
||||||
|
w(f"| h5dump | {h5dump_v} (CVE corpus only) |")
|
||||||
|
if meta_run:
|
||||||
|
w(f"| limits | {meta_run.get('timeout_s')} s timeout (SIGKILL), {int(meta_run.get('mem_kb', 0)) // 1024} MiB address space, per process; {meta_run.get('jobs')} files in parallel |")
|
||||||
|
w(f"| runtime | {meta_run.get('probe_seconds')} s probing + comparing ({meta_run.get('build_seconds')} s fetch/build before it) |")
|
||||||
|
w("")
|
||||||
|
w("## Results")
|
||||||
|
w("")
|
||||||
|
w("A file's class is the first that applies:")
|
||||||
|
w("")
|
||||||
|
w("- **panic / hang / crash / oom** — clawhdf5 panicked (caught per object or not), hit the timeout, died on a signal, or failed an allocation. The CI gate fails on any of these.")
|
||||||
|
w("- **h5py-cannot-read** — libhdf5 could not open the file (or itself crashed or hung). Nothing to compare against; most are the deliberately malformed CVE reproducers.")
|
||||||
|
w("- **our-error** — clawhdf5 returned an error for something h5py reads.")
|
||||||
|
w("- **mismatch** — both read it, but the shapes, values, object set or attribute set differ.")
|
||||||
|
w("- **ok** — every object h5py reads, clawhdf5 reads identically.")
|
||||||
|
w("")
|
||||||
|
w("| corpus | files | " + " | ".join(CLASSES) + " |")
|
||||||
|
w("|---" * (len(CLASSES) + 2) + "|")
|
||||||
|
for c in sorted(by_corpus):
|
||||||
|
cnt = by_corpus[c]
|
||||||
|
w(f"| {c} | {sum(cnt.values())} | " + " | ".join(str(cnt.get(k, 0)) for k in CLASSES) + " |")
|
||||||
|
w(f"| **all** | **{len(rows)}** | " + " | ".join(f"**{total.get(k, 0)}**" for k in CLASSES) + " |")
|
||||||
|
w("")
|
||||||
|
if known["h5py-be-vlen"]:
|
||||||
|
w(f"{len(known['h5py-be-vlen'])} of the {total.get('mismatch', 0)} mismatches are a known h5py bug, "
|
||||||
|
"not ours (see *Known not-our-bug*).")
|
||||||
|
w("")
|
||||||
|
if known["libhdf5-2.0"]:
|
||||||
|
w(f"{len(known['libhdf5-2.0'])} of the {total.get('our-error', 0)} our-errors are corrupt data that "
|
||||||
|
"HDF5 2.0 reads only through a bug and clawhdf5 refuses (see *Known not-our-bug*).")
|
||||||
|
w("")
|
||||||
|
w("Corpora (fetched by `conformance/fetch-corpus.sh` into the gitignored `conformance/.cache/`):")
|
||||||
|
w("")
|
||||||
|
w("| corpus | source | commit |")
|
||||||
|
w("|---|---|---|")
|
||||||
|
for name, url, rev, root in pins:
|
||||||
|
w(f"| {name} | {url.removesuffix('.git')}" + ("" if root == "." else f" (`{root}`)") + f" | `{rev[:12]}` |")
|
||||||
|
w("")
|
||||||
|
|
||||||
|
w("## Panics, hangs, crashes, out-of-memory")
|
||||||
|
w("")
|
||||||
|
if not res["panics"]:
|
||||||
|
w("None.")
|
||||||
|
else:
|
||||||
|
for p in res["panics"]:
|
||||||
|
w(f"- `{p['file']}` [{p['class']}] {p['detail']}")
|
||||||
|
w("")
|
||||||
|
|
||||||
|
w("## Our-error root causes")
|
||||||
|
w("")
|
||||||
|
w("Grouped by normalised error message. *files* counts files whose class this cause affects.")
|
||||||
|
w("")
|
||||||
|
w("| files | objects | error | examples |")
|
||||||
|
w("|---:|---:|---|---|")
|
||||||
|
for k, v in res["root_causes"].items():
|
||||||
|
w(f"| {v['files']} | {v['count']} | `{k.replace('|', '/')}` | {ex_list(v['file_list'])} |")
|
||||||
|
w("")
|
||||||
|
w("## Mismatch root causes")
|
||||||
|
w("")
|
||||||
|
w("| files | objects | cause | examples |")
|
||||||
|
w("|---:|---:|---|---|")
|
||||||
|
for k, v in res["mismatch_causes"].items():
|
||||||
|
w(f"| {v['files']} | {v['count']} | `{k.replace('|', '/')}` | {ex_list(v['file_list'])} |")
|
||||||
|
w("")
|
||||||
|
|
||||||
|
w("## CVE corpus: clawhdf5 vs h5dump vs h5py")
|
||||||
|
w("")
|
||||||
|
w(f"The {len(cve_rows)} files of [HDFGroup/cve_hdf5](https://github.com/HDFGroup/cve_hdf5) — reproducers for")
|
||||||
|
w("published libhdf5 CVEs and fuzzer finds. *read* = produced output (possibly with per-object")
|
||||||
|
w("errors), *error* = refused cleanly. h5dump exits non-zero on any error anywhere in a file, so")
|
||||||
|
w("its read/error split is not comparable with the other two rows; the panic, crash, hang and oom")
|
||||||
|
w("columns are.")
|
||||||
|
w("")
|
||||||
|
w("| tool | read | error | panic | crash | hang | oom |")
|
||||||
|
w("|---|---:|---:|---:|---:|---:|---:|")
|
||||||
|
for tool, label in (("clawhdf5", "clawhdf5"), ("h5dump", f"h5dump {h5dump_v.split()[-1] if h5dump_v else ''}"),
|
||||||
|
("h5py", f"h5py {h5py.__version__} / HDF5 {h5py.version.hdf5_version}")):
|
||||||
|
b = buckets[tool]
|
||||||
|
w(f"| {label} | " + " | ".join(str(b.get(k, 0)) for k in ("read", "error", "panic", "crash", "hang", "oom")) + " |")
|
||||||
|
w("")
|
||||||
|
w("<details><summary>Per-file outcomes</summary>")
|
||||||
|
w("")
|
||||||
|
w("| file | h5dump | h5py | clawhdf5 | class |")
|
||||||
|
w("|---|---|---|---|---|")
|
||||||
|
for f, d, p, o, cls in cve_rows:
|
||||||
|
w(f"| {f} | {d} | {p} | {o} | {cls} |")
|
||||||
|
w("")
|
||||||
|
w("</details>")
|
||||||
|
w("")
|
||||||
|
|
||||||
|
w("## Known not-our-bug")
|
||||||
|
w("")
|
||||||
|
w("- **h5py big-endian variable-length sequences.** h5py returns the elements of a VL sequence")
|
||||||
|
w(" whose base type is big-endian with the file's big-endian bytes but a native (little-endian)")
|
||||||
|
w(" numpy dtype, so the values it reports are byte-swapped garbage; `h5dump` prints the values")
|
||||||
|
w(" clawhdf5 reads. Reproducer: `h5py.vlen_dtype(np.dtype('>f4'))` dataset holding `[1.0, 2.0]`")
|
||||||
|
w(" reads back in h5py as `[4.6e-41, 9.0e-44]`. Affected here: "
|
||||||
|
+ (ex_list(sorted(known["h5py-be-vlen"]), 10) if known["h5py-be-vlen"] else "none") + ".")
|
||||||
|
w("- **Non-IEEE floats and partial-precision integers (N-Bit).** libhdf5 converts a float whose")
|
||||||
|
w(" bit layout is not IEEE (e.g. `H5Tset_precision` for the N-Bit filter) or an integer with a")
|
||||||
|
w(" bit offset / reduced precision into the plain numpy type of the same size. The probe")
|
||||||
|
w(" compares such values as converted numbers, not raw file bytes (before 2026-09-25 it compared")
|
||||||
|
w(" raw bytes, which reported every N-Bit float dataset as a mismatch).")
|
||||||
|
if res["incomparable"]:
|
||||||
|
w("- **Types h5py widens.** Where h5py reads a type into a numpy type of a different size")
|
||||||
|
w(" (FP8 -> float16, bfloat16 -> float32, x87 long double -> float128) the values are not")
|
||||||
|
w(" compared (shape and presence still are): "
|
||||||
|
+ ", ".join(f"{k} ({n}x)" for k, n in res["incomparable"]) + ".")
|
||||||
|
w("- **Corrupt data HDF5 2.0 reads through a bug.** clawhdf5 refuses these objects; h5py 3.16 /")
|
||||||
|
w(" HDF5 2.0 returns values for them that the file does not hold:")
|
||||||
|
for (f, obj), why in sorted(LIBHDF5_BUGS.items()):
|
||||||
|
here = "" if f in known["libhdf5-2.0"] else " (not an our-error in this run)"
|
||||||
|
w(f" - `{f}` `{obj}`: {why}{here}.")
|
||||||
|
w("- **References** are compared by presence only (`R`), not by target.")
|
||||||
|
w("")
|
||||||
|
if res.get("ref_only_errors"):
|
||||||
|
w("## Objects h5py fails on but clawhdf5 reads")
|
||||||
|
w("")
|
||||||
|
for k, n in res["ref_only_errors"][:15]:
|
||||||
|
w(f"- {n} x `{k}`")
|
||||||
|
w("")
|
||||||
|
w("## Reproduce")
|
||||||
|
w("")
|
||||||
|
w("```sh")
|
||||||
|
w("# needs: Rust, python3 with h5py numpy hdf5plugin (conformance/requirements.txt), h5dump (hdf5-tools), git")
|
||||||
|
w("CLAWHDF5_PYTHON=/path/to/venv/bin/python conformance/run.sh")
|
||||||
|
w("```")
|
||||||
|
w("")
|
||||||
|
w("The corpus (about 450 MB of sparse checkouts) is cached in `conformance/.cache/`; results for")
|
||||||
|
w("every file, both sides' raw JSON and stderr, are in `conformance/.cache/results/`.")
|
||||||
|
w("`conformance/baseline.json` holds the ok files the nightly CI job (`.gitea/workflows/conformance.yml`)")
|
||||||
|
w("must keep; `conformance/run.sh --update-baseline` rewrites it.")
|
||||||
|
|
||||||
|
with open(OUT_MD, "w") as fh:
|
||||||
|
fh.write("\n".join(L) + "\n")
|
||||||
@@ -0,0 +1,6 @@
|
|||||||
|
# The reference side of the conformance sweep. Pinned so the nightly job and a
|
||||||
|
# local run compare against the same libhdf5 (h5py wheels bundle it).
|
||||||
|
h5py==3.16.0
|
||||||
|
numpy==2.5.3
|
||||||
|
hdf5plugin==7.1.0
|
||||||
|
netCDF4==1.7.4
|
||||||
Executable
+88
@@ -0,0 +1,88 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# conformance/run.sh — the clawhdf5 conformance sweep, end to end.
|
||||||
|
#
|
||||||
|
# fetch the pinned corpora (cached) -> build the probe -> probe every file
|
||||||
|
# with clawhdf5 and with h5py (and h5dump for the CVE corpus), each under a
|
||||||
|
# timeout and a memory limit -> compare -> write CONFORMANCE.md -> check the
|
||||||
|
# result against conformance/baseline.json.
|
||||||
|
#
|
||||||
|
# Usage: conformance/run.sh [--no-fetch] [--no-report] [--update-baseline]
|
||||||
|
#
|
||||||
|
# Environment:
|
||||||
|
# CLAWHDF5_PYTHON python with h5py, numpy, hdf5plugin (default: repo .venv, then python3)
|
||||||
|
# CONFORMANCE_CACHE corpus / build / results cache (default: conformance/.cache)
|
||||||
|
# CONFORMANCE_OUT results directory (default: $CONFORMANCE_CACHE/results)
|
||||||
|
# CONFORMANCE_REPORT report path (default: CONFORMANCE.md at the repo root)
|
||||||
|
# JOBS parallel files (default: nproc)
|
||||||
|
# CONFORMANCE_PROBE use this prebuilt probe binary instead of building one
|
||||||
|
# TMO / MEM_KB per-process timeout in seconds (20) / address-space limit in KiB (4 GiB)
|
||||||
|
#
|
||||||
|
# Exit status: 0 = gate passed; 1 = a panic/hang/crash/oom in clawhdf5, or the
|
||||||
|
# ok count fell below the baseline, or a baseline-ok file regressed; 2 = setup error.
|
||||||
|
set -euo pipefail
|
||||||
|
HERE="$(cd "$(dirname "$0")" && pwd)"
|
||||||
|
ROOT="$(cd "$HERE/.." && pwd)"
|
||||||
|
FETCH=1 REPORT=1 UPDATE=0
|
||||||
|
for a in "$@"; do
|
||||||
|
case "$a" in
|
||||||
|
--no-fetch) FETCH=0 ;;
|
||||||
|
--no-report) REPORT=0 ;;
|
||||||
|
--update-baseline) UPDATE=1 ;;
|
||||||
|
-h|--help) sed -n '2,23p' "$0"; exit 0 ;;
|
||||||
|
*) echo "unknown argument: $a" >&2; exit 2 ;;
|
||||||
|
esac
|
||||||
|
done
|
||||||
|
|
||||||
|
export PATH="$HOME/.cargo/bin:$PATH"
|
||||||
|
CACHE="${CONFORMANCE_CACHE:-$HERE/.cache}"
|
||||||
|
mkdir -p "$CACHE"; CACHE="$(cd "$CACHE" && pwd)"
|
||||||
|
OUT="${CONFORMANCE_OUT:-$CACHE/results}"
|
||||||
|
REPORT_PATH="${CONFORMANCE_REPORT:-$ROOT/CONFORMANCE.md}"
|
||||||
|
JOBS="${JOBS:-$(nproc 2>/dev/null || echo 4)}"
|
||||||
|
if [ -n "${CLAWHDF5_PYTHON:-}" ]; then PY="$CLAWHDF5_PYTHON"
|
||||||
|
elif [ -x "$ROOT/.venv/bin/python" ]; then PY="$ROOT/.venv/bin/python"
|
||||||
|
else PY="$(command -v python3)"; fi
|
||||||
|
export PY TMO="${TMO:-20}" MEM_KB="${MEM_KB:-4194304}"
|
||||||
|
command -v h5dump >/dev/null || { echo "error: h5dump not found (install hdf5-tools)" >&2; exit 2; }
|
||||||
|
"$PY" -c 'import h5py, numpy, hdf5plugin' || { echo "error: $PY lacks h5py/numpy/hdf5plugin" >&2; exit 2; }
|
||||||
|
|
||||||
|
t0=$(date +%s)
|
||||||
|
[ "$FETCH" = 1 ] && bash "$HERE/fetch-corpus.sh" "$CACHE"
|
||||||
|
C="$CACHE/corpus"
|
||||||
|
[ -d "$C" ] || { echo "error: no corpus in $C (run without --no-fetch)" >&2; exit 2; }
|
||||||
|
|
||||||
|
if [ -n "${CONFORMANCE_PROBE:-}" ]; then
|
||||||
|
export PROBE="$CONFORMANCE_PROBE" # a prebuilt probe, e.g. an older one for a before/after
|
||||||
|
else
|
||||||
|
echo "== building the probe"
|
||||||
|
CARGO_TARGET_DIR="${CARGO_TARGET_DIR:-$CACHE/target}" \
|
||||||
|
cargo build -q --release --manifest-path "$HERE/probe/Cargo.toml"
|
||||||
|
export PROBE="${CARGO_TARGET_DIR:-$CACHE/target}/release/conformance-probe"
|
||||||
|
fi
|
||||||
|
t1=$(date +%s)
|
||||||
|
|
||||||
|
rm -rf "$OUT"; mkdir -p "$OUT"
|
||||||
|
"$PY" "$HERE/list_files.py" "$C" > "$OUT/files.txt"
|
||||||
|
echo "== probing $(wc -l <"$OUT/files.txt") files, $JOBS at a time (timeout ${TMO}s, limit $((MEM_KB / 1024)) MiB)"
|
||||||
|
export C OUT HERE
|
||||||
|
# The shell's "Segmentation fault (core dumped)" notices go to probe.log; the
|
||||||
|
# signals themselves are recorded in each side's .rc.
|
||||||
|
xargs -a "$OUT/files.txt" -d '\n' -P "$JOBS" -I{} bash -c '
|
||||||
|
f="$1"; d="$OUT/runs/${f//\//__}"
|
||||||
|
case "$f" in cve_hdf5/*) export WITH_H5DUMP=1 ;; esac
|
||||||
|
"$HERE/run_one.sh" "$C/$f" "$d"' _ {} 2>"$OUT/probe.log"
|
||||||
|
echo "== comparing"
|
||||||
|
"$PY" "$HERE/compare.py" "$OUT" >/dev/null
|
||||||
|
t2=$(date +%s)
|
||||||
|
cat > "$OUT/meta.json" <<EOF
|
||||||
|
{"build_seconds": $((t1 - t0)), "probe_seconds": $((t2 - t1)), "jobs": $JOBS, "timeout_s": $TMO, "mem_kb": $MEM_KB}
|
||||||
|
EOF
|
||||||
|
export CONFORMANCE_CMD="${CONFORMANCE_CMD:-conformance/run.sh${*:+ $*}}"
|
||||||
|
if [ "$REPORT" = 1 ]; then
|
||||||
|
"$PY" "$HERE/report.py" "$OUT" "$REPORT_PATH" "$C"
|
||||||
|
echo "== wrote $REPORT_PATH"
|
||||||
|
fi
|
||||||
|
if [ "$UPDATE" = 1 ]; then
|
||||||
|
"$PY" "$HERE/check.py" "$OUT" "$HERE/baseline.json" --update
|
||||||
|
fi
|
||||||
|
"$PY" "$HERE/check.py" "$OUT" "$HERE/baseline.json"
|
||||||
Executable
+27
@@ -0,0 +1,27 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# run_one.sh <file> <outdir>
|
||||||
|
#
|
||||||
|
# Probe one file with clawhdf5 (PROBE) and with h5py (PY ref.py), and with
|
||||||
|
# h5dump too when WITH_H5DUMP is set. Each side runs under a timeout (TMO
|
||||||
|
# seconds, SIGKILL) and an address-space limit (MEM_KB), with core dumps off.
|
||||||
|
# Writes <outdir>/<side>.{json,err,rc}; rc 137 = killed by the timeout.
|
||||||
|
set -u
|
||||||
|
f="$1"; out="$2"; mkdir -p "$out"
|
||||||
|
HERE="$(cd "$(dirname "$0")" && pwd)"
|
||||||
|
: "${PROBE:?PROBE must name the conformance-probe binary}"
|
||||||
|
: "${PY:?PY must name a python with h5py}"
|
||||||
|
TMO="${TMO:-20}"
|
||||||
|
MEM_KB="${MEM_KB:-4194304}"
|
||||||
|
run() { # name cmd...
|
||||||
|
local name=$1; shift
|
||||||
|
( ulimit -v "$MEM_KB"; ulimit -c 0; RUST_BACKTRACE=1 exec timeout -s KILL "$TMO" "$@" ) \
|
||||||
|
>"$out/$name.json" 2>"$out/$name.err"
|
||||||
|
echo $? >"$out/$name.rc"
|
||||||
|
}
|
||||||
|
run ours "$PROBE" "$f"
|
||||||
|
run ref "$PY" "$HERE/ref.py" "$f"
|
||||||
|
if [ -n "${WITH_H5DUMP:-}" ]; then
|
||||||
|
run h5dump h5dump "$f"
|
||||||
|
: >"$out/h5dump.json" # h5dump's text dump is not compared, only its exit status
|
||||||
|
fi
|
||||||
|
exit 0
|
||||||
@@ -1,7 +1,8 @@
|
|||||||
[package]
|
[package]
|
||||||
name = "clawhdf5-accel"
|
name = "clawhdf5-accel"
|
||||||
version = "2.5.0"
|
version = "2.7.0"
|
||||||
edition = "2024"
|
edition = "2024"
|
||||||
|
rust-version.workspace = true
|
||||||
description = "SIMD-accelerated operations for rustyhdf5"
|
description = "SIMD-accelerated operations for rustyhdf5"
|
||||||
license = "MIT"
|
license = "MIT"
|
||||||
repository = "https://git.redclaw.dev/quantumclaw/clawhdf5"
|
repository = "https://git.redclaw.dev/quantumclaw/clawhdf5"
|
||||||
|
|||||||
@@ -25,6 +25,55 @@ unsafe fn hsum_256(v: __m256) -> f32 {
|
|||||||
_mm_cvtss_f32(result)
|
_mm_cvtss_f32(result)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// AVX2 dot product of two `i8` slices, widened to `i32`.
|
||||||
|
///
|
||||||
|
/// Each 16-byte half is sign-extended to sixteen `i16` lanes and multiplied
|
||||||
|
/// pairwise with `madd_epi16`, which sums adjacent products straight into
|
||||||
|
/// eight `i32` lanes — the widening that an autovectorised scalar loop does
|
||||||
|
/// in several shuffles is one instruction here. A pair sum is at most
|
||||||
|
/// `2 * 127 * 127`, far inside `i32`.
|
||||||
|
///
|
||||||
|
/// # Safety
|
||||||
|
/// Caller must verify is_x86_feature_detected!("avx2").
|
||||||
|
// SAFETY: Caller must have verified AVX2 via is_x86_feature_detected!.
|
||||||
|
#[target_feature(enable = "avx2")]
|
||||||
|
pub unsafe fn dot_i8(a: &[i8], b: &[i8]) -> i32 {
|
||||||
|
// SAFETY: Caller guarantees AVX2 is available per the # Safety contract;
|
||||||
|
// every load reads 32 bytes at an index checked against `len` first.
|
||||||
|
unsafe {
|
||||||
|
assert_eq!(a.len(), b.len());
|
||||||
|
let len = a.len();
|
||||||
|
let mut i = 0;
|
||||||
|
let mut acc0 = _mm256_setzero_si256();
|
||||||
|
let mut acc1 = _mm256_setzero_si256();
|
||||||
|
|
||||||
|
while i + 32 <= len {
|
||||||
|
let va = _mm256_loadu_si256(a.as_ptr().add(i).cast());
|
||||||
|
let vb = _mm256_loadu_si256(b.as_ptr().add(i).cast());
|
||||||
|
let a_lo = _mm256_cvtepi8_epi16(_mm256_castsi256_si128(va));
|
||||||
|
let b_lo = _mm256_cvtepi8_epi16(_mm256_castsi256_si128(vb));
|
||||||
|
let a_hi = _mm256_cvtepi8_epi16(_mm256_extracti128_si256(va, 1));
|
||||||
|
let b_hi = _mm256_cvtepi8_epi16(_mm256_extracti128_si256(vb, 1));
|
||||||
|
acc0 = _mm256_add_epi32(acc0, _mm256_madd_epi16(a_lo, b_lo));
|
||||||
|
acc1 = _mm256_add_epi32(acc1, _mm256_madd_epi16(a_hi, b_hi));
|
||||||
|
i += 32;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Horizontal sum of the eight i32 lanes.
|
||||||
|
let v = _mm256_add_epi32(acc0, acc1);
|
||||||
|
let s128 = _mm_add_epi32(_mm256_castsi256_si128(v), _mm256_extracti128_si256(v, 1));
|
||||||
|
let s64 = _mm_add_epi32(s128, _mm_unpackhi_epi64(s128, s128));
|
||||||
|
let s32 = _mm_add_epi32(s64, _mm_shuffle_epi32(s64, 0b01));
|
||||||
|
let mut sum = _mm_cvtsi128_si32(s32);
|
||||||
|
|
||||||
|
while i < len {
|
||||||
|
sum += i32::from(a[i]) * i32::from(b[i]);
|
||||||
|
i += 1;
|
||||||
|
}
|
||||||
|
sum
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// AVX2 dot product for f32 slices.
|
/// AVX2 dot product for f32 slices.
|
||||||
///
|
///
|
||||||
/// # Safety
|
/// # Safety
|
||||||
|
|||||||
@@ -122,6 +122,36 @@ pub fn dot_product(a: &[f32], b: &[f32]) -> f32 {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Dot product of two `i8` slices, widened to `i32`.
|
||||||
|
///
|
||||||
|
/// The kernel behind int8-quantised vector search. On x86-64 it uses the AVX2
|
||||||
|
/// path whenever AVX2 is present (including on AVX-512 machines, where it is
|
||||||
|
/// what the f32 kernels use too on a default build). On aarch64 it uses the
|
||||||
|
/// ARMv8.2 `SDOT` instruction when the CPU has the dot-product extension, and
|
||||||
|
/// plain NEON otherwise.
|
||||||
|
pub fn dot_i8(a: &[i8], b: &[i8]) -> i32 {
|
||||||
|
match detect_backend() {
|
||||||
|
#[cfg(target_arch = "aarch64")]
|
||||||
|
Backend::Neon => {
|
||||||
|
if std::arch::is_aarch64_feature_detected!("dotprod") {
|
||||||
|
// SAFETY: the dotprod extension was just detected at runtime.
|
||||||
|
unsafe { neon::dot_i8_dotprod(a, b) }
|
||||||
|
} else {
|
||||||
|
// SAFETY: NEON is always available on aarch64.
|
||||||
|
unsafe { neon::dot_i8(a, b) }
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(target_arch = "x86_64")]
|
||||||
|
// SAFETY: both variants imply AVX2 was detected at runtime (the
|
||||||
|
// AVX-512 backend is only selected on CPUs that also have AVX2).
|
||||||
|
Backend::Avx2 | Backend::Avx512 if is_x86_feature_detected!("avx2") => unsafe {
|
||||||
|
avx2::dot_i8(a, b)
|
||||||
|
},
|
||||||
|
_ => scalar::dot_i8(a, b),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// Compute the L2 norm (magnitude) of a vector.
|
/// Compute the L2 norm (magnitude) of a vector.
|
||||||
pub fn vector_norm(v: &[f32]) -> f32 {
|
pub fn vector_norm(v: &[f32]) -> f32 {
|
||||||
dot_product(v, v).sqrt()
|
dot_product(v, v).sqrt()
|
||||||
@@ -207,7 +237,10 @@ pub fn f16_to_f32_batch(input: &[u16], output: &mut [f32]) {
|
|||||||
convert::f16_to_f32_batch(input, output);
|
convert::f16_to_f32_batch(input, output);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Compute Fletcher-32 checksum.
|
/// Compute a textbook Fletcher-32 checksum (both sums start at 0xffff).
|
||||||
|
///
|
||||||
|
/// This is not HDF5's checksum; the Fletcher-32 I/O filter uses
|
||||||
|
/// `clawhdf5_format::checksum::fletcher32`.
|
||||||
pub fn checksum_fletcher32(data: &[u8]) -> u32 {
|
pub fn checksum_fletcher32(data: &[u8]) -> u32 {
|
||||||
checksum::checksum_fletcher32(data)
|
checksum::checksum_fletcher32(data)
|
||||||
}
|
}
|
||||||
@@ -713,3 +746,78 @@ mod tests {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod dot_i8_tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
fn codes(n: usize, seed: u64) -> Vec<i8> {
|
||||||
|
let mut state = seed;
|
||||||
|
(0..n)
|
||||||
|
.map(|_| {
|
||||||
|
state = state
|
||||||
|
.wrapping_mul(6_364_136_223_846_793_005)
|
||||||
|
.wrapping_add(1_442_695_040_888_963_407);
|
||||||
|
// Full range, including the extremes.
|
||||||
|
((state >> 56) as u8) as i8
|
||||||
|
})
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn dispatched_kernel_matches_scalar_exactly() {
|
||||||
|
// Integer arithmetic: the SIMD path must agree bit for bit, at every
|
||||||
|
// length — including ones that are not multiples of the 32-byte block,
|
||||||
|
// which exercise the tail.
|
||||||
|
for len in [0, 1, 7, 31, 32, 33, 63, 64, 100, 384, 385, 1536] {
|
||||||
|
let a = codes(len, 1 + len as u64);
|
||||||
|
let b = codes(len, 1000 + len as u64);
|
||||||
|
assert_eq!(dot_i8(&a, &b), scalar::dot_i8(&a, &b), "len {len}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Dispatch only ever takes one path on a given CPU, so on a machine with
|
||||||
|
/// the dot-product extension the plain-NEON kernel would otherwise go
|
||||||
|
/// untested. Check each aarch64 kernel against scalar directly.
|
||||||
|
#[cfg(target_arch = "aarch64")]
|
||||||
|
#[test]
|
||||||
|
fn every_aarch64_kernel_matches_scalar_exactly() {
|
||||||
|
for len in [0, 1, 7, 15, 16, 17, 31, 32, 33, 63, 64, 100, 384, 385, 1536] {
|
||||||
|
let a = codes(len, 7 + len as u64);
|
||||||
|
let b = codes(len, 7000 + len as u64);
|
||||||
|
let want = scalar::dot_i8(&a, &b);
|
||||||
|
// SAFETY: NEON is always available on aarch64.
|
||||||
|
assert_eq!(unsafe { neon::dot_i8(&a, &b) }, want, "neon, len {len}");
|
||||||
|
if std::arch::is_aarch64_feature_detected!("dotprod") {
|
||||||
|
// SAFETY: the dotprod extension was just detected.
|
||||||
|
assert_eq!(
|
||||||
|
unsafe { neon::dot_i8_dotprod(&a, &b) },
|
||||||
|
want,
|
||||||
|
"dotprod, len {len}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// The extremes, through both kernels.
|
||||||
|
let lo = vec![-128i8; 4096];
|
||||||
|
let hi = vec![127i8; 4096];
|
||||||
|
// SAFETY: NEON is always available on aarch64.
|
||||||
|
assert_eq!(unsafe { neon::dot_i8(&lo, &lo) }, 4096 * 128 * 128);
|
||||||
|
// SAFETY: NEON is always available on aarch64.
|
||||||
|
assert_eq!(unsafe { neon::dot_i8(&lo, &hi) }, -4096 * 128 * 127);
|
||||||
|
if std::arch::is_aarch64_feature_detected!("dotprod") {
|
||||||
|
// SAFETY: the dotprod extension was just detected.
|
||||||
|
assert_eq!(unsafe { neon::dot_i8_dotprod(&lo, &lo) }, 4096 * 128 * 128);
|
||||||
|
// SAFETY: the dotprod extension was just detected.
|
||||||
|
assert_eq!(unsafe { neon::dot_i8_dotprod(&lo, &hi) }, -4096 * 128 * 127);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn extremes_do_not_overflow() {
|
||||||
|
// -128 * -128 is the largest product; a long run of it must still fit.
|
||||||
|
let a = vec![-128i8; 4096];
|
||||||
|
assert_eq!(dot_i8(&a, &a), 4096 * 128 * 128);
|
||||||
|
let b = vec![127i8; 4096];
|
||||||
|
assert_eq!(dot_i8(&a, &b), -4096 * 128 * 127);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -180,3 +180,130 @@ pub fn checksum_fletcher32(data: &[u8]) -> u32 {
|
|||||||
|
|
||||||
(sum2 << 16) | sum1
|
(sum2 << 16) | sum1
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// NEON dot product of two `i8` slices, widened to `i32`, for any aarch64 CPU.
|
||||||
|
///
|
||||||
|
/// `vmull_s8` multiplies eight lanes into `i16` — even `-128 * -128` is 16 384,
|
||||||
|
/// inside `i16` — and `vpadalq_s16` adds adjacent pairs of those into `i32`
|
||||||
|
/// accumulators, so nothing can overflow before the final horizontal sum.
|
||||||
|
///
|
||||||
|
/// CPUs with the ARMv8.2 dot-product extension should use
|
||||||
|
/// [`dot_i8_dotprod`], which does the multiply and the accumulate in one
|
||||||
|
/// instruction.
|
||||||
|
///
|
||||||
|
/// # Safety
|
||||||
|
/// Caller must ensure aarch64 target (NEON always available).
|
||||||
|
// SAFETY: NEON is always available on aarch64 targets; caller guarantees aarch64.
|
||||||
|
#[target_feature(enable = "neon")]
|
||||||
|
pub unsafe fn dot_i8(a: &[i8], b: &[i8]) -> i32 {
|
||||||
|
assert_eq!(a.len(), b.len());
|
||||||
|
let len = a.len();
|
||||||
|
let mut i = 0;
|
||||||
|
let mut acc0 = vdupq_n_s32(0);
|
||||||
|
let mut acc1 = vdupq_n_s32(0);
|
||||||
|
|
||||||
|
while i + 16 <= len {
|
||||||
|
// SAFETY: NEON is available per the # Safety contract, and both
|
||||||
|
// 16-byte loads start at an index checked against `len` above.
|
||||||
|
unsafe {
|
||||||
|
let va = vld1q_s8(a.as_ptr().add(i));
|
||||||
|
let vb = vld1q_s8(b.as_ptr().add(i));
|
||||||
|
acc0 = vpadalq_s16(acc0, vmull_s8(vget_low_s8(va), vget_low_s8(vb)));
|
||||||
|
acc1 = vpadalq_s16(acc1, vmull_high_s8(va, vb));
|
||||||
|
}
|
||||||
|
i += 16;
|
||||||
|
}
|
||||||
|
|
||||||
|
let mut sum = vaddvq_s32(vaddq_s32(acc0, acc1));
|
||||||
|
while i < len {
|
||||||
|
sum += i32::from(a[i]) * i32::from(b[i]);
|
||||||
|
i += 1;
|
||||||
|
}
|
||||||
|
sum
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One `SDOT`: for each of the four `i32` lanes of `acc`, add the dot
|
||||||
|
/// product of the corresponding four `i8` pairs from `a` and `b`.
|
||||||
|
///
|
||||||
|
/// Written as inline assembly because the `vdotq_s32` intrinsic is still
|
||||||
|
/// behind the unstable `stdarch_neon_dotprod` feature; inline assembly is
|
||||||
|
/// stable on aarch64.
|
||||||
|
///
|
||||||
|
/// # Safety
|
||||||
|
/// Caller must ensure the CPU supports the `dotprod` extension.
|
||||||
|
#[inline]
|
||||||
|
#[target_feature(enable = "neon,dotprod")]
|
||||||
|
unsafe fn sdot(acc: int32x4_t, a: int8x16_t, b: int8x16_t) -> int32x4_t {
|
||||||
|
let mut acc = acc;
|
||||||
|
// SAFETY: `dotprod` is enabled for this function and the caller
|
||||||
|
// guarantees the CPU supports it. The instruction reads only its three
|
||||||
|
// vector registers and touches no memory.
|
||||||
|
unsafe {
|
||||||
|
std::arch::asm!(
|
||||||
|
"sdot {acc:v}.4s, {a:v}.16b, {b:v}.16b",
|
||||||
|
acc = inout(vreg) acc,
|
||||||
|
a = in(vreg) a,
|
||||||
|
b = in(vreg) b,
|
||||||
|
options(pure, nomem, nostack),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
acc
|
||||||
|
}
|
||||||
|
|
||||||
|
/// NEON dot product of two `i8` slices using the ARMv8.2 dot-product
|
||||||
|
/// extension (`SDOT`): sixteen multiply-accumulates per instruction, straight
|
||||||
|
/// into `i32` lanes.
|
||||||
|
///
|
||||||
|
/// Present on the cores this crate actually runs on — Cortex-A76 and later
|
||||||
|
/// (Raspberry Pi 5, current Android phones), Neoverse-N1 (Graviton2, Ampere
|
||||||
|
/// Altra), and every Apple Silicon generation.
|
||||||
|
///
|
||||||
|
/// # Safety
|
||||||
|
/// Caller must verify `is_aarch64_feature_detected!("dotprod")`.
|
||||||
|
// SAFETY: caller has verified the dotprod extension at runtime.
|
||||||
|
#[target_feature(enable = "neon,dotprod")]
|
||||||
|
pub unsafe fn dot_i8_dotprod(a: &[i8], b: &[i8]) -> i32 {
|
||||||
|
assert_eq!(a.len(), b.len());
|
||||||
|
let len = a.len();
|
||||||
|
let mut i = 0;
|
||||||
|
let mut acc0 = vdupq_n_s32(0);
|
||||||
|
let mut acc1 = vdupq_n_s32(0);
|
||||||
|
|
||||||
|
// Two independent accumulators so consecutive SDOTs are not serialised on
|
||||||
|
// one register.
|
||||||
|
while i + 32 <= len {
|
||||||
|
// SAFETY: dotprod is available per the # Safety contract, and every
|
||||||
|
// 16-byte load starts at an index checked against `len` above.
|
||||||
|
unsafe {
|
||||||
|
acc0 = sdot(
|
||||||
|
acc0,
|
||||||
|
vld1q_s8(a.as_ptr().add(i)),
|
||||||
|
vld1q_s8(b.as_ptr().add(i)),
|
||||||
|
);
|
||||||
|
acc1 = sdot(
|
||||||
|
acc1,
|
||||||
|
vld1q_s8(a.as_ptr().add(i + 16)),
|
||||||
|
vld1q_s8(b.as_ptr().add(i + 16)),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
i += 32;
|
||||||
|
}
|
||||||
|
if i + 16 <= len {
|
||||||
|
// SAFETY: as above; the load is bounds-checked by this condition.
|
||||||
|
unsafe {
|
||||||
|
acc0 = sdot(
|
||||||
|
acc0,
|
||||||
|
vld1q_s8(a.as_ptr().add(i)),
|
||||||
|
vld1q_s8(b.as_ptr().add(i)),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
i += 16;
|
||||||
|
}
|
||||||
|
|
||||||
|
let mut sum = vaddvq_s32(vaddq_s32(acc0, acc1));
|
||||||
|
while i < len {
|
||||||
|
sum += i32::from(a[i]) * i32::from(b[i]);
|
||||||
|
i += 1;
|
||||||
|
}
|
||||||
|
sum
|
||||||
|
}
|
||||||
|
|||||||
@@ -140,3 +140,33 @@ fn f16_to_f32_soft(h: u16) -> f32 {
|
|||||||
|
|
||||||
f32::from_bits(f32_bits)
|
f32::from_bits(f32_bits)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Dot product of two `i8` slices, widened to `i32`.
|
||||||
|
///
|
||||||
|
/// `dim` terms of at most `127 * 127` fit an `i32` for any realistic
|
||||||
|
/// dimension (over 130 000 terms before overflow is possible).
|
||||||
|
pub fn dot_i8(a: &[i8], b: &[i8]) -> i32 {
|
||||||
|
assert_eq!(a.len(), b.len());
|
||||||
|
// Four independent accumulators over 32-lane blocks: the widening product
|
||||||
|
// has to sit in a fixed-length chunk for the vectoriser to see it, and the
|
||||||
|
// separate accumulators keep it off one dependency chain.
|
||||||
|
const LANE: usize = 8;
|
||||||
|
let (a_blocks, a_tail) = a.as_chunks::<{ LANE * 4 }>();
|
||||||
|
let (b_blocks, b_tail) = b.as_chunks::<{ LANE * 4 }>();
|
||||||
|
let mut acc = [0i32; 4];
|
||||||
|
for (x, y) in a_blocks.iter().zip(b_blocks) {
|
||||||
|
for (lane, slot) in acc.iter_mut().enumerate() {
|
||||||
|
let mut sum = 0i32;
|
||||||
|
for k in 0..LANE {
|
||||||
|
sum += i32::from(x[lane * LANE + k]) * i32::from(y[lane * LANE + k]);
|
||||||
|
}
|
||||||
|
*slot += sum;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let tail: i32 = a_tail
|
||||||
|
.iter()
|
||||||
|
.zip(b_tail)
|
||||||
|
.map(|(&x, &y)| i32::from(x) * i32::from(y))
|
||||||
|
.sum();
|
||||||
|
acc[0] + acc[1] + acc[2] + acc[3] + tail
|
||||||
|
}
|
||||||
|
|||||||
@@ -1,7 +1,8 @@
|
|||||||
[package]
|
[package]
|
||||||
name = "clawhdf5-agent"
|
name = "clawhdf5-agent"
|
||||||
version = "2.5.0"
|
version = "2.7.0"
|
||||||
edition = "2024"
|
edition = "2024"
|
||||||
|
rust-version.workspace = true
|
||||||
description = "HDF5-backed persistent memory store for on-device AI agents"
|
description = "HDF5-backed persistent memory store for on-device AI agents"
|
||||||
license = "MIT"
|
license = "MIT"
|
||||||
repository = "https://git.redclaw.dev/quantumclaw/clawhdf5"
|
repository = "https://git.redclaw.dev/quantumclaw/clawhdf5"
|
||||||
@@ -10,14 +11,18 @@ keywords = ["agent", "memory", "hdf5", "vector-search", "embedding"]
|
|||||||
categories = ["database", "science", "algorithms"]
|
categories = ["database", "science", "algorithms"]
|
||||||
|
|
||||||
[dependencies]
|
[dependencies]
|
||||||
clawhdf5-format = { path = "../clawhdf5-format", version = "2.5.0", features = ["parallel", "fast-checksum"] }
|
clawhdf5-format = { path = "../clawhdf5-format", version = "2.7.0", features = ["parallel", "fast-checksum"] }
|
||||||
clawhdf5 = { path = "../clawhdf5", version = "2.5.0" }
|
clawhdf5 = { path = "../clawhdf5", version = "2.7.0" }
|
||||||
clawhdf5-io = { path = "../clawhdf5-io", version = "2.5.0", features = ["mmap"] }
|
clawhdf5-io = { path = "../clawhdf5-io", version = "2.7.0", features = ["mmap"] }
|
||||||
clawhdf5-accel = { path = "../clawhdf5-accel", version = "2.5.0" }
|
clawhdf5-accel = { path = "../clawhdf5-accel", version = "2.7.0" }
|
||||||
clawhdf5-ann = { path = "../clawhdf5-ann", version = "2.5.0", optional = true }
|
clawhdf5-ann = { path = "../clawhdf5-ann", version = "2.7.0", optional = true }
|
||||||
clawhdf5-gpu = { path = "../clawhdf5-gpu", version = "2.5.0", optional = true, default-features = false }
|
clawhdf5-gpu = { path = "../clawhdf5-gpu", version = "2.7.0", optional = true, default-features = false }
|
||||||
serde = { workspace = true }
|
serde = { workspace = true }
|
||||||
byteorder = "1"
|
byteorder = "1"
|
||||||
|
# Signed checkpoints (MemoryConfig-independent; see `signing`). Pure Rust.
|
||||||
|
ed25519-dalek = { version = "2", features = ["rand_core"] }
|
||||||
|
sha2 = "0.10"
|
||||||
|
rand_core = { version = "0.6", features = ["getrandom"] }
|
||||||
half = { workspace = true, optional = true }
|
half = { workspace = true, optional = true }
|
||||||
rayon = { version = "1", optional = true }
|
rayon = { version = "1", optional = true }
|
||||||
matrixmultiply = { version = "0.3", optional = true }
|
matrixmultiply = { version = "0.3", optional = true }
|
||||||
@@ -44,6 +49,10 @@ harness = false
|
|||||||
name = "memory_bench"
|
name = "memory_bench"
|
||||||
harness = false
|
harness = false
|
||||||
|
|
||||||
|
[[bench]]
|
||||||
|
name = "multimodal_bench"
|
||||||
|
harness = false
|
||||||
|
|
||||||
[features]
|
[features]
|
||||||
default = ["float16", "hnsw", "parallel"]
|
default = ["float16", "hnsw", "parallel"]
|
||||||
float16 = ["half"]
|
float16 = ["half"]
|
||||||
@@ -59,7 +68,6 @@ zstd = ["clawhdf5/zstd"]
|
|||||||
# `--no-default-features` (plus re-enabling other defaults) to force the exact
|
# `--no-default-features` (plus re-enabling other defaults) to force the exact
|
||||||
# linear cosine scan.
|
# linear cosine scan.
|
||||||
hnsw = ["clawhdf5-ann"]
|
hnsw = ["clawhdf5-ann"]
|
||||||
agent = []
|
|
||||||
gpu = ["clawhdf5-gpu/gpu-wgpu"]
|
gpu = ["clawhdf5-gpu/gpu-wgpu"]
|
||||||
fast-math = ["matrixmultiply"]
|
fast-math = ["matrixmultiply"]
|
||||||
accelerate = ["accelerate-src", "cblas-sys"]
|
accelerate = ["accelerate-src", "cblas-sys"]
|
||||||
|
|||||||
@@ -0,0 +1,107 @@
|
|||||||
|
//! Multi-modal memory search benchmarks (`clawhdf5_agent::multimodal`).
|
||||||
|
//!
|
||||||
|
//! Covers `MultiModalStore::search_cross_modal` (every embedding of every
|
||||||
|
//! record, whatever its modality) and, for comparison,
|
||||||
|
//! `MultiModalStore::search_by_modality` restricted to one modality.
|
||||||
|
//!
|
||||||
|
//! Corpus: N records (1K and 10K), each carrying two 384-dim embeddings —
|
||||||
|
//! a text embedding of its caption plus one embedding of its primary modality,
|
||||||
|
//! cycling Image / Audio / Video — so a cross-modal query scores 2N vectors.
|
||||||
|
//! All data comes from a fixed-seed LCG, so every run sees the same corpus.
|
||||||
|
//!
|
||||||
|
//! Run: `cargo bench -p clawhdf5-agent --bench multimodal_bench`
|
||||||
|
|
||||||
|
use std::collections::HashMap;
|
||||||
|
|
||||||
|
use clawhdf5_agent::multimodal::{
|
||||||
|
MediaRef, ModalEmbedding, Modality, MultiModalRecord, MultiModalStore,
|
||||||
|
};
|
||||||
|
use criterion::{BenchmarkId, Criterion, criterion_group, criterion_main};
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Simple deterministic PRNG (LCG), same as the other agent benches
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
struct Rng(u32);
|
||||||
|
|
||||||
|
impl Rng {
|
||||||
|
fn new(seed: u32) -> Self {
|
||||||
|
Self(seed)
|
||||||
|
}
|
||||||
|
fn next_u32(&mut self) -> u32 {
|
||||||
|
self.0 = self.0.wrapping_mul(1103515245).wrapping_add(12345);
|
||||||
|
self.0 >> 16
|
||||||
|
}
|
||||||
|
fn next_f32(&mut self) -> f32 {
|
||||||
|
self.next_u32() as f32 / 65536.0 - 0.5
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn make_vec(rng: &mut Rng, dim: usize) -> Vec<f32> {
|
||||||
|
(0..dim).map(|_| rng.next_f32()).collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Corpus
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
const DIM: usize = 384;
|
||||||
|
const K: usize = 10;
|
||||||
|
|
||||||
|
const MEDIA: [(Modality, &str, &str); 3] = [
|
||||||
|
(Modality::Image, "image/png", "clip-vit-base"),
|
||||||
|
(Modality::Audio, "audio/wav", "clap-base"),
|
||||||
|
(Modality::Video, "video/mp4", "xclip-base"),
|
||||||
|
];
|
||||||
|
|
||||||
|
fn build_store(n: usize, seed: u32) -> MultiModalStore {
|
||||||
|
let mut rng = Rng::new(seed);
|
||||||
|
let mut store = MultiModalStore::new();
|
||||||
|
for i in 0..n {
|
||||||
|
let (modality, mime, model) = &MEDIA[i % MEDIA.len()];
|
||||||
|
let embeddings = vec![
|
||||||
|
ModalEmbedding::new(Modality::Text, make_vec(&mut rng, DIM), "minilm-l6"),
|
||||||
|
ModalEmbedding::new(modality.clone(), make_vec(&mut rng, DIM), *model),
|
||||||
|
];
|
||||||
|
store.add_record(MultiModalRecord {
|
||||||
|
id: 0,
|
||||||
|
primary_modality: modality.clone(),
|
||||||
|
text_content: Some(format!("{modality} memory {i}")),
|
||||||
|
media_ref: Some(MediaRef::path(format!("/media/{i}"), *mime)),
|
||||||
|
embeddings,
|
||||||
|
observation: None,
|
||||||
|
timestamp: 1_700_000_000.0 + i as f64,
|
||||||
|
metadata: HashMap::new(),
|
||||||
|
});
|
||||||
|
}
|
||||||
|
store
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Benchmarks
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
fn multimodal_search_benches(c: &mut Criterion) {
|
||||||
|
let query = make_vec(&mut Rng::new(99), DIM);
|
||||||
|
|
||||||
|
let mut group = c.benchmark_group("multimodal_search");
|
||||||
|
group.sample_size(50);
|
||||||
|
|
||||||
|
for (label, n) in [("1k", 1_000usize), ("10k", 10_000)] {
|
||||||
|
let store = build_store(n, 42);
|
||||||
|
assert_eq!(store.count(), n);
|
||||||
|
|
||||||
|
group.bench_with_input(BenchmarkId::new("cross_modal", label), &n, |b, _| {
|
||||||
|
b.iter(|| store.search_cross_modal(&query, K));
|
||||||
|
});
|
||||||
|
|
||||||
|
group.bench_with_input(BenchmarkId::new("by_modality_image", label), &n, |b, _| {
|
||||||
|
b.iter(|| store.search_by_modality(&Modality::Image, &query, K));
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
group.finish();
|
||||||
|
}
|
||||||
|
|
||||||
|
criterion_group!(multimodal_benches, multimodal_search_benches);
|
||||||
|
criterion_main!(multimodal_benches);
|
||||||
@@ -118,6 +118,10 @@ mod tests {
|
|||||||
created_at: "2025-01-01T00:00:00Z".to_string(),
|
created_at: "2025-01-01T00:00:00Z".to_string(),
|
||||||
wal_enabled: false,
|
wal_enabled: false,
|
||||||
wal_max_entries: 500,
|
wal_max_entries: 500,
|
||||||
|
quantized_index: false,
|
||||||
|
hnsw_m: 16,
|
||||||
|
hnsw_ef_construction: 64,
|
||||||
|
hnsw_ef_search: 0,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -88,8 +88,11 @@ impl BM25Index {
|
|||||||
/// Search the index for a query, returning the top `k` results
|
/// Search the index for a query, returning the top `k` results
|
||||||
/// as `(doc_id, score)` pairs sorted by score descending.
|
/// as `(doc_id, score)` pairs sorted by score descending.
|
||||||
///
|
///
|
||||||
/// Uses Block-Max WAND for early termination when remaining documents
|
/// Scores every matching document exhaustively, then keeps the top `k`.
|
||||||
/// cannot beat the current top-k threshold.
|
/// There is no early termination (WAND, MaxScore): the store's hot path
|
||||||
|
/// is [`scores`](Self::scores), because score fusion normalises over the
|
||||||
|
/// whole matching set and so needs every score, which no pruning scheme
|
||||||
|
/// can skip. This method is for BM25-only callers.
|
||||||
pub fn search(&self, query: &str, k: usize) -> Vec<(usize, f32)> {
|
pub fn search(&self, query: &str, k: usize) -> Vec<(usize, f32)> {
|
||||||
if k == 0 {
|
if k == 0 {
|
||||||
return Vec::new();
|
return Vec::new();
|
||||||
@@ -561,8 +564,9 @@ mod tests {
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn wand_returns_same_results_as_exhaustive() {
|
fn top_k_search_matches_ranking_every_score() {
|
||||||
// WAND-style search should produce same scores as exhaustive
|
// `search` must agree with ranking the full `scores` set — the
|
||||||
|
// bounded heap is an optimisation over sorting, not an approximation.
|
||||||
let docs: Vec<String> = (0..100)
|
let docs: Vec<String> = (0..100)
|
||||||
.map(|i| {
|
.map(|i| {
|
||||||
if i % 3 == 0 {
|
if i % 3 == 0 {
|
||||||
|
|||||||
@@ -1,17 +1,145 @@
|
|||||||
//! In-memory cache for memory entries, sessions, and knowledge graph.
|
//! In-memory cache for memory entries, sessions, and knowledge graph.
|
||||||
|
|
||||||
use crate::vector_search;
|
use crate::vector_search;
|
||||||
|
use clawhdf5_format::float16::round_to_f16;
|
||||||
|
|
||||||
|
/// Every entry's embedding, in one contiguous `[N x dim]` buffer.
|
||||||
|
///
|
||||||
|
/// Rows are always exactly `dim` long: a shorter one is zero-padded, a longer
|
||||||
|
/// one truncated. The previous `Vec<Vec<f32>>` allowed ragged rows, which
|
||||||
|
/// silently misaligned the flattened copy that the batched kernels read — a
|
||||||
|
/// single wrong-length embedding shifted every row after it. Padding makes
|
||||||
|
/// that unrepresentable. A record stored without an embedding therefore holds
|
||||||
|
/// a zero row, and is told apart by its norm being zero rather than by length.
|
||||||
|
///
|
||||||
|
/// This used to be two fields — a `Vec<Vec<f32>>` and a flattened copy kept in
|
||||||
|
/// lock-step — which stored the whole corpus twice and cost one heap
|
||||||
|
/// allocation per entry on top. At 100k 384-dim entries that duplicate was
|
||||||
|
/// ~150 MiB. Indexing yields a `&[f32]` row, so `embeddings[i]` still reads
|
||||||
|
/// the same way.
|
||||||
|
#[derive(Debug, Clone, Default)]
|
||||||
|
pub struct Embeddings {
|
||||||
|
flat: Vec<f32>,
|
||||||
|
dim: usize,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Embeddings {
|
||||||
|
pub fn new(dim: usize) -> Self {
|
||||||
|
Self {
|
||||||
|
flat: Vec::new(),
|
||||||
|
dim,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Number of embeddings.
|
||||||
|
pub fn len(&self) -> usize {
|
||||||
|
self.flat.len().checked_div(self.dim).unwrap_or(0)
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn is_empty(&self) -> bool {
|
||||||
|
self.len() == 0
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The whole buffer, `[N x dim]` row-major — what batched kernels read.
|
||||||
|
pub fn as_flat(&self) -> &[f32] {
|
||||||
|
&self.flat
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn dim(&self) -> usize {
|
||||||
|
self.dim
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Row `i`, or `None` if out of range.
|
||||||
|
pub fn get(&self, i: usize) -> Option<&[f32]> {
|
||||||
|
let start = i.checked_mul(self.dim)?;
|
||||||
|
self.flat.get(start..start.checked_add(self.dim)?)
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn iter(&self) -> impl ExactSizeIterator<Item = &[f32]> {
|
||||||
|
self.flat.chunks_exact(self.dim.max(1))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Append one embedding. A row whose length doesn't match `dim` is padded
|
||||||
|
/// or truncated, so the buffer stays rectangular whatever a caller passes.
|
||||||
|
pub fn push(&mut self, embedding: &[f32]) {
|
||||||
|
if self.dim == 0 {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
let take = embedding.len().min(self.dim);
|
||||||
|
self.flat.extend_from_slice(&embedding[..take]);
|
||||||
|
self.flat.resize(self.flat.len() + (self.dim - take), 0.0);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Replace row `i`. Out-of-range indices are ignored.
|
||||||
|
pub fn set(&mut self, i: usize, embedding: &[f32]) {
|
||||||
|
let Some(start) = i.checked_mul(self.dim) else {
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
if start + self.dim > self.flat.len() {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
let take = embedding.len().min(self.dim);
|
||||||
|
self.flat[start..start + take].copy_from_slice(&embedding[..take]);
|
||||||
|
self.flat[start + take..start + self.dim].fill(0.0);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Keep only the rows `keep` returns true for, preserving order.
|
||||||
|
pub fn retain(&mut self, mut keep: impl FnMut(usize) -> bool) {
|
||||||
|
if self.dim == 0 {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
let mut write = 0usize;
|
||||||
|
for read in 0..self.len() {
|
||||||
|
if keep(read) {
|
||||||
|
if write != read {
|
||||||
|
let (dst, src) = (write * self.dim, read * self.dim);
|
||||||
|
self.flat.copy_within(src..src + self.dim, dst);
|
||||||
|
}
|
||||||
|
write += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
self.flat.truncate(write * self.dim);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Replace the contents with `rows`.
|
||||||
|
pub fn reset_from(&mut self, dim: usize, rows: impl IntoIterator<Item = Vec<f32>>) {
|
||||||
|
self.dim = dim;
|
||||||
|
self.flat.clear();
|
||||||
|
for row in rows {
|
||||||
|
self.push(&row);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Adopt an already-flat buffer, trimming any partial trailing row.
|
||||||
|
pub fn set_flat(&mut self, dim: usize, mut flat: Vec<f32>) {
|
||||||
|
self.dim = dim;
|
||||||
|
match flat.len().checked_div(dim) {
|
||||||
|
Some(rows) => flat.truncate(rows * dim),
|
||||||
|
None => flat.clear(),
|
||||||
|
}
|
||||||
|
self.flat = flat;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl PartialEq for Embeddings {
|
||||||
|
fn eq(&self, other: &Self) -> bool {
|
||||||
|
self.dim == other.dim && self.flat == other.flat
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl std::ops::Index<usize> for Embeddings {
|
||||||
|
type Output = [f32];
|
||||||
|
|
||||||
|
fn index(&self, i: usize) -> &[f32] {
|
||||||
|
self.get(i).expect("embedding index out of range")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// In-memory cache for the /memory group data.
|
/// In-memory cache for the /memory group data.
|
||||||
#[derive(Debug, Clone)]
|
#[derive(Debug, Clone)]
|
||||||
pub struct MemoryCache {
|
pub struct MemoryCache {
|
||||||
pub chunks: Vec<String>,
|
pub chunks: Vec<String>,
|
||||||
pub embeddings: Vec<Vec<f32>>,
|
pub embeddings: Embeddings,
|
||||||
/// `embeddings` flattened into one contiguous `[N × embedding_dim]`
|
|
||||||
/// buffer, maintained incrementally alongside `embeddings` (push/update/
|
|
||||||
/// compact) so BLAS/Accelerate batch search can read it directly instead
|
|
||||||
/// of re-flattening the whole corpus on every query.
|
|
||||||
pub embeddings_flat: Vec<f32>,
|
|
||||||
pub source_channels: Vec<String>,
|
pub source_channels: Vec<String>,
|
||||||
pub timestamps: Vec<f64>,
|
pub timestamps: Vec<f64>,
|
||||||
pub session_ids: Vec<String>,
|
pub session_ids: Vec<String>,
|
||||||
@@ -22,14 +150,18 @@ pub struct MemoryCache {
|
|||||||
pub norms: Vec<f32>,
|
pub norms: Vec<f32>,
|
||||||
/// Hebbian activation weights (default 1.0 per entry).
|
/// Hebbian activation weights (default 1.0 per entry).
|
||||||
pub activation_weights: Vec<f32>,
|
pub activation_weights: Vec<f32>,
|
||||||
|
/// Round every embedding to IEEE half precision as it enters the cache,
|
||||||
|
/// so the cache holds exactly what a `float16` store writes to disk. Set
|
||||||
|
/// it with [`MemoryCache::set_half_precision`], which also rounds the
|
||||||
|
/// rows already held.
|
||||||
|
pub half_precision: bool,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl MemoryCache {
|
impl MemoryCache {
|
||||||
pub fn new(embedding_dim: usize) -> Self {
|
pub fn new(embedding_dim: usize) -> Self {
|
||||||
Self {
|
Self {
|
||||||
chunks: Vec::new(),
|
chunks: Vec::new(),
|
||||||
embeddings: Vec::new(),
|
embeddings: Embeddings::new(embedding_dim),
|
||||||
embeddings_flat: Vec::new(),
|
|
||||||
source_channels: Vec::new(),
|
source_channels: Vec::new(),
|
||||||
timestamps: Vec::new(),
|
timestamps: Vec::new(),
|
||||||
session_ids: Vec::new(),
|
session_ids: Vec::new(),
|
||||||
@@ -38,18 +170,52 @@ impl MemoryCache {
|
|||||||
embedding_dim,
|
embedding_dim,
|
||||||
norms: Vec::new(),
|
norms: Vec::new(),
|
||||||
activation_weights: Vec::new(),
|
activation_weights: Vec::new(),
|
||||||
|
half_precision: false,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Rebuild `embeddings_flat` from `embeddings` from scratch. Callers that
|
/// Switch half-precision rounding on or off. Turning it on rounds every
|
||||||
/// populate `embeddings` directly (bulk loads) must call this afterward.
|
/// embedding already held (and recomputes norms where one changed) —
|
||||||
pub fn rebuild_flat(&mut self) {
|
/// e.g. a `float16` store whose last checkpoint predates half-precision
|
||||||
self.embeddings_flat.clear();
|
/// storage and so is still `f32` on disk.
|
||||||
self.embeddings_flat
|
pub fn set_half_precision(&mut self, on: bool) {
|
||||||
.reserve(self.embeddings.len() * self.embedding_dim);
|
self.half_precision = on;
|
||||||
for emb in &self.embeddings {
|
if !on {
|
||||||
self.embeddings_flat.extend_from_slice(emb);
|
return;
|
||||||
}
|
}
|
||||||
|
for i in 0..self.embeddings.len() {
|
||||||
|
let row = &self.embeddings[i];
|
||||||
|
if row
|
||||||
|
.iter()
|
||||||
|
.all(|&v| round_to_f16(v).to_bits() == v.to_bits())
|
||||||
|
{
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let rounded: Vec<f32> = row.iter().map(|&v| round_to_f16(v)).collect();
|
||||||
|
self.norms[i] = vector_search::compute_norm(&rounded);
|
||||||
|
self.embeddings.set(i, &rounded);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The embedding as the cache will hold it: rounded to half precision
|
||||||
|
/// when [`Self::half_precision`] is on, otherwise unchanged.
|
||||||
|
fn stored_form(&self, mut embedding: Vec<f32>) -> Vec<f32> {
|
||||||
|
if self.half_precision {
|
||||||
|
for v in &mut embedding {
|
||||||
|
*v = round_to_f16(*v);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
embedding
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Kept for callers that used to have to re-flatten after a bulk load.
|
||||||
|
/// The buffer is always flat now, so there is nothing to rebuild.
|
||||||
|
#[deprecated(note = "embeddings are stored flat; this is a no-op")]
|
||||||
|
pub fn rebuild_flat(&mut self) {}
|
||||||
|
|
||||||
|
/// The embeddings as one contiguous `[N x dim]` buffer.
|
||||||
|
pub fn flat_embeddings(&self) -> &[f32] {
|
||||||
|
self.embeddings.as_flat()
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Total number of entries (including tombstoned).
|
/// Total number of entries (including tombstoned).
|
||||||
@@ -77,10 +243,10 @@ impl MemoryCache {
|
|||||||
tags: String,
|
tags: String,
|
||||||
) -> usize {
|
) -> usize {
|
||||||
let idx = self.chunks.len();
|
let idx = self.chunks.len();
|
||||||
|
let embedding = self.stored_form(embedding);
|
||||||
let norm = vector_search::compute_norm(&embedding);
|
let norm = vector_search::compute_norm(&embedding);
|
||||||
self.chunks.push(chunk);
|
self.chunks.push(chunk);
|
||||||
self.embeddings_flat.extend_from_slice(&embedding);
|
self.embeddings.push(&embedding);
|
||||||
self.embeddings.push(embedding);
|
|
||||||
self.source_channels.push(source_channel);
|
self.source_channels.push(source_channel);
|
||||||
self.timestamps.push(timestamp);
|
self.timestamps.push(timestamp);
|
||||||
self.session_ids.push(session_id);
|
self.session_ids.push(session_id);
|
||||||
@@ -116,22 +282,10 @@ impl MemoryCache {
|
|||||||
session_id: String,
|
session_id: String,
|
||||||
) {
|
) {
|
||||||
if idx < self.chunks.len() {
|
if idx < self.chunks.len() {
|
||||||
|
let embedding = self.stored_form(embedding);
|
||||||
let norm = vector_search::compute_norm(&embedding);
|
let norm = vector_search::compute_norm(&embedding);
|
||||||
self.chunks[idx] = chunk;
|
self.chunks[idx] = chunk;
|
||||||
let dim = self.embedding_dim;
|
self.embeddings.set(idx, &embedding);
|
||||||
let flat_start = idx * dim;
|
|
||||||
let matches_dim =
|
|
||||||
embedding.len() == dim && flat_start + dim <= self.embeddings_flat.len();
|
|
||||||
self.embeddings[idx] = embedding;
|
|
||||||
if matches_dim {
|
|
||||||
self.embeddings_flat[flat_start..flat_start + dim]
|
|
||||||
.copy_from_slice(&self.embeddings[idx]);
|
|
||||||
} else {
|
|
||||||
// Embedding length doesn't match embedding_dim (shouldn't
|
|
||||||
// happen in practice) — fall back to a full rebuild rather
|
|
||||||
// than leave embeddings_flat misaligned with embeddings.
|
|
||||||
self.rebuild_flat();
|
|
||||||
}
|
|
||||||
self.source_channels[idx] = source_channel;
|
self.source_channels[idx] = source_channel;
|
||||||
self.timestamps[idx] = timestamp;
|
self.timestamps[idx] = timestamp;
|
||||||
self.session_ids[idx] = session_id;
|
self.session_ids[idx] = session_id;
|
||||||
@@ -183,7 +337,7 @@ impl MemoryCache {
|
|||||||
new_idx += 1;
|
new_idx += 1;
|
||||||
let norm = vector_search::compute_norm(&self.embeddings[i]);
|
let norm = vector_search::compute_norm(&self.embeddings[i]);
|
||||||
new_chunks.push(self.chunks[i].clone());
|
new_chunks.push(self.chunks[i].clone());
|
||||||
new_embeddings.push(self.embeddings[i].clone());
|
new_embeddings.push(self.embeddings[i].to_vec());
|
||||||
new_source_channels.push(self.source_channels[i].clone());
|
new_source_channels.push(self.source_channels[i].clone());
|
||||||
new_timestamps.push(self.timestamps[i]);
|
new_timestamps.push(self.timestamps[i]);
|
||||||
new_session_ids.push(self.session_ids[i].clone());
|
new_session_ids.push(self.session_ids[i].clone());
|
||||||
@@ -196,7 +350,8 @@ impl MemoryCache {
|
|||||||
|
|
||||||
let removed = old_len - new_chunks.len();
|
let removed = old_len - new_chunks.len();
|
||||||
self.chunks = new_chunks;
|
self.chunks = new_chunks;
|
||||||
self.embeddings = new_embeddings;
|
self.embeddings
|
||||||
|
.reset_from(self.embedding_dim, new_embeddings);
|
||||||
self.source_channels = new_source_channels;
|
self.source_channels = new_source_channels;
|
||||||
self.timestamps = new_timestamps;
|
self.timestamps = new_timestamps;
|
||||||
self.session_ids = new_session_ids;
|
self.session_ids = new_session_ids;
|
||||||
@@ -204,16 +359,14 @@ impl MemoryCache {
|
|||||||
self.tombstones = new_tombstones;
|
self.tombstones = new_tombstones;
|
||||||
self.norms = new_norms;
|
self.norms = new_norms;
|
||||||
self.activation_weights = new_activation_weights;
|
self.activation_weights = new_activation_weights;
|
||||||
self.rebuild_flat();
|
|
||||||
|
|
||||||
(removed, index_map)
|
(removed, index_map)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Flatten all embeddings into a single Vec<f32> for HDF5 storage.
|
/// All embeddings as one owned `[N x dim]` buffer, for HDF5 storage.
|
||||||
/// `embeddings_flat` is already maintained incrementally, so this just
|
/// Prefer [`MemoryCache::flat_embeddings`] where a borrow will do.
|
||||||
/// clones it — kept as a method for callers that want an owned copy.
|
pub fn flat_embeddings_owned(&self) -> Vec<f32> {
|
||||||
pub fn flat_embeddings(&self) -> Vec<f32> {
|
self.embeddings.as_flat().to_vec()
|
||||||
self.embeddings_flat.clone()
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -224,7 +377,7 @@ mod tests {
|
|||||||
/// `embeddings_flat` must always equal a from-scratch flatten of `embeddings`.
|
/// `embeddings_flat` must always equal a from-scratch flatten of `embeddings`.
|
||||||
fn assert_flat_in_sync(cache: &MemoryCache) {
|
fn assert_flat_in_sync(cache: &MemoryCache) {
|
||||||
let expected: Vec<f32> = cache.embeddings.iter().flatten().copied().collect();
|
let expected: Vec<f32> = cache.embeddings.iter().flatten().copied().collect();
|
||||||
assert_eq!(cache.embeddings_flat, expected);
|
assert_eq!(cache.embeddings.as_flat(), expected);
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
@@ -247,7 +400,10 @@ mod tests {
|
|||||||
String::new(),
|
String::new(),
|
||||||
);
|
);
|
||||||
assert_flat_in_sync(&cache);
|
assert_flat_in_sync(&cache);
|
||||||
assert_eq!(cache.embeddings_flat, vec![1.0, 2.0, 3.0, 4.0, 5.0, 6.0]);
|
assert_eq!(
|
||||||
|
cache.embeddings.as_flat(),
|
||||||
|
vec![1.0, 2.0, 3.0, 4.0, 5.0, 6.0]
|
||||||
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
@@ -279,7 +435,7 @@ mod tests {
|
|||||||
);
|
);
|
||||||
assert_flat_in_sync(&cache);
|
assert_flat_in_sync(&cache);
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
cache.embeddings_flat,
|
cache.embeddings.as_flat(),
|
||||||
vec![7.0, 8.0, 9.0, 4.0, 5.0, 6.0],
|
vec![7.0, 8.0, 9.0, 4.0, 5.0, 6.0],
|
||||||
"update must overwrite the correct flat slice, not just append"
|
"update must overwrite the correct flat slice, not just append"
|
||||||
);
|
);
|
||||||
@@ -315,14 +471,71 @@ mod tests {
|
|||||||
cache.mark_deleted(1);
|
cache.mark_deleted(1);
|
||||||
cache.compact();
|
cache.compact();
|
||||||
assert_flat_in_sync(&cache);
|
assert_flat_in_sync(&cache);
|
||||||
assert_eq!(cache.embeddings_flat, vec![1.0, 1.0, 3.0, 3.0]);
|
assert_eq!(cache.embeddings.as_flat(), vec![1.0, 1.0, 3.0, 3.0]);
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn rebuild_flat_matches_manual_flatten() {
|
fn rebuild_flat_matches_manual_flatten() {
|
||||||
let mut cache = MemoryCache::new(2);
|
let mut cache = MemoryCache::new(2);
|
||||||
cache.embeddings = vec![vec![1.0, 2.0], vec![3.0, 4.0]];
|
cache
|
||||||
cache.rebuild_flat();
|
.embeddings
|
||||||
assert_eq!(cache.embeddings_flat, vec![1.0, 2.0, 3.0, 4.0]);
|
.reset_from(2, vec![vec![1.0, 2.0], vec![3.0, 4.0]]);
|
||||||
|
assert_eq!(cache.embeddings.as_flat(), vec![1.0, 2.0, 3.0, 4.0]);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn set_half_precision_rounds_existing_rows_and_their_norms() {
|
||||||
|
// A store with float16 set whose checkpoint is still f32 on disk
|
||||||
|
// loads full-precision rows; switching rounding on must bring them to
|
||||||
|
// exactly what the next checkpoint will write.
|
||||||
|
let mut cache = MemoryCache::new(3);
|
||||||
|
cache.push(
|
||||||
|
"a".into(),
|
||||||
|
vec![0.1, 0.2, 0.3],
|
||||||
|
"c".into(),
|
||||||
|
0.0,
|
||||||
|
"s".into(),
|
||||||
|
"".into(),
|
||||||
|
);
|
||||||
|
cache.push(
|
||||||
|
"b".into(),
|
||||||
|
vec![0.5, 0.25, 1.0],
|
||||||
|
"c".into(),
|
||||||
|
0.0,
|
||||||
|
"s".into(),
|
||||||
|
"".into(),
|
||||||
|
);
|
||||||
|
let exact_norm = cache.norms[0];
|
||||||
|
|
||||||
|
cache.set_half_precision(true);
|
||||||
|
let row0: Vec<f32> = [0.1f32, 0.2, 0.3]
|
||||||
|
.iter()
|
||||||
|
.map(|&v| round_to_f16(v))
|
||||||
|
.collect();
|
||||||
|
assert_eq!(&cache.embeddings[0], row0.as_slice());
|
||||||
|
assert_eq!(cache.norms[0], vector_search::compute_norm(&row0));
|
||||||
|
assert_ne!(cache.norms[0], exact_norm);
|
||||||
|
// Already representable: untouched.
|
||||||
|
assert_eq!(&cache.embeddings[1], &[0.5, 0.25, 1.0]);
|
||||||
|
|
||||||
|
// New rows are rounded as they arrive, and updates too.
|
||||||
|
cache.push(
|
||||||
|
"c".into(),
|
||||||
|
vec![0.1, 0.0, 0.0],
|
||||||
|
"c".into(),
|
||||||
|
0.0,
|
||||||
|
"s".into(),
|
||||||
|
"".into(),
|
||||||
|
);
|
||||||
|
assert_eq!(cache.embeddings[2][0], round_to_f16(0.1));
|
||||||
|
cache.update(
|
||||||
|
2,
|
||||||
|
"c".into(),
|
||||||
|
vec![0.3, 0.0, 0.0],
|
||||||
|
"c".into(),
|
||||||
|
0.0,
|
||||||
|
"s".into(),
|
||||||
|
);
|
||||||
|
assert_eq!(cache.embeddings[2][0], round_to_f16(0.3));
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -144,9 +144,44 @@ pub struct ConsolidationStats {
|
|||||||
|
|
||||||
pub struct ImportanceScorer;
|
pub struct ImportanceScorer;
|
||||||
|
|
||||||
|
/// Sum of squares, in 8-wide lanes so it vectorises.
|
||||||
|
fn sum_of_squares(a: &[f32]) -> f32 {
|
||||||
|
let (blocks, tail) = a.as_chunks::<8>();
|
||||||
|
let mut acc = [0.0f32; 8];
|
||||||
|
for b in blocks {
|
||||||
|
for i in 0..8 {
|
||||||
|
acc[i] += b[i] * b[i];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
acc.iter().sum::<f32>() + tail.iter().map(|x| x * x).sum::<f32>()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `(a · b, |b|²)` in one pass over equal-length slices, in 8-wide lanes.
|
||||||
|
fn dot_and_norm2(a: &[f32], b: &[f32]) -> (f32, f32) {
|
||||||
|
let (a_blocks, a_tail) = a.as_chunks::<8>();
|
||||||
|
let (b_blocks, b_tail) = b.as_chunks::<8>();
|
||||||
|
let mut dot = [0.0f32; 8];
|
||||||
|
let mut nb = [0.0f32; 8];
|
||||||
|
for (x, y) in a_blocks.iter().zip(b_blocks) {
|
||||||
|
for i in 0..8 {
|
||||||
|
dot[i] += x[i] * y[i];
|
||||||
|
nb[i] += y[i] * y[i];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let mut d = dot.iter().sum::<f32>();
|
||||||
|
let mut n = nb.iter().sum::<f32>();
|
||||||
|
for (x, y) in a_tail.iter().zip(b_tail) {
|
||||||
|
d += x * y;
|
||||||
|
n += y * y;
|
||||||
|
}
|
||||||
|
(d, n)
|
||||||
|
}
|
||||||
|
|
||||||
impl ImportanceScorer {
|
impl ImportanceScorer {
|
||||||
/// Cosine similarity between two embedding slices.
|
/// Cosine similarity between two embedding slices.
|
||||||
/// Returns 0.0 if either norm is zero.
|
/// Returns 0.0 if either norm is zero. The reference that
|
||||||
|
/// [`Self::score_surprise`] is tested against.
|
||||||
|
#[cfg(test)]
|
||||||
fn cosine_similarity(a: &[f32], b: &[f32]) -> f32 {
|
fn cosine_similarity(a: &[f32], b: &[f32]) -> f32 {
|
||||||
let len = a.len().min(b.len());
|
let len = a.len().min(b.len());
|
||||||
if len == 0 {
|
if len == 0 {
|
||||||
@@ -167,13 +202,54 @@ impl ImportanceScorer {
|
|||||||
|
|
||||||
/// Novelty score: 1.0 − max cosine similarity against all existing records.
|
/// Novelty score: 1.0 − max cosine similarity against all existing records.
|
||||||
/// Returns 1.0 when there are no existing memories.
|
/// Returns 1.0 when there are no existing memories.
|
||||||
|
///
|
||||||
|
/// Same result as the reference cosine similarity against each record, but the
|
||||||
|
/// new embedding's norm is computed once rather than per record, each
|
||||||
|
/// record costs one fused pass (dot product and its norm together) rather
|
||||||
|
/// than three, and a large working set is scored in parallel. Every insert
|
||||||
|
/// scores against the whole working tier, so this is what an unbounded
|
||||||
|
/// working tier pays for: at 100K records it was the difference between a
|
||||||
|
/// benchmark finishing and not (`BENCHMARKS.md`, "Consolidation Efficiency").
|
||||||
pub fn score_surprise(embedding: &[f32], existing_memories: &[&MemoryRecord]) -> f32 {
|
pub fn score_surprise(embedding: &[f32], existing_memories: &[&MemoryRecord]) -> f32 {
|
||||||
if existing_memories.is_empty() {
|
if existing_memories.is_empty() {
|
||||||
return 1.0;
|
return 1.0;
|
||||||
}
|
}
|
||||||
|
let query_norm2 = sum_of_squares(embedding);
|
||||||
|
let similarity = |r: &&MemoryRecord| -> f32 {
|
||||||
|
let other = &r.embedding;
|
||||||
|
let len = embedding.len().min(other.len());
|
||||||
|
if len == 0 {
|
||||||
|
return 0.0;
|
||||||
|
}
|
||||||
|
let (dot, other_norm2) = dot_and_norm2(&embedding[..len], &other[..len]);
|
||||||
|
// A shorter record compares against the query's matching prefix.
|
||||||
|
let q2 = if len == embedding.len() {
|
||||||
|
query_norm2
|
||||||
|
} else {
|
||||||
|
sum_of_squares(&embedding[..len])
|
||||||
|
};
|
||||||
|
if q2 == 0.0 || other_norm2 == 0.0 {
|
||||||
|
return 0.0;
|
||||||
|
}
|
||||||
|
dot / (q2.sqrt() * other_norm2.sqrt())
|
||||||
|
};
|
||||||
|
#[cfg(feature = "parallel")]
|
||||||
|
let max_sim = if existing_memories.len() >= 4096 {
|
||||||
|
use rayon::prelude::*;
|
||||||
|
existing_memories
|
||||||
|
.par_iter()
|
||||||
|
.map(similarity)
|
||||||
|
.reduce(|| f32::NEG_INFINITY, f32::max)
|
||||||
|
} else {
|
||||||
|
existing_memories
|
||||||
|
.iter()
|
||||||
|
.map(similarity)
|
||||||
|
.fold(f32::NEG_INFINITY, f32::max)
|
||||||
|
};
|
||||||
|
#[cfg(not(feature = "parallel"))]
|
||||||
let max_sim = existing_memories
|
let max_sim = existing_memories
|
||||||
.iter()
|
.iter()
|
||||||
.map(|r| Self::cosine_similarity(embedding, &r.embedding))
|
.map(similarity)
|
||||||
.fold(f32::NEG_INFINITY, f32::max);
|
.fold(f32::NEG_INFINITY, f32::max);
|
||||||
(1.0 - max_sim).clamp(0.0, 1.0)
|
(1.0 - max_sim).clamp(0.0, 1.0)
|
||||||
}
|
}
|
||||||
@@ -471,6 +547,54 @@ impl ConsolidationEngine {
|
|||||||
mod tests {
|
mod tests {
|
||||||
use super::*;
|
use super::*;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn score_surprise_matches_the_reference_cosine() {
|
||||||
|
let mut x = 0x2545_F491_4F6C_DD1Du64;
|
||||||
|
let mut next = || {
|
||||||
|
x ^= x << 13;
|
||||||
|
x ^= x >> 7;
|
||||||
|
x ^= x << 17;
|
||||||
|
(x >> 40) as f32 / (1u64 << 24) as f32 - 0.5
|
||||||
|
};
|
||||||
|
let make = |id: u64, v: Vec<f32>| MemoryRecord {
|
||||||
|
id,
|
||||||
|
chunk: String::new(),
|
||||||
|
embedding: v,
|
||||||
|
tier: MemoryTier::Working,
|
||||||
|
importance: 0.0,
|
||||||
|
access_count: 0,
|
||||||
|
last_accessed: 0.0,
|
||||||
|
created_at: 0.0,
|
||||||
|
source: MemorySource::User,
|
||||||
|
};
|
||||||
|
// Ordinary rows, a shorter one, an empty one and a zero vector; and
|
||||||
|
// enough rows to take the parallel path too.
|
||||||
|
for n in [5usize, 5000] {
|
||||||
|
let mut recs: Vec<MemoryRecord> = (0..n as u64)
|
||||||
|
.map(|i| make(i, (0..37).map(|_| next()).collect()))
|
||||||
|
.collect();
|
||||||
|
recs.push(make(9_000, (0..20).map(|_| next()).collect()));
|
||||||
|
recs.push(make(9_001, Vec::new()));
|
||||||
|
recs.push(make(9_002, vec![0.0; 37]));
|
||||||
|
let refs: Vec<&MemoryRecord> = recs.iter().collect();
|
||||||
|
for _ in 0..5 {
|
||||||
|
let q: Vec<f32> = (0..37).map(|_| next()).collect();
|
||||||
|
let expected = (1.0
|
||||||
|
- refs
|
||||||
|
.iter()
|
||||||
|
.map(|r| ImportanceScorer::cosine_similarity(&q, &r.embedding))
|
||||||
|
.fold(f32::NEG_INFINITY, f32::max))
|
||||||
|
.clamp(0.0, 1.0);
|
||||||
|
let got = ImportanceScorer::score_surprise(&q, &refs);
|
||||||
|
assert!((got - expected).abs() < 1e-5, "n={n}: {got} vs {expected}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
assert_eq!(
|
||||||
|
ImportanceScorer::score_surprise(&[0.0; 4], &[&make(1, vec![1.0; 4])]),
|
||||||
|
1.0
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
// Helper: build a simple normalised embedding of given dimension.
|
// Helper: build a simple normalised embedding of given dimension.
|
||||||
fn unit_vec(dim: usize, hot: usize) -> Vec<f32> {
|
fn unit_vec(dim: usize, hot: usize) -> Vec<f32> {
|
||||||
let mut v = vec![0.0f32; dim];
|
let mut v = vec![0.0f32; dim];
|
||||||
|
|||||||
@@ -28,7 +28,7 @@ use crate::vector_search;
|
|||||||
pub fn hybrid_search(
|
pub fn hybrid_search(
|
||||||
query_embedding: &[f32],
|
query_embedding: &[f32],
|
||||||
query_text: &str,
|
query_text: &str,
|
||||||
vectors: &[Vec<f32>],
|
vectors: &(impl crate::vector_search::VectorSet + Sync + ?Sized),
|
||||||
chunks: &[String],
|
chunks: &[String],
|
||||||
tombstones: &[u8],
|
tombstones: &[u8],
|
||||||
bm25_index: &BM25Index,
|
bm25_index: &BM25Index,
|
||||||
@@ -56,7 +56,7 @@ pub fn hybrid_search(
|
|||||||
pub fn hybrid_search_fused(
|
pub fn hybrid_search_fused(
|
||||||
query_embedding: &[f32],
|
query_embedding: &[f32],
|
||||||
query_text: &str,
|
query_text: &str,
|
||||||
vectors: &[Vec<f32>],
|
vectors: &(impl crate::vector_search::VectorSet + Sync + ?Sized),
|
||||||
_chunks: &[String],
|
_chunks: &[String],
|
||||||
tombstones: &[u8],
|
tombstones: &[u8],
|
||||||
bm25_index: &BM25Index,
|
bm25_index: &BM25Index,
|
||||||
@@ -65,31 +65,34 @@ pub fn hybrid_search_fused(
|
|||||||
) -> Vec<(usize, f32)> {
|
) -> Vec<(usize, f32)> {
|
||||||
// Get raw scores from both systems. Request all results so normalization
|
// Get raw scores from both systems. Request all results so normalization
|
||||||
// covers the full distribution.
|
// covers the full distribution.
|
||||||
// Use parallel search when rayon feature is enabled and vector count > 10K.
|
let vec_scores = exact_vector_scores(query_embedding, vectors, tombstones);
|
||||||
let vec_scores = {
|
|
||||||
#[cfg(feature = "parallel")]
|
|
||||||
{
|
|
||||||
if vectors.len() > 10_000 {
|
|
||||||
vector_search::parallel_cosine_batch(
|
|
||||||
query_embedding,
|
|
||||||
vectors,
|
|
||||||
tombstones,
|
|
||||||
vectors.len(),
|
|
||||||
)
|
|
||||||
} else {
|
|
||||||
vector_search::cosine_similarity_batch(query_embedding, vectors, tombstones)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
#[cfg(not(feature = "parallel"))]
|
|
||||||
{
|
|
||||||
vector_search::cosine_similarity_batch(query_embedding, vectors, tombstones)
|
|
||||||
}
|
|
||||||
};
|
|
||||||
let kw_scores = bm25_index.scores(query_text);
|
let kw_scores = bm25_index.scores(query_text);
|
||||||
|
|
||||||
fuse(vec_scores, kw_scores, fusion, k)
|
fuse(vec_scores, kw_scores, fusion, k)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Cosine similarity of `query_embedding` to every vector whose `skip` byte is
|
||||||
|
/// 0 (a tombstone, or any other exclusion mask). Parallel above 10K vectors
|
||||||
|
/// when the `parallel` feature is on.
|
||||||
|
pub fn exact_vector_scores(
|
||||||
|
query_embedding: &[f32],
|
||||||
|
vectors: &(impl crate::vector_search::VectorSet + Sync + ?Sized),
|
||||||
|
skip: &[u8],
|
||||||
|
) -> Vec<(usize, f32)> {
|
||||||
|
#[cfg(feature = "parallel")]
|
||||||
|
{
|
||||||
|
if vectors.count() > 10_000 {
|
||||||
|
return vector_search::parallel_cosine_batch(
|
||||||
|
query_embedding,
|
||||||
|
vectors,
|
||||||
|
skip,
|
||||||
|
vectors.count(),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
vector_search::cosine_similarity_batch(query_embedding, vectors, skip)
|
||||||
|
}
|
||||||
|
|
||||||
/// Merge pre-computed vector-similarity and keyword scores into a single ranking.
|
/// Merge pre-computed vector-similarity and keyword scores into a single ranking.
|
||||||
///
|
///
|
||||||
/// Both score sets are independently min-max normalized to [0, 1] and combined
|
/// Both score sets are independently min-max normalized to [0, 1] and combined
|
||||||
@@ -270,7 +273,7 @@ fn normalize_scores(scores: &[(usize, f32)]) -> Vec<(usize, f32)> {
|
|||||||
pub fn rrf_hybrid_search(
|
pub fn rrf_hybrid_search(
|
||||||
query_embedding: &[f32],
|
query_embedding: &[f32],
|
||||||
query_text: &str,
|
query_text: &str,
|
||||||
vectors: &[Vec<f32>],
|
vectors: &(impl crate::vector_search::VectorSet + Sync + ?Sized),
|
||||||
_chunks: &[String],
|
_chunks: &[String],
|
||||||
tombstones: &[u8],
|
tombstones: &[u8],
|
||||||
bm25_index: &BM25Index,
|
bm25_index: &BM25Index,
|
||||||
@@ -282,12 +285,12 @@ pub fn rrf_hybrid_search(
|
|||||||
let mut vec_scores = {
|
let mut vec_scores = {
|
||||||
#[cfg(feature = "parallel")]
|
#[cfg(feature = "parallel")]
|
||||||
{
|
{
|
||||||
if vectors.len() > 10_000 {
|
if vectors.count() > 10_000 {
|
||||||
vector_search::parallel_cosine_batch(
|
vector_search::parallel_cosine_batch(
|
||||||
query_embedding,
|
query_embedding,
|
||||||
vectors,
|
vectors,
|
||||||
tombstones,
|
tombstones,
|
||||||
vectors.len(),
|
vectors.count(),
|
||||||
)
|
)
|
||||||
} else {
|
} else {
|
||||||
vector_search::cosine_similarity_batch(query_embedding, vectors, tombstones)
|
vector_search::cosine_similarity_batch(query_embedding, vectors, tombstones)
|
||||||
@@ -298,7 +301,7 @@ pub fn rrf_hybrid_search(
|
|||||||
vector_search::cosine_similarity_batch(query_embedding, vectors, tombstones)
|
vector_search::cosine_similarity_batch(query_embedding, vectors, tombstones)
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
let mut kw_scores = bm25_index.search(query_text, vectors.len());
|
let mut kw_scores = bm25_index.search(query_text, vectors.count());
|
||||||
|
|
||||||
// Sort both lists descending so rank 1 = best.
|
// Sort both lists descending so rank 1 = best.
|
||||||
vec_scores.sort_by(|a, b| b.1.partial_cmp(&a.1).unwrap_or(std::cmp::Ordering::Equal));
|
vec_scores.sort_by(|a, b| b.1.partial_cmp(&a.1).unwrap_or(std::cmp::Ordering::Equal));
|
||||||
|
|||||||
@@ -163,12 +163,13 @@ fn levenshtein(a: &str, b: &str) -> usize {
|
|||||||
/// entities-slice-index map, and an entity-id -> relation-indices map (edges
|
/// entities-slice-index map, and an entity-id -> relation-indices map (edges
|
||||||
/// touching that entity as either source or target).
|
/// touching that entity as either source or target).
|
||||||
///
|
///
|
||||||
/// Built fresh per traversal call rather than cached on `KnowledgeCache`:
|
/// Cached on `KnowledgeCache` and checked against a fingerprint of the graph
|
||||||
/// entities/relations are plain `pub` `Vec`s that get pushed to directly
|
/// on every use ([`graph_fingerprint`]). entities/relations are plain `pub`
|
||||||
/// (e.g. `schema.rs`'s load path bypasses `add_entity`/`add_relation`), so a
|
/// `Vec`s that get changed directly (e.g. `schema.rs`'s load path bypasses
|
||||||
/// persistent index would need extra bookkeeping to avoid drifting stale. A
|
/// `add_entity`/`add_relation`), so the cache cannot rely on being told about
|
||||||
/// one-off O(V+E) build per call is still a large win over the O(V·E) (BFS)
|
/// changes; the fingerprint notices any of them. Rebuilding it on every
|
||||||
/// / O(steps·active·E) (spreading activation) scans it replaces.
|
/// traversal instead made a 2-hop BFS over 1K entities 6.5x slower than the
|
||||||
|
/// scan it replaced (24 -> 155 µs; `BENCHMARKS.md`, "Knowledge Graph").
|
||||||
struct AdjacencyIndex {
|
struct AdjacencyIndex {
|
||||||
entity_index: HashMap<u64, usize>,
|
entity_index: HashMap<u64, usize>,
|
||||||
by_entity: HashMap<u64, Vec<usize>>,
|
by_entity: HashMap<u64, Vec<usize>>,
|
||||||
@@ -204,6 +205,45 @@ impl AdjacencyIndex {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// A hash of everything [`AdjacencyIndex`] depends on — each entity's id and
|
||||||
|
/// position, each relation's endpoints and position. One linear pass, no
|
||||||
|
/// allocation: far cheaper than building the index, which hashes the same
|
||||||
|
/// values into two maps.
|
||||||
|
fn graph_fingerprint(entities: &[Entity], relations: &[Relation]) -> u64 {
|
||||||
|
// splitmix64-style mixing; order matters, so positions are covered.
|
||||||
|
fn mix(h: u64, v: u64) -> u64 {
|
||||||
|
let mut z = (h ^ v).wrapping_add(0x9E37_79B9_7F4A_7C15);
|
||||||
|
z = (z ^ (z >> 30)).wrapping_mul(0xBF58_476D_1CE4_E5B9);
|
||||||
|
z = (z ^ (z >> 27)).wrapping_mul(0x94D0_49BB_1331_11EB);
|
||||||
|
z ^ (z >> 31)
|
||||||
|
}
|
||||||
|
let mut h = mix(entities.len() as u64, relations.len() as u64);
|
||||||
|
for e in entities {
|
||||||
|
h = mix(h, e.id);
|
||||||
|
}
|
||||||
|
for r in relations {
|
||||||
|
h = mix(mix(h, r.src), r.tgt);
|
||||||
|
}
|
||||||
|
h
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The cached [`AdjacencyIndex`] and the fingerprint it was built for.
|
||||||
|
/// Cloning a `KnowledgeCache` starts the clone with an empty cache.
|
||||||
|
#[derive(Default)]
|
||||||
|
struct AdjacencyCache(std::sync::Mutex<Option<(u64, std::sync::Arc<AdjacencyIndex>)>>);
|
||||||
|
|
||||||
|
impl Clone for AdjacencyCache {
|
||||||
|
fn clone(&self) -> Self {
|
||||||
|
Self::default()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl std::fmt::Debug for AdjacencyCache {
|
||||||
|
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||||
|
f.write_str("AdjacencyCache")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// KnowledgeCache
|
// KnowledgeCache
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
@@ -216,6 +256,7 @@ pub struct KnowledgeCache {
|
|||||||
pub alias_strings: Vec<String>,
|
pub alias_strings: Vec<String>,
|
||||||
pub alias_entity_ids: Vec<i64>,
|
pub alias_entity_ids: Vec<i64>,
|
||||||
next_entity_id: u64,
|
next_entity_id: u64,
|
||||||
|
adjacency: AdjacencyCache,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl KnowledgeCache {
|
impl KnowledgeCache {
|
||||||
@@ -226,6 +267,7 @@ impl KnowledgeCache {
|
|||||||
alias_strings: Vec::new(),
|
alias_strings: Vec::new(),
|
||||||
alias_entity_ids: Vec::new(),
|
alias_entity_ids: Vec::new(),
|
||||||
next_entity_id: 0,
|
next_entity_id: 0,
|
||||||
|
adjacency: AdjacencyCache::default(),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -236,9 +278,29 @@ impl KnowledgeCache {
|
|||||||
alias_strings: Vec::new(),
|
alias_strings: Vec::new(),
|
||||||
alias_entity_ids: Vec::new(),
|
alias_entity_ids: Vec::new(),
|
||||||
next_entity_id: next_id,
|
next_entity_id: next_id,
|
||||||
|
adjacency: AdjacencyCache::default(),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The adjacency index for the graph as it is now: the cached one if the
|
||||||
|
/// graph's fingerprint still matches, otherwise rebuilt and cached.
|
||||||
|
fn adjacency_index(&self) -> std::sync::Arc<AdjacencyIndex> {
|
||||||
|
let fp = graph_fingerprint(&self.entities, &self.relations);
|
||||||
|
let mut slot = self
|
||||||
|
.adjacency
|
||||||
|
.0
|
||||||
|
.lock()
|
||||||
|
.unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||||
|
if let Some((cached_fp, idx)) = slot.as_ref()
|
||||||
|
&& *cached_fp == fp
|
||||||
|
{
|
||||||
|
return idx.clone();
|
||||||
|
}
|
||||||
|
let idx = std::sync::Arc::new(AdjacencyIndex::build(&self.entities, &self.relations));
|
||||||
|
*slot = Some((fp, idx.clone()));
|
||||||
|
idx
|
||||||
|
}
|
||||||
|
|
||||||
// -----------------------------------------------------------------------
|
// -----------------------------------------------------------------------
|
||||||
// Entity management
|
// Entity management
|
||||||
// -----------------------------------------------------------------------
|
// -----------------------------------------------------------------------
|
||||||
@@ -397,7 +459,7 @@ impl KnowledgeCache {
|
|||||||
/// together with their discovered depth. The seed entity itself is NOT
|
/// together with their discovered depth. The seed entity itself is NOT
|
||||||
/// included. Traversal follows both outgoing and incoming relation edges.
|
/// included. Traversal follows both outgoing and incoming relation edges.
|
||||||
pub fn bfs_neighbors(&self, entity_id: u64, max_depth: usize) -> Vec<(Entity, usize)> {
|
pub fn bfs_neighbors(&self, entity_id: u64, max_depth: usize) -> Vec<(Entity, usize)> {
|
||||||
let idx = AdjacencyIndex::build(&self.entities, &self.relations);
|
let idx = self.adjacency_index();
|
||||||
let mut visited: HashSet<u64> = HashSet::new();
|
let mut visited: HashSet<u64> = HashSet::new();
|
||||||
let mut queue: VecDeque<(u64, usize)> = VecDeque::new();
|
let mut queue: VecDeque<(u64, usize)> = VecDeque::new();
|
||||||
let mut results: Vec<(Entity, usize)> = Vec::new();
|
let mut results: Vec<(Entity, usize)> = Vec::new();
|
||||||
@@ -502,7 +564,7 @@ impl KnowledgeCache {
|
|||||||
min_activation: f32,
|
min_activation: f32,
|
||||||
max_steps: usize,
|
max_steps: usize,
|
||||||
) -> Vec<(u64, f32)> {
|
) -> Vec<(u64, f32)> {
|
||||||
let idx = AdjacencyIndex::build(&self.entities, &self.relations);
|
let idx = self.adjacency_index();
|
||||||
let mut activation: HashMap<u64, f32> = HashMap::new();
|
let mut activation: HashMap<u64, f32> = HashMap::new();
|
||||||
|
|
||||||
// Initialise seeds with activation 1.0.
|
// Initialise seeds with activation 1.0.
|
||||||
@@ -631,6 +693,51 @@ impl Default for KnowledgeCache {
|
|||||||
mod tests {
|
mod tests {
|
||||||
use super::*;
|
use super::*;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn cached_adjacency_sees_direct_changes_to_the_graph() {
|
||||||
|
// The index is cached across traversals, but entities/relations are
|
||||||
|
// pub Vecs anyone can edit; every kind of edit must be seen.
|
||||||
|
let mut kg = KnowledgeCache::new();
|
||||||
|
let a = kg.add_entity("a", "t", -1);
|
||||||
|
let b = kg.add_entity("b", "t", -1);
|
||||||
|
let c = kg.add_entity("c", "t", -1);
|
||||||
|
kg.add_relation(a, b, "r", 1.0);
|
||||||
|
let ids = |kg: &KnowledgeCache| -> Vec<u64> {
|
||||||
|
let mut v: Vec<u64> = kg.bfs_neighbors(a, 3).iter().map(|(e, _)| e.id).collect();
|
||||||
|
v.sort();
|
||||||
|
v
|
||||||
|
};
|
||||||
|
assert_eq!(ids(&kg), vec![b]);
|
||||||
|
assert_eq!(ids(&kg), vec![b], "cached index reused");
|
||||||
|
|
||||||
|
// Pushed directly, bypassing add_relation.
|
||||||
|
kg.relations.push(Relation {
|
||||||
|
src: b,
|
||||||
|
tgt: c,
|
||||||
|
..Relation::default()
|
||||||
|
});
|
||||||
|
assert_eq!(ids(&kg), vec![b, c]);
|
||||||
|
|
||||||
|
// Rewired in place: same lengths, different edge.
|
||||||
|
kg.relations[1].tgt = a;
|
||||||
|
assert_eq!(ids(&kg), vec![b]);
|
||||||
|
|
||||||
|
// Removed and replaced: same lengths again.
|
||||||
|
kg.relations.pop();
|
||||||
|
kg.relations.push(Relation {
|
||||||
|
src: a,
|
||||||
|
tgt: c,
|
||||||
|
..Relation::default()
|
||||||
|
});
|
||||||
|
assert_eq!(ids(&kg), vec![b, c]);
|
||||||
|
let act: Vec<u64> = kg
|
||||||
|
.spreading_activation(&[a], 0.5, 0.0, 2)
|
||||||
|
.iter()
|
||||||
|
.map(|(id, _)| *id)
|
||||||
|
.collect();
|
||||||
|
assert!(act.contains(&c));
|
||||||
|
}
|
||||||
|
|
||||||
// -----------------------------------------------------------------------
|
// -----------------------------------------------------------------------
|
||||||
// Original tests — must remain passing
|
// Original tests — must remain passing
|
||||||
// -----------------------------------------------------------------------
|
// -----------------------------------------------------------------------
|
||||||
|
|||||||
@@ -1,4 +1,4 @@
|
|||||||
//! ZeroClaw agent memory HDF5 backend.
|
//! Agent memory stored in a single HDF5 file.
|
||||||
//!
|
//!
|
||||||
//! Provides persistent memory storage for AI agents using HDF5 files.
|
//! Provides persistent memory storage for AI agents using HDF5 files.
|
||||||
//! All data is cached in-memory for fast access and flushed to disk
|
//! All data is cached in-memory for fast access and flushed to disk
|
||||||
@@ -36,6 +36,7 @@ pub mod reranker;
|
|||||||
pub mod schema;
|
pub mod schema;
|
||||||
pub mod search;
|
pub mod search;
|
||||||
pub mod session;
|
pub mod session;
|
||||||
|
pub mod signing;
|
||||||
pub mod storage;
|
pub mod storage;
|
||||||
mod store_lock;
|
mod store_lock;
|
||||||
pub mod temporal;
|
pub mod temporal;
|
||||||
@@ -62,26 +63,23 @@ use std::path::{Path, PathBuf};
|
|||||||
|
|
||||||
use cache::MemoryCache;
|
use cache::MemoryCache;
|
||||||
#[cfg(feature = "hnsw")]
|
#[cfg(feature = "hnsw")]
|
||||||
use clawhdf5_ann::{DistanceMetric, HnswIndex};
|
use clawhdf5_ann::{DistanceMetric, HnswIndex, Storage};
|
||||||
|
use clawhdf5_format::float16::round_to_f16;
|
||||||
use ephemeral::{EphemeralConfig, EphemeralStore};
|
use ephemeral::{EphemeralConfig, EphemeralStore};
|
||||||
|
|
||||||
/// HNSW construction parameters used for the agent's vector index. Cosine is the
|
|
||||||
/// agent's similarity metric, so the index is built with cosine distance.
|
|
||||||
#[cfg(feature = "hnsw")]
|
|
||||||
const HNSW_M: usize = 16;
|
|
||||||
#[cfg(feature = "hnsw")]
|
|
||||||
const HNSW_EF_CONSTRUCTION: usize = 64;
|
|
||||||
// EphemeralEntry and EphemeralStats are part of the crate public API via
|
// EphemeralEntry and EphemeralStats are part of the crate public API via
|
||||||
// the `ephemeral` module; they are not needed directly in lib.rs internals.
|
// the `ephemeral` module; they are not needed directly in lib.rs internals.
|
||||||
#[allow(unused_imports)]
|
#[allow(unused_imports)]
|
||||||
pub use ephemeral::{EphemeralEntry, EphemeralStats};
|
pub use ephemeral::{EphemeralEntry, EphemeralStats};
|
||||||
use knowledge::KnowledgeCache;
|
use knowledge::KnowledgeCache;
|
||||||
use memory_strategy::{Exchange, MemoryStrategy, StrategyOutput};
|
use memory_strategy::{Exchange, MemoryStrategy, StrategyOutput};
|
||||||
use session::SessionCache;
|
pub use search::SearchOptions;
|
||||||
|
pub use session::{SessionCache, SessionEntry};
|
||||||
|
|
||||||
// --- Error type ---
|
// --- Error type ---
|
||||||
|
|
||||||
#[derive(Debug)]
|
#[derive(Debug)]
|
||||||
|
#[non_exhaustive]
|
||||||
pub enum MemoryError {
|
pub enum MemoryError {
|
||||||
Io(std::io::Error),
|
Io(std::io::Error),
|
||||||
Hdf5(String),
|
Hdf5(String),
|
||||||
@@ -89,6 +87,14 @@ pub enum MemoryError {
|
|||||||
NotFound(String),
|
NotFound(String),
|
||||||
/// Another `HDF5Memory` (in this or another process) has the store open.
|
/// Another `HDF5Memory` (in this or another process) has the store open.
|
||||||
Locked(String),
|
Locked(String),
|
||||||
|
/// A record the store cannot hold as given, e.g. an embedding value
|
||||||
|
/// outside the half-precision range of a `float16` store.
|
||||||
|
InvalidEntry(String),
|
||||||
|
/// The store's checkpoints are signed and no signing key is set, so a
|
||||||
|
/// checkpoint would leave it unsigned. Set the key with
|
||||||
|
/// [`HDF5Memory::set_signing_key`], or drop the signature on purpose with
|
||||||
|
/// [`HDF5Memory::remove_signature`].
|
||||||
|
SigningKeyRequired(String),
|
||||||
}
|
}
|
||||||
|
|
||||||
impl std::fmt::Display for MemoryError {
|
impl std::fmt::Display for MemoryError {
|
||||||
@@ -99,6 +105,8 @@ impl std::fmt::Display for MemoryError {
|
|||||||
MemoryError::Schema(e) => write!(f, "schema error: {e}"),
|
MemoryError::Schema(e) => write!(f, "schema error: {e}"),
|
||||||
MemoryError::NotFound(e) => write!(f, "not found: {e}"),
|
MemoryError::NotFound(e) => write!(f, "not found: {e}"),
|
||||||
MemoryError::Locked(e) => write!(f, "store is locked: {e}"),
|
MemoryError::Locked(e) => write!(f, "store is locked: {e}"),
|
||||||
|
MemoryError::InvalidEntry(e) => write!(f, "invalid entry: {e}"),
|
||||||
|
MemoryError::SigningKeyRequired(e) => write!(f, "signing key required: {e}"),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -130,6 +138,19 @@ pub struct MemoryConfig {
|
|||||||
pub embedding_dim: usize,
|
pub embedding_dim: usize,
|
||||||
pub chunk_size: usize,
|
pub chunk_size: usize,
|
||||||
pub overlap: usize,
|
pub overlap: usize,
|
||||||
|
/// Store embeddings as IEEE half precision (numpy `float16`): half the
|
||||||
|
/// bytes of the embeddings dataset on disk. Every embedding is rounded to
|
||||||
|
/// the nearest half as it enters the store, in memory as well as on disk,
|
||||||
|
/// so search results are the same before and after a reopen. Values must
|
||||||
|
/// lie within ±65504; a save outside that is `MemoryError::InvalidEntry`.
|
||||||
|
/// Fixed when the store is created (persisted in `/meta`).
|
||||||
|
///
|
||||||
|
/// **On by default for new stores**: on the full LongMemEval haystack with
|
||||||
|
/// real MiniLM embeddings every retrieval metric matched `f32`, and at
|
||||||
|
/// 100K records the file is 48% smaller (`BENCHMARKS.md`). Existing
|
||||||
|
/// stores keep the setting they were created with. Set it to `false` for
|
||||||
|
/// full-precision embeddings, e.g. for unnormalised vectors that may
|
||||||
|
/// exceed the half-precision range.
|
||||||
pub float16: bool,
|
pub float16: bool,
|
||||||
pub compression: bool,
|
pub compression: bool,
|
||||||
pub compression_level: u32,
|
pub compression_level: u32,
|
||||||
@@ -139,6 +160,40 @@ pub struct MemoryConfig {
|
|||||||
pub created_at: String,
|
pub created_at: String,
|
||||||
pub wal_enabled: bool,
|
pub wal_enabled: bool,
|
||||||
pub wal_max_entries: usize,
|
pub wal_max_entries: usize,
|
||||||
|
/// Store the vector index's own copy of the embeddings as int8 rather than
|
||||||
|
/// f32, a quarter of the memory. **On by default** for new stores.
|
||||||
|
///
|
||||||
|
/// The index's copy is the single largest part of a loaded store's
|
||||||
|
/// footprint. Quantised distances are approximate, so the candidate pool
|
||||||
|
/// is re-scored against the cache's exact embeddings before fusion, which
|
||||||
|
/// holds recall at the f32 index's level. It is also faster, not slower:
|
||||||
|
/// at equal recall, 1.63x the queries per second on x86-64 (AVX2) and
|
||||||
|
/// 1.18x on a Raspberry Pi 5 (NEON `SDOT`), with builds 1.8x and 2.3x
|
||||||
|
/// faster. See `BENCHMARKS.md`.
|
||||||
|
///
|
||||||
|
/// Persisted with the store. Stores written before this setting existed
|
||||||
|
/// have no stored value and open as `false`, so reopening an old store
|
||||||
|
/// never changes how its index is held.
|
||||||
|
///
|
||||||
|
/// Has no effect without the `hnsw` feature.
|
||||||
|
pub quantized_index: bool,
|
||||||
|
/// HNSW graph degree. Higher means a denser graph: better recall, more
|
||||||
|
/// memory and slower builds. Clamped to at least 2 when the index is
|
||||||
|
/// built, since a graph with fewer connections is not one.
|
||||||
|
///
|
||||||
|
/// Has no effect without the `hnsw` feature.
|
||||||
|
pub hnsw_m: usize,
|
||||||
|
/// Candidate list size while building the HNSW graph. Higher means a
|
||||||
|
/// better graph and a slower build; it does not affect query cost.
|
||||||
|
///
|
||||||
|
/// Has no effect without the `hnsw` feature.
|
||||||
|
pub hnsw_ef_construction: usize,
|
||||||
|
/// Candidate list size for a query, trading throughput for recall. `0`
|
||||||
|
/// keeps the default, which scales with the requested `k`
|
||||||
|
/// (`max(k * 8, 64)`) so that fusion still sees a useful pool.
|
||||||
|
///
|
||||||
|
/// Has no effect without the `hnsw` feature.
|
||||||
|
pub hnsw_ef_search: usize,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl MemoryConfig {
|
impl MemoryConfig {
|
||||||
@@ -151,7 +206,7 @@ impl MemoryConfig {
|
|||||||
embedding_dim,
|
embedding_dim,
|
||||||
chunk_size: 512,
|
chunk_size: 512,
|
||||||
overlap: 50,
|
overlap: 50,
|
||||||
float16: false,
|
float16: true,
|
||||||
compression: false,
|
compression: false,
|
||||||
compression_level: 0,
|
compression_level: 0,
|
||||||
compact_threshold: 0.3,
|
compact_threshold: 0.3,
|
||||||
@@ -160,6 +215,10 @@ impl MemoryConfig {
|
|||||||
created_at,
|
created_at,
|
||||||
wal_enabled: true,
|
wal_enabled: true,
|
||||||
wal_max_entries: 500,
|
wal_max_entries: 500,
|
||||||
|
quantized_index: true,
|
||||||
|
hnsw_m: 16,
|
||||||
|
hnsw_ef_construction: 64,
|
||||||
|
hnsw_ef_search: 0,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -266,6 +325,12 @@ pub struct HDF5Memory {
|
|||||||
activations_dirty: bool,
|
activations_dirty: bool,
|
||||||
/// Opened with [`HDF5Memory::open_read_only`]: nothing may reach the disk.
|
/// Opened with [`HDF5Memory::open_read_only`]: nothing may reach the disk.
|
||||||
read_only: bool,
|
read_only: bool,
|
||||||
|
/// Key that signs every checkpoint; never persisted. See
|
||||||
|
/// [`HDF5Memory::set_signing_key`].
|
||||||
|
signing_key: Option<signing::SigningKey>,
|
||||||
|
/// Checkpoints of this store are signed: the file on disk is, or a key
|
||||||
|
/// has been set. A checkpoint without a key is then refused.
|
||||||
|
signed: bool,
|
||||||
/// A WAL that `open()` could not read and moved aside; see
|
/// A WAL that `open()` could not read and moved aside; see
|
||||||
/// [`HDF5Memory::quarantined_wal`].
|
/// [`HDF5Memory::quarantined_wal`].
|
||||||
quarantined_wal: Option<PathBuf>,
|
quarantined_wal: Option<PathBuf>,
|
||||||
@@ -285,7 +350,8 @@ impl HDF5Memory {
|
|||||||
/// Create a new HDF5 memory file with the given configuration.
|
/// Create a new HDF5 memory file with the given configuration.
|
||||||
pub fn create(config: MemoryConfig) -> Result<Self> {
|
pub fn create(config: MemoryConfig) -> Result<Self> {
|
||||||
let lock = store_lock::StoreLock::acquire(&config.path)?;
|
let lock = store_lock::StoreLock::acquire(&config.path)?;
|
||||||
let cache = MemoryCache::new(config.embedding_dim);
|
let mut cache = MemoryCache::new(config.embedding_dim);
|
||||||
|
cache.set_half_precision(config.float16);
|
||||||
let sessions = SessionCache::new();
|
let sessions = SessionCache::new();
|
||||||
let knowledge = KnowledgeCache::new();
|
let knowledge = KnowledgeCache::new();
|
||||||
|
|
||||||
@@ -320,6 +386,8 @@ impl HDF5Memory {
|
|||||||
bm25_filter: bm25::TokenFilter::default(),
|
bm25_filter: bm25::TokenFilter::default(),
|
||||||
activations_dirty: false,
|
activations_dirty: false,
|
||||||
read_only: false,
|
read_only: false,
|
||||||
|
signing_key: None,
|
||||||
|
signed: false,
|
||||||
quarantined_wal: None,
|
quarantined_wal: None,
|
||||||
_lock: Some(lock),
|
_lock: Some(lock),
|
||||||
})
|
})
|
||||||
@@ -444,7 +512,17 @@ impl HDF5Memory {
|
|||||||
|
|
||||||
#[cfg(feature = "hnsw")]
|
#[cfg(feature = "hnsw")]
|
||||||
let loaded_index = if replay_only_appended {
|
let loaded_index = if replay_only_appended {
|
||||||
Self::load_vector_index(path, checkpoint.ann_generation, &cache, n_checkpoint)
|
Self::load_vector_index(
|
||||||
|
path,
|
||||||
|
checkpoint.ann_generation,
|
||||||
|
&cache,
|
||||||
|
n_checkpoint,
|
||||||
|
if config.quantized_index {
|
||||||
|
Storage::Int8
|
||||||
|
} else {
|
||||||
|
Storage::Float32
|
||||||
|
},
|
||||||
|
)
|
||||||
} else {
|
} else {
|
||||||
None
|
None
|
||||||
};
|
};
|
||||||
@@ -488,6 +566,8 @@ impl HDF5Memory {
|
|||||||
bm25_filter: bm25::TokenFilter::default(),
|
bm25_filter: bm25::TokenFilter::default(),
|
||||||
activations_dirty: false,
|
activations_dirty: false,
|
||||||
read_only,
|
read_only,
|
||||||
|
signing_key: None,
|
||||||
|
signed: checkpoint.signed,
|
||||||
quarantined_wal,
|
quarantined_wal,
|
||||||
_lock: lock,
|
_lock: lock,
|
||||||
})
|
})
|
||||||
@@ -558,6 +638,7 @@ impl HDF5Memory {
|
|||||||
generation: Option<u64>,
|
generation: Option<u64>,
|
||||||
cache: &MemoryCache,
|
cache: &MemoryCache,
|
||||||
n_checkpoint: usize,
|
n_checkpoint: usize,
|
||||||
|
storage: Storage,
|
||||||
) -> Option<HnswIndex> {
|
) -> Option<HnswIndex> {
|
||||||
let generation = generation?;
|
let generation = generation?;
|
||||||
let bytes = std::fs::read(Self::vector_index_path(store)).ok()?;
|
let bytes = std::fs::read(Self::vector_index_path(store)).ok()?;
|
||||||
@@ -565,15 +646,17 @@ impl HDF5Memory {
|
|||||||
if u64::from_le_bytes(stamp.try_into().ok()?) != generation {
|
if u64::from_le_bytes(stamp.try_into().ok()?) != generation {
|
||||||
return None;
|
return None;
|
||||||
}
|
}
|
||||||
let vectors = cache.embeddings.get(..n_checkpoint)?.to_vec();
|
let vectors: Vec<Vec<f32>> = (0..n_checkpoint)
|
||||||
let mut index = HnswIndex::from_graph_bytes(graph, vectors).ok()?;
|
.map(|i| cache.embeddings.get(i).map(<[f32]>::to_vec))
|
||||||
|
.collect::<Option<_>>()?;
|
||||||
|
let mut index = HnswIndex::from_graph_bytes_with(graph, vectors, storage).ok()?;
|
||||||
if index.dimension() != cache.embedding_dim {
|
if index.dimension() != cache.embedding_dim {
|
||||||
return None;
|
return None;
|
||||||
}
|
}
|
||||||
// Records appended since (replayed from the WAL) join incrementally.
|
// Records appended since (replayed from the WAL) join incrementally.
|
||||||
for id in n_checkpoint..cache.embeddings.len() {
|
for id in n_checkpoint..cache.embeddings.len() {
|
||||||
if cache.embeddings[id].len() != index.dimension()
|
if cache.embeddings[id].len() != index.dimension()
|
||||||
|| index.insert(cache.embeddings[id].clone()) != id
|
|| index.insert(cache.embeddings[id].to_vec()) != id
|
||||||
{
|
{
|
||||||
return None;
|
return None;
|
||||||
}
|
}
|
||||||
@@ -645,6 +728,39 @@ impl HDF5Memory {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Sign every checkpoint from now on with `key` (Ed25519). The key is
|
||||||
|
/// never written anywhere; set it again after every `open`. Once a store
|
||||||
|
/// is signed, a checkpoint without the key is refused
|
||||||
|
/// ([`MemoryError::SigningKeyRequired`]) rather than silently leaving it
|
||||||
|
/// unsigned. Setting a different key re-signs the store under that key
|
||||||
|
/// from the next checkpoint; a verifier trusting the old key will then
|
||||||
|
/// reject it, which is the point. Call [`AgentMemory::flush_wal`] to sign
|
||||||
|
/// right away.
|
||||||
|
pub fn set_signing_key(&mut self, key: signing::SigningKey) {
|
||||||
|
self.signing_key = Some(key);
|
||||||
|
self.signed = true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Stop signing: the next checkpoint writes the store unsigned. The
|
||||||
|
/// deliberate way out of [`MemoryError::SigningKeyRequired`].
|
||||||
|
pub fn remove_signature(&mut self) {
|
||||||
|
self.signing_key = None;
|
||||||
|
self.signed = false;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Checkpoints of this store are signed (on disk, or from the next
|
||||||
|
/// checkpoint because a key has been set).
|
||||||
|
pub fn is_signed(&self) -> bool {
|
||||||
|
self.signed
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Check the checkpoint at `path` against the public key the caller
|
||||||
|
/// trusts; see [`signing::verify_store`]. Reads the file only: it works
|
||||||
|
/// on a store another process has open.
|
||||||
|
pub fn verify(path: &Path, trusted: &signing::VerifyingKey) -> Result<signing::VerifyReport> {
|
||||||
|
signing::verify_store(path, trusted)
|
||||||
|
}
|
||||||
|
|
||||||
/// Flush current state to disk and truncate the WAL.
|
/// Flush current state to disk and truncate the WAL.
|
||||||
///
|
///
|
||||||
/// Every code path that persists the full cache to the .h5 file must
|
/// Every code path that persists the full cache to the .h5 file must
|
||||||
@@ -660,10 +776,28 @@ impl HDF5Memory {
|
|||||||
// Record which WAL prefix this checkpoint contains, so a crash before
|
// Record which WAL prefix this checkpoint contains, so a crash before
|
||||||
// the truncate below can't replay those entries a second time.
|
// the truncate below can't replay those entries a second time.
|
||||||
let wal_applied = self.wal.as_ref().map(|w| w.mark());
|
let wal_applied = self.wal.as_ref().map(|w| w.mark());
|
||||||
|
let signature = match &self.signing_key {
|
||||||
|
Some(key) => Some(signing::sign(
|
||||||
|
key,
|
||||||
|
&self.config,
|
||||||
|
&self.cache,
|
||||||
|
&self.sessions,
|
||||||
|
&self.knowledge,
|
||||||
|
wal_applied,
|
||||||
|
)),
|
||||||
|
None if self.signed => {
|
||||||
|
return Err(MemoryError::SigningKeyRequired(format!(
|
||||||
|
"{} is signed; set its signing key before a checkpoint \
|
||||||
|
(saves so far are held in the WAL or in memory)",
|
||||||
|
self.config.path.display()
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
None => None,
|
||||||
|
};
|
||||||
// Written before the .h5 so a crash in between leaves a sidecar whose
|
// Written before the .h5 so a crash in between leaves a sidecar whose
|
||||||
// generation matches no checkpoint (ignored), never the reverse.
|
// generation matches no checkpoint (ignored), never the reverse.
|
||||||
let ann_generation = self.persist_vector_index();
|
let ann_generation = self.persist_vector_index();
|
||||||
storage::write_to_disk_with_meta(
|
storage::write_to_disk_signed(
|
||||||
&self.config.path,
|
&self.config.path,
|
||||||
&self.config,
|
&self.config,
|
||||||
&self.cache,
|
&self.cache,
|
||||||
@@ -672,7 +806,9 @@ impl HDF5Memory {
|
|||||||
&schema::CheckpointMeta {
|
&schema::CheckpointMeta {
|
||||||
wal_applied,
|
wal_applied,
|
||||||
ann_generation,
|
ann_generation,
|
||||||
|
signed: signature.is_some(),
|
||||||
},
|
},
|
||||||
|
signature.as_ref(),
|
||||||
)?;
|
)?;
|
||||||
if let Some(ref mut w) = self.wal {
|
if let Some(ref mut w) = self.wal {
|
||||||
w.truncate()?;
|
w.truncate()?;
|
||||||
@@ -801,6 +937,42 @@ impl HDF5Memory {
|
|||||||
// the index length drifts from the cache length (covering any mutation path
|
// the index length drifts from the cache length (covering any mutation path
|
||||||
// that doesn't call a hook, e.g. consolidation pushes).
|
// that doesn't call a hook, e.g. consolidation pushes).
|
||||||
|
|
||||||
|
/// Graph degree for the index, never below the 2 the builder requires:
|
||||||
|
/// a config value of 0 or 1 would otherwise panic inside `clawhdf5-ann`.
|
||||||
|
#[cfg(feature = "hnsw")]
|
||||||
|
fn hnsw_m(&self) -> usize {
|
||||||
|
self.config.hnsw_m.max(2)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Build-time candidate list size, never below the graph degree — a
|
||||||
|
/// smaller one cannot fill a node's connections.
|
||||||
|
#[cfg(feature = "hnsw")]
|
||||||
|
fn hnsw_ef_construction(&self) -> usize {
|
||||||
|
self.config.hnsw_ef_construction.max(self.hnsw_m())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Query-time candidate list size for a `k`-result search. `0` means the
|
||||||
|
/// default, which scales with `k`.
|
||||||
|
#[cfg(feature = "hnsw")]
|
||||||
|
pub(crate) fn hnsw_ef_search(&self, k: usize) -> usize {
|
||||||
|
let default = (k * 8).max(64);
|
||||||
|
if self.config.hnsw_ef_search == 0 {
|
||||||
|
default
|
||||||
|
} else {
|
||||||
|
self.config.hnsw_ef_search.max(k)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// How the index should store its copy of the vectors, per the config.
|
||||||
|
#[cfg(feature = "hnsw")]
|
||||||
|
fn index_storage(&self) -> Storage {
|
||||||
|
if self.config.quantized_index {
|
||||||
|
Storage::Int8
|
||||||
|
} else {
|
||||||
|
Storage::Float32
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// Build an HNSW index over the entire cache, re-applying tombstones as
|
/// Build an HNSW index over the entire cache, re-applying tombstones as
|
||||||
/// soft-deletions so node ids stay aligned with cache indices.
|
/// soft-deletions so node ids stay aligned with cache indices.
|
||||||
///
|
///
|
||||||
@@ -816,11 +988,15 @@ impl HDF5Memory {
|
|||||||
if self.cache.embeddings.iter().any(|e| e.len() != dim) {
|
if self.cache.embeddings.iter().any(|e| e.len() != dim) {
|
||||||
return None;
|
return None;
|
||||||
}
|
}
|
||||||
let mut index = HnswIndex::build_with_metric(
|
// The index owns its vectors, so it needs rows rather than the cache's
|
||||||
&self.cache.embeddings,
|
// flat buffer. This copy is the index's own; the cache keeps one.
|
||||||
HNSW_M,
|
let rows: Vec<Vec<f32>> = self.cache.embeddings.iter().map(<[f32]>::to_vec).collect();
|
||||||
HNSW_EF_CONSTRUCTION,
|
let mut index = HnswIndex::build_with(
|
||||||
|
&rows,
|
||||||
|
self.hnsw_m(),
|
||||||
|
self.hnsw_ef_construction(),
|
||||||
DistanceMetric::Cosine,
|
DistanceMetric::Cosine,
|
||||||
|
self.index_storage(),
|
||||||
);
|
);
|
||||||
for (i, &t) in self.cache.tombstones.iter().enumerate() {
|
for (i, &t) in self.cache.tombstones.iter().enumerate() {
|
||||||
if t != 0 {
|
if t != 0 {
|
||||||
@@ -846,7 +1022,7 @@ impl HDF5Memory {
|
|||||||
let dim = index.dimension();
|
let dim = index.dimension();
|
||||||
let appended = (self.hnsw_synced_len..n).all(|id| {
|
let appended = (self.hnsw_synced_len..n).all(|id| {
|
||||||
self.cache.embeddings[id].len() == dim
|
self.cache.embeddings[id].len() == dim
|
||||||
&& index.insert(self.cache.embeddings[id].clone()) == id
|
&& index.insert(self.cache.embeddings[id].to_vec()) == id
|
||||||
});
|
});
|
||||||
if appended {
|
if appended {
|
||||||
for id in self.hnsw_synced_len..n {
|
for id in self.hnsw_synced_len..n {
|
||||||
@@ -877,7 +1053,7 @@ impl HDF5Memory {
|
|||||||
let emb_len = self.cache.embeddings[idx].len();
|
let emb_len = self.cache.embeddings[idx].len();
|
||||||
match self.hnsw.as_mut() {
|
match self.hnsw.as_mut() {
|
||||||
Some(index) if emb_len == index.dimension() => {
|
Some(index) if emb_len == index.dimension() => {
|
||||||
let id = index.insert(self.cache.embeddings[idx].clone());
|
let id = index.insert(self.cache.embeddings[idx].to_vec());
|
||||||
if id == idx {
|
if id == idx {
|
||||||
self.hnsw_synced_len = self.cache.embeddings.len();
|
self.hnsw_synced_len = self.cache.embeddings.len();
|
||||||
} else {
|
} else {
|
||||||
@@ -923,6 +1099,18 @@ impl HDF5Memory {
|
|||||||
&self.config
|
&self.config
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The sessions recorded in this store.
|
||||||
|
pub fn sessions(&self) -> &SessionCache {
|
||||||
|
&self.sessions
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Mutable access to the sessions, e.g. to add many at once. Changes
|
||||||
|
/// reach the disk at the next checkpoint (any flushing call, such as
|
||||||
|
/// [`HDF5Memory::flush_wal`] or `save_batch`), not immediately.
|
||||||
|
pub fn sessions_mut(&mut self) -> &mut SessionCache {
|
||||||
|
&mut self.sessions
|
||||||
|
}
|
||||||
|
|
||||||
/// Get a reference to the knowledge cache.
|
/// Get a reference to the knowledge cache.
|
||||||
pub fn knowledge(&self) -> &KnowledgeCache {
|
pub fn knowledge(&self) -> &KnowledgeCache {
|
||||||
&self.knowledge
|
&self.knowledge
|
||||||
@@ -985,7 +1173,29 @@ impl HDF5Memory {
|
|||||||
/// Upsert: if an active entry with the same tags (key) exists, update it in-place.
|
/// Upsert: if an active entry with the same tags (key) exists, update it in-place.
|
||||||
/// Otherwise append a new entry. Use this for key-based memory stores where
|
/// Otherwise append a new entry. Use this for key-based memory stores where
|
||||||
/// the same key should not create duplicates.
|
/// the same key should not create duplicates.
|
||||||
|
/// A `float16` store holds embeddings as IEEE half precision, which has no
|
||||||
|
/// finite value beyond ±65504. Refuse such an embedding rather than
|
||||||
|
/// silently store infinity. (Values that are already infinite or NaN are
|
||||||
|
/// stored as they are, as in an `f32` store.)
|
||||||
|
fn check_embedding(&self, embedding: &[f32]) -> Result<()> {
|
||||||
|
if !self.config.float16 {
|
||||||
|
return Ok(());
|
||||||
|
}
|
||||||
|
let overflow = embedding
|
||||||
|
.iter()
|
||||||
|
.enumerate()
|
||||||
|
.find(|&(_, &v)| v.is_finite() && round_to_f16(v).is_infinite());
|
||||||
|
match overflow {
|
||||||
|
None => Ok(()),
|
||||||
|
Some((i, v)) => Err(MemoryError::InvalidEntry(format!(
|
||||||
|
"embedding[{i}] = {v} is outside the half-precision range (±65504) \
|
||||||
|
of this float16 store"
|
||||||
|
))),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
pub fn save_or_update(&mut self, entry: MemoryEntry) -> Result<usize> {
|
pub fn save_or_update(&mut self, entry: MemoryEntry) -> Result<usize> {
|
||||||
|
self.check_embedding(&entry.embedding)?;
|
||||||
if let Some(existing_idx) = self.cache.find_by_tags(&entry.tags) {
|
if let Some(existing_idx) = self.cache.find_by_tags(&entry.tags) {
|
||||||
if let Some(ref mut w) = self.wal {
|
if let Some(ref mut w) = self.wal {
|
||||||
let wal_entry = wal::WalEntry {
|
let wal_entry = wal::WalEntry {
|
||||||
@@ -1041,6 +1251,7 @@ impl HDF5Memory {
|
|||||||
|
|
||||||
impl AgentMemory for HDF5Memory {
|
impl AgentMemory for HDF5Memory {
|
||||||
fn save(&mut self, entry: MemoryEntry) -> Result<usize> {
|
fn save(&mut self, entry: MemoryEntry) -> Result<usize> {
|
||||||
|
self.check_embedding(&entry.embedding)?;
|
||||||
if let Some(ref mut w) = self.wal {
|
if let Some(ref mut w) = self.wal {
|
||||||
let wal_entry = wal::WalEntry {
|
let wal_entry = wal::WalEntry {
|
||||||
entry_type: wal::WalEntryType::Save,
|
entry_type: wal::WalEntryType::Save,
|
||||||
@@ -1082,6 +1293,10 @@ impl AgentMemory for HDF5Memory {
|
|||||||
}
|
}
|
||||||
|
|
||||||
fn save_batch(&mut self, entries: Vec<MemoryEntry>) -> Result<Vec<usize>> {
|
fn save_batch(&mut self, entries: Vec<MemoryEntry>) -> Result<Vec<usize>> {
|
||||||
|
// All or nothing: check every entry before storing any.
|
||||||
|
for entry in &entries {
|
||||||
|
self.check_embedding(&entry.embedding)?;
|
||||||
|
}
|
||||||
let mut indices = Vec::with_capacity(entries.len());
|
let mut indices = Vec::with_capacity(entries.len());
|
||||||
for entry in entries {
|
for entry in entries {
|
||||||
let idx = self.cache.push(
|
let idx = self.cache.push(
|
||||||
@@ -1242,6 +1457,9 @@ impl HDF5Memory {
|
|||||||
})?;
|
})?;
|
||||||
let view = memory_strategy::CacheStoreView::new(&self.cache, &self.knowledge);
|
let view = memory_strategy::CacheStoreView::new(&self.cache, &self.knowledge);
|
||||||
let output = strat.evaluate(&exchange, &view);
|
let output = strat.evaluate(&exchange, &view);
|
||||||
|
for e in &output.entries {
|
||||||
|
self.check_embedding(&e.embedding)?;
|
||||||
|
}
|
||||||
for e in &output.entries {
|
for e in &output.entries {
|
||||||
self.cache.push(
|
self.cache.push(
|
||||||
e.chunk.clone(),
|
e.chunk.clone(),
|
||||||
@@ -1266,6 +1484,35 @@ impl HDF5Memory {
|
|||||||
}
|
}
|
||||||
|
|
||||||
impl HDF5Memory {
|
impl HDF5Memory {
|
||||||
|
/// Delete many records with a single checkpoint, where
|
||||||
|
/// [`AgentMemory::delete`] checkpoints once per record.
|
||||||
|
///
|
||||||
|
/// All or nothing: if any id is out of range or already deleted (or
|
||||||
|
/// repeated), nothing is deleted and `MemoryError::NotFound` is returned.
|
||||||
|
/// Unlike `delete`, this never auto-compacts, so the records stay in the
|
||||||
|
/// store as tombstones (their indices unchanged) until [`AgentMemory::compact`]
|
||||||
|
/// is called — importers use it to carry over records that were already
|
||||||
|
/// deleted in the source.
|
||||||
|
pub fn delete_batch(&mut self, ids: &[usize]) -> Result<()> {
|
||||||
|
let mut seen = std::collections::HashSet::with_capacity(ids.len());
|
||||||
|
for &id in ids {
|
||||||
|
if self.cache.tombstones.get(id).copied() != Some(0) || !seen.insert(id) {
|
||||||
|
return Err(MemoryError::NotFound(format!(
|
||||||
|
"entry {id} not found or already deleted"
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if ids.is_empty() {
|
||||||
|
return Ok(());
|
||||||
|
}
|
||||||
|
for &id in ids {
|
||||||
|
self.cache.mark_deleted(id);
|
||||||
|
self.hnsw_on_delete(id);
|
||||||
|
self.bm25_on_delete(id);
|
||||||
|
}
|
||||||
|
self.flush()
|
||||||
|
}
|
||||||
|
|
||||||
pub fn tick_session(&mut self) -> Result<()> {
|
pub fn tick_session(&mut self) -> Result<()> {
|
||||||
let d = self.config.decay_factor;
|
let d = self.config.decay_factor;
|
||||||
for w in self.cache.activation_weights.iter_mut() {
|
for w in self.cache.activation_weights.iter_mut() {
|
||||||
@@ -1326,6 +1573,16 @@ impl HDF5Memory {
|
|||||||
let mut promoted = 0;
|
let mut promoted = 0;
|
||||||
|
|
||||||
for key in candidates {
|
for key in candidates {
|
||||||
|
// Check before taking, so a rejected entry stays in the ephemeral
|
||||||
|
// tier rather than being lost.
|
||||||
|
if let Some(emb) = self
|
||||||
|
.ephemeral
|
||||||
|
.as_ref()
|
||||||
|
.and_then(|s| s.get_entry(&key))
|
||||||
|
.and_then(|e| e.embedding.as_deref())
|
||||||
|
{
|
||||||
|
self.check_embedding(emb)?;
|
||||||
|
}
|
||||||
let entry = match self
|
let entry = match self
|
||||||
.ephemeral
|
.ephemeral
|
||||||
.as_mut()
|
.as_mut()
|
||||||
@@ -1454,6 +1711,79 @@ mod tests {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn delete_batch_tombstones_without_compacting() {
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let path = dir.path().join("test.h5");
|
||||||
|
let mut mem = HDF5Memory::create(make_config(&dir)).unwrap();
|
||||||
|
mem.save_batch(
|
||||||
|
(0..4)
|
||||||
|
.map(|i| make_entry(&format!("record {i}"), &[i as f32, 1.0, 0.0, 0.0]))
|
||||||
|
.collect(),
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
// 3 of 4 is far past compact_threshold (0.3): delete() would compact.
|
||||||
|
mem.delete_batch(&[0, 1, 3]).unwrap();
|
||||||
|
assert_eq!(mem.count(), 4);
|
||||||
|
assert_eq!(mem.count_active(), 1);
|
||||||
|
drop(mem);
|
||||||
|
|
||||||
|
let mut mem = HDF5Memory::open(&path).unwrap();
|
||||||
|
assert_eq!(mem.cache.tombstones, vec![1, 1, 0, 1]);
|
||||||
|
let hits = mem.hybrid_search(&[0.0, 1.0, 0.0, 0.0], "record", 0.5, 0.5, 10);
|
||||||
|
assert!(
|
||||||
|
hits.iter().all(|r| r.index == 2),
|
||||||
|
"tombstoned record returned"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn delete_batch_is_all_or_nothing() {
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let mut mem = HDF5Memory::create(make_config(&dir)).unwrap();
|
||||||
|
mem.save_batch(vec![
|
||||||
|
make_entry("a", &[1.0, 0.0, 0.0, 0.0]),
|
||||||
|
make_entry("b", &[0.0, 1.0, 0.0, 0.0]),
|
||||||
|
])
|
||||||
|
.unwrap();
|
||||||
|
for bad in [&[0, 5][..], &[1, 1][..]] {
|
||||||
|
assert!(matches!(
|
||||||
|
mem.delete_batch(bad),
|
||||||
|
Err(MemoryError::NotFound(_))
|
||||||
|
));
|
||||||
|
assert_eq!(mem.count_active(), 2, "{bad:?} deleted something");
|
||||||
|
}
|
||||||
|
mem.delete_batch(&[]).unwrap();
|
||||||
|
assert_eq!(mem.count_active(), 2);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn sessions_mut_add_at_keeps_timestamp_across_reopen() {
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let path = dir.path().join("test.h5");
|
||||||
|
let mut mem = HDF5Memory::create(make_config(&dir)).unwrap();
|
||||||
|
mem.sessions_mut()
|
||||||
|
.add_at("s-old", 2, 7, "discord", "old summary", 1.7e15);
|
||||||
|
mem.flush_wal().unwrap();
|
||||||
|
drop(mem);
|
||||||
|
|
||||||
|
let mem = HDF5Memory::open_read_only(&path).unwrap();
|
||||||
|
let s = mem.sessions();
|
||||||
|
assert_eq!(s.len(), 1);
|
||||||
|
let e = &s.entries[0];
|
||||||
|
assert_eq!(
|
||||||
|
(
|
||||||
|
e.id.as_str(),
|
||||||
|
e.start_idx,
|
||||||
|
e.end_idx,
|
||||||
|
e.channel.as_str(),
|
||||||
|
e.ts
|
||||||
|
),
|
||||||
|
("s-old", 2, 7, "discord", 1.7e15)
|
||||||
|
);
|
||||||
|
assert_eq!(s.summaries[0], "old summary");
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn create_new_file() {
|
fn create_new_file() {
|
||||||
let dir = TempDir::new().unwrap();
|
let dir = TempDir::new().unwrap();
|
||||||
|
|||||||
@@ -1,10 +1,12 @@
|
|||||||
//! OpenClaw Integration Layer.
|
//! A Markdown-oriented memory backend over [`crate::HDF5Memory`].
|
||||||
//!
|
//!
|
||||||
//! Bridge between OpenClaw agent gateway (Markdown + sqlite-vec) and the
|
//! Named for OpenClaw, whose workspace memory is Markdown, but **not an
|
||||||
//! clawhdf5 HDF5-backed memory backend. Provides:
|
//! OpenClaw plugin**: nothing here registers with OpenClaw, and the
|
||||||
|
//! integration it was written for never worked (see `docs/openclaw.md`).
|
||||||
|
//! Provides:
|
||||||
//!
|
//!
|
||||||
//! - [`MemoryBackend`] — the trait OpenClaw implements against.
|
//! - [`MemoryBackend`] — search / read back / write / ingest / export.
|
||||||
//! - [`ClawhdfBackend`] — concrete HDF5-backed implementation.
|
//! - [`ClawhdfBackend`] — the HDF5-backed implementation.
|
||||||
//! - [`MarkdownParser`] — splits Markdown into [`MarkdownSection`] records.
|
//! - [`MarkdownParser`] — splits Markdown into [`MarkdownSection`] records.
|
||||||
//! - [`MarkdownExporter`] — renders sections back to Markdown text.
|
//! - [`MarkdownExporter`] — renders sections back to Markdown text.
|
||||||
|
|
||||||
@@ -13,9 +15,8 @@ use std::path::{Path, PathBuf};
|
|||||||
use std::time::{SystemTime, UNIX_EPOCH};
|
use std::time::{SystemTime, UNIX_EPOCH};
|
||||||
|
|
||||||
use crate::{
|
use crate::{
|
||||||
AgentMemory, HDF5Memory, MemoryConfig, MemoryEntry,
|
AgentMemory, HDF5Memory, MemoryConfig, MemoryEntry, SearchOptions,
|
||||||
confidence::{ConfidenceConfig, ScoredResult, reject_low_confidence},
|
confidence::ConfidenceConfig, reranker::ReRankConfig,
|
||||||
reranker::{ReRankConfig, RerankInput, rerank},
|
|
||||||
};
|
};
|
||||||
|
|
||||||
// ─────────────────────────────────────────────────────────────────────────────
|
// ─────────────────────────────────────────────────────────────────────────────
|
||||||
@@ -62,7 +63,8 @@ pub struct BackendStats {
|
|||||||
// MemoryBackend trait
|
// MemoryBackend trait
|
||||||
// ─────────────────────────────────────────────────────────────────────────────
|
// ─────────────────────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
/// Interface that OpenClaw uses to interact with a memory backend.
|
/// A Markdown-oriented memory backend: search, read back by path, write,
|
||||||
|
/// ingest and export.
|
||||||
///
|
///
|
||||||
/// Implementors provide persistent storage, full-text + vector search,
|
/// Implementors provide persistent storage, full-text + vector search,
|
||||||
/// Markdown ingestion / export, and statistics.
|
/// Markdown ingestion / export, and statistics.
|
||||||
@@ -319,7 +321,7 @@ impl MarkdownExporter {
|
|||||||
///
|
///
|
||||||
/// # Path mapping
|
/// # Path mapping
|
||||||
///
|
///
|
||||||
/// OpenClaw addresses memories by file path (e.g. `"memory/user.md"`).
|
/// Memories are addressed by file path (e.g. `"memory/user.md"`).
|
||||||
/// Internally every [`MemoryEntry`] stores the originating path as its
|
/// Internally every [`MemoryEntry`] stores the originating path as its
|
||||||
/// `source_channel`. Section sub-paths are stored as
|
/// `source_channel`. Section sub-paths are stored as
|
||||||
/// `"<path>::<heading>"`.
|
/// `"<path>::<heading>"`.
|
||||||
@@ -422,7 +424,7 @@ impl ClawhdfBackend {
|
|||||||
|
|
||||||
// ── Compaction & Consolidation hooks (7.6) ────────────────────────────
|
// ── Compaction & Consolidation hooks (7.6) ────────────────────────────
|
||||||
|
|
||||||
/// Run a compaction cycle — called by OpenClaw during session compaction.
|
/// Run a compaction cycle (decay, compaction, WAL flush).
|
||||||
///
|
///
|
||||||
/// Sequence:
|
/// Sequence:
|
||||||
/// 1. `tick_session()` — apply Hebbian decay to all activation weights.
|
/// 1. `tick_session()` — apply Hebbian decay to all activation weights.
|
||||||
@@ -466,7 +468,7 @@ impl ClawhdfBackend {
|
|||||||
let record = MemoryRecord {
|
let record = MemoryRecord {
|
||||||
id: i as u64,
|
id: i as u64,
|
||||||
chunk: cache.chunks[i].clone(),
|
chunk: cache.chunks[i].clone(),
|
||||||
embedding: cache.embeddings[i].clone(),
|
embedding: cache.embeddings[i].to_vec(),
|
||||||
tier: MemoryTier::Working,
|
tier: MemoryTier::Working,
|
||||||
importance: cache.activation_weights[i],
|
importance: cache.activation_weights[i],
|
||||||
access_count: 0,
|
access_count: 0,
|
||||||
@@ -524,70 +526,27 @@ impl ClawhdfBackend {
|
|||||||
|
|
||||||
impl MemoryBackend for ClawhdfBackend {
|
impl MemoryBackend for ClawhdfBackend {
|
||||||
/// Search using hybrid vector + BM25 retrieval, then re-rank and
|
/// Search using hybrid vector + BM25 retrieval, then re-rank and
|
||||||
/// confidence-filter.
|
/// confidence-filter — [`HDF5Memory::search`] with both stages on.
|
||||||
fn search(
|
fn search(
|
||||||
&mut self,
|
&mut self,
|
||||||
query_text: &str,
|
query_text: &str,
|
||||||
query_embedding: &[f32],
|
query_embedding: &[f32],
|
||||||
k: usize,
|
k: usize,
|
||||||
) -> Vec<MemorySearchResult> {
|
) -> Vec<MemorySearchResult> {
|
||||||
// 1. Hybrid retrieval (vector + BM25, fused by score).
|
let options = SearchOptions::new(k)
|
||||||
let candidates = k.saturating_mul(3).max(10);
|
.with_rerank(self.rerank_config)
|
||||||
let raw = self.memory.hybrid_search_with(
|
.with_confidence(self.confidence_config.clone())
|
||||||
query_embedding,
|
.at_time(Self::now_secs());
|
||||||
query_text,
|
self.memory
|
||||||
crate::hybrid::DEFAULT_FUSION,
|
.search(query_embedding, query_text, &options)
|
||||||
candidates,
|
|
||||||
);
|
|
||||||
|
|
||||||
if raw.is_empty() {
|
|
||||||
return Vec::new();
|
|
||||||
}
|
|
||||||
|
|
||||||
let now = Self::now_secs();
|
|
||||||
|
|
||||||
// 2. Re-rank using temporal recency, source authority, Hebbian weight.
|
|
||||||
let rerank_inputs: Vec<RerankInput> = raw
|
|
||||||
.iter()
|
|
||||||
.map(|r| RerankInput {
|
|
||||||
index: r.index,
|
|
||||||
timestamp: r.timestamp,
|
|
||||||
source_channel: r.source_channel.clone(),
|
|
||||||
raw_activation: r.activation,
|
|
||||||
})
|
|
||||||
.collect();
|
|
||||||
|
|
||||||
let reranked = rerank(&rerank_inputs, &self.rerank_config, now);
|
|
||||||
|
|
||||||
// 3. Confidence rejection.
|
|
||||||
let scored: Vec<ScoredResult> = reranked
|
|
||||||
.iter()
|
|
||||||
.map(|r| ScoredResult {
|
|
||||||
index: r.index,
|
|
||||||
score: r.combined_score,
|
|
||||||
})
|
|
||||||
.collect();
|
|
||||||
|
|
||||||
let confident = reject_low_confidence(&scored, &self.confidence_config);
|
|
||||||
|
|
||||||
// 4. Map back to MemorySearchResult; preserve raw text via index lookup.
|
|
||||||
let raw_by_idx: HashMap<usize, &crate::SearchResult> =
|
|
||||||
raw.iter().map(|r| (r.index, r)).collect();
|
|
||||||
|
|
||||||
confident
|
|
||||||
.into_iter()
|
.into_iter()
|
||||||
.take(k)
|
.map(|r| MemorySearchResult {
|
||||||
.filter_map(|sr| {
|
text: r.chunk,
|
||||||
let r = raw_by_idx.get(&sr.index)?;
|
score: r.score,
|
||||||
let path = r.source_channel.clone();
|
path: r.source_channel.clone(),
|
||||||
Some(MemorySearchResult {
|
line_range: None,
|
||||||
text: r.chunk.clone(),
|
timestamp: Some(r.timestamp),
|
||||||
score: sr.score,
|
source: r.source_channel,
|
||||||
path: path.clone(),
|
|
||||||
line_range: None,
|
|
||||||
timestamp: Some(r.timestamp),
|
|
||||||
source: path,
|
|
||||||
})
|
|
||||||
})
|
})
|
||||||
.collect()
|
.collect()
|
||||||
}
|
}
|
||||||
@@ -716,11 +675,13 @@ impl MemoryBackend for ClawhdfBackend {
|
|||||||
|
|
||||||
let total_records = cache.count_active();
|
let total_records = cache.count_active();
|
||||||
|
|
||||||
|
// A record saved without an embedding occupies a zero row, so "has an
|
||||||
|
// embedding" is "has a non-zero norm" rather than "row is non-empty".
|
||||||
let total_embeddings = cache
|
let total_embeddings = cache
|
||||||
.embeddings
|
.norms
|
||||||
.iter()
|
.iter()
|
||||||
.enumerate()
|
.enumerate()
|
||||||
.filter(|(i, emb)| cache.tombstones[*i] == 0 && !emb.is_empty())
|
.filter(|(i, norm)| cache.tombstones[*i] == 0 && **norm > 0.0)
|
||||||
.count();
|
.count();
|
||||||
|
|
||||||
let file_size_bytes = std::fs::metadata(&self.hdf5_path)
|
let file_size_bytes = std::fs::metadata(&self.hdf5_path)
|
||||||
|
|||||||
@@ -2,7 +2,7 @@
|
|||||||
//!
|
//!
|
||||||
//! Records the origin, authorship, and a content hash of every memory chunk
|
//! Records the origin, authorship, and a content hash of every memory chunk
|
||||||
//! so the system can detect *accidental* corruption and trace data lineage.
|
//! so the system can detect *accidental* corruption and trace data lineage.
|
||||||
//! The hash is unkeyed (see [`fnv1a_64`]) — this is not a tamper-evidence or
|
//! The hash is unkeyed (FNV-1a) — this is not a tamper-evidence or
|
||||||
//! authenticity guarantee.
|
//! authenticity guarantee.
|
||||||
|
|
||||||
use std::collections::HashMap;
|
use std::collections::HashMap;
|
||||||
|
|||||||
@@ -4,8 +4,10 @@
|
|||||||
//! into a single composite score for each retrieved result.
|
//! into a single composite score for each retrieved result.
|
||||||
|
|
||||||
/// Configuration for the multi-factor re-ranker.
|
/// Configuration for the multi-factor re-ranker.
|
||||||
#[derive(Debug, Clone)]
|
#[derive(Debug, Clone, Copy)]
|
||||||
pub struct ReRankConfig {
|
pub struct ReRankConfig {
|
||||||
|
/// Weight applied to the retrieval score the candidate arrived with.
|
||||||
|
pub relevance_weight: f32,
|
||||||
/// Weight applied to the temporal decay score (0.0–1.0).
|
/// Weight applied to the temporal decay score (0.0–1.0).
|
||||||
pub temporal_weight: f32,
|
pub temporal_weight: f32,
|
||||||
/// Weight applied to the source authority score (0.0–1.0).
|
/// Weight applied to the source authority score (0.0–1.0).
|
||||||
@@ -20,6 +22,9 @@ pub struct ReRankConfig {
|
|||||||
impl Default for ReRankConfig {
|
impl Default for ReRankConfig {
|
||||||
fn default() -> Self {
|
fn default() -> Self {
|
||||||
Self {
|
Self {
|
||||||
|
// Relevance leads: the metadata signals break ties and nudge, they
|
||||||
|
// do not decide. See `BENCHMARKS.md`, "Recency discrimination".
|
||||||
|
relevance_weight: 1.0,
|
||||||
temporal_weight: 0.3,
|
temporal_weight: 0.3,
|
||||||
authority_weight: 0.2,
|
authority_weight: 0.2,
|
||||||
activation_weight: 0.5,
|
activation_weight: 0.5,
|
||||||
@@ -41,6 +46,8 @@ pub struct ReRankResult {
|
|||||||
pub authority_score: f32,
|
pub authority_score: f32,
|
||||||
/// Normalised Hebbian activation score in [0, 1].
|
/// Normalised Hebbian activation score in [0, 1].
|
||||||
pub activation_score: f32,
|
pub activation_score: f32,
|
||||||
|
/// The retrieval score carried through from the input.
|
||||||
|
pub relevance_score: f32,
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Compute an exponential decay temporal score.
|
/// Compute an exponential decay temporal score.
|
||||||
@@ -105,6 +112,15 @@ pub struct RerankInput {
|
|||||||
pub source_channel: String,
|
pub source_channel: String,
|
||||||
/// Raw Hebbian activation weight for this entry.
|
/// Raw Hebbian activation weight for this entry.
|
||||||
pub raw_activation: f32,
|
pub raw_activation: f32,
|
||||||
|
/// The retrieval score that put this entry in the candidate list.
|
||||||
|
///
|
||||||
|
/// Re-ranking is meant to *adjust* the retriever's ordering with signals
|
||||||
|
/// it does not have, not to replace it. Without this the combined score
|
||||||
|
/// was made of recency, authority and activation alone, so a candidate
|
||||||
|
/// pool came back ordered by age with its relevance ordering discarded.
|
||||||
|
/// Callers with no meaningful score can pass the same value for every
|
||||||
|
/// entry, which reduces to the old behaviour.
|
||||||
|
pub relevance: f32,
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Re-rank a list of retrieval results using multi-factor scoring.
|
/// Re-rank a list of retrieval results using multi-factor scoring.
|
||||||
@@ -138,7 +154,8 @@ pub fn rerank(
|
|||||||
let auth = source_authority_score(&inp.source_channel);
|
let auth = source_authority_score(&inp.source_channel);
|
||||||
let act = activation_score(inp.raw_activation);
|
let act = activation_score(inp.raw_activation);
|
||||||
|
|
||||||
let combined = config.temporal_weight * ts
|
let combined = config.relevance_weight * inp.relevance
|
||||||
|
+ config.temporal_weight * ts
|
||||||
+ config.authority_weight * auth
|
+ config.authority_weight * auth
|
||||||
+ config.activation_weight * act;
|
+ config.activation_weight * act;
|
||||||
|
|
||||||
@@ -148,6 +165,7 @@ pub fn rerank(
|
|||||||
temporal_score: ts,
|
temporal_score: ts,
|
||||||
authority_score: auth,
|
authority_score: auth,
|
||||||
activation_score: act,
|
activation_score: act,
|
||||||
|
relevance_score: inp.relevance,
|
||||||
}
|
}
|
||||||
})
|
})
|
||||||
.collect();
|
.collect();
|
||||||
@@ -253,22 +271,51 @@ mod tests {
|
|||||||
timestamp: 0.0, // very old
|
timestamp: 0.0, // very old
|
||||||
source_channel: "other".to_string(),
|
source_channel: "other".to_string(),
|
||||||
raw_activation: 0.1,
|
raw_activation: 0.1,
|
||||||
|
relevance: 0.0,
|
||||||
},
|
},
|
||||||
RerankInput {
|
RerankInput {
|
||||||
index: 1,
|
index: 1,
|
||||||
timestamp: 86_400.0, // one day ago
|
timestamp: 86_400.0, // one day ago
|
||||||
source_channel: "conversation".to_string(),
|
source_channel: "conversation".to_string(),
|
||||||
raw_activation: 0.5,
|
raw_activation: 0.5,
|
||||||
|
relevance: 0.0,
|
||||||
},
|
},
|
||||||
RerankInput {
|
RerankInput {
|
||||||
index: 2,
|
index: 2,
|
||||||
timestamp: 172_800.0, // "now"
|
timestamp: 172_800.0, // "now"
|
||||||
source_channel: "user_correction".to_string(),
|
source_channel: "user_correction".to_string(),
|
||||||
raw_activation: 1.0,
|
raw_activation: 1.0,
|
||||||
|
relevance: 0.0,
|
||||||
},
|
},
|
||||||
]
|
]
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn relevance_leads_but_recency_breaks_near_ties() {
|
||||||
|
let entry = |index, timestamp, relevance| RerankInput {
|
||||||
|
index,
|
||||||
|
timestamp,
|
||||||
|
source_channel: "conversation".to_string(),
|
||||||
|
raw_activation: 1.0,
|
||||||
|
relevance,
|
||||||
|
};
|
||||||
|
let now = 10.0 * 86_400.0;
|
||||||
|
let config = ReRankConfig::default();
|
||||||
|
|
||||||
|
// A clearly better match wins despite being much older. Before
|
||||||
|
// `relevance` existed the combined score ignored it entirely, so this
|
||||||
|
// returned the newer, irrelevant entry.
|
||||||
|
let ranked = rerank(&[entry(0, 0.0, 1.0), entry(1, now, 0.1)], &config, now);
|
||||||
|
assert_eq!(ranked[0].index, 0, "{ranked:?}");
|
||||||
|
|
||||||
|
// Between near-equal matches, the newer one wins.
|
||||||
|
let ranked = rerank(&[entry(0, 0.0, 0.80), entry(1, now, 0.79)], &config, now);
|
||||||
|
assert_eq!(ranked[0].index, 1, "{ranked:?}");
|
||||||
|
|
||||||
|
// The breakdown carries the relevance through.
|
||||||
|
assert_eq!(ranked[0].relevance_score, 0.79);
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn rerank_returns_all_entries() {
|
fn rerank_returns_all_entries() {
|
||||||
let inputs = make_inputs();
|
let inputs = make_inputs();
|
||||||
@@ -302,6 +349,7 @@ mod tests {
|
|||||||
#[test]
|
#[test]
|
||||||
fn rerank_score_breakdown_matches_manual_calculation() {
|
fn rerank_score_breakdown_matches_manual_calculation() {
|
||||||
let config = ReRankConfig {
|
let config = ReRankConfig {
|
||||||
|
relevance_weight: 0.0,
|
||||||
temporal_weight: 1.0,
|
temporal_weight: 1.0,
|
||||||
authority_weight: 0.0,
|
authority_weight: 0.0,
|
||||||
activation_weight: 0.0,
|
activation_weight: 0.0,
|
||||||
@@ -312,6 +360,7 @@ mod tests {
|
|||||||
timestamp: 0.0,
|
timestamp: 0.0,
|
||||||
source_channel: "other".to_string(),
|
source_channel: "other".to_string(),
|
||||||
raw_activation: 0.5,
|
raw_activation: 0.5,
|
||||||
|
relevance: 0.0,
|
||||||
}];
|
}];
|
||||||
let now = 3600.0_f64; // exactly one half-life later
|
let now = 3600.0_f64; // exactly one half-life later
|
||||||
let results = rerank(&inputs, &config, now);
|
let results = rerank(&inputs, &config, now);
|
||||||
|
|||||||
@@ -15,6 +15,9 @@ use crate::session::SessionCache;
|
|||||||
use crate::wal::WalMark;
|
use crate::wal::WalMark;
|
||||||
|
|
||||||
pub const SCHEMA_VERSION: &str = "1.0";
|
pub const SCHEMA_VERSION: &str = "1.0";
|
||||||
|
/// Writer-version tag stored in `/meta` as `edgehdf5_version`. Kept for file
|
||||||
|
/// compatibility; despite the name it has nothing to do with ZeroClaw, which
|
||||||
|
/// does not use clawhdf5.
|
||||||
pub const ZEROCLAW_VERSION: &str = "0.8.0";
|
pub const ZEROCLAW_VERSION: &str = "0.8.0";
|
||||||
|
|
||||||
/// `/meta` attributes holding the [`WalMark`] of the WAL prefix already folded
|
/// `/meta` attributes holding the [`WalMark`] of the WAL prefix already folded
|
||||||
@@ -23,6 +26,7 @@ pub const ZEROCLAW_VERSION: &str = "0.8.0";
|
|||||||
const WAL_APPLIED_LEN_ATTR: &str = "wal_applied_len";
|
const WAL_APPLIED_LEN_ATTR: &str = "wal_applied_len";
|
||||||
const WAL_APPLIED_CRC_ATTR: &str = "wal_applied_crc";
|
const WAL_APPLIED_CRC_ATTR: &str = "wal_applied_crc";
|
||||||
const ANN_GENERATION_ATTR: &str = "ann_generation";
|
const ANN_GENERATION_ATTR: &str = "ann_generation";
|
||||||
|
const SIG_VERSION_ATTR: &str = "sig_version";
|
||||||
|
|
||||||
/// Build a complete HDF5 file from the in-memory state.
|
/// Build a complete HDF5 file from the in-memory state.
|
||||||
pub fn build_hdf5_file(
|
pub fn build_hdf5_file(
|
||||||
@@ -46,7 +50,7 @@ pub fn build_hdf5_file_with_mark(
|
|||||||
) -> Result<Vec<u8>, MemoryError> {
|
) -> Result<Vec<u8>, MemoryError> {
|
||||||
let meta = CheckpointMeta {
|
let meta = CheckpointMeta {
|
||||||
wal_applied,
|
wal_applied,
|
||||||
ann_generation: None,
|
..CheckpointMeta::default()
|
||||||
};
|
};
|
||||||
build_hdf5_file_with_meta(config, cache, sessions, knowledge, &meta)
|
build_hdf5_file_with_meta(config, cache, sessions, knowledge, &meta)
|
||||||
}
|
}
|
||||||
@@ -61,6 +65,10 @@ pub struct CheckpointMeta {
|
|||||||
/// one left over from another checkpoint can never be attached to records
|
/// one left over from another checkpoint can never be attached to records
|
||||||
/// it wasn't built from.
|
/// it wasn't built from.
|
||||||
pub ann_generation: Option<u64>,
|
pub ann_generation: Option<u64>,
|
||||||
|
/// The checkpoint carries an Ed25519 signature (see [`crate::signing`]).
|
||||||
|
/// Read-only: whether a checkpoint is *written* signed is decided by the
|
||||||
|
/// signature passed to [`build_hdf5_file_signed`].
|
||||||
|
pub signed: bool,
|
||||||
}
|
}
|
||||||
|
|
||||||
/// [`build_hdf5_file`] with checkpoint bookkeeping.
|
/// [`build_hdf5_file`] with checkpoint bookkeeping.
|
||||||
@@ -70,6 +78,19 @@ pub fn build_hdf5_file_with_meta(
|
|||||||
sessions: &SessionCache,
|
sessions: &SessionCache,
|
||||||
knowledge: &KnowledgeCache,
|
knowledge: &KnowledgeCache,
|
||||||
checkpoint: &CheckpointMeta,
|
checkpoint: &CheckpointMeta,
|
||||||
|
) -> Result<Vec<u8>, MemoryError> {
|
||||||
|
build_hdf5_file_signed(config, cache, sessions, knowledge, checkpoint, None)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`build_hdf5_file_with_meta`], plus a signed manifest of the contents
|
||||||
|
/// (see [`crate::signing`]).
|
||||||
|
pub fn build_hdf5_file_signed(
|
||||||
|
config: &MemoryConfig,
|
||||||
|
cache: &MemoryCache,
|
||||||
|
sessions: &SessionCache,
|
||||||
|
knowledge: &KnowledgeCache,
|
||||||
|
checkpoint: &CheckpointMeta,
|
||||||
|
signature: Option<&crate::signing::StoredSignature>,
|
||||||
) -> Result<Vec<u8>, MemoryError> {
|
) -> Result<Vec<u8>, MemoryError> {
|
||||||
let wal_applied = checkpoint.wal_applied;
|
let wal_applied = checkpoint.wal_applied;
|
||||||
let mut builder = clawhdf5::FileBuilder::new();
|
let mut builder = clawhdf5::FileBuilder::new();
|
||||||
@@ -104,6 +125,19 @@ pub fn build_hdf5_file_with_meta(
|
|||||||
"wal_max_entries",
|
"wal_max_entries",
|
||||||
AttrValue::I64(config.wal_max_entries as i64),
|
AttrValue::I64(config.wal_max_entries as i64),
|
||||||
);
|
);
|
||||||
|
meta.set_attr(
|
||||||
|
"quantized_index",
|
||||||
|
AttrValue::I64(config.quantized_index.into()),
|
||||||
|
);
|
||||||
|
meta.set_attr("hnsw_m", AttrValue::I64(config.hnsw_m as i64));
|
||||||
|
meta.set_attr(
|
||||||
|
"hnsw_ef_construction",
|
||||||
|
AttrValue::I64(config.hnsw_ef_construction as i64),
|
||||||
|
);
|
||||||
|
meta.set_attr(
|
||||||
|
"hnsw_ef_search",
|
||||||
|
AttrValue::I64(config.hnsw_ef_search as i64),
|
||||||
|
);
|
||||||
meta.set_attr(
|
meta.set_attr(
|
||||||
"edgehdf5_version",
|
"edgehdf5_version",
|
||||||
AttrValue::String(ZEROCLAW_VERSION.into()),
|
AttrValue::String(ZEROCLAW_VERSION.into()),
|
||||||
@@ -117,11 +151,42 @@ pub fn build_hdf5_file_with_meta(
|
|||||||
// round trip through every reader.
|
// round trip through every reader.
|
||||||
meta.set_attr(ANN_GENERATION_ATTR, AttrValue::I64(generation as i64));
|
meta.set_attr(ANN_GENERATION_ATTR, AttrValue::I64(generation as i64));
|
||||||
}
|
}
|
||||||
|
if let Some(sig) = signature {
|
||||||
|
use crate::signing::to_hex;
|
||||||
|
let m = &sig.manifest;
|
||||||
|
meta.set_attr(
|
||||||
|
SIG_VERSION_ATTR,
|
||||||
|
AttrValue::I64(crate::signing::MANIFEST_VERSION),
|
||||||
|
);
|
||||||
|
meta.set_attr("sig_algorithm", AttrValue::String("ed25519".into()));
|
||||||
|
meta.set_attr("sig_public_key", AttrValue::String(to_hex(&sig.public_key)));
|
||||||
|
meta.set_attr("sig_signature", AttrValue::String(to_hex(&sig.signature)));
|
||||||
|
meta.set_attr("sig_record_count", AttrValue::I64(m.record_count as i64));
|
||||||
|
meta.set_attr(
|
||||||
|
"sig_records_root",
|
||||||
|
AttrValue::String(to_hex(&m.records_root)),
|
||||||
|
);
|
||||||
|
meta.set_attr("sig_settings", AttrValue::String(to_hex(&m.settings)));
|
||||||
|
meta.set_attr("sig_sessions", AttrValue::String(to_hex(&m.sessions)));
|
||||||
|
meta.set_attr("sig_graph", AttrValue::String(to_hex(&m.graph)));
|
||||||
|
}
|
||||||
// Need at least one dataset in the group for it to be a proper group
|
// Need at least one dataset in the group for it to be a proper group
|
||||||
meta.create_dataset("_marker").with_u8_data(&[1]).compact();
|
meta.create_dataset("_marker").with_u8_data(&[1]).compact();
|
||||||
let finished_meta = meta.finish();
|
let finished_meta = meta.finish();
|
||||||
builder.add_group(finished_meta);
|
builder.add_group(finished_meta);
|
||||||
|
|
||||||
|
// /integrity: the signed per-record hashes, so verification can say
|
||||||
|
// which records changed.
|
||||||
|
if let Some(sig) = signature {
|
||||||
|
let mut group = builder.create_group("integrity");
|
||||||
|
let flat: Vec<u8> = sig.record_hashes.iter().flatten().copied().collect();
|
||||||
|
group
|
||||||
|
.create_dataset("record_hashes")
|
||||||
|
.with_u8_data(&flat)
|
||||||
|
.with_shape(&[sig.record_hashes.len() as u64, 32]);
|
||||||
|
builder.add_group(group.finish());
|
||||||
|
}
|
||||||
|
|
||||||
// /memory group
|
// /memory group
|
||||||
build_memory_group(&mut builder, config, cache)?;
|
build_memory_group(&mut builder, config, cache)?;
|
||||||
|
|
||||||
@@ -146,20 +211,27 @@ fn build_memory_group(
|
|||||||
// chunks: fixed-length string array
|
// chunks: fixed-length string array
|
||||||
write_string_dataset(&mut group, "chunks", &cache.chunks);
|
write_string_dataset(&mut group, "chunks", &cache.chunks);
|
||||||
|
|
||||||
// embeddings: f32 [N x D]
|
// embeddings: [N x D], f32 — or IEEE half precision for a `float16`
|
||||||
|
// store. The cache already holds half-rounded values then, so this
|
||||||
|
// conversion is exact and a reopened store sees the same numbers.
|
||||||
let n = cache.embeddings.len() as u64;
|
let n = cache.embeddings.len() as u64;
|
||||||
let d = cache.embedding_dim as u64;
|
let d = cache.embedding_dim as u64;
|
||||||
let flat = cache.flat_embeddings();
|
let flat = cache.flat_embeddings();
|
||||||
{
|
{
|
||||||
let ds = group
|
let ds = group.create_dataset("embeddings");
|
||||||
.create_dataset("embeddings")
|
let elem_bytes: u64 = if config.float16 {
|
||||||
.with_f32_data(&flat)
|
ds.with_f16_data(flat);
|
||||||
.with_shape(&[n, d]);
|
2
|
||||||
|
} else {
|
||||||
|
ds.with_f32_data(flat);
|
||||||
|
4
|
||||||
|
};
|
||||||
|
ds.with_shape(&[n, d]);
|
||||||
|
|
||||||
// Chunk size tuning: target ~256KB per chunk for optimal I/O
|
// Chunk size tuning: target ~256KB per chunk for optimal I/O
|
||||||
if n > 0 && d > 0 {
|
if n > 0 && d > 0 {
|
||||||
let target_chunk_bytes: u64 = 256 * 1024;
|
let target_chunk_bytes: u64 = 256 * 1024;
|
||||||
let rows_per_chunk = (target_chunk_bytes / (d * 4)).max(1).min(n);
|
let rows_per_chunk = (target_chunk_bytes / (d * elem_bytes)).max(1).min(n);
|
||||||
ds.with_chunks(&[rows_per_chunk, d]);
|
ds.with_chunks(&[rows_per_chunk, d]);
|
||||||
|
|
||||||
// Compression. Shuffle is applied automatically (auto-shuffle
|
// Compression. Shuffle is applied automatically (auto-shuffle
|
||||||
@@ -405,10 +477,34 @@ fn write_string_dataset(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// `/meta`'s attributes, failing if any of them cannot be read.
|
||||||
|
///
|
||||||
|
/// `Group::attrs` leaves out an attribute it cannot decode. For the store's
|
||||||
|
/// settings that would silently fall back to defaults (e.g. `float16`, the
|
||||||
|
/// WAL mark), so an unreadable attribute is an error here, as it was before
|
||||||
|
/// `attrs` became tolerant.
|
||||||
|
fn meta_attrs(
|
||||||
|
file: &clawhdf5::File,
|
||||||
|
) -> Result<std::collections::HashMap<String, AttrValue>, MemoryError> {
|
||||||
|
let meta = file
|
||||||
|
.group("meta")
|
||||||
|
.map_err(|e| MemoryError::Schema(format!("missing /meta group: {e}")))?;
|
||||||
|
let (attrs, errors) = meta
|
||||||
|
.attrs_with_errors()
|
||||||
|
.map_err(|e| MemoryError::Schema(format!("cannot read /meta attrs: {e}")))?;
|
||||||
|
if let Some(e) = errors.first() {
|
||||||
|
return Err(MemoryError::Schema(format!(
|
||||||
|
"cannot read /meta attrs: {} unreadable, first: {e}",
|
||||||
|
errors.len()
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
Ok(attrs)
|
||||||
|
}
|
||||||
|
|
||||||
/// Validate an HDF5 file has the correct schema and load all data.
|
/// Validate an HDF5 file has the correct schema and load all data.
|
||||||
/// Read the checkpoint's [`WalMark`] from `/meta`, if it has one.
|
/// Read the checkpoint's [`WalMark`] from `/meta`, if it has one.
|
||||||
pub fn read_wal_mark(file: &clawhdf5::File) -> Option<WalMark> {
|
pub fn read_wal_mark(file: &clawhdf5::File) -> Option<WalMark> {
|
||||||
let attrs = file.group("meta").ok()?.attrs().ok()?;
|
let attrs = meta_attrs(file).ok()?;
|
||||||
let len = match attrs.get(WAL_APPLIED_LEN_ATTR)? {
|
let len = match attrs.get(WAL_APPLIED_LEN_ATTR)? {
|
||||||
AttrValue::I64(v) => u64::try_from(*v).ok()?,
|
AttrValue::I64(v) => u64::try_from(*v).ok()?,
|
||||||
_ => return None,
|
_ => return None,
|
||||||
@@ -420,19 +516,75 @@ pub fn read_wal_mark(file: &clawhdf5::File) -> Option<WalMark> {
|
|||||||
Some(WalMark { len, crc })
|
Some(WalMark { len, crc })
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Read a checkpoint's signature, if it has one. A signature whose
|
||||||
|
/// attributes are present but malformed is an error, not "unsigned".
|
||||||
|
pub fn read_signature(
|
||||||
|
file: &clawhdf5::File,
|
||||||
|
) -> Result<Option<crate::signing::StoredSignature>, MemoryError> {
|
||||||
|
use crate::signing::{Manifest, StoredSignature, from_hex};
|
||||||
|
let attrs = meta_attrs(file)?;
|
||||||
|
let version = match attrs.get(SIG_VERSION_ATTR) {
|
||||||
|
None => return Ok(None),
|
||||||
|
Some(AttrValue::I64(v)) => *v,
|
||||||
|
Some(_) => return Err(MemoryError::Schema("malformed sig_version".into())),
|
||||||
|
};
|
||||||
|
if version != crate::signing::MANIFEST_VERSION {
|
||||||
|
return Err(MemoryError::Schema(format!(
|
||||||
|
"unsupported signature version {version}"
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
fn hex<const N: usize>(
|
||||||
|
attrs: &std::collections::HashMap<String, AttrValue>,
|
||||||
|
name: &str,
|
||||||
|
) -> Result<[u8; N], MemoryError> {
|
||||||
|
match attrs.get(name) {
|
||||||
|
Some(AttrValue::String(s)) => from_hex::<N>(s),
|
||||||
|
_ => None,
|
||||||
|
}
|
||||||
|
.ok_or_else(|| MemoryError::Schema(format!("malformed or missing {name}")))
|
||||||
|
}
|
||||||
|
let record_count = match attrs.get("sig_record_count") {
|
||||||
|
Some(AttrValue::I64(v)) if *v >= 0 => *v as u64,
|
||||||
|
_ => return Err(MemoryError::Schema("malformed sig_record_count".into())),
|
||||||
|
};
|
||||||
|
let group = file
|
||||||
|
.group("integrity")
|
||||||
|
.map_err(|e| MemoryError::Schema(format!("signed checkpoint without /integrity: {e}")))?;
|
||||||
|
let flat = read_u8_dataset(&group, "record_hashes")?;
|
||||||
|
if flat.len() % 32 != 0 {
|
||||||
|
return Err(MemoryError::Schema(
|
||||||
|
"/integrity/record_hashes is not a whole number of hashes".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
let record_hashes = flat.as_chunks::<32>().0.to_vec();
|
||||||
|
Ok(Some(StoredSignature {
|
||||||
|
manifest: Manifest {
|
||||||
|
record_count,
|
||||||
|
records_root: hex::<32>(&attrs, "sig_records_root")?,
|
||||||
|
settings: hex::<32>(&attrs, "sig_settings")?,
|
||||||
|
sessions: hex::<32>(&attrs, "sig_sessions")?,
|
||||||
|
graph: hex::<32>(&attrs, "sig_graph")?,
|
||||||
|
},
|
||||||
|
record_hashes,
|
||||||
|
public_key: hex::<32>(&attrs, "sig_public_key")?,
|
||||||
|
signature: hex::<64>(&attrs, "sig_signature")?,
|
||||||
|
}))
|
||||||
|
}
|
||||||
|
|
||||||
/// Read the checkpoint bookkeeping from `/meta`.
|
/// Read the checkpoint bookkeeping from `/meta`.
|
||||||
pub fn read_checkpoint_meta(file: &clawhdf5::File) -> CheckpointMeta {
|
pub fn read_checkpoint_meta(file: &clawhdf5::File) -> CheckpointMeta {
|
||||||
let ann_generation = file
|
let ann_generation =
|
||||||
.group("meta")
|
meta_attrs(file)
|
||||||
.ok()
|
.ok()
|
||||||
.and_then(|g| g.attrs().ok())
|
.and_then(|attrs| match attrs.get(ANN_GENERATION_ATTR) {
|
||||||
.and_then(|attrs| match attrs.get(ANN_GENERATION_ATTR) {
|
Some(AttrValue::I64(v)) => Some(*v as u64),
|
||||||
Some(AttrValue::I64(v)) => Some(*v as u64),
|
_ => None,
|
||||||
_ => None,
|
});
|
||||||
});
|
let signed = meta_attrs(file).is_ok_and(|attrs| attrs.contains_key(SIG_VERSION_ATTR));
|
||||||
CheckpointMeta {
|
CheckpointMeta {
|
||||||
wal_applied: read_wal_mark(file),
|
wal_applied: read_wal_mark(file),
|
||||||
ann_generation,
|
ann_generation,
|
||||||
|
signed,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -440,12 +592,7 @@ pub fn validate_and_load(
|
|||||||
file: &clawhdf5::File,
|
file: &clawhdf5::File,
|
||||||
) -> Result<(MemoryConfig, MemoryCache, SessionCache, KnowledgeCache), MemoryError> {
|
) -> Result<(MemoryConfig, MemoryCache, SessionCache, KnowledgeCache), MemoryError> {
|
||||||
// Read /meta group attributes
|
// Read /meta group attributes
|
||||||
let meta = file
|
let attrs = meta_attrs(file)?;
|
||||||
.group("meta")
|
|
||||||
.map_err(|e| MemoryError::Schema(format!("missing /meta group: {e}")))?;
|
|
||||||
let attrs = meta
|
|
||||||
.attrs()
|
|
||||||
.map_err(|e| MemoryError::Schema(format!("cannot read /meta attrs: {e}")))?;
|
|
||||||
|
|
||||||
let schema_version = match attrs.get("schema_version") {
|
let schema_version = match attrs.get("schema_version") {
|
||||||
Some(AttrValue::String(s)) => s.clone(),
|
Some(AttrValue::String(s)) => s.clone(),
|
||||||
@@ -484,10 +631,32 @@ pub fn validate_and_load(
|
|||||||
wal_max_entries: optional_i64_attr(&attrs, "wal_max_entries")
|
wal_max_entries: optional_i64_attr(&attrs, "wal_max_entries")
|
||||||
.and_then(|v| usize::try_from(v).ok())
|
.and_then(|v| usize::try_from(v).ok())
|
||||||
.unwrap_or(500),
|
.unwrap_or(500),
|
||||||
|
// `false`, not the new-store default: a store written before this
|
||||||
|
// setting existed was built with an f32 index, and reopening it must
|
||||||
|
// not silently change that.
|
||||||
|
quantized_index: optional_bool_attr(&attrs, "quantized_index", false),
|
||||||
|
hnsw_m: optional_i64_attr(&attrs, "hnsw_m")
|
||||||
|
.and_then(|v| usize::try_from(v).ok())
|
||||||
|
.unwrap_or(16),
|
||||||
|
hnsw_ef_construction: optional_i64_attr(&attrs, "hnsw_ef_construction")
|
||||||
|
.and_then(|v| usize::try_from(v).ok())
|
||||||
|
.unwrap_or(64),
|
||||||
|
hnsw_ef_search: optional_i64_attr(&attrs, "hnsw_ef_search")
|
||||||
|
.and_then(|v| usize::try_from(v).ok())
|
||||||
|
.unwrap_or(0),
|
||||||
};
|
};
|
||||||
|
|
||||||
// Load /memory group
|
// Load /memory group
|
||||||
let memory_cache = load_memory_group(file, embedding_dim)?;
|
let mut memory_cache = load_memory_group(file, embedding_dim)?;
|
||||||
|
// A float16 store's cache holds half-rounded embeddings. Embeddings read
|
||||||
|
// from an f16 dataset already are; a float16 store whose last checkpoint
|
||||||
|
// predates half-precision storage is still f32 on disk and is rounded
|
||||||
|
// here.
|
||||||
|
if config.float16 && embeddings_are_f16(file) {
|
||||||
|
memory_cache.half_precision = true;
|
||||||
|
} else {
|
||||||
|
memory_cache.set_half_precision(config.float16);
|
||||||
|
}
|
||||||
|
|
||||||
// Load /sessions group
|
// Load /sessions group
|
||||||
let session_cache = load_sessions_group(file)?;
|
let session_cache = load_sessions_group(file)?;
|
||||||
@@ -563,12 +732,7 @@ fn load_memory_group(
|
|||||||
.collect(),
|
.collect(),
|
||||||
};
|
};
|
||||||
|
|
||||||
// Unflatten embeddings
|
// No unflattening: the cache stores the buffer as it is on disk.
|
||||||
let embeddings: Vec<Vec<f32>> = flat_embeddings
|
|
||||||
.chunks(embedding_dim)
|
|
||||||
.map(|c| c.to_vec())
|
|
||||||
.collect();
|
|
||||||
|
|
||||||
// Read activation_weights if present, default to vec![1.0; N] for backward compat
|
// Read activation_weights if present, default to vec![1.0; N] for backward compat
|
||||||
let activation_weights = match read_f32_dataset(&group, "activation_weights") {
|
let activation_weights = match read_f32_dataset(&group, "activation_weights") {
|
||||||
Ok(w) if w.len() == n => w,
|
Ok(w) if w.len() == n => w,
|
||||||
@@ -576,7 +740,7 @@ fn load_memory_group(
|
|||||||
};
|
};
|
||||||
|
|
||||||
cache.chunks = chunks;
|
cache.chunks = chunks;
|
||||||
cache.embeddings = embeddings;
|
cache.embeddings.set_flat(embedding_dim, flat_embeddings);
|
||||||
cache.source_channels = source_channels;
|
cache.source_channels = source_channels;
|
||||||
cache.timestamps = timestamps;
|
cache.timestamps = timestamps;
|
||||||
cache.session_ids = session_ids;
|
cache.session_ids = session_ids;
|
||||||
@@ -584,7 +748,6 @@ fn load_memory_group(
|
|||||||
cache.tombstones = tombstones;
|
cache.tombstones = tombstones;
|
||||||
cache.norms = norms;
|
cache.norms = norms;
|
||||||
cache.activation_weights = activation_weights;
|
cache.activation_weights = activation_weights;
|
||||||
cache.rebuild_flat();
|
|
||||||
|
|
||||||
Ok(cache)
|
Ok(cache)
|
||||||
}
|
}
|
||||||
@@ -736,6 +899,13 @@ fn read_string_dataset_from_group(
|
|||||||
.map_err(|e| MemoryError::Hdf5(format!("cannot read strings from {name}: {e}")))
|
.map_err(|e| MemoryError::Hdf5(format!("cannot read strings from {name}: {e}")))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Whether `/memory/embeddings` is stored as IEEE half precision.
|
||||||
|
fn embeddings_are_f16(file: &clawhdf5::File) -> bool {
|
||||||
|
file.dataset("memory/embeddings")
|
||||||
|
.and_then(|ds| ds.dtype())
|
||||||
|
.is_ok_and(|dt| matches!(dt, clawhdf5::DType::Other(ref s) if s == "float16"))
|
||||||
|
}
|
||||||
|
|
||||||
fn read_f32_dataset(group: &clawhdf5::Group<'_>, name: &str) -> Result<Vec<f32>, MemoryError> {
|
fn read_f32_dataset(group: &clawhdf5::Group<'_>, name: &str) -> Result<Vec<f32>, MemoryError> {
|
||||||
let ds = group
|
let ds = group
|
||||||
.dataset(name)
|
.dataset(name)
|
||||||
|
|||||||
@@ -2,18 +2,107 @@
|
|||||||
|
|
||||||
use std::path::Path;
|
use std::path::Path;
|
||||||
|
|
||||||
|
use std::collections::HashSet;
|
||||||
|
|
||||||
use crate::bm25;
|
use crate::bm25;
|
||||||
|
use crate::confidence::{ConfidenceConfig, ScoredResult, reject_low_confidence};
|
||||||
use crate::hybrid;
|
use crate::hybrid;
|
||||||
|
use crate::reranker::{ReRankConfig, RerankInput, rerank};
|
||||||
use crate::{HDF5Memory, MAX_ACTIVATION_WEIGHT, MemoryError, Result, SearchResult};
|
use crate::{HDF5Memory, MAX_ACTIVATION_WEIGHT, MemoryError, Result, SearchResult};
|
||||||
|
|
||||||
|
/// Options for [`HDF5Memory::search`].
|
||||||
|
///
|
||||||
|
/// [`SearchOptions::new`] is plain hybrid search with the tuned default
|
||||||
|
/// fusion — the same as `hybrid_search_with(.., hybrid::DEFAULT_FUSION, k)`.
|
||||||
|
/// Every stage beyond that is opt-in.
|
||||||
|
#[derive(Debug, Clone)]
|
||||||
|
pub struct SearchOptions {
|
||||||
|
/// Number of results to return.
|
||||||
|
pub k: usize,
|
||||||
|
/// How the vector and keyword stages are combined.
|
||||||
|
pub fusion: hybrid::Fusion,
|
||||||
|
/// Only consider records whose `source_channel` is one of these. The
|
||||||
|
/// filter applies *before* ranking, so a filtered search still returns up
|
||||||
|
/// to `k` results and scores are normalised over the records it can
|
||||||
|
/// return. `None` searches everything; an empty list matches nothing.
|
||||||
|
pub source_channels: Option<Vec<String>>,
|
||||||
|
/// Re-rank a candidate pool by retrieval relevance, recency, source
|
||||||
|
/// authority and activation — the pipeline the OpenClaw backend runs.
|
||||||
|
pub rerank: Option<ReRankConfig>,
|
||||||
|
/// Candidates retrieved for re-ranking; 0 means `max(3k, 10)`.
|
||||||
|
pub rerank_pool: usize,
|
||||||
|
/// Drop low-confidence results (after re-ranking, when that is on).
|
||||||
|
pub confidence: Option<ConfidenceConfig>,
|
||||||
|
/// The time recency is measured from, in seconds since the epoch.
|
||||||
|
/// `None` uses the system clock.
|
||||||
|
pub now: Option<f64>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl SearchOptions {
|
||||||
|
pub fn new(k: usize) -> Self {
|
||||||
|
Self {
|
||||||
|
k,
|
||||||
|
fusion: hybrid::DEFAULT_FUSION,
|
||||||
|
source_channels: None,
|
||||||
|
rerank: None,
|
||||||
|
rerank_pool: 0,
|
||||||
|
confidence: None,
|
||||||
|
now: None,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn with_fusion(mut self, fusion: hybrid::Fusion) -> Self {
|
||||||
|
self.fusion = fusion;
|
||||||
|
self
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Search only records from these source channels.
|
||||||
|
pub fn with_sources<S: Into<String>>(mut self, channels: impl IntoIterator<Item = S>) -> Self {
|
||||||
|
self.source_channels = Some(channels.into_iter().map(Into::into).collect());
|
||||||
|
self
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn with_rerank(mut self, config: ReRankConfig) -> Self {
|
||||||
|
self.rerank = Some(config);
|
||||||
|
self
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn with_confidence(mut self, config: ConfidenceConfig) -> Self {
|
||||||
|
self.confidence = Some(config);
|
||||||
|
self
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Measure recency from `now` (seconds since the epoch) instead of the
|
||||||
|
/// system clock — for reproducible results and tests.
|
||||||
|
pub fn at_time(mut self, now: f64) -> Self {
|
||||||
|
self.now = Some(now);
|
||||||
|
self
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Default for SearchOptions {
|
||||||
|
fn default() -> Self {
|
||||||
|
Self::new(10)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
impl HDF5Memory {
|
impl HDF5Memory {
|
||||||
/// Vector + keyword scoring stage of [`HDF5Memory::hybrid_search`].
|
/// Vector + keyword scoring stage of [`HDF5Memory::search`].
|
||||||
///
|
///
|
||||||
/// Without the `hnsw` feature this is a full linear cosine scan (the exact
|
/// Without the `hnsw` feature this is a full linear cosine scan (the exact
|
||||||
/// previous behaviour, also used as the correctness oracle in tests). With
|
/// previous behaviour, also used as the correctness oracle in tests). With
|
||||||
/// `hnsw` enabled and an index available, the vector candidates come from an
|
/// `hnsw` enabled and an index available, the vector candidates come from an
|
||||||
/// approximate-nearest-neighbour search over an over-fetched pool, then merge
|
/// approximate-nearest-neighbour search over an over-fetched pool, then merge
|
||||||
/// with BM25 via the shared [`hybrid::merge_vector_keyword`].
|
/// with BM25 via the shared [`hybrid::merge_vector_keyword`].
|
||||||
|
///
|
||||||
|
/// `exclude`, when given, marks records that must not be returned (1 =
|
||||||
|
/// excluded; it covers tombstones too). The index is over-fetched in
|
||||||
|
/// proportion to how much the mask removes. Surfacing `pool` candidates
|
||||||
|
/// costs the index roughly `pool × M` distance evaluations, while an exact
|
||||||
|
/// scan of the allowed records costs one each — so whenever that scan is
|
||||||
|
/// the cheaper of the two it is used instead, and it is also the fallback
|
||||||
|
/// if the pool comes back with too few allowed hits (the allowed records
|
||||||
|
/// sit away from the query). A filtered search never comes back short.
|
||||||
#[cfg(feature = "hnsw")]
|
#[cfg(feature = "hnsw")]
|
||||||
fn vector_keyword_search(
|
fn vector_keyword_search(
|
||||||
&mut self,
|
&mut self,
|
||||||
@@ -22,24 +111,103 @@ impl HDF5Memory {
|
|||||||
bm25: &bm25::BM25Index,
|
bm25: &bm25::BM25Index,
|
||||||
fusion: hybrid::Fusion,
|
fusion: hybrid::Fusion,
|
||||||
k: usize,
|
k: usize,
|
||||||
|
exclude: Option<&[u8]>,
|
||||||
) -> Vec<(usize, f32)> {
|
) -> Vec<(usize, f32)> {
|
||||||
self.ensure_hnsw_fresh();
|
self.ensure_hnsw_fresh();
|
||||||
|
let n = self.cache.len();
|
||||||
|
// Over-fetch so the merge sees a useful vector pool. `ef` is
|
||||||
|
// configurable, but the pool the fusion stage sees is not tied to it:
|
||||||
|
// a caller lowering `ef` for speed should not silently narrow what
|
||||||
|
// fusion has to work with.
|
||||||
|
let mut pool = (k * 8).max(64);
|
||||||
|
let mut allowed = n;
|
||||||
|
if let Some(ex) = exclude {
|
||||||
|
allowed = ex.iter().filter(|&&e| e == 0).count();
|
||||||
|
if allowed == 0 {
|
||||||
|
return Vec::new();
|
||||||
|
}
|
||||||
|
// Expect `pool` allowed hits if the filter is independent of the
|
||||||
|
// query's neighbourhood.
|
||||||
|
pool = pool.saturating_mul(n).div_ceil(allowed);
|
||||||
|
if allowed <= pool.saturating_mul(self.hnsw_m()) {
|
||||||
|
return self.exact_masked_search(query_embedding, query_text, bm25, fusion, k, ex);
|
||||||
|
}
|
||||||
|
}
|
||||||
match self.hnsw.as_ref() {
|
match self.hnsw.as_ref() {
|
||||||
Some(index) if !index.is_empty() && index.dimension() == query_embedding.len() => {
|
Some(index) if !index.is_empty() && index.dimension() == query_embedding.len() => {
|
||||||
// Over-fetch so the merge sees a useful vector pool; cosine
|
let ef = self.hnsw_ef_search(k).max(pool);
|
||||||
// distance from the index converts back to similarity (1 - d).
|
let candidates = index.search(query_embedding, pool, ef);
|
||||||
let pool = (k * 8).max(64);
|
// A quantised index returns approximate distances, and no
|
||||||
let vec_scores: Vec<(usize, f32)> = index
|
// amount of `ef` fixes that — the loss is in the distances,
|
||||||
.search(query_embedding, pool, pool)
|
// not the graph. Re-score the pool against the cache's exact
|
||||||
|
// embeddings, which cost nothing extra to keep: recall then
|
||||||
|
// matches an f32 index. See `BENCHMARKS.md`.
|
||||||
|
let exact = index.storage() == clawhdf5_ann::Storage::Int8;
|
||||||
|
let vec_scores: Vec<(usize, f32)> = candidates
|
||||||
.into_iter()
|
.into_iter()
|
||||||
.map(|(id, dist)| (id, 1.0 - dist))
|
.filter(|(id, _)| exclude.is_none_or(|ex| ex[*id] == 0))
|
||||||
|
.map(|(id, dist)| {
|
||||||
|
let score = if exact {
|
||||||
|
crate::vector_search::cosine_similarity(
|
||||||
|
query_embedding,
|
||||||
|
&self.cache.embeddings[id],
|
||||||
|
)
|
||||||
|
} else {
|
||||||
|
1.0 - dist
|
||||||
|
};
|
||||||
|
(id, score)
|
||||||
|
})
|
||||||
.collect();
|
.collect();
|
||||||
// Fusion normalises over every keyword match, so it needs all
|
// Fusion normalises over every keyword match, so it needs all
|
||||||
// the scores — but not ranked.
|
// the scores — but not ranked.
|
||||||
let kw_scores = bm25.scores(query_text);
|
let mut kw_scores = bm25.scores(query_text);
|
||||||
|
if let Some(ex) = exclude {
|
||||||
|
if vec_scores.len() < k.min(allowed) {
|
||||||
|
// The allowed records are not where the index looked.
|
||||||
|
return self.exact_masked_search(
|
||||||
|
query_embedding,
|
||||||
|
query_text,
|
||||||
|
bm25,
|
||||||
|
fusion,
|
||||||
|
k,
|
||||||
|
ex,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
kw_scores.retain(|(id, _)| ex[*id] == 0);
|
||||||
|
}
|
||||||
hybrid::fuse(vec_scores, kw_scores, fusion, k)
|
hybrid::fuse(vec_scores, kw_scores, fusion, k)
|
||||||
}
|
}
|
||||||
_ => hybrid::hybrid_search_fused(
|
_ => match exclude {
|
||||||
|
Some(ex) => {
|
||||||
|
self.exact_masked_search(query_embedding, query_text, bm25, fusion, k, ex)
|
||||||
|
}
|
||||||
|
None => hybrid::hybrid_search_fused(
|
||||||
|
query_embedding,
|
||||||
|
query_text,
|
||||||
|
&self.cache.embeddings,
|
||||||
|
&self.cache.chunks,
|
||||||
|
&self.cache.tombstones,
|
||||||
|
bm25,
|
||||||
|
fusion,
|
||||||
|
k,
|
||||||
|
),
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(not(feature = "hnsw"))]
|
||||||
|
fn vector_keyword_search(
|
||||||
|
&mut self,
|
||||||
|
query_embedding: &[f32],
|
||||||
|
query_text: &str,
|
||||||
|
bm25: &bm25::BM25Index,
|
||||||
|
fusion: hybrid::Fusion,
|
||||||
|
k: usize,
|
||||||
|
exclude: Option<&[u8]>,
|
||||||
|
) -> Vec<(usize, f32)> {
|
||||||
|
match exclude {
|
||||||
|
Some(ex) => self.exact_masked_search(query_embedding, query_text, bm25, fusion, k, ex),
|
||||||
|
None => hybrid::hybrid_search_fused(
|
||||||
query_embedding,
|
query_embedding,
|
||||||
query_text,
|
query_text,
|
||||||
&self.cache.embeddings,
|
&self.cache.embeddings,
|
||||||
@@ -52,25 +220,33 @@ impl HDF5Memory {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
#[cfg(not(feature = "hnsw"))]
|
/// Exact hybrid search over the records `exclude` leaves (0 = allowed).
|
||||||
fn vector_keyword_search(
|
fn exact_masked_search(
|
||||||
&mut self,
|
&self,
|
||||||
query_embedding: &[f32],
|
query_embedding: &[f32],
|
||||||
query_text: &str,
|
query_text: &str,
|
||||||
bm25: &bm25::BM25Index,
|
bm25: &bm25::BM25Index,
|
||||||
fusion: hybrid::Fusion,
|
fusion: hybrid::Fusion,
|
||||||
k: usize,
|
k: usize,
|
||||||
|
exclude: &[u8],
|
||||||
) -> Vec<(usize, f32)> {
|
) -> Vec<(usize, f32)> {
|
||||||
hybrid::hybrid_search_fused(
|
let vec_scores =
|
||||||
query_embedding,
|
hybrid::exact_vector_scores(query_embedding, &self.cache.embeddings, exclude);
|
||||||
query_text,
|
let mut kw_scores = bm25.scores(query_text);
|
||||||
&self.cache.embeddings,
|
kw_scores.retain(|(id, _)| exclude.get(*id) == Some(&0));
|
||||||
&self.cache.chunks,
|
hybrid::fuse(vec_scores, kw_scores, fusion, k)
|
||||||
&self.cache.tombstones,
|
}
|
||||||
bm25,
|
|
||||||
fusion,
|
/// The exclusion mask for a source-channel filter: 1 for a tombstoned
|
||||||
k,
|
/// record or one from a channel not in `channels`.
|
||||||
)
|
fn source_mask(&self, channels: &[String]) -> Vec<u8> {
|
||||||
|
let allowed: HashSet<&str> = channels.iter().map(String::as_str).collect();
|
||||||
|
self.cache
|
||||||
|
.source_channels
|
||||||
|
.iter()
|
||||||
|
.zip(&self.cache.tombstones)
|
||||||
|
.map(|(ch, &t)| u8::from(t != 0 || !allowed.contains(ch.as_str())))
|
||||||
|
.collect()
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Perform hybrid search combining cosine vector similarity and BM25 keyword search.
|
/// Perform hybrid search combining cosine vector similarity and BM25 keyword search.
|
||||||
@@ -105,12 +281,53 @@ impl HDF5Memory {
|
|||||||
fusion: hybrid::Fusion,
|
fusion: hybrid::Fusion,
|
||||||
k: usize,
|
k: usize,
|
||||||
) -> Vec<SearchResult> {
|
) -> Vec<SearchResult> {
|
||||||
|
self.search(
|
||||||
|
query_embedding,
|
||||||
|
query_text,
|
||||||
|
&SearchOptions::new(k).with_fusion(fusion),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Hybrid search with optional source filtering, re-ranking and
|
||||||
|
/// confidence rejection — see [`SearchOptions`].
|
||||||
|
///
|
||||||
|
/// Stages, in order: vector + keyword retrieval over the records the
|
||||||
|
/// source filter allows; fusion; scaling by Hebbian activation; re-ranking
|
||||||
|
/// (if on) of a `rerank_pool` of candidates; confidence rejection (if on);
|
||||||
|
/// the top `k`. The records returned with a positive score get their
|
||||||
|
/// Hebbian boost.
|
||||||
|
pub fn search(
|
||||||
|
&mut self,
|
||||||
|
query_embedding: &[f32],
|
||||||
|
query_text: &str,
|
||||||
|
options: &SearchOptions,
|
||||||
|
) -> Vec<SearchResult> {
|
||||||
|
let k = options.k;
|
||||||
|
let fetch = match options.rerank {
|
||||||
|
Some(_) if options.rerank_pool > 0 => options.rerank_pool.max(k),
|
||||||
|
Some(_) => k.saturating_mul(3).max(10),
|
||||||
|
None => k,
|
||||||
|
};
|
||||||
|
let exclude = options
|
||||||
|
.source_channels
|
||||||
|
.as_deref()
|
||||||
|
.map(|channels| self.source_mask(channels));
|
||||||
|
|
||||||
// The keyword index lives for the life of the store and is updated
|
// The keyword index lives for the life of the store and is updated
|
||||||
// incrementally. Take it out for the duration of the call so the
|
// incrementally. Take it out for the duration of the call so the
|
||||||
// vector stage can borrow `self` mutably, then put it back.
|
// vector stage can borrow `self` mutably, then put it back.
|
||||||
self.ensure_bm25_fresh();
|
self.ensure_bm25_fresh();
|
||||||
let bm25 = self.bm25.take().expect("ensure_bm25_fresh leaves an index");
|
let bm25 = self.bm25.take().expect("ensure_bm25_fresh leaves an index");
|
||||||
let scored = self.vector_keyword_search(query_embedding, query_text, &bm25, fusion, k);
|
let scored = self.vector_keyword_search(
|
||||||
|
query_embedding,
|
||||||
|
query_text,
|
||||||
|
&bm25,
|
||||||
|
options.fusion,
|
||||||
|
fetch,
|
||||||
|
exclude.as_deref(),
|
||||||
|
);
|
||||||
|
self.bm25 = Some(bm25);
|
||||||
|
|
||||||
let mut results: Vec<SearchResult> = scored
|
let mut results: Vec<SearchResult> = scored
|
||||||
.into_iter()
|
.into_iter()
|
||||||
.map(|(idx, score)| {
|
.map(|(idx, score)| {
|
||||||
@@ -134,6 +351,25 @@ impl HDF5Memory {
|
|||||||
.then(a.index.cmp(&b.index))
|
.then(a.index.cmp(&b.index))
|
||||||
});
|
});
|
||||||
|
|
||||||
|
if let Some(config) = &options.rerank {
|
||||||
|
results = Self::rerank_results(results, config, options.now);
|
||||||
|
}
|
||||||
|
if let Some(config) = &options.confidence {
|
||||||
|
let scored: Vec<ScoredResult> = results
|
||||||
|
.iter()
|
||||||
|
.map(|r| ScoredResult {
|
||||||
|
index: r.index,
|
||||||
|
score: r.score,
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
let keep: HashSet<usize> = reject_low_confidence(&scored, config)
|
||||||
|
.into_iter()
|
||||||
|
.map(|r| r.index)
|
||||||
|
.collect();
|
||||||
|
results.retain(|r| keep.contains(&r.index));
|
||||||
|
}
|
||||||
|
results.truncate(k);
|
||||||
|
|
||||||
// Only reinforce records that actually matched. When fewer than `k`
|
// Only reinforce records that actually matched. When fewer than `k`
|
||||||
// records are relevant, the rest of the list is zero-score filler;
|
// records are relevant, the rest of the list is zero-score filler;
|
||||||
// boosting it would teach the store that arbitrary records are
|
// boosting it would teach the store that arbitrary records are
|
||||||
@@ -144,11 +380,45 @@ impl HDF5Memory {
|
|||||||
.map(|r| r.index)
|
.map(|r| r.index)
|
||||||
.collect();
|
.collect();
|
||||||
self.apply_hebbian_boost(&hit_indices);
|
self.apply_hebbian_boost(&hit_indices);
|
||||||
self.bm25 = Some(bm25);
|
|
||||||
|
|
||||||
results
|
results
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Reorder by the re-ranker's combined score, which also becomes each
|
||||||
|
/// result's `score`.
|
||||||
|
fn rerank_results(
|
||||||
|
results: Vec<SearchResult>,
|
||||||
|
config: &ReRankConfig,
|
||||||
|
now: Option<f64>,
|
||||||
|
) -> Vec<SearchResult> {
|
||||||
|
let now = now.unwrap_or_else(|| {
|
||||||
|
std::time::SystemTime::now()
|
||||||
|
.duration_since(std::time::UNIX_EPOCH)
|
||||||
|
.map(|d| d.as_secs_f64())
|
||||||
|
.unwrap_or(0.0)
|
||||||
|
});
|
||||||
|
let inputs: Vec<RerankInput> = results
|
||||||
|
.iter()
|
||||||
|
.map(|r| RerankInput {
|
||||||
|
index: r.index,
|
||||||
|
timestamp: r.timestamp,
|
||||||
|
source_channel: r.source_channel.clone(),
|
||||||
|
raw_activation: r.activation,
|
||||||
|
relevance: r.score,
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
let mut by_index: std::collections::HashMap<usize, SearchResult> =
|
||||||
|
results.into_iter().map(|r| (r.index, r)).collect();
|
||||||
|
rerank(&inputs, config, now)
|
||||||
|
.into_iter()
|
||||||
|
.filter_map(|rr| {
|
||||||
|
let mut r = by_index.remove(&rr.index)?;
|
||||||
|
r.score = rr.combined_score;
|
||||||
|
Some(r)
|
||||||
|
})
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
/// Reinforce the records a query returned. The new weights are persisted by
|
/// Reinforce the records a query returned. The new weights are persisted by
|
||||||
/// the next checkpoint (any write that flushes, `flush_wal`, or drop) — not
|
/// the next checkpoint (any write that flushes, `flush_wal`, or drop) — not
|
||||||
/// by rewriting the whole store inside the query, which is what made
|
/// by rewriting the whole store inside the query, which is what made
|
||||||
|
|||||||
@@ -33,7 +33,7 @@ impl SessionCache {
|
|||||||
self.entries.is_empty()
|
self.entries.is_empty()
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Add a new session with its summary.
|
/// Add a new session with its summary, timestamped now.
|
||||||
pub fn add(
|
pub fn add(
|
||||||
&mut self,
|
&mut self,
|
||||||
id: &str,
|
id: &str,
|
||||||
@@ -47,6 +47,21 @@ impl SessionCache {
|
|||||||
.unwrap_or_default()
|
.unwrap_or_default()
|
||||||
.as_secs_f64()
|
.as_secs_f64()
|
||||||
* 1_000_000.0; // microseconds
|
* 1_000_000.0; // microseconds
|
||||||
|
self.add_at(id, start_idx, end_idx, channel, summary, ts);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Add a session with an explicit timestamp (Unix **microseconds**, the
|
||||||
|
/// unit [`SessionEntry::ts`] uses) — for importers carrying sessions over
|
||||||
|
/// from another store, whose original time should be kept.
|
||||||
|
pub fn add_at(
|
||||||
|
&mut self,
|
||||||
|
id: &str,
|
||||||
|
start_idx: usize,
|
||||||
|
end_idx: usize,
|
||||||
|
channel: &str,
|
||||||
|
summary: &str,
|
||||||
|
ts: f64,
|
||||||
|
) {
|
||||||
self.entries.push(SessionEntry {
|
self.entries.push(SessionEntry {
|
||||||
id: id.to_string(),
|
id: id.to_string(),
|
||||||
start_idx: start_idx as u64,
|
start_idx: start_idx as u64,
|
||||||
|
|||||||
@@ -0,0 +1,419 @@
|
|||||||
|
//! Ed25519-signed checkpoints.
|
||||||
|
//!
|
||||||
|
//! When a signing key is set ([`crate::HDF5Memory::set_signing_key`]), every
|
||||||
|
//! checkpoint writes a signed manifest of the store: a SHA-256 per memory
|
||||||
|
//! record rolled into a Merkle root, plus hashes of the store's settings, its
|
||||||
|
//! sessions and its knowledge graph. [`verify_store`] recomputes all of it from
|
||||||
|
//! the file and checks the signature against a public key the caller trusts,
|
||||||
|
//! so any change to the checkpointed file — a record's text or embedding, a
|
||||||
|
//! setting, a session, a graph edge, made through this crate or any other HDF5
|
||||||
|
//! tool — is detected, and the per-record hashes say which records changed.
|
||||||
|
//!
|
||||||
|
//! What it does not cover: saves still only in the WAL (made since the last
|
||||||
|
//! checkpoint). [`VerifyReport::wal_entries_unsigned`] counts them.
|
||||||
|
//!
|
||||||
|
//! The hashes cover exactly what the file persists, in the form the loader
|
||||||
|
//! returns it, so a store verifies after any number of reopen/checkpoint
|
||||||
|
//! cycles. Derived data (L2 norms, the vector index) is not covered; it is
|
||||||
|
//! recomputed from covered data.
|
||||||
|
|
||||||
|
use ed25519_dalek::{Signature, Signer, Verifier};
|
||||||
|
pub use ed25519_dalek::{SigningKey, VerifyingKey};
|
||||||
|
use sha2::{Digest, Sha256};
|
||||||
|
|
||||||
|
use crate::MemoryConfig;
|
||||||
|
use crate::cache::MemoryCache;
|
||||||
|
use crate::knowledge::KnowledgeCache;
|
||||||
|
use crate::session::SessionCache;
|
||||||
|
use crate::wal::WalMark;
|
||||||
|
|
||||||
|
/// Version of the manifest encoding; part of what is signed.
|
||||||
|
pub const MANIFEST_VERSION: i64 = 1;
|
||||||
|
|
||||||
|
type Hash = [u8; 32];
|
||||||
|
|
||||||
|
/// The hashes a signature covers.
|
||||||
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||||
|
pub struct Manifest {
|
||||||
|
pub record_count: u64,
|
||||||
|
/// Merkle root over the per-record hashes.
|
||||||
|
pub records_root: Hash,
|
||||||
|
/// Settings persisted in `/meta`, plus the checkpoint's WAL mark.
|
||||||
|
pub settings: Hash,
|
||||||
|
pub sessions: Hash,
|
||||||
|
pub graph: Hash,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Manifest {
|
||||||
|
/// The exact bytes that are signed.
|
||||||
|
pub fn signed_bytes(&self) -> Vec<u8> {
|
||||||
|
let mut m = Vec::with_capacity(160);
|
||||||
|
m.extend_from_slice(b"clawhdf5-agent signed checkpoint\0");
|
||||||
|
m.extend_from_slice(&MANIFEST_VERSION.to_le_bytes());
|
||||||
|
m.extend_from_slice(&self.record_count.to_le_bytes());
|
||||||
|
m.extend_from_slice(&self.records_root);
|
||||||
|
m.extend_from_slice(&self.settings);
|
||||||
|
m.extend_from_slice(&self.sessions);
|
||||||
|
m.extend_from_slice(&self.graph);
|
||||||
|
m
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A signature as stored in a checkpoint.
|
||||||
|
#[derive(Debug, Clone)]
|
||||||
|
pub struct StoredSignature {
|
||||||
|
pub manifest: Manifest,
|
||||||
|
pub record_hashes: Vec<Hash>,
|
||||||
|
pub public_key: [u8; 32],
|
||||||
|
pub signature: [u8; 64],
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Build the manifest (and per-record hashes) for the state about to be
|
||||||
|
/// checkpointed, and sign it.
|
||||||
|
pub fn sign(
|
||||||
|
key: &SigningKey,
|
||||||
|
config: &MemoryConfig,
|
||||||
|
cache: &MemoryCache,
|
||||||
|
sessions: &SessionCache,
|
||||||
|
knowledge: &KnowledgeCache,
|
||||||
|
wal_applied: Option<WalMark>,
|
||||||
|
) -> StoredSignature {
|
||||||
|
let (manifest, record_hashes) = manifest(config, cache, sessions, knowledge, wal_applied);
|
||||||
|
let signature = key.sign(&manifest.signed_bytes()).to_bytes();
|
||||||
|
StoredSignature {
|
||||||
|
manifest,
|
||||||
|
record_hashes,
|
||||||
|
public_key: key.verifying_key().to_bytes(),
|
||||||
|
signature,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Compute the manifest of a store's state.
|
||||||
|
pub fn manifest(
|
||||||
|
config: &MemoryConfig,
|
||||||
|
cache: &MemoryCache,
|
||||||
|
sessions: &SessionCache,
|
||||||
|
knowledge: &KnowledgeCache,
|
||||||
|
wal_applied: Option<WalMark>,
|
||||||
|
) -> (Manifest, Vec<Hash>) {
|
||||||
|
let record_hashes: Vec<Hash> = (0..cache.len()).map(|i| record_hash(cache, i)).collect();
|
||||||
|
let manifest = Manifest {
|
||||||
|
record_count: cache.len() as u64,
|
||||||
|
records_root: merkle_root(&record_hashes),
|
||||||
|
settings: settings_hash(config, wal_applied),
|
||||||
|
sessions: sessions_hash(sessions),
|
||||||
|
graph: graph_hash(knowledge),
|
||||||
|
};
|
||||||
|
(manifest, record_hashes)
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Canonical encoding
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// A SHA-256 over length-prefixed fields, so no two different field lists
|
||||||
|
/// hash the same bytes.
|
||||||
|
struct Fields(Sha256);
|
||||||
|
|
||||||
|
impl Fields {
|
||||||
|
fn new(domain: &str) -> Self {
|
||||||
|
let mut h = Sha256::new();
|
||||||
|
h.update((domain.len() as u64).to_le_bytes());
|
||||||
|
h.update(domain.as_bytes());
|
||||||
|
Self(h)
|
||||||
|
}
|
||||||
|
fn bytes(&mut self, b: &[u8]) -> &mut Self {
|
||||||
|
self.0.update((b.len() as u64).to_le_bytes());
|
||||||
|
self.0.update(b);
|
||||||
|
self
|
||||||
|
}
|
||||||
|
/// Strings as the loader returns them: stored null-padded, so a trailing
|
||||||
|
/// NUL cannot survive a round trip and must not be part of the hash.
|
||||||
|
fn str(&mut self, s: &str) -> &mut Self {
|
||||||
|
self.bytes(s.trim_end_matches('\0').as_bytes())
|
||||||
|
}
|
||||||
|
fn u64(&mut self, v: u64) -> &mut Self {
|
||||||
|
self.0.update(v.to_le_bytes());
|
||||||
|
self
|
||||||
|
}
|
||||||
|
fn f64(&mut self, v: f64) -> &mut Self {
|
||||||
|
self.0.update(v.to_bits().to_le_bytes());
|
||||||
|
self
|
||||||
|
}
|
||||||
|
fn f32(&mut self, v: f32) -> &mut Self {
|
||||||
|
self.0.update(v.to_bits().to_le_bytes());
|
||||||
|
self
|
||||||
|
}
|
||||||
|
fn finish(self) -> Hash {
|
||||||
|
self.0.finalize().into()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Everything persisted about record `i`, including its position. The
|
||||||
|
/// embedding is hashed as the cache holds it — for a `float16` store that is
|
||||||
|
/// the half-rounded value the file holds.
|
||||||
|
fn record_hash(cache: &MemoryCache, i: usize) -> Hash {
|
||||||
|
let mut f = Fields::new("clawhdf5-agent/record");
|
||||||
|
f.u64(i as u64).str(&cache.chunks[i]);
|
||||||
|
let emb: Vec<u8> = cache.embeddings[i]
|
||||||
|
.iter()
|
||||||
|
.flat_map(|v| v.to_bits().to_le_bytes())
|
||||||
|
.collect();
|
||||||
|
f.bytes(&emb)
|
||||||
|
.str(&cache.source_channels[i])
|
||||||
|
.f64(cache.timestamps[i])
|
||||||
|
.str(&cache.session_ids[i])
|
||||||
|
.str(&cache.tags[i])
|
||||||
|
.u64(u64::from(cache.tombstones[i]))
|
||||||
|
.f32(cache.activation_weights[i]);
|
||||||
|
f.finish()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Binary Merkle tree: leaves are the record hashes; a parent hashes its two
|
||||||
|
/// children with a node prefix; an odd node is carried up unchanged.
|
||||||
|
fn merkle_root(leaves: &[Hash]) -> Hash {
|
||||||
|
if leaves.is_empty() {
|
||||||
|
return Fields::new("clawhdf5-agent/merkle-empty").finish();
|
||||||
|
}
|
||||||
|
let mut level: Vec<Hash> = leaves.to_vec();
|
||||||
|
while level.len() > 1 {
|
||||||
|
level = level
|
||||||
|
.chunks(2)
|
||||||
|
.map(|pair| match pair {
|
||||||
|
[l, r] => {
|
||||||
|
let mut h = Sha256::new();
|
||||||
|
h.update([1u8]);
|
||||||
|
h.update(l);
|
||||||
|
h.update(r);
|
||||||
|
h.finalize().into()
|
||||||
|
}
|
||||||
|
[only] => *only,
|
||||||
|
_ => unreachable!(),
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
}
|
||||||
|
level[0]
|
||||||
|
}
|
||||||
|
|
||||||
|
fn settings_hash(c: &MemoryConfig, wal_applied: Option<WalMark>) -> Hash {
|
||||||
|
let mut f = Fields::new("clawhdf5-agent/settings");
|
||||||
|
f.str(crate::schema::SCHEMA_VERSION)
|
||||||
|
.str(&c.created_at)
|
||||||
|
.str(&c.agent_id)
|
||||||
|
.str(&c.embedder)
|
||||||
|
.u64(c.embedding_dim as u64)
|
||||||
|
.u64(c.chunk_size as u64)
|
||||||
|
.u64(c.overlap as u64)
|
||||||
|
.u64(u64::from(c.float16))
|
||||||
|
.u64(u64::from(c.compression))
|
||||||
|
.u64(u64::from(c.compression_level))
|
||||||
|
.f32(c.compact_threshold)
|
||||||
|
.f32(c.hebbian_boost)
|
||||||
|
.f32(c.decay_factor)
|
||||||
|
.u64(u64::from(c.wal_enabled))
|
||||||
|
.u64(c.wal_max_entries as u64)
|
||||||
|
.u64(u64::from(c.quantized_index))
|
||||||
|
.u64(c.hnsw_m as u64)
|
||||||
|
.u64(c.hnsw_ef_construction as u64)
|
||||||
|
.u64(c.hnsw_ef_search as u64);
|
||||||
|
// An empty mark is not written to the file, so it must hash as none.
|
||||||
|
match wal_applied.filter(|m| m.len > 0) {
|
||||||
|
Some(m) => f.u64(1).u64(m.len).u64(u64::from(m.crc)),
|
||||||
|
None => f.u64(0),
|
||||||
|
};
|
||||||
|
f.finish()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn sessions_hash(s: &SessionCache) -> Hash {
|
||||||
|
let mut f = Fields::new("clawhdf5-agent/sessions");
|
||||||
|
f.u64(s.entries.len() as u64);
|
||||||
|
for (i, e) in s.entries.iter().enumerate() {
|
||||||
|
f.str(&e.id)
|
||||||
|
.u64(e.start_idx)
|
||||||
|
.u64(e.end_idx)
|
||||||
|
.str(&e.channel)
|
||||||
|
.f64(e.ts)
|
||||||
|
.str(s.summaries.get(i).map(String::as_str).unwrap_or(""));
|
||||||
|
}
|
||||||
|
f.finish()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn graph_hash(k: &KnowledgeCache) -> Hash {
|
||||||
|
let mut f = Fields::new("clawhdf5-agent/graph");
|
||||||
|
f.u64(k.entities.len() as u64);
|
||||||
|
for e in &k.entities {
|
||||||
|
f.u64(e.id)
|
||||||
|
.str(&e.name)
|
||||||
|
.str(&e.entity_type)
|
||||||
|
.u64(e.embedding_idx as u64);
|
||||||
|
}
|
||||||
|
f.u64(k.relations.len() as u64);
|
||||||
|
for r in &k.relations {
|
||||||
|
f.u64(r.src)
|
||||||
|
.u64(r.tgt)
|
||||||
|
.str(&r.relation)
|
||||||
|
.f32(r.weight)
|
||||||
|
.f64(r.ts);
|
||||||
|
}
|
||||||
|
f.u64(k.alias_strings.len() as u64);
|
||||||
|
for (s, id) in k.alias_strings.iter().zip(&k.alias_entity_ids) {
|
||||||
|
f.str(s).u64(*id as u64);
|
||||||
|
}
|
||||||
|
f.finish()
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Verification
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// The outcome of [`verify_store`].
|
||||||
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||||
|
pub struct VerifyReport {
|
||||||
|
/// The checkpoint carries a signature.
|
||||||
|
pub signed: bool,
|
||||||
|
/// The signature was made by the key the caller trusts.
|
||||||
|
pub key_matches: bool,
|
||||||
|
/// The signature over the stored manifest is valid.
|
||||||
|
pub signature_valid: bool,
|
||||||
|
/// The file's current contents match the signed manifest.
|
||||||
|
pub records_match: bool,
|
||||||
|
pub settings_match: bool,
|
||||||
|
pub sessions_match: bool,
|
||||||
|
pub graph_match: bool,
|
||||||
|
/// Records whose contents differ from what was signed (by position),
|
||||||
|
/// when the stored per-record hashes are themselves authentic.
|
||||||
|
pub changed_records: Vec<usize>,
|
||||||
|
/// Records in the file versus in the signed manifest.
|
||||||
|
pub record_count: u64,
|
||||||
|
pub signed_record_count: u64,
|
||||||
|
/// The public key the checkpoint claims to be signed by.
|
||||||
|
pub public_key: Option<[u8; 32]>,
|
||||||
|
/// Saves in the WAL after the checkpoint: not covered by the signature.
|
||||||
|
pub wal_entries_unsigned: usize,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl VerifyReport {
|
||||||
|
/// Signed by the trusted key, signature valid, and every part of the
|
||||||
|
/// file unchanged since it was signed.
|
||||||
|
pub fn is_valid(&self) -> bool {
|
||||||
|
self.signed
|
||||||
|
&& self.key_matches
|
||||||
|
&& self.signature_valid
|
||||||
|
&& self.records_match
|
||||||
|
&& self.settings_match
|
||||||
|
&& self.sessions_match
|
||||||
|
&& self.graph_match
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Check a store file against the public key the caller trusts.
|
||||||
|
///
|
||||||
|
/// Reads the checkpoint (not the WAL), recomputes every hash from its
|
||||||
|
/// contents and checks the signature. Never writes.
|
||||||
|
pub fn verify_store(
|
||||||
|
path: &std::path::Path,
|
||||||
|
trusted: &VerifyingKey,
|
||||||
|
) -> Result<VerifyReport, crate::MemoryError> {
|
||||||
|
let file = clawhdf5::File::open(path)
|
||||||
|
.map_err(|e| crate::MemoryError::Hdf5(format!("cannot open {}: {e}", path.display())))?;
|
||||||
|
let (config, cache, sessions, knowledge) = crate::schema::validate_and_load(&file)?;
|
||||||
|
let checkpoint = crate::schema::read_checkpoint_meta(&file);
|
||||||
|
let stored = crate::schema::read_signature(&file)?;
|
||||||
|
let wal_entries_unsigned = count_wal_entries_after(path, checkpoint.wal_applied);
|
||||||
|
|
||||||
|
let (current, current_hashes) = manifest(
|
||||||
|
&config,
|
||||||
|
&cache,
|
||||||
|
&sessions,
|
||||||
|
&knowledge,
|
||||||
|
checkpoint.wal_applied,
|
||||||
|
);
|
||||||
|
|
||||||
|
let Some(stored) = stored else {
|
||||||
|
return Ok(VerifyReport {
|
||||||
|
signed: false,
|
||||||
|
key_matches: false,
|
||||||
|
signature_valid: false,
|
||||||
|
records_match: false,
|
||||||
|
settings_match: false,
|
||||||
|
sessions_match: false,
|
||||||
|
graph_match: false,
|
||||||
|
changed_records: Vec::new(),
|
||||||
|
record_count: current.record_count,
|
||||||
|
signed_record_count: 0,
|
||||||
|
public_key: None,
|
||||||
|
wal_entries_unsigned,
|
||||||
|
});
|
||||||
|
};
|
||||||
|
|
||||||
|
let key_matches = stored.public_key == trusted.to_bytes();
|
||||||
|
let signature_valid = trusted
|
||||||
|
.verify(
|
||||||
|
&stored.manifest.signed_bytes(),
|
||||||
|
&Signature::from_bytes(&stored.signature),
|
||||||
|
)
|
||||||
|
.is_ok();
|
||||||
|
// The stored per-record hashes can localise a change only if they are
|
||||||
|
// the ones that were signed.
|
||||||
|
let hashes_authentic = signature_valid
|
||||||
|
&& stored.record_hashes.len() as u64 == stored.manifest.record_count
|
||||||
|
&& merkle_root(&stored.record_hashes) == stored.manifest.records_root;
|
||||||
|
let changed_records = if hashes_authentic {
|
||||||
|
let n = current_hashes.len().max(stored.record_hashes.len());
|
||||||
|
(0..n)
|
||||||
|
.filter(|&i| current_hashes.get(i) != stored.record_hashes.get(i))
|
||||||
|
.collect()
|
||||||
|
} else {
|
||||||
|
Vec::new()
|
||||||
|
};
|
||||||
|
|
||||||
|
Ok(VerifyReport {
|
||||||
|
signed: true,
|
||||||
|
key_matches,
|
||||||
|
signature_valid,
|
||||||
|
records_match: signature_valid
|
||||||
|
&& current.record_count == stored.manifest.record_count
|
||||||
|
&& current.records_root == stored.manifest.records_root,
|
||||||
|
settings_match: signature_valid && current.settings == stored.manifest.settings,
|
||||||
|
sessions_match: signature_valid && current.sessions == stored.manifest.sessions,
|
||||||
|
graph_match: signature_valid && current.graph == stored.manifest.graph,
|
||||||
|
changed_records,
|
||||||
|
record_count: current.record_count,
|
||||||
|
signed_record_count: stored.manifest.record_count,
|
||||||
|
public_key: Some(stored.public_key),
|
||||||
|
wal_entries_unsigned,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
fn count_wal_entries_after(store: &std::path::Path, mark: Option<WalMark>) -> usize {
|
||||||
|
let wal = store.with_extension("h5.wal");
|
||||||
|
if !wal.exists() {
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
crate::wal::WalFile::read_entries_for_migration(&wal, mark)
|
||||||
|
.map(|e| e.len())
|
||||||
|
.unwrap_or(0)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A new random signing key from the operating system's RNG.
|
||||||
|
pub fn generate_key() -> SigningKey {
|
||||||
|
SigningKey::generate(&mut rand_core::OsRng)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Hex encoding for keys and signatures in attributes and the CLI.
|
||||||
|
pub fn to_hex(bytes: &[u8]) -> String {
|
||||||
|
bytes.iter().map(|b| format!("{b:02x}")).collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Parse hex into exactly `N` bytes.
|
||||||
|
pub fn from_hex<const N: usize>(s: &str) -> Option<[u8; N]> {
|
||||||
|
let s = s.trim();
|
||||||
|
if s.len() != 2 * N {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
let mut out = [0u8; N];
|
||||||
|
for (i, byte) in out.iter_mut().enumerate() {
|
||||||
|
*byte = u8::from_str_radix(&s[2 * i..2 * i + 2], 16).ok()?;
|
||||||
|
}
|
||||||
|
Some(out)
|
||||||
|
}
|
||||||
@@ -36,7 +36,7 @@ pub fn write_to_disk_with_mark(
|
|||||||
) -> Result<(), MemoryError> {
|
) -> Result<(), MemoryError> {
|
||||||
let meta = schema::CheckpointMeta {
|
let meta = schema::CheckpointMeta {
|
||||||
wal_applied,
|
wal_applied,
|
||||||
ann_generation: None,
|
..schema::CheckpointMeta::default()
|
||||||
};
|
};
|
||||||
write_to_disk_with_meta(path, config, cache, sessions, knowledge, &meta)
|
write_to_disk_with_meta(path, config, cache, sessions, knowledge, &meta)
|
||||||
}
|
}
|
||||||
@@ -50,7 +50,21 @@ pub fn write_to_disk_with_meta(
|
|||||||
knowledge: &KnowledgeCache,
|
knowledge: &KnowledgeCache,
|
||||||
checkpoint: &schema::CheckpointMeta,
|
checkpoint: &schema::CheckpointMeta,
|
||||||
) -> Result<(), MemoryError> {
|
) -> Result<(), MemoryError> {
|
||||||
let bytes = schema::build_hdf5_file_with_meta(config, cache, sessions, knowledge, checkpoint)?;
|
write_to_disk_signed(path, config, cache, sessions, knowledge, checkpoint, None)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`write_to_disk_with_meta`] with a signed manifest of the contents.
|
||||||
|
pub fn write_to_disk_signed(
|
||||||
|
path: &Path,
|
||||||
|
config: &MemoryConfig,
|
||||||
|
cache: &MemoryCache,
|
||||||
|
sessions: &SessionCache,
|
||||||
|
knowledge: &KnowledgeCache,
|
||||||
|
checkpoint: &schema::CheckpointMeta,
|
||||||
|
signature: Option<&crate::signing::StoredSignature>,
|
||||||
|
) -> Result<(), MemoryError> {
|
||||||
|
let bytes =
|
||||||
|
schema::build_hdf5_file_signed(config, cache, sessions, knowledge, checkpoint, signature)?;
|
||||||
|
|
||||||
if bytes.is_empty() {
|
if bytes.is_empty() {
|
||||||
return Err(MemoryError::Hdf5("build_hdf5_file produced 0 bytes".into()));
|
return Err(MemoryError::Hdf5("build_hdf5_file produced 0 bytes".into()));
|
||||||
@@ -113,13 +127,11 @@ pub type StoreState = (MemoryConfig, MemoryCache, SessionCache, KnowledgeCache);
|
|||||||
/// [`read_from_disk`], plus the checkpoint's [`WalMark`] (if any) so the
|
/// [`read_from_disk`], plus the checkpoint's [`WalMark`] (if any) so the
|
||||||
/// caller can skip WAL entries this file already contains.
|
/// caller can skip WAL entries this file already contains.
|
||||||
pub fn read_from_disk_with_mark(path: &Path) -> Result<(StoreState, Option<WalMark>), MemoryError> {
|
pub fn read_from_disk_with_mark(path: &Path) -> Result<(StoreState, Option<WalMark>), MemoryError> {
|
||||||
let mmap = clawhdf5_io::MmapReader::open(path).map_err(MemoryError::Io)?;
|
// `File::open` memory-maps the file itself (the facade's `mmap` feature is
|
||||||
|
// on by default). Mapping it here and handing over `as_bytes().to_vec()`
|
||||||
// Advise the OS we'll need the whole file for parsing
|
// did the same work and then copied the whole store — a second full copy
|
||||||
mmap.advise_willneed(0, mmap.len());
|
// of the file, live for the whole parse, on top of the mapping.
|
||||||
|
let file = clawhdf5::File::open(path)
|
||||||
// Parse the HDF5 file from the mmap'd bytes
|
|
||||||
let file = clawhdf5::File::from_bytes(mmap.as_bytes().to_vec())
|
|
||||||
.map_err(|e| MemoryError::Hdf5(format!("cannot open {}: {e}", path.display())))?;
|
.map_err(|e| MemoryError::Hdf5(format!("cannot open {}: {e}", path.display())))?;
|
||||||
|
|
||||||
let (mut config, cache, sessions, knowledge) = schema::validate_and_load(&file)?;
|
let (mut config, cache, sessions, knowledge) = schema::validate_and_load(&file)?;
|
||||||
@@ -133,9 +145,7 @@ pub fn read_from_disk_with_mark(path: &Path) -> Result<(StoreState, Option<WalMa
|
|||||||
pub fn read_from_disk_with_meta(
|
pub fn read_from_disk_with_meta(
|
||||||
path: &Path,
|
path: &Path,
|
||||||
) -> Result<(StoreState, schema::CheckpointMeta), MemoryError> {
|
) -> Result<(StoreState, schema::CheckpointMeta), MemoryError> {
|
||||||
let mmap = clawhdf5_io::MmapReader::open(path).map_err(MemoryError::Io)?;
|
let file = clawhdf5::File::open(path)
|
||||||
mmap.advise_willneed(0, mmap.len());
|
|
||||||
let file = clawhdf5::File::from_bytes(mmap.as_bytes().to_vec())
|
|
||||||
.map_err(|e| MemoryError::Hdf5(format!("cannot open {}: {e}", path.display())))?;
|
.map_err(|e| MemoryError::Hdf5(format!("cannot open {}: {e}", path.display())))?;
|
||||||
let (mut config, cache, sessions, knowledge) = schema::validate_and_load(&file)?;
|
let (mut config, cache, sessions, knowledge) = schema::validate_and_load(&file)?;
|
||||||
config.path = path.to_path_buf();
|
config.path = path.to_path_buf();
|
||||||
|
|||||||
@@ -4,6 +4,44 @@
|
|||||||
//! `clawhdf5_accel`, with optional float16 support via the `half` crate.
|
//! `clawhdf5_accel`, with optional float16 support via the `half` crate.
|
||||||
//! Supports pre-computed norms for eliminating redundant norm computations.
|
//! Supports pre-computed norms for eliminating redundant norm computations.
|
||||||
|
|
||||||
|
/// A corpus of equal-length embeddings addressable by index.
|
||||||
|
///
|
||||||
|
/// Lets the batch kernels read either the cache's flat `[N x dim]` buffer or a
|
||||||
|
/// plain `Vec<Vec<f32>>` without either side owning a second copy.
|
||||||
|
pub trait VectorSet {
|
||||||
|
/// Number of embeddings.
|
||||||
|
fn count(&self) -> usize;
|
||||||
|
/// Embedding `i`; callers only index below [`VectorSet::count`].
|
||||||
|
fn row(&self, i: usize) -> &[f32];
|
||||||
|
}
|
||||||
|
|
||||||
|
impl VectorSet for [Vec<f32>] {
|
||||||
|
fn count(&self) -> usize {
|
||||||
|
self.len()
|
||||||
|
}
|
||||||
|
fn row(&self, i: usize) -> &[f32] {
|
||||||
|
&self[i]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl VectorSet for Vec<Vec<f32>> {
|
||||||
|
fn count(&self) -> usize {
|
||||||
|
self.len()
|
||||||
|
}
|
||||||
|
fn row(&self, i: usize) -> &[f32] {
|
||||||
|
&self[i]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl VectorSet for crate::cache::Embeddings {
|
||||||
|
fn count(&self) -> usize {
|
||||||
|
self.len()
|
||||||
|
}
|
||||||
|
fn row(&self, i: usize) -> &[f32] {
|
||||||
|
&self[i]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// Compute cosine similarity between two f32 slices.
|
/// Compute cosine similarity between two f32 slices.
|
||||||
///
|
///
|
||||||
/// Returns 0.0 if either vector has zero magnitude.
|
/// Returns 0.0 if either vector has zero magnitude.
|
||||||
@@ -22,7 +60,7 @@ pub fn cosine_similarity(a: &[f32], b: &[f32]) -> f32 {
|
|||||||
/// Returns `(index, score)` pairs sorted by score descending.
|
/// Returns `(index, score)` pairs sorted by score descending.
|
||||||
pub fn cosine_similarity_batch(
|
pub fn cosine_similarity_batch(
|
||||||
query: &[f32],
|
query: &[f32],
|
||||||
vectors: &[Vec<f32>],
|
vectors: &(impl VectorSet + ?Sized),
|
||||||
tombstones: &[u8],
|
tombstones: &[u8],
|
||||||
) -> Vec<(usize, f32)> {
|
) -> Vec<(usize, f32)> {
|
||||||
let query_norm = clawhdf5_accel::vector_norm(query);
|
let query_norm = clawhdf5_accel::vector_norm(query);
|
||||||
@@ -30,7 +68,7 @@ pub fn cosine_similarity_batch(
|
|||||||
return Vec::new();
|
return Vec::new();
|
||||||
}
|
}
|
||||||
|
|
||||||
let n = vectors.len();
|
let n = vectors.count();
|
||||||
let mut results: Vec<(usize, f32)> = Vec::with_capacity(n);
|
let mut results: Vec<(usize, f32)> = Vec::with_capacity(n);
|
||||||
|
|
||||||
// Process 4 vectors at a time where possible
|
// Process 4 vectors at a time where possible
|
||||||
@@ -42,8 +80,9 @@ pub fn cosine_similarity_batch(
|
|||||||
if i < tombstones.len() && tombstones[i] != 0 {
|
if i < tombstones.len() && tombstones[i] != 0 {
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
let vec_norm = clawhdf5_accel::vector_norm(&vectors[i]);
|
let vec_norm = clawhdf5_accel::vector_norm(vectors.row(i));
|
||||||
let score = crate::cosine_similarity_prenorm(query, query_norm, &vectors[i], vec_norm);
|
let score =
|
||||||
|
crate::cosine_similarity_prenorm(query, query_norm, vectors.row(i), vec_norm);
|
||||||
results.push((i, score));
|
results.push((i, score));
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -53,8 +92,8 @@ pub fn cosine_similarity_batch(
|
|||||||
if i < tombstones.len() && tombstones[i] != 0 {
|
if i < tombstones.len() && tombstones[i] != 0 {
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
let vec_norm = clawhdf5_accel::vector_norm(&vectors[i]);
|
let vec_norm = clawhdf5_accel::vector_norm(vectors.row(i));
|
||||||
let score = crate::cosine_similarity_prenorm(query, query_norm, &vectors[i], vec_norm);
|
let score = crate::cosine_similarity_prenorm(query, query_norm, vectors.row(i), vec_norm);
|
||||||
results.push((i, score));
|
results.push((i, score));
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -68,7 +107,7 @@ pub fn cosine_similarity_batch(
|
|||||||
/// collections. Uses `score = dot(query, vec) / (query_norm * stored_norm)`.
|
/// collections. Uses `score = dot(query, vec) / (query_norm * stored_norm)`.
|
||||||
pub fn cosine_similarity_batch_prenorm(
|
pub fn cosine_similarity_batch_prenorm(
|
||||||
query: &[f32],
|
query: &[f32],
|
||||||
vectors: &[Vec<f32>],
|
vectors: &(impl VectorSet + ?Sized),
|
||||||
norms: &[f32],
|
norms: &[f32],
|
||||||
tombstones: &[u8],
|
tombstones: &[u8],
|
||||||
) -> Vec<(usize, f32)> {
|
) -> Vec<(usize, f32)> {
|
||||||
@@ -77,7 +116,7 @@ pub fn cosine_similarity_batch_prenorm(
|
|||||||
return Vec::new();
|
return Vec::new();
|
||||||
}
|
}
|
||||||
|
|
||||||
let n = vectors.len();
|
let n = vectors.count();
|
||||||
let mut results: Vec<(usize, f32)> = Vec::with_capacity(n);
|
let mut results: Vec<(usize, f32)> = Vec::with_capacity(n);
|
||||||
|
|
||||||
for i in 0..n {
|
for i in 0..n {
|
||||||
@@ -85,7 +124,7 @@ pub fn cosine_similarity_batch_prenorm(
|
|||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
let vec_norm = norms[i];
|
let vec_norm = norms[i];
|
||||||
let score = crate::cosine_similarity_prenorm(query, query_norm, &vectors[i], vec_norm);
|
let score = crate::cosine_similarity_prenorm(query, query_norm, vectors.row(i), vec_norm);
|
||||||
results.push((i, score));
|
results.push((i, score));
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -162,7 +201,7 @@ pub fn cosine_similarity_f16(
|
|||||||
#[cfg(feature = "parallel")]
|
#[cfg(feature = "parallel")]
|
||||||
pub fn parallel_cosine_batch(
|
pub fn parallel_cosine_batch(
|
||||||
query: &[f32],
|
query: &[f32],
|
||||||
vectors: &[Vec<f32>],
|
vectors: &(impl VectorSet + Sync + ?Sized),
|
||||||
tombstones: &[u8],
|
tombstones: &[u8],
|
||||||
k: usize,
|
k: usize,
|
||||||
) -> Vec<(usize, f32)> {
|
) -> Vec<(usize, f32)> {
|
||||||
@@ -174,24 +213,27 @@ pub fn parallel_cosine_batch(
|
|||||||
}
|
}
|
||||||
|
|
||||||
let num_cores = rayon::current_num_threads().max(1);
|
let num_cores = rayon::current_num_threads().max(1);
|
||||||
let chunk_size = vectors.len().div_ceil(num_cores);
|
let chunk_size = vectors.count().div_ceil(num_cores);
|
||||||
if chunk_size == 0 {
|
if chunk_size == 0 {
|
||||||
return Vec::new();
|
return Vec::new();
|
||||||
}
|
}
|
||||||
|
|
||||||
let mut all_results: Vec<(usize, f32)> = vectors
|
// Chunk over index ranges: the corpus may be one flat buffer rather than
|
||||||
.par_chunks(chunk_size)
|
// a slice of rows, so there is nothing to `par_chunks` over.
|
||||||
.enumerate()
|
let n = vectors.count();
|
||||||
.flat_map(|(chunk_idx, chunk)| {
|
let mut all_results: Vec<(usize, f32)> = (0..n.div_ceil(chunk_size))
|
||||||
|
.into_par_iter()
|
||||||
|
.flat_map(|chunk_idx| {
|
||||||
let base = chunk_idx * chunk_size;
|
let base = chunk_idx * chunk_size;
|
||||||
let mut local: Vec<(usize, f32)> = Vec::with_capacity(chunk.len());
|
let end = (base + chunk_size).min(n);
|
||||||
for (j, vec) in chunk.iter().enumerate() {
|
let mut local: Vec<(usize, f32)> = Vec::with_capacity(end - base);
|
||||||
let i = base + j;
|
for i in base..end {
|
||||||
if i < tombstones.len() && tombstones[i] != 0 {
|
if i < tombstones.len() && tombstones[i] != 0 {
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
let vec_norm = clawhdf5_accel::vector_norm(vec);
|
let vec_norm = clawhdf5_accel::vector_norm(vectors.row(i));
|
||||||
let score = crate::cosine_similarity_prenorm(query, query_norm, vec, vec_norm);
|
let score =
|
||||||
|
crate::cosine_similarity_prenorm(query, query_norm, vectors.row(i), vec_norm);
|
||||||
local.push((i, score));
|
local.push((i, score));
|
||||||
}
|
}
|
||||||
local.sort_by(|a, b| b.1.partial_cmp(&a.1).unwrap_or(std::cmp::Ordering::Equal));
|
local.sort_by(|a, b| b.1.partial_cmp(&a.1).unwrap_or(std::cmp::Ordering::Equal));
|
||||||
@@ -209,7 +251,7 @@ pub fn parallel_cosine_batch(
|
|||||||
#[cfg(feature = "parallel")]
|
#[cfg(feature = "parallel")]
|
||||||
pub fn parallel_cosine_batch_prenorm(
|
pub fn parallel_cosine_batch_prenorm(
|
||||||
query: &[f32],
|
query: &[f32],
|
||||||
vectors: &[Vec<f32>],
|
vectors: &(impl VectorSet + Sync + ?Sized),
|
||||||
norms: &[f32],
|
norms: &[f32],
|
||||||
tombstones: &[u8],
|
tombstones: &[u8],
|
||||||
k: usize,
|
k: usize,
|
||||||
@@ -222,23 +264,26 @@ pub fn parallel_cosine_batch_prenorm(
|
|||||||
}
|
}
|
||||||
|
|
||||||
let num_cores = rayon::current_num_threads().max(1);
|
let num_cores = rayon::current_num_threads().max(1);
|
||||||
let chunk_size = vectors.len().div_ceil(num_cores);
|
let chunk_size = vectors.count().div_ceil(num_cores);
|
||||||
if chunk_size == 0 {
|
if chunk_size == 0 {
|
||||||
return Vec::new();
|
return Vec::new();
|
||||||
}
|
}
|
||||||
|
|
||||||
let mut all_results: Vec<(usize, f32)> = vectors
|
// Chunk over index ranges: the corpus may be one flat buffer rather than
|
||||||
.par_chunks(chunk_size)
|
// a slice of rows, so there is nothing to `par_chunks` over.
|
||||||
.enumerate()
|
let n = vectors.count();
|
||||||
.flat_map(|(chunk_idx, chunk)| {
|
let mut all_results: Vec<(usize, f32)> = (0..n.div_ceil(chunk_size))
|
||||||
|
.into_par_iter()
|
||||||
|
.flat_map(|chunk_idx| {
|
||||||
let base = chunk_idx * chunk_size;
|
let base = chunk_idx * chunk_size;
|
||||||
let mut local: Vec<(usize, f32)> = Vec::with_capacity(chunk.len());
|
let end = (base + chunk_size).min(n);
|
||||||
for (j, vec) in chunk.iter().enumerate() {
|
let mut local: Vec<(usize, f32)> = Vec::with_capacity(end - base);
|
||||||
let i = base + j;
|
for i in base..end {
|
||||||
if i < tombstones.len() && tombstones[i] != 0 {
|
if i < tombstones.len() && tombstones[i] != 0 {
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
let score = crate::cosine_similarity_prenorm(query, query_norm, vec, norms[i]);
|
let score =
|
||||||
|
crate::cosine_similarity_prenorm(query, query_norm, vectors.row(i), norms[i]);
|
||||||
local.push((i, score));
|
local.push((i, score));
|
||||||
}
|
}
|
||||||
local.sort_by(|a, b| b.1.partial_cmp(&a.1).unwrap_or(std::cmp::Ordering::Equal));
|
local.sort_by(|a, b| b.1.partial_cmp(&a.1).unwrap_or(std::cmp::Ordering::Equal));
|
||||||
|
|||||||
Binary file not shown.
@@ -0,0 +1,300 @@
|
|||||||
|
//! `MemoryConfig::float16`: embeddings stored as IEEE half precision.
|
||||||
|
//!
|
||||||
|
//! The setting used to be recorded in `/meta` and otherwise ignored — the
|
||||||
|
//! embeddings dataset was always `f32`. These tests pin what it now does: the
|
||||||
|
//! dataset is `float16`, the in-memory cache holds exactly the values the file
|
||||||
|
//! holds (so search results survive a reopen bit for bit), and a value half
|
||||||
|
//! precision cannot represent is refused rather than stored as infinity.
|
||||||
|
|
||||||
|
use std::path::{Path, PathBuf};
|
||||||
|
|
||||||
|
use clawhdf5_agent::{AgentMemory, HDF5Memory, MemoryConfig, MemoryEntry, MemoryError};
|
||||||
|
use clawhdf5_format::float16::round_to_f16;
|
||||||
|
use tempfile::TempDir;
|
||||||
|
|
||||||
|
const DIM: usize = 64;
|
||||||
|
|
||||||
|
/// Deterministic, embedding-like unit vectors.
|
||||||
|
fn embedding(seed: u64) -> Vec<f32> {
|
||||||
|
let mut x = seed.wrapping_mul(0x9E37_79B9_7F4A_7C15) | 1;
|
||||||
|
let v: Vec<f32> = (0..DIM)
|
||||||
|
.map(|_| {
|
||||||
|
x ^= x << 13;
|
||||||
|
x ^= x >> 7;
|
||||||
|
x ^= x << 17;
|
||||||
|
(x >> 40) as f32 / (1u64 << 24) as f32 - 0.5
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
let norm = v.iter().map(|a| a * a).sum::<f32>().sqrt();
|
||||||
|
v.iter().map(|a| a / norm).collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn entry(i: u64) -> MemoryEntry {
|
||||||
|
MemoryEntry {
|
||||||
|
chunk: format!("memory number {i} about topic {}", i % 7),
|
||||||
|
embedding: embedding(i),
|
||||||
|
source_channel: "test".into(),
|
||||||
|
timestamp: i as f64,
|
||||||
|
session_id: "s".into(),
|
||||||
|
tags: format!("t{i}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn config(dir: &TempDir, name: &str, float16: bool) -> MemoryConfig {
|
||||||
|
let mut c = MemoryConfig::new(dir.path().join(name), "agent", DIM);
|
||||||
|
c.float16 = float16;
|
||||||
|
c
|
||||||
|
}
|
||||||
|
|
||||||
|
fn embeddings_dtype_and_values(path: &Path) -> (String, Vec<f32>) {
|
||||||
|
let file = clawhdf5::File::open(path).unwrap();
|
||||||
|
let ds = file.dataset("memory/embeddings").unwrap();
|
||||||
|
(format!("{:?}", ds.dtype().unwrap()), ds.read_f32().unwrap())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn search_bits(m: &mut HDF5Memory, q: u64) -> Vec<(usize, u32)> {
|
||||||
|
m.hybrid_search(&embedding(q), "memory topic 3", 0.4, 0.6, 10)
|
||||||
|
.iter()
|
||||||
|
.map(|r| (r.index, r.score.to_bits()))
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn float16_store_writes_half_precision_and_reopens_identically() {
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
// Two identical stores. Search is not read-only (it boosts the Hebbian
|
||||||
|
// activation of what it returns, and checkpoints persist that), so each
|
||||||
|
// is queried exactly once: one live, one after a checkpoint and reopen.
|
||||||
|
let live_cfg = config(&dir, "live.h5", true);
|
||||||
|
let cfg = config(&dir, "f16.h5", true);
|
||||||
|
let path: PathBuf = cfg.path.clone();
|
||||||
|
|
||||||
|
let mut live = HDF5Memory::create(live_cfg).unwrap();
|
||||||
|
live.save_batch((0..200).map(entry).collect()).unwrap();
|
||||||
|
let mut m = HDF5Memory::create(cfg).unwrap();
|
||||||
|
m.save_batch((0..200).map(entry).collect()).unwrap();
|
||||||
|
drop(m);
|
||||||
|
|
||||||
|
// On disk: a genuine float16 dataset holding the rounded inputs.
|
||||||
|
let (dtype, values) = embeddings_dtype_and_values(&path);
|
||||||
|
assert_eq!(dtype, "Other(\"float16\")");
|
||||||
|
let expected: Vec<u32> = (0..200)
|
||||||
|
.flat_map(|i| embedding(i).into_iter().map(|v| round_to_f16(v).to_bits()))
|
||||||
|
.collect();
|
||||||
|
let got: Vec<u32> = values.iter().map(|v| v.to_bits()).collect();
|
||||||
|
assert_eq!(got, expected);
|
||||||
|
|
||||||
|
// Reopened, the store answers exactly as the live one does: the cache
|
||||||
|
// held the half-rounded values before the checkpoint.
|
||||||
|
let mut reopened = HDF5Memory::open(&path).unwrap();
|
||||||
|
for q in 0..5 {
|
||||||
|
assert_eq!(
|
||||||
|
search_bits(&mut live, 1000 + q),
|
||||||
|
search_bits(&mut reopened, 1000 + q),
|
||||||
|
"query {q}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn float16_halves_the_embeddings_on_disk() {
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let mut sizes = Vec::new();
|
||||||
|
for float16 in [false, true] {
|
||||||
|
let cfg = config(&dir, &format!("s{float16}.h5"), float16);
|
||||||
|
let path = cfg.path.clone();
|
||||||
|
let mut m = HDF5Memory::create(cfg).unwrap();
|
||||||
|
m.save_batch((0..2000).map(entry).collect()).unwrap();
|
||||||
|
drop(m);
|
||||||
|
sizes.push(std::fs::metadata(&path).unwrap().len());
|
||||||
|
}
|
||||||
|
let embedding_bytes_f32 = (2000 * DIM * 4) as u64;
|
||||||
|
let saved = sizes[0] - sizes[1];
|
||||||
|
// Half of the f32 embeddings, give or take metadata and alignment.
|
||||||
|
assert!(
|
||||||
|
saved.abs_diff(embedding_bytes_f32 / 2) < 16 * 1024,
|
||||||
|
"f32 {} B, f16 {} B, saved {saved} B, expected ~{} B",
|
||||||
|
sizes[0],
|
||||||
|
sizes[1],
|
||||||
|
embedding_bytes_f32 / 2
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn f32_store_is_unchanged() {
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let cfg = config(&dir, "f32.h5", false);
|
||||||
|
let path = cfg.path.clone();
|
||||||
|
let mut m = HDF5Memory::create(cfg).unwrap();
|
||||||
|
m.save_batch((0..50).map(entry).collect()).unwrap();
|
||||||
|
drop(m);
|
||||||
|
let (dtype, values) = embeddings_dtype_and_values(&path);
|
||||||
|
assert_eq!(dtype, "F32");
|
||||||
|
let expected: Vec<f32> = (0..50).flat_map(embedding).collect();
|
||||||
|
assert_eq!(values, expected);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn out_of_range_values_are_refused_not_stored_as_infinity() {
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let mut cfg = config(&dir, "range.h5", true);
|
||||||
|
cfg.wal_enabled = true;
|
||||||
|
let path = cfg.path.clone();
|
||||||
|
let mut m = HDF5Memory::create(cfg).unwrap();
|
||||||
|
m.save(entry(1)).unwrap();
|
||||||
|
|
||||||
|
let mut bad = entry(2);
|
||||||
|
bad.embedding[5] = 70_000.0;
|
||||||
|
match m.save(bad.clone()) {
|
||||||
|
Err(MemoryError::InvalidEntry(msg)) => assert!(msg.contains("embedding[5]"), "{msg}"),
|
||||||
|
other => panic!("expected InvalidEntry, got {other:?}"),
|
||||||
|
}
|
||||||
|
assert!(matches!(
|
||||||
|
m.save_or_update(bad.clone()),
|
||||||
|
Err(MemoryError::InvalidEntry(_))
|
||||||
|
));
|
||||||
|
// A batch is all or nothing.
|
||||||
|
assert!(matches!(
|
||||||
|
m.save_batch(vec![entry(3), bad.clone(), entry(4)]),
|
||||||
|
Err(MemoryError::InvalidEntry(_))
|
||||||
|
));
|
||||||
|
assert_eq!(m.count(), 1);
|
||||||
|
|
||||||
|
// The largest finite half, and values that round down to it, are fine.
|
||||||
|
let mut edge = entry(5);
|
||||||
|
edge.embedding[0] = 65504.0;
|
||||||
|
edge.embedding[1] = -65519.0;
|
||||||
|
m.save(edge).unwrap();
|
||||||
|
assert_eq!(m.count(), 2);
|
||||||
|
drop(m);
|
||||||
|
|
||||||
|
// Nothing rejected reached the WAL or the file.
|
||||||
|
let m = HDF5Memory::open(&path).unwrap();
|
||||||
|
assert_eq!(m.count(), 2);
|
||||||
|
|
||||||
|
// An f32 store takes the same value as it always did.
|
||||||
|
let mut m32 = HDF5Memory::create(config(&dir, "range32.h5", false)).unwrap();
|
||||||
|
m32.save(bad).unwrap();
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn wal_replay_rounds_like_a_live_save() {
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let mut cfg = config(&dir, "wal.h5", true);
|
||||||
|
cfg.wal_enabled = true;
|
||||||
|
cfg.wal_max_entries = 10_000; // keep everything in the WAL
|
||||||
|
let path = cfg.path.clone();
|
||||||
|
let mut m = HDF5Memory::create(cfg).unwrap();
|
||||||
|
for i in 0..30 {
|
||||||
|
m.save(entry(i)).unwrap();
|
||||||
|
}
|
||||||
|
let live = search_bits(&mut m, 77);
|
||||||
|
|
||||||
|
// Crash image: the .h5 is still the empty checkpoint; everything is in
|
||||||
|
// the WAL, which holds the caller's f32 values.
|
||||||
|
let crash = TempDir::new().unwrap();
|
||||||
|
let image = crash.path().join("image.h5");
|
||||||
|
std::fs::copy(&path, &image).unwrap();
|
||||||
|
std::fs::copy(
|
||||||
|
path.with_extension("h5.wal"),
|
||||||
|
image.with_extension("h5.wal"),
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
drop(m);
|
||||||
|
|
||||||
|
let mut recovered = HDF5Memory::open(&image).unwrap();
|
||||||
|
assert_eq!(recovered.count(), 30);
|
||||||
|
assert_eq!(search_bits(&mut recovered, 77), live);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn new_stores_default_to_float16() {
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let path = dir.path().join("default.h5");
|
||||||
|
let mut m = HDF5Memory::create(MemoryConfig::new(path.clone(), "agent", DIM)).unwrap();
|
||||||
|
assert!(m.config().float16);
|
||||||
|
m.save_batch((0..10).map(entry).collect()).unwrap();
|
||||||
|
drop(m);
|
||||||
|
assert_eq!(embeddings_dtype_and_values(&path).0, "Other(\"float16\")");
|
||||||
|
assert!(HDF5Memory::open(&path).unwrap().config().float16);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn an_existing_f32_store_stays_f32() {
|
||||||
|
// Written by the v2.5.0 CLI, with `float16 = 0` in /meta (every agent
|
||||||
|
// store has recorded it). Flipping the default for new stores must not
|
||||||
|
// reach back and round an existing store's embeddings.
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let path = dir.path().join("legacy.h5");
|
||||||
|
std::fs::copy(
|
||||||
|
concat!(
|
||||||
|
env!("CARGO_MANIFEST_DIR"),
|
||||||
|
"/tests/fixtures/store_v2_5_0.h5"
|
||||||
|
),
|
||||||
|
&path,
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
let before = embeddings_dtype_and_values(&path);
|
||||||
|
assert_eq!(before.0, "F32");
|
||||||
|
|
||||||
|
let mut m = HDF5Memory::open(&path).unwrap();
|
||||||
|
assert!(!m.config().float16, "an old store must reopen as f32");
|
||||||
|
let dim = m.config().embedding_dim;
|
||||||
|
let odd: Vec<f32> = (0..dim).map(|i| 0.1 + i as f32 * 1e-4).collect();
|
||||||
|
m.save_batch(vec![MemoryEntry {
|
||||||
|
chunk: "added after the upgrade".into(),
|
||||||
|
embedding: odd.clone(),
|
||||||
|
source_channel: "test".into(),
|
||||||
|
timestamp: 1.0,
|
||||||
|
session_id: "s".into(),
|
||||||
|
tags: String::new(),
|
||||||
|
}])
|
||||||
|
.unwrap();
|
||||||
|
drop(m);
|
||||||
|
|
||||||
|
// Checkpointed: still f32, the old rows untouched and the new one exact.
|
||||||
|
let (dtype, values) = embeddings_dtype_and_values(&path);
|
||||||
|
assert_eq!(dtype, "F32");
|
||||||
|
assert_eq!(&values[..before.1.len()], before.1.as_slice());
|
||||||
|
assert_eq!(&values[before.1.len()..], odd.as_slice());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `Group::attrs` leaves out an attribute it cannot decode. A store whose
|
||||||
|
/// `float16` setting is unreadable must not open as `float16 = false` (or with
|
||||||
|
/// any other default in place of a setting it has): it is an error.
|
||||||
|
#[test]
|
||||||
|
fn unreadable_meta_attribute_fails_open_instead_of_defaulting() {
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let path = dir.path().join("store.h5");
|
||||||
|
{
|
||||||
|
let mut m = HDF5Memory::create(config(&dir, "store.h5", true)).unwrap();
|
||||||
|
m.save(entry(1)).unwrap();
|
||||||
|
m.flush_wal().unwrap();
|
||||||
|
}
|
||||||
|
assert!(HDF5Memory::open_read_only(&path).is_ok());
|
||||||
|
|
||||||
|
// Give the `float16` attribute message an unknown version (the name is
|
||||||
|
// at +8 in a version-1 message and +9 in a version-3 one).
|
||||||
|
let mut bytes = std::fs::read(&path).unwrap();
|
||||||
|
let name = b"float16\0";
|
||||||
|
let mut hit = false;
|
||||||
|
let positions: Vec<usize> = (9..bytes.len() - name.len())
|
||||||
|
.filter(|&p| &bytes[p..p + name.len()] == name)
|
||||||
|
.collect();
|
||||||
|
for pos in positions {
|
||||||
|
for (back, version) in [(8, 1u8), (9, 3u8)] {
|
||||||
|
if bytes[pos - back] == version {
|
||||||
|
bytes[pos - back] = 0x7f;
|
||||||
|
hit = true;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
assert!(hit, "float16 attribute message not found");
|
||||||
|
std::fs::write(&path, &bytes).unwrap();
|
||||||
|
|
||||||
|
match HDF5Memory::open_read_only(&path) {
|
||||||
|
Err(MemoryError::Schema(msg)) => assert!(msg.contains("/meta"), "{msg}"),
|
||||||
|
Err(e) => panic!("unexpected error: {e}"),
|
||||||
|
Ok(_) => panic!("store opened with an unreadable float16 setting"),
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,155 @@
|
|||||||
|
//! An agent store is a standard HDF5 file: h5py can open it and read every
|
||||||
|
//! dataset.
|
||||||
|
//!
|
||||||
|
//! It could not: the float datatype's sign-bit position was hard-coded for
|
||||||
|
//! f64, so every f32 dataset (embeddings, norms, activation weights) made
|
||||||
|
//! libhdf5 refuse the file with "sign bit position out of bounds".
|
||||||
|
|
||||||
|
use std::process::Command;
|
||||||
|
|
||||||
|
use clawhdf5_agent::{AgentMemory, HDF5Memory, MemoryConfig, MemoryEntry};
|
||||||
|
|
||||||
|
fn python() -> String {
|
||||||
|
std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn h5py_available() -> bool {
|
||||||
|
Command::new(python())
|
||||||
|
.args(["-c", "import h5py"])
|
||||||
|
.output()
|
||||||
|
.map(|o| o.status.success())
|
||||||
|
.unwrap_or(false)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn h5py_reads_every_dataset_of_an_agent_store() {
|
||||||
|
if !h5py_available() {
|
||||||
|
assert!(
|
||||||
|
std::env::var("CLAWHDF5_REQUIRE_INTEROP").as_deref() != Ok("1"),
|
||||||
|
"CLAWHDF5_REQUIRE_INTEROP=1 but python3 with h5py is not available"
|
||||||
|
);
|
||||||
|
eprintln!("SKIP: python3 with h5py not available");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
let dir = tempfile::tempdir().unwrap();
|
||||||
|
for float16 in [false, true] {
|
||||||
|
let path = dir.path().join(format!("store_{float16}.h5"));
|
||||||
|
let mut cfg = MemoryConfig::new(path.clone(), "agent", 8);
|
||||||
|
cfg.float16 = float16;
|
||||||
|
let mut m = HDF5Memory::create(cfg).unwrap();
|
||||||
|
// save_batch checkpoints, so the records are in the .h5, not the WAL.
|
||||||
|
m.save_batch(
|
||||||
|
(0..20)
|
||||||
|
.map(|i| MemoryEntry {
|
||||||
|
chunk: format!("memory {i}"),
|
||||||
|
embedding: (0..8).map(|j| ((i * 8 + j) as f32).sin()).collect(),
|
||||||
|
source_channel: "test".into(),
|
||||||
|
timestamp: i as f64,
|
||||||
|
session_id: "s".into(),
|
||||||
|
tags: String::new(),
|
||||||
|
})
|
||||||
|
.collect(),
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
drop(m);
|
||||||
|
|
||||||
|
// Exact expected values, as bits: numpy's sin need not match Rust's
|
||||||
|
// to the last place.
|
||||||
|
let bits = (0..160)
|
||||||
|
.map(|k| (k as f32).sin().to_bits().to_string())
|
||||||
|
.collect::<Vec<_>>()
|
||||||
|
.join(",");
|
||||||
|
let script = format!(
|
||||||
|
r#"
|
||||||
|
import h5py, numpy as np
|
||||||
|
want = np.float16 if {py_bool} else np.float32
|
||||||
|
with h5py.File("{path}", "r") as f:
|
||||||
|
names = []
|
||||||
|
f.visititems(lambda n, o: names.append(n) if isinstance(o, h5py.Dataset) else None)
|
||||||
|
for n in names:
|
||||||
|
f[n][()] # every dataset must decode
|
||||||
|
e = f["memory/embeddings"]
|
||||||
|
assert e.dtype == want, e.dtype
|
||||||
|
assert e.shape == (20, 8), e.shape
|
||||||
|
ref = np.array([{bits}], dtype=np.uint32).view(np.float32).astype(want).reshape(20, 8)
|
||||||
|
assert (e[()] == ref).all()
|
||||||
|
assert f["memory/norms"].dtype == np.float32
|
||||||
|
print(len(names))
|
||||||
|
"#,
|
||||||
|
py_bool = if float16 { "True" } else { "False" },
|
||||||
|
path = path.display()
|
||||||
|
);
|
||||||
|
let out = Command::new(python())
|
||||||
|
.args(["-c", &script])
|
||||||
|
.output()
|
||||||
|
.unwrap();
|
||||||
|
assert!(
|
||||||
|
out.status.success(),
|
||||||
|
"float16={float16}: {}",
|
||||||
|
String::from_utf8_lossy(&out.stderr)
|
||||||
|
);
|
||||||
|
let n: usize = String::from_utf8_lossy(&out.stdout).trim().parse().unwrap();
|
||||||
|
assert!(n >= 10, "only {n} datasets");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn an_edit_made_with_h5py_breaks_the_signature_and_names_the_record() {
|
||||||
|
if !h5py_available() {
|
||||||
|
assert!(
|
||||||
|
std::env::var("CLAWHDF5_REQUIRE_INTEROP").as_deref() != Ok("1"),
|
||||||
|
"CLAWHDF5_REQUIRE_INTEROP=1 but python3 with h5py is not available"
|
||||||
|
);
|
||||||
|
eprintln!("SKIP: python3 with h5py not available");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
use clawhdf5_agent::signing::SigningKey;
|
||||||
|
let dir = tempfile::tempdir().unwrap();
|
||||||
|
let path = dir.path().join("signed.h5");
|
||||||
|
let key = SigningKey::from_bytes(&[42; 32]);
|
||||||
|
let mut m = HDF5Memory::create(MemoryConfig::new(path.clone(), "agent", 8)).unwrap();
|
||||||
|
m.set_signing_key(key.clone());
|
||||||
|
m.save_batch(
|
||||||
|
(0..10)
|
||||||
|
.map(|i| MemoryEntry {
|
||||||
|
chunk: format!("memory {i}"),
|
||||||
|
embedding: (0..8).map(|j| ((i * 8 + j) as f32).cos()).collect(),
|
||||||
|
source_channel: "test".into(),
|
||||||
|
timestamp: i as f64,
|
||||||
|
session_id: "s".into(),
|
||||||
|
tags: String::new(),
|
||||||
|
})
|
||||||
|
.collect(),
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
drop(m);
|
||||||
|
assert!(
|
||||||
|
HDF5Memory::verify(&path, &key.verifying_key())
|
||||||
|
.unwrap()
|
||||||
|
.is_valid()
|
||||||
|
);
|
||||||
|
|
||||||
|
// Someone edits one timestamp in place with h5py.
|
||||||
|
let script = format!(
|
||||||
|
r#"
|
||||||
|
import h5py
|
||||||
|
with h5py.File("{}", "r+") as f:
|
||||||
|
ts = f["memory/timestamps"]
|
||||||
|
ts[3] = 12345.0
|
||||||
|
"#,
|
||||||
|
path.display()
|
||||||
|
);
|
||||||
|
let out = Command::new(python())
|
||||||
|
.args(["-c", &script])
|
||||||
|
.output()
|
||||||
|
.unwrap();
|
||||||
|
assert!(
|
||||||
|
out.status.success(),
|
||||||
|
"{}",
|
||||||
|
String::from_utf8_lossy(&out.stderr)
|
||||||
|
);
|
||||||
|
|
||||||
|
let r = HDF5Memory::verify(&path, &key.verifying_key()).unwrap();
|
||||||
|
assert!(r.signature_valid && !r.is_valid(), "{r:?}");
|
||||||
|
assert_eq!(r.changed_records, vec![3]);
|
||||||
|
}
|
||||||
@@ -165,3 +165,182 @@ fn save_batch_then_search_is_consistent() {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn quantized_index_matches_the_f32_index_after_re_scoring() {
|
||||||
|
// A quantised index holds approximate vectors, but the store still has the
|
||||||
|
// exact ones, so the query path re-scores the candidate pool before
|
||||||
|
// fusion. The results a caller sees should therefore be the same.
|
||||||
|
let dim = 64;
|
||||||
|
let n = 400;
|
||||||
|
let mut seed = 0x5EED_1234_5678_9ABC;
|
||||||
|
let vectors: Vec<Vec<f32>> = (0..n).map(|_| make_vector(&mut seed, dim)).collect();
|
||||||
|
let queries: Vec<Vec<f32>> = (0..20).map(|_| make_vector(&mut seed, dim)).collect();
|
||||||
|
|
||||||
|
let build = |dir: &TempDir, quantized: bool| {
|
||||||
|
let mut config = MemoryConfig::new(dir.path().join("mem.h5"), "agent", dim);
|
||||||
|
config.quantized_index = quantized;
|
||||||
|
let mut mem = HDF5Memory::create(config).unwrap();
|
||||||
|
for (i, v) in vectors.iter().enumerate() {
|
||||||
|
mem.save(entry(&format!("chunk {i}"), v.clone(), &format!("k{i}")))
|
||||||
|
.unwrap();
|
||||||
|
}
|
||||||
|
mem
|
||||||
|
};
|
||||||
|
|
||||||
|
let exact_dir = TempDir::new().unwrap();
|
||||||
|
let quant_dir = TempDir::new().unwrap();
|
||||||
|
let mut exact = build(&exact_dir, false);
|
||||||
|
let mut quantized = build(&quant_dir, true);
|
||||||
|
|
||||||
|
let k = 10;
|
||||||
|
let mut agree = 0;
|
||||||
|
for q in &queries {
|
||||||
|
let want: Vec<usize> = exact
|
||||||
|
.hybrid_search(q, "", 1.0, 0.0, k)
|
||||||
|
.iter()
|
||||||
|
.map(|r| r.index)
|
||||||
|
.collect();
|
||||||
|
agree += quantized
|
||||||
|
.hybrid_search(q, "", 1.0, 0.0, k)
|
||||||
|
.iter()
|
||||||
|
.filter(|r| want.contains(&r.index))
|
||||||
|
.count();
|
||||||
|
}
|
||||||
|
let overlap = agree as f64 / (k * queries.len()) as f64;
|
||||||
|
assert!(
|
||||||
|
overlap >= 0.95,
|
||||||
|
"quantised store should match the f32 one: {overlap}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn quantized_index_setting_survives_a_reopen() {
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let path = dir.path().join("mem.h5");
|
||||||
|
let mut config = MemoryConfig::new(path.clone(), "agent", 8);
|
||||||
|
config.quantized_index = true;
|
||||||
|
let mut mem = HDF5Memory::create(config).unwrap();
|
||||||
|
let mut seed = 7;
|
||||||
|
for i in 0..30 {
|
||||||
|
mem.save(entry(&format!("c{i}"), make_vector(&mut seed, 8), "t"))
|
||||||
|
.unwrap();
|
||||||
|
}
|
||||||
|
mem.flush_wal().unwrap();
|
||||||
|
drop(mem);
|
||||||
|
|
||||||
|
// Reopening must not silently quadruple the index's memory, so the flag
|
||||||
|
// is part of the stored config rather than a per-session choice.
|
||||||
|
let reopened = HDF5Memory::open(&path).unwrap();
|
||||||
|
assert!(reopened.config().quantized_index);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn hnsw_parameters_are_configurable_and_persisted() {
|
||||||
|
// The graph degree and both candidate-list sizes used to be constants, so
|
||||||
|
// a deployment could not trade recall against memory or speed at all.
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let path = dir.path().join("mem.h5");
|
||||||
|
let mut config = MemoryConfig::new(path.clone(), "agent", 16);
|
||||||
|
config.hnsw_m = 8;
|
||||||
|
config.hnsw_ef_construction = 32;
|
||||||
|
config.hnsw_ef_search = 128;
|
||||||
|
let mut mem = HDF5Memory::create(config).unwrap();
|
||||||
|
|
||||||
|
let mut seed = 99;
|
||||||
|
let vectors: Vec<Vec<f32>> = (0..300).map(|_| make_vector(&mut seed, 16)).collect();
|
||||||
|
for (i, v) in vectors.iter().enumerate() {
|
||||||
|
mem.save(entry(&format!("c{i}"), v.clone(), "t")).unwrap();
|
||||||
|
}
|
||||||
|
// Still correct with a smaller graph: an exact match must rank first.
|
||||||
|
let top = mem.hybrid_search(&vectors[42], "", 1.0, 0.0, 1);
|
||||||
|
assert_eq!(top[0].index, 42);
|
||||||
|
|
||||||
|
mem.flush_wal().unwrap();
|
||||||
|
drop(mem);
|
||||||
|
let reopened = HDF5Memory::open(&path).unwrap();
|
||||||
|
assert_eq!(reopened.config().hnsw_m, 8);
|
||||||
|
assert_eq!(reopened.config().hnsw_ef_construction, 32);
|
||||||
|
assert_eq!(reopened.config().hnsw_ef_search, 128);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn degenerate_hnsw_parameters_do_not_panic() {
|
||||||
|
// `clawhdf5-ann` asserts m >= 2, so a zero from a config file — or from a
|
||||||
|
// caller who assumed 0 meant "default" — would abort the process inside
|
||||||
|
// the index builder. The store clamps instead.
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let mut config = MemoryConfig::new(dir.path().join("mem.h5"), "agent", 8);
|
||||||
|
config.hnsw_m = 0;
|
||||||
|
config.hnsw_ef_construction = 0;
|
||||||
|
config.hnsw_ef_search = 1;
|
||||||
|
let mut mem = HDF5Memory::create(config).unwrap();
|
||||||
|
|
||||||
|
let mut seed = 5;
|
||||||
|
let vectors: Vec<Vec<f32>> = (0..50).map(|_| make_vector(&mut seed, 8)).collect();
|
||||||
|
for (i, v) in vectors.iter().enumerate() {
|
||||||
|
mem.save(entry(&format!("c{i}"), v.clone(), "t")).unwrap();
|
||||||
|
}
|
||||||
|
let results = mem.hybrid_search(&vectors[7], "", 1.0, 0.0, 5);
|
||||||
|
assert_eq!(results[0].index, 7, "exact match should still rank first");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn new_stores_default_to_the_quantized_index() {
|
||||||
|
// int8 is the default because it is smaller and, with an exact re-score,
|
||||||
|
// faster at equal recall on every platform measured (see BENCHMARKS.md).
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let config = MemoryConfig::new(dir.path().join("mem.h5"), "agent", 8);
|
||||||
|
assert!(config.quantized_index);
|
||||||
|
|
||||||
|
let path = config.path.clone();
|
||||||
|
let mut mem = HDF5Memory::create(config).unwrap();
|
||||||
|
let mut seed = 3;
|
||||||
|
let vectors: Vec<Vec<f32>> = (0..40).map(|_| make_vector(&mut seed, 8)).collect();
|
||||||
|
for (i, v) in vectors.iter().enumerate() {
|
||||||
|
mem.save(entry(&format!("c{i}"), v.clone(), "t")).unwrap();
|
||||||
|
}
|
||||||
|
assert_eq!(
|
||||||
|
mem.hybrid_search(&vectors[11], "", 1.0, 0.0, 1)[0].index,
|
||||||
|
11
|
||||||
|
);
|
||||||
|
mem.flush_wal().unwrap();
|
||||||
|
drop(mem);
|
||||||
|
assert!(HDF5Memory::open(&path).unwrap().config().quantized_index);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn a_store_written_before_the_setting_existed_stays_f32() {
|
||||||
|
// `store_v2_5_0.h5` was written by the v2.5.0 CLI, before
|
||||||
|
// `quantized_index` or the HNSW parameters were persisted, so it carries
|
||||||
|
// none of them. Flipping the default for new stores must not reach back
|
||||||
|
// and change how an existing store's index is held.
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let path = dir.path().join("legacy.h5");
|
||||||
|
std::fs::copy(
|
||||||
|
concat!(
|
||||||
|
env!("CARGO_MANIFEST_DIR"),
|
||||||
|
"/tests/fixtures/store_v2_5_0.h5"
|
||||||
|
),
|
||||||
|
&path,
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
|
let bytes = std::fs::read(&path).unwrap();
|
||||||
|
assert!(
|
||||||
|
!bytes.windows(15).any(|w| w == b"quantized_index"),
|
||||||
|
"the fixture must predate the setting, or it tests nothing"
|
||||||
|
);
|
||||||
|
|
||||||
|
let mut mem = HDF5Memory::open(&path).unwrap();
|
||||||
|
assert!(
|
||||||
|
!mem.config().quantized_index,
|
||||||
|
"an old store must reopen with an f32 index"
|
||||||
|
);
|
||||||
|
assert_eq!(mem.config().hnsw_m, 16);
|
||||||
|
assert_eq!(mem.config().hnsw_ef_construction, 64);
|
||||||
|
assert_eq!(mem.count(), 6);
|
||||||
|
// And it still searches: entry 3's own embedding finds it first.
|
||||||
|
let hit = mem.hybrid_search(&[3.0, 1.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0], "", 1.0, 0.0, 1);
|
||||||
|
assert_eq!(hit[0].index, 3);
|
||||||
|
}
|
||||||
|
|||||||
@@ -0,0 +1,344 @@
|
|||||||
|
//! `HDF5Memory::search` with `SearchOptions`: source filtering, re-ranking and
|
||||||
|
//! confidence rejection in the store's own search path.
|
||||||
|
|
||||||
|
use std::collections::HashSet;
|
||||||
|
|
||||||
|
use clawhdf5_agent::confidence::ConfidenceConfig;
|
||||||
|
use clawhdf5_agent::reranker::ReRankConfig;
|
||||||
|
use clawhdf5_agent::{AgentMemory, HDF5Memory, MemoryConfig, MemoryEntry, SearchOptions, hybrid};
|
||||||
|
use tempfile::TempDir;
|
||||||
|
|
||||||
|
const DIM: usize = 32;
|
||||||
|
const N: usize = 3000;
|
||||||
|
const CLUSTERS: usize = 20;
|
||||||
|
|
||||||
|
struct Rng(u64);
|
||||||
|
impl Rng {
|
||||||
|
fn next(&mut self) -> u64 {
|
||||||
|
self.0 = self.0.wrapping_add(0x9E37_79B9_7F4A_7C15);
|
||||||
|
let mut z = self.0;
|
||||||
|
z = (z ^ (z >> 30)).wrapping_mul(0xBF58_476D_1CE4_E5B9);
|
||||||
|
z = (z ^ (z >> 27)).wrapping_mul(0x94D0_49BB_1331_11EB);
|
||||||
|
z ^ (z >> 31)
|
||||||
|
}
|
||||||
|
fn unit(&mut self) -> f32 {
|
||||||
|
(self.next() >> 40) as f32 / (1u64 << 24) as f32 - 0.5
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn normalize(v: &mut [f32]) {
|
||||||
|
let n = v.iter().map(|x| x * x).sum::<f32>().sqrt();
|
||||||
|
v.iter_mut().for_each(|x| *x /= n);
|
||||||
|
}
|
||||||
|
|
||||||
|
struct Data {
|
||||||
|
vectors: Vec<Vec<f32>>,
|
||||||
|
cluster: Vec<usize>,
|
||||||
|
centres: Vec<Vec<f32>>,
|
||||||
|
}
|
||||||
|
|
||||||
|
fn data() -> Data {
|
||||||
|
let mut rng = Rng(42);
|
||||||
|
let centres: Vec<Vec<f32>> = (0..CLUSTERS)
|
||||||
|
.map(|_| {
|
||||||
|
let mut c: Vec<f32> = (0..DIM).map(|_| rng.unit()).collect();
|
||||||
|
normalize(&mut c);
|
||||||
|
c
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
let mut vectors = Vec::new();
|
||||||
|
let mut cluster = Vec::new();
|
||||||
|
for i in 0..N {
|
||||||
|
let c = i % CLUSTERS;
|
||||||
|
let mut v: Vec<f32> = centres[c].iter().map(|x| x + rng.unit() * 0.3).collect();
|
||||||
|
normalize(&mut v);
|
||||||
|
vectors.push(v);
|
||||||
|
cluster.push(c);
|
||||||
|
}
|
||||||
|
Data {
|
||||||
|
vectors,
|
||||||
|
cluster,
|
||||||
|
centres,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Channel of record `i` for a filter keeping `percent`% of the store at
|
||||||
|
/// random (independent of the vectors).
|
||||||
|
fn random_channel(i: usize, rng_seed: u64, percent: u64) -> String {
|
||||||
|
let mut r = Rng(rng_seed ^ (i as u64 * 7919));
|
||||||
|
if r.next() % 100 < percent {
|
||||||
|
"keep".into()
|
||||||
|
} else {
|
||||||
|
"other".into()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn build(data: &Data, channel: impl Fn(usize) -> String) -> (TempDir, HDF5Memory) {
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let mut cfg = MemoryConfig::new(dir.path().join("s.h5"), "agent", DIM);
|
||||||
|
cfg.hebbian_boost = 0.0; // every query sees the same store
|
||||||
|
let mut m = HDF5Memory::create(cfg).unwrap();
|
||||||
|
let entries = data
|
||||||
|
.vectors
|
||||||
|
.iter()
|
||||||
|
.enumerate()
|
||||||
|
.map(|(i, v)| MemoryEntry {
|
||||||
|
chunk: format!("record {i} cluster {}", data.cluster[i]),
|
||||||
|
embedding: v.clone(),
|
||||||
|
source_channel: channel(i),
|
||||||
|
timestamp: i as f64,
|
||||||
|
session_id: "s".into(),
|
||||||
|
tags: format!("t{i}"),
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
m.save_batch(entries).unwrap();
|
||||||
|
(dir, m)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Exact top-k by cosine among the records `allowed` keeps.
|
||||||
|
fn exact_top(data: &Data, q: &[f32], k: usize, allowed: impl Fn(usize) -> bool) -> Vec<usize> {
|
||||||
|
let mut s: Vec<(usize, f32)> = (0..N)
|
||||||
|
.filter(|&i| allowed(i))
|
||||||
|
.map(|i| (i, data.vectors[i].iter().zip(q).map(|(a, b)| a * b).sum()))
|
||||||
|
.collect();
|
||||||
|
s.sort_by(|a, b| b.1.total_cmp(&a.1).then(a.0.cmp(&b.0)));
|
||||||
|
s.into_iter().take(k).map(|(i, _)| i).collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn query(data: &Data, i: usize) -> Vec<f32> {
|
||||||
|
let mut rng = Rng(1000 + i as u64);
|
||||||
|
let mut q: Vec<f32> = data.centres[i % CLUSTERS]
|
||||||
|
.iter()
|
||||||
|
.map(|x| x + rng.unit() * 0.3)
|
||||||
|
.collect();
|
||||||
|
normalize(&mut q);
|
||||||
|
q
|
||||||
|
}
|
||||||
|
|
||||||
|
fn vector_only(k: usize) -> SearchOptions {
|
||||||
|
SearchOptions::new(k).with_fusion(hybrid::Fusion::Weighted {
|
||||||
|
vector: 1.0,
|
||||||
|
keyword: 0.0,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn source_filter_returns_only_allowed_records_and_a_full_page() {
|
||||||
|
let d = data();
|
||||||
|
// At N = 3000 and k = 10 the index serves a filter only when that is
|
||||||
|
// cheaper than scanning the allowed records: pool = 80 * N / allowed
|
||||||
|
// candidates at ~M = 16 distances each, against `allowed` distances. So
|
||||||
|
// 90% goes through the index, 50% and 1% to the exact scan.
|
||||||
|
for percent in [90, 50, 1] {
|
||||||
|
let (_dir, mut m) = build(&d, |i| random_channel(i, 5, percent));
|
||||||
|
let allowed = |i: usize| random_channel(i, 5, percent) == "keep";
|
||||||
|
let mut hits = 0;
|
||||||
|
for qi in 0..40 {
|
||||||
|
let q = query(&d, qi);
|
||||||
|
let got = m.search(&q, "", &vector_only(10).with_sources(["keep"]));
|
||||||
|
assert_eq!(got.len(), 10, "{percent}%: short page");
|
||||||
|
assert!(got.iter().all(|r| r.source_channel == "keep"));
|
||||||
|
let want: HashSet<usize> = exact_top(&d, &q, 10, allowed).into_iter().collect();
|
||||||
|
hits += got.iter().filter(|r| want.contains(&r.index)).count();
|
||||||
|
}
|
||||||
|
let recall = hits as f64 / 400.0;
|
||||||
|
let floor = if percent == 90 { 0.95 } else { 1.0 };
|
||||||
|
assert!(recall >= floor, "{percent}%: recall@10 {recall}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn filter_away_from_the_query_falls_back_to_an_exact_scan() {
|
||||||
|
// Channel = cluster, and the filter keeps two clusters (10% of the
|
||||||
|
// store) that are not the query's: the index's neighbourhood of the
|
||||||
|
// query holds none of them. The search must still return the exact
|
||||||
|
// top 10 among the allowed records, not a short or empty page.
|
||||||
|
let d = data();
|
||||||
|
let (_dir, mut m) = build(&d, |i| format!("c{}", d.cluster[i]));
|
||||||
|
for qi in 0..20 {
|
||||||
|
let q = query(&d, qi);
|
||||||
|
let a = format!("c{}", (qi + 7) % CLUSTERS);
|
||||||
|
let b = format!("c{}", (qi + 13) % CLUSTERS);
|
||||||
|
let got: Vec<usize> = m
|
||||||
|
.search(
|
||||||
|
&q,
|
||||||
|
"",
|
||||||
|
&vector_only(10).with_sources([a.clone(), b.clone()]),
|
||||||
|
)
|
||||||
|
.iter()
|
||||||
|
.map(|r| r.index)
|
||||||
|
.collect();
|
||||||
|
let want = exact_top(&d, &q, 10, |i| {
|
||||||
|
let c = format!("c{}", d.cluster[i]);
|
||||||
|
c == a || c == b
|
||||||
|
});
|
||||||
|
assert_eq!(got, want, "query {qi}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn filter_edge_cases() {
|
||||||
|
let d = data();
|
||||||
|
let (_dir, mut m) = build(&d, |i| random_channel(i, 9, 50));
|
||||||
|
let q = query(&d, 0);
|
||||||
|
assert!(
|
||||||
|
m.search(
|
||||||
|
&q,
|
||||||
|
"cluster",
|
||||||
|
&SearchOptions::new(10).with_sources(Vec::<String>::new())
|
||||||
|
)
|
||||||
|
.is_empty()
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
m.search(
|
||||||
|
&q,
|
||||||
|
"cluster",
|
||||||
|
&SearchOptions::new(10).with_sources(["nope"])
|
||||||
|
)
|
||||||
|
.is_empty()
|
||||||
|
);
|
||||||
|
// Keyword matches from other channels are filtered too.
|
||||||
|
let got = m.search(
|
||||||
|
&q,
|
||||||
|
"record cluster",
|
||||||
|
&SearchOptions::new(50).with_sources(["keep"]),
|
||||||
|
);
|
||||||
|
assert_eq!(got.len(), 50);
|
||||||
|
assert!(got.iter().all(|r| r.source_channel == "keep"));
|
||||||
|
// Deleted records never come back, filtered or not.
|
||||||
|
let first = got[0].index;
|
||||||
|
m.delete(first).unwrap();
|
||||||
|
let again = m.search(
|
||||||
|
&q,
|
||||||
|
"record cluster",
|
||||||
|
&SearchOptions::new(50).with_sources(["keep"]),
|
||||||
|
);
|
||||||
|
assert!(again.iter().all(|r| r.index != first));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn plain_options_equal_hybrid_search_with() {
|
||||||
|
// Two identical stores, so neither query sees the other's boosts.
|
||||||
|
let d = data();
|
||||||
|
let (_a, mut a) = build(&d, |i| random_channel(i, 3, 50));
|
||||||
|
let (_b, mut b) = build(&d, |i| random_channel(i, 3, 50));
|
||||||
|
for qi in 0..10 {
|
||||||
|
let q = query(&d, qi);
|
||||||
|
let x: Vec<(usize, u32)> = a
|
||||||
|
.search(&q, "record cluster 3", &SearchOptions::new(10))
|
||||||
|
.iter()
|
||||||
|
.map(|r| (r.index, r.score.to_bits()))
|
||||||
|
.collect();
|
||||||
|
let y: Vec<(usize, u32)> = b
|
||||||
|
.hybrid_search_with(&q, "record cluster 3", hybrid::DEFAULT_FUSION, 10)
|
||||||
|
.iter()
|
||||||
|
.map(|r| (r.index, r.score.to_bits()))
|
||||||
|
.collect();
|
||||||
|
assert_eq!(x, y);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn small_store(entries: &[(&str, &str, f64)]) -> (TempDir, HDF5Memory) {
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let mut m = HDF5Memory::create(MemoryConfig::new(dir.path().join("r.h5"), "a", 4)).unwrap();
|
||||||
|
m.save_batch(
|
||||||
|
entries
|
||||||
|
.iter()
|
||||||
|
.map(|(chunk, channel, ts)| MemoryEntry {
|
||||||
|
chunk: chunk.to_string(),
|
||||||
|
embedding: vec![1.0, 0.0, 0.0, 0.0],
|
||||||
|
source_channel: channel.to_string(),
|
||||||
|
timestamp: *ts,
|
||||||
|
session_id: "s".into(),
|
||||||
|
tags: String::new(),
|
||||||
|
})
|
||||||
|
.collect(),
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
(dir, m)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn rerank_breaks_relevance_ties_by_recency() {
|
||||||
|
// Identical text and vectors, so retrieval ties; re-ranking must put the
|
||||||
|
// newer record first and report the combined score.
|
||||||
|
let now = 1_000_000.0;
|
||||||
|
let (_d, mut m) = small_store(&[
|
||||||
|
("user prefers dark mode", "chat", now - 30.0 * 86_400.0),
|
||||||
|
("user prefers dark mode", "chat", now - 60.0),
|
||||||
|
]);
|
||||||
|
let q = [1.0, 0.0, 0.0, 0.0];
|
||||||
|
let plain = m.search(&q, "dark mode", &SearchOptions::new(2));
|
||||||
|
assert_eq!(plain[0].index, 0, "ties break by index without re-ranking");
|
||||||
|
let reranked = m.search(
|
||||||
|
&q,
|
||||||
|
"dark mode",
|
||||||
|
&SearchOptions::new(2)
|
||||||
|
.with_rerank(ReRankConfig::default())
|
||||||
|
.at_time(now),
|
||||||
|
);
|
||||||
|
assert_eq!(reranked[0].index, 1);
|
||||||
|
assert!(reranked[0].score > reranked[1].score);
|
||||||
|
assert_ne!(reranked[0].score.to_bits(), plain[0].score.to_bits());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn confidence_rejects_when_nothing_is_good_enough() {
|
||||||
|
let (_d, mut m) = small_store(&[("alpha", "chat", 0.0), ("beta", "chat", 0.0)]);
|
||||||
|
let q = [1.0, 0.0, 0.0, 0.0];
|
||||||
|
let strict = ConfidenceConfig {
|
||||||
|
min_score: 10.0,
|
||||||
|
..ConfidenceConfig::default()
|
||||||
|
};
|
||||||
|
assert!(
|
||||||
|
m.search(&q, "alpha", &SearchOptions::new(2).with_confidence(strict))
|
||||||
|
.is_empty()
|
||||||
|
);
|
||||||
|
let lenient = ConfidenceConfig {
|
||||||
|
min_score: 0.0,
|
||||||
|
min_gap: f32::INFINITY,
|
||||||
|
max_results: 1,
|
||||||
|
};
|
||||||
|
assert_eq!(
|
||||||
|
m.search(&q, "alpha", &SearchOptions::new(2).with_confidence(lenient))
|
||||||
|
.len(),
|
||||||
|
1
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn only_returned_results_are_reinforced() {
|
||||||
|
// With re-ranking, a pool of max(3k, 10) candidates is retrieved; only
|
||||||
|
// the k returned should gain activation.
|
||||||
|
let d = data();
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let path = dir.path().join("h.h5");
|
||||||
|
let mut m = HDF5Memory::create(MemoryConfig::new(path, "a", DIM)).unwrap();
|
||||||
|
m.save_batch(
|
||||||
|
(0..200)
|
||||||
|
.map(|i| MemoryEntry {
|
||||||
|
chunk: format!("record {i}"),
|
||||||
|
embedding: d.vectors[i].clone(),
|
||||||
|
source_channel: "chat".into(),
|
||||||
|
timestamp: i as f64,
|
||||||
|
session_id: "s".into(),
|
||||||
|
tags: String::new(),
|
||||||
|
})
|
||||||
|
.collect(),
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
let q = query(&d, 0);
|
||||||
|
let got = m.search(
|
||||||
|
&q,
|
||||||
|
"record",
|
||||||
|
&SearchOptions::new(3).with_rerank(ReRankConfig::default()),
|
||||||
|
);
|
||||||
|
assert_eq!(got.len(), 3);
|
||||||
|
let returned: HashSet<usize> = got.iter().map(|r| r.index).collect();
|
||||||
|
// A second plain search reports each record's current activation.
|
||||||
|
let all = m.search(&q, "record", &SearchOptions::new(200));
|
||||||
|
for r in &all {
|
||||||
|
let boosted = r.activation > 1.0;
|
||||||
|
assert_eq!(boosted, returned.contains(&r.index), "record {}", r.index);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,330 @@
|
|||||||
|
//! Ed25519-signed checkpoints: `HDF5Memory::set_signing_key` and
|
||||||
|
//! `HDF5Memory::verify`.
|
||||||
|
|
||||||
|
use std::path::Path;
|
||||||
|
|
||||||
|
use clawhdf5_agent::signing::{SigningKey, VerifyReport, VerifyingKey};
|
||||||
|
use clawhdf5_agent::storage;
|
||||||
|
use clawhdf5_agent::{AgentMemory, HDF5Memory, MemoryConfig, MemoryEntry, MemoryError, schema};
|
||||||
|
use tempfile::TempDir;
|
||||||
|
|
||||||
|
const DIM: usize = 16;
|
||||||
|
|
||||||
|
fn key(seed: u8) -> SigningKey {
|
||||||
|
SigningKey::from_bytes(&[seed; 32])
|
||||||
|
}
|
||||||
|
|
||||||
|
fn entry(i: usize, chunk: &str) -> MemoryEntry {
|
||||||
|
MemoryEntry {
|
||||||
|
chunk: chunk.to_string(),
|
||||||
|
embedding: (0..DIM)
|
||||||
|
.map(|j| ((i * DIM + j) as f32 * 0.37).sin())
|
||||||
|
.collect(),
|
||||||
|
source_channel: "chat".into(),
|
||||||
|
timestamp: 1_700_000_000.0 + i as f64,
|
||||||
|
session_id: format!("s{}", i % 3),
|
||||||
|
tags: format!("t{i}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Awkward strings on purpose: they must hash the same after a round trip.
|
||||||
|
const TEXTS: [&str; 6] = [
|
||||||
|
"plain text",
|
||||||
|
"ünïcödé — 日本語 🙂",
|
||||||
|
"",
|
||||||
|
"trailing spaces ",
|
||||||
|
"tab\tand\nnewline",
|
||||||
|
"x",
|
||||||
|
];
|
||||||
|
|
||||||
|
fn signed_store(dir: &TempDir, float16: bool, k: &SigningKey) -> std::path::PathBuf {
|
||||||
|
let mut cfg = MemoryConfig::new(dir.path().join("s.h5"), "agent", DIM);
|
||||||
|
cfg.float16 = float16;
|
||||||
|
let path = cfg.path.clone();
|
||||||
|
let mut m = HDF5Memory::create(cfg).unwrap();
|
||||||
|
m.set_signing_key(k.clone());
|
||||||
|
let entries = (0..30).map(|i| entry(i, TEXTS[i % TEXTS.len()])).collect();
|
||||||
|
m.save_batch(entries).unwrap();
|
||||||
|
// Some graph and a deleted record, so every part of the manifest is used.
|
||||||
|
let a = m.knowledge_mut().add_entity("Alice", "person", 0);
|
||||||
|
let b = m.knowledge_mut().add_entity("Acme", "org", -1);
|
||||||
|
m.knowledge_mut().add_relation(a, b, "works_at", 0.75);
|
||||||
|
m.sessions_mut()
|
||||||
|
.add_at("s0", 0, 9, "chat", "first session", 1_700_000_000.0);
|
||||||
|
m.delete(4).unwrap();
|
||||||
|
m.flush_wal().unwrap();
|
||||||
|
path
|
||||||
|
}
|
||||||
|
|
||||||
|
fn verify(path: &Path, k: &SigningKey) -> VerifyReport {
|
||||||
|
HDF5Memory::verify(path, &k.verifying_key()).unwrap()
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn a_signed_store_verifies_through_reopen_and_checkpoint_cycles() {
|
||||||
|
for float16 in [true, false] {
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let k = key(7);
|
||||||
|
let path = signed_store(&dir, float16, &k);
|
||||||
|
let r = verify(&path, &k);
|
||||||
|
assert!(r.is_valid(), "float16={float16}: {r:?}");
|
||||||
|
assert_eq!(r.public_key, Some(k.verifying_key().to_bytes()));
|
||||||
|
assert_eq!(r.record_count, 30);
|
||||||
|
assert!(r.changed_records.is_empty());
|
||||||
|
|
||||||
|
// Reopen, change nothing, checkpoint again (with the key): still valid.
|
||||||
|
for _ in 0..3 {
|
||||||
|
let mut m = HDF5Memory::open(&path).unwrap();
|
||||||
|
assert!(m.is_signed());
|
||||||
|
m.set_signing_key(k.clone());
|
||||||
|
m.flush_wal().unwrap();
|
||||||
|
drop(m);
|
||||||
|
assert!(verify(&path, &k).is_valid());
|
||||||
|
}
|
||||||
|
// And after real changes, re-signed.
|
||||||
|
let mut m = HDF5Memory::open(&path).unwrap();
|
||||||
|
m.set_signing_key(k.clone());
|
||||||
|
m.save(entry(99, "added later")).unwrap();
|
||||||
|
m.hybrid_search(&entry(1, "").embedding, "text", 0.4, 0.6, 5);
|
||||||
|
m.flush_wal().unwrap();
|
||||||
|
drop(m);
|
||||||
|
let r = verify(&path, &k);
|
||||||
|
assert!(r.is_valid(), "{r:?}");
|
||||||
|
assert_eq!(r.record_count, 31);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn a_signed_store_refuses_to_checkpoint_without_its_key() {
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let k = key(1);
|
||||||
|
let path = signed_store(&dir, true, &k);
|
||||||
|
|
||||||
|
let mut m = HDF5Memory::open(&path).unwrap();
|
||||||
|
m.save(entry(50, "pending")).unwrap();
|
||||||
|
match m.flush_wal() {
|
||||||
|
Err(MemoryError::SigningKeyRequired(msg)) => assert!(msg.contains("signed"), "{msg}"),
|
||||||
|
other => panic!("expected SigningKeyRequired, got {other:?}"),
|
||||||
|
}
|
||||||
|
// The file is untouched and still valid; the save is still in the WAL.
|
||||||
|
let r = verify(&path, &k);
|
||||||
|
assert!(r.is_valid());
|
||||||
|
assert_eq!(r.wal_entries_unsigned, 1);
|
||||||
|
|
||||||
|
// Supplying the key lets the checkpoint through, signed.
|
||||||
|
m.set_signing_key(k.clone());
|
||||||
|
m.flush_wal().unwrap();
|
||||||
|
drop(m);
|
||||||
|
let r = verify(&path, &k);
|
||||||
|
assert!(r.is_valid());
|
||||||
|
assert_eq!((r.record_count, r.wal_entries_unsigned), (31, 0));
|
||||||
|
|
||||||
|
// Removing the signature on purpose writes it unsigned.
|
||||||
|
let mut m = HDF5Memory::open(&path).unwrap();
|
||||||
|
m.remove_signature();
|
||||||
|
m.flush_wal().unwrap();
|
||||||
|
drop(m);
|
||||||
|
let r = verify(&path, &k);
|
||||||
|
assert!(!r.signed && !r.is_valid());
|
||||||
|
assert!(!HDF5Memory::open(&path).unwrap().is_signed());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn the_wrong_key_does_not_verify_and_a_new_key_re_signs() {
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let (a, b) = (key(1), key(2));
|
||||||
|
let path = signed_store(&dir, true, &a);
|
||||||
|
let r = verify(&path, &b);
|
||||||
|
assert!(r.signed && !r.key_matches && !r.signature_valid && !r.is_valid());
|
||||||
|
|
||||||
|
let mut m = HDF5Memory::open(&path).unwrap();
|
||||||
|
m.set_signing_key(b.clone());
|
||||||
|
m.flush_wal().unwrap();
|
||||||
|
drop(m);
|
||||||
|
assert!(verify(&path, &b).is_valid());
|
||||||
|
assert!(!verify(&path, &a).is_valid());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Rewrite the store with changed contents but the *old* signature — what
|
||||||
|
/// someone with write access to the file, but not the key, can do.
|
||||||
|
fn tamper(path: &Path, change: impl FnOnce(&mut Tampered)) {
|
||||||
|
let file = clawhdf5::File::open(path).unwrap();
|
||||||
|
let (config, cache, sessions, knowledge) = schema::validate_and_load(&file).unwrap();
|
||||||
|
let checkpoint = schema::read_checkpoint_meta(&file);
|
||||||
|
let signature = schema::read_signature(&file).unwrap().unwrap();
|
||||||
|
drop(file);
|
||||||
|
let mut t = Tampered {
|
||||||
|
config,
|
||||||
|
cache,
|
||||||
|
sessions,
|
||||||
|
knowledge,
|
||||||
|
};
|
||||||
|
change(&mut t);
|
||||||
|
storage::write_to_disk_signed(
|
||||||
|
path,
|
||||||
|
&t.config,
|
||||||
|
&t.cache,
|
||||||
|
&t.sessions,
|
||||||
|
&t.knowledge,
|
||||||
|
&checkpoint,
|
||||||
|
Some(&signature),
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
}
|
||||||
|
|
||||||
|
struct Tampered {
|
||||||
|
config: MemoryConfig,
|
||||||
|
cache: clawhdf5_agent::cache::MemoryCache,
|
||||||
|
sessions: clawhdf5_agent::SessionCache,
|
||||||
|
knowledge: clawhdf5_agent::knowledge::KnowledgeCache,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn every_kind_of_edit_is_detected_and_located() {
|
||||||
|
let k = key(3);
|
||||||
|
type Edit = Box<dyn FnOnce(&mut Tampered)>;
|
||||||
|
type Case = (&'static str, Edit, fn(&VerifyReport) -> bool);
|
||||||
|
let cases: Vec<Case> = vec![
|
||||||
|
(
|
||||||
|
"record text",
|
||||||
|
Box::new(|t: &mut Tampered| t.cache.chunks[7] = "rewritten".into()),
|
||||||
|
|r| !r.records_match && r.changed_records == vec![7],
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"one embedding value",
|
||||||
|
Box::new(|t: &mut Tampered| {
|
||||||
|
let mut e = t.cache.embeddings[12].to_vec();
|
||||||
|
e[3] = 0.5;
|
||||||
|
t.cache.embeddings.set(12, &e);
|
||||||
|
}),
|
||||||
|
|r| r.changed_records == vec![12],
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"undelete",
|
||||||
|
Box::new(|t: &mut Tampered| t.cache.tombstones[4] = 0),
|
||||||
|
|r| r.changed_records == vec![4],
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"timestamp",
|
||||||
|
Box::new(|t: &mut Tampered| t.cache.timestamps[20] += 1.0),
|
||||||
|
|r| r.changed_records == vec![20],
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"record appended",
|
||||||
|
Box::new(|t: &mut Tampered| {
|
||||||
|
t.cache.push(
|
||||||
|
"new".into(),
|
||||||
|
vec![0.1; DIM],
|
||||||
|
"x".into(),
|
||||||
|
1.0,
|
||||||
|
"s".into(),
|
||||||
|
"".into(),
|
||||||
|
);
|
||||||
|
}),
|
||||||
|
|r| !r.records_match && r.changed_records == vec![30] && r.record_count == 31,
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"setting",
|
||||||
|
Box::new(|t: &mut Tampered| t.config.agent_id = "someone-else".into()),
|
||||||
|
|r| !r.settings_match && r.records_match,
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"session summary",
|
||||||
|
Box::new(|t: &mut Tampered| t.sessions.summaries[0] = "edited".into()),
|
||||||
|
|r| !r.sessions_match && r.records_match,
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"graph edge",
|
||||||
|
Box::new(|t: &mut Tampered| t.knowledge.relations[0].weight = 1.0),
|
||||||
|
|r| !r.graph_match && r.records_match,
|
||||||
|
),
|
||||||
|
];
|
||||||
|
for (name, edit, check) in cases {
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let path = signed_store(&dir, true, &k);
|
||||||
|
tamper(&path, edit);
|
||||||
|
let r = verify(&path, &k);
|
||||||
|
assert!(
|
||||||
|
r.signed && r.key_matches && r.signature_valid,
|
||||||
|
"{name}: {r:?}"
|
||||||
|
);
|
||||||
|
assert!(!r.is_valid(), "{name}: edit not detected: {r:?}");
|
||||||
|
assert!(check(&r), "{name}: {r:?}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn a_forged_manifest_fails_the_signature() {
|
||||||
|
// Recomputing the hashes for tampered contents does not help without the
|
||||||
|
// key: the signature no longer matches the manifest.
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let k = key(5);
|
||||||
|
let path = signed_store(&dir, true, &k);
|
||||||
|
let file = clawhdf5::File::open(&path).unwrap();
|
||||||
|
let (config, mut cache, sessions, knowledge) = schema::validate_and_load(&file).unwrap();
|
||||||
|
let checkpoint = schema::read_checkpoint_meta(&file);
|
||||||
|
let mut sig = schema::read_signature(&file).unwrap().unwrap();
|
||||||
|
drop(file);
|
||||||
|
cache.chunks[0] = "forged".into();
|
||||||
|
// Re-sign with an attacker key, then splice the victim's public key back.
|
||||||
|
let forged = clawhdf5_agent::signing::sign(
|
||||||
|
&key(66),
|
||||||
|
&config,
|
||||||
|
&cache,
|
||||||
|
&sessions,
|
||||||
|
&knowledge,
|
||||||
|
checkpoint.wal_applied,
|
||||||
|
);
|
||||||
|
sig.manifest = forged.manifest;
|
||||||
|
sig.record_hashes = forged.record_hashes;
|
||||||
|
storage::write_to_disk_signed(
|
||||||
|
&path,
|
||||||
|
&config,
|
||||||
|
&cache,
|
||||||
|
&sessions,
|
||||||
|
&knowledge,
|
||||||
|
&checkpoint,
|
||||||
|
Some(&sig),
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
let r = verify(&path, &k);
|
||||||
|
assert!(
|
||||||
|
r.key_matches && !r.signature_valid && !r.is_valid(),
|
||||||
|
"{r:?}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn an_unsigned_store_reports_unsigned() {
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let mut m = HDF5Memory::create(MemoryConfig::new(dir.path().join("u.h5"), "a", DIM)).unwrap();
|
||||||
|
m.save_batch(vec![entry(0, "hello")]).unwrap();
|
||||||
|
drop(m);
|
||||||
|
let r = HDF5Memory::verify(&dir.path().join("u.h5"), &VerifyingKey::from(&key(1))).unwrap();
|
||||||
|
assert!(!r.signed && !r.is_valid());
|
||||||
|
assert_eq!(r.record_count, 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn nul_bytes_in_text_still_verify() {
|
||||||
|
// Strings are stored null-padded; the hash must follow what a reopened
|
||||||
|
// store actually holds, or an untouched store would fail to verify.
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let k = key(9);
|
||||||
|
let mut m = HDF5Memory::create(MemoryConfig::new(dir.path().join("n.h5"), "a", DIM)).unwrap();
|
||||||
|
m.set_signing_key(k.clone());
|
||||||
|
m.save_batch(vec![
|
||||||
|
entry(0, "inner\0nul"),
|
||||||
|
entry(1, "trailing nul\0"),
|
||||||
|
entry(2, "\0leading"),
|
||||||
|
])
|
||||||
|
.unwrap();
|
||||||
|
drop(m);
|
||||||
|
let r = verify(&dir.path().join("n.h5"), &k);
|
||||||
|
assert!(r.is_valid(), "{r:?}");
|
||||||
|
let m = HDF5Memory::open(&dir.path().join("n.h5")).unwrap();
|
||||||
|
eprintln!(
|
||||||
|
"reloaded: {:?}",
|
||||||
|
(0..3).map(|i| m.get_chunk(i)).collect::<Vec<_>>()
|
||||||
|
);
|
||||||
|
}
|
||||||
@@ -1,7 +1,8 @@
|
|||||||
[package]
|
[package]
|
||||||
name = "clawhdf5-android"
|
name = "clawhdf5-android"
|
||||||
version = "2.5.0"
|
version = "2.7.0"
|
||||||
edition = "2024"
|
edition = "2024"
|
||||||
|
rust-version.workspace = true
|
||||||
description = "Android JNI bridge for edgehdf5-memory HDF5 backend"
|
description = "Android JNI bridge for edgehdf5-memory HDF5 backend"
|
||||||
license = "MIT"
|
license = "MIT"
|
||||||
|
|
||||||
|
|||||||
@@ -1,7 +1,8 @@
|
|||||||
[package]
|
[package]
|
||||||
name = "clawhdf5-ann"
|
name = "clawhdf5-ann"
|
||||||
version = "2.5.0"
|
version = "2.7.0"
|
||||||
edition = "2024"
|
edition = "2024"
|
||||||
|
rust-version.workspace = true
|
||||||
description = "HNSW approximate nearest neighbor index stored as HDF5"
|
description = "HNSW approximate nearest neighbor index stored as HDF5"
|
||||||
license = "MIT"
|
license = "MIT"
|
||||||
repository = "https://git.redclaw.dev/quantumclaw/clawhdf5"
|
repository = "https://git.redclaw.dev/quantumclaw/clawhdf5"
|
||||||
@@ -10,9 +11,9 @@ keywords = ["hdf5", "ann", "hnsw", "nearest-neighbor"]
|
|||||||
categories = ["algorithms", "science"]
|
categories = ["algorithms", "science"]
|
||||||
|
|
||||||
[dependencies]
|
[dependencies]
|
||||||
clawhdf5-format = { path = "../clawhdf5-format", version = "2.5.0" }
|
clawhdf5-format = { path = "../clawhdf5-format", version = "2.7.0" }
|
||||||
clawhdf5-io = { path = "../clawhdf5-io", version = "2.5.0" }
|
clawhdf5-io = { path = "../clawhdf5-io", version = "2.7.0" }
|
||||||
clawhdf5-accel = { path = "../clawhdf5-accel", version = "2.5.0" }
|
clawhdf5-accel = { path = "../clawhdf5-accel", version = "2.7.0" }
|
||||||
rayon = { version = "1", optional = true }
|
rayon = { version = "1", optional = true }
|
||||||
|
|
||||||
[features]
|
[features]
|
||||||
|
|||||||
+416
-65
@@ -13,7 +13,7 @@ use clawhdf5_format::filter_pipeline::FilterPipeline;
|
|||||||
use clawhdf5_format::group_v2::resolve_path_any;
|
use clawhdf5_format::group_v2::resolve_path_any;
|
||||||
use clawhdf5_format::message_type::MessageType;
|
use clawhdf5_format::message_type::MessageType;
|
||||||
use clawhdf5_format::object_header::ObjectHeader;
|
use clawhdf5_format::object_header::ObjectHeader;
|
||||||
use clawhdf5_format::signature::find_signature;
|
use clawhdf5_format::signature::split_user_block;
|
||||||
use clawhdf5_format::superblock::Superblock;
|
use clawhdf5_format::superblock::Superblock;
|
||||||
use clawhdf5_io::FileWriter as IoFileWriter;
|
use clawhdf5_io::FileWriter as IoFileWriter;
|
||||||
|
|
||||||
@@ -154,6 +154,218 @@ impl Ord for FarCandidate {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// How the index keeps its copy of the vectors.
|
||||||
|
#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
|
||||||
|
pub enum Storage {
|
||||||
|
/// Exactly as given: `dim * 4` bytes per vector.
|
||||||
|
#[default]
|
||||||
|
Float32,
|
||||||
|
/// Each component scaled to an `i8`: `dim` bytes per vector, a quarter of
|
||||||
|
/// the space, at some cost in precision.
|
||||||
|
///
|
||||||
|
/// Only meaningful for [`DistanceMetric::Cosine`]: rows are stored
|
||||||
|
/// unit-length, so a quantised dot product reconstructs the similarity
|
||||||
|
/// directly. Requesting it for `L2` keeps `Float32`, because an L2
|
||||||
|
/// distance cannot be recovered from a dot product alone.
|
||||||
|
Int8,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The index's copy of the vectors, flat and row-major.
|
||||||
|
#[derive(Debug, Clone)]
|
||||||
|
enum Vectors {
|
||||||
|
F32 {
|
||||||
|
dim: usize,
|
||||||
|
flat: Vec<f32>,
|
||||||
|
},
|
||||||
|
/// `flat[i * dim + j]` is component `j` of vector `i` divided by
|
||||||
|
/// `scales[i]`; multiplying back recovers it.
|
||||||
|
///
|
||||||
|
/// The scale is per row rather than global. A unit-length row in `d`
|
||||||
|
/// dimensions has components around `1/sqrt(d)`, so a fixed `[-1, 1]`
|
||||||
|
/// scale spends fewer than 12 of the 255 levels on a 128-dimensional
|
||||||
|
/// vector and the reconstruction error swamps the gaps between near
|
||||||
|
/// neighbours — measured at 0.35 top-10 overlap with the exact ranking.
|
||||||
|
/// Scaling each row by its own largest component uses the full range.
|
||||||
|
Int8 {
|
||||||
|
dim: usize,
|
||||||
|
flat: Vec<i8>,
|
||||||
|
scales: Vec<f32>,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Levels either side of zero. 127, not 128, so the range is symmetric.
|
||||||
|
const INT8_LEVELS: f32 = 127.0;
|
||||||
|
|
||||||
|
/// Quantise one row, returning the codes and the scale that inverts them.
|
||||||
|
fn quantise_row(v: &[f32], out: &mut Vec<i8>) -> f32 {
|
||||||
|
let max_abs = v.iter().fold(0.0f32, |m, x| m.max(x.abs()));
|
||||||
|
if max_abs <= f32::MIN_POSITIVE {
|
||||||
|
out.extend(core::iter::repeat_n(0i8, v.len()));
|
||||||
|
return 0.0;
|
||||||
|
}
|
||||||
|
let inv = INT8_LEVELS / max_abs;
|
||||||
|
out.extend(
|
||||||
|
v.iter()
|
||||||
|
.map(|x| (x * inv).round().clamp(-INT8_LEVELS, INT8_LEVELS) as i8),
|
||||||
|
);
|
||||||
|
max_abs / INT8_LEVELS
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Vectors {
|
||||||
|
fn new(dim: usize, storage: Storage, metric: DistanceMetric) -> Self {
|
||||||
|
match storage {
|
||||||
|
Storage::Int8 if metric == DistanceMetric::Cosine => Vectors::Int8 {
|
||||||
|
dim,
|
||||||
|
flat: Vec::new(),
|
||||||
|
scales: Vec::new(),
|
||||||
|
},
|
||||||
|
_ => Vectors::F32 {
|
||||||
|
dim,
|
||||||
|
flat: Vec::new(),
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn dim(&self) -> usize {
|
||||||
|
match self {
|
||||||
|
Vectors::F32 { dim, .. } | Vectors::Int8 { dim, .. } => *dim,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn storage(&self) -> Storage {
|
||||||
|
match self {
|
||||||
|
Vectors::F32 { .. } => Storage::Float32,
|
||||||
|
Vectors::Int8 { .. } => Storage::Int8,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn len(&self) -> usize {
|
||||||
|
let dim = self.dim();
|
||||||
|
if dim == 0 {
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
match self {
|
||||||
|
Vectors::F32 { flat, .. } => flat.len() / dim,
|
||||||
|
Vectors::Int8 { flat, .. } => flat.len() / dim,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Set the row width, for a store seeded empty by `new`.
|
||||||
|
fn set_dim(&mut self, new_dim: usize) {
|
||||||
|
match self {
|
||||||
|
Vectors::F32 { dim, .. } | Vectors::Int8 { dim, .. } => *dim = new_dim,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn push(&mut self, vector: &[f32]) {
|
||||||
|
match self {
|
||||||
|
Vectors::F32 { flat, .. } => flat.extend_from_slice(vector),
|
||||||
|
Vectors::Int8 { flat, scales, .. } => scales.push(quantise_row(vector, flat)),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Row `i` as `f32`, for callers that need the values back (serialization,
|
||||||
|
/// and the f32 fast paths). Quantised rows are reconstructed, so this is
|
||||||
|
/// lossy in exactly the way the storage is.
|
||||||
|
fn row(&self, i: usize) -> Vec<f32> {
|
||||||
|
let dim = self.dim();
|
||||||
|
let start = i * dim;
|
||||||
|
match self {
|
||||||
|
Vectors::F32 { flat, .. } => flat[start..start + dim].to_vec(),
|
||||||
|
Vectors::Int8 { flat, scales, .. } => flat[start..start + dim]
|
||||||
|
.iter()
|
||||||
|
.map(|&q| f32::from(q) * scales[i])
|
||||||
|
.collect(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Distance between two stored vectors.
|
||||||
|
fn dist(&self, a: usize, b: usize, metric: DistanceMetric) -> f32 {
|
||||||
|
let dim = self.dim();
|
||||||
|
match self {
|
||||||
|
Vectors::F32 { flat, .. } => {
|
||||||
|
let (x, y) = (a * dim, b * dim);
|
||||||
|
compute_distance(&flat[x..x + dim], &flat[y..y + dim], metric)
|
||||||
|
}
|
||||||
|
Vectors::Int8 { flat, scales, .. } => {
|
||||||
|
let (x, y) = (a * dim, b * dim);
|
||||||
|
let dot = dot_i8(&flat[x..x + dim], &flat[y..y + dim]);
|
||||||
|
1.0 - dot as f32 * scales[a] * scales[b]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Distance from a prepared query to stored vector `i`.
|
||||||
|
fn dist_query(&self, query: &Query, i: usize, metric: DistanceMetric) -> f32 {
|
||||||
|
let dim = self.dim();
|
||||||
|
let start = i * dim;
|
||||||
|
match (self, query) {
|
||||||
|
(Vectors::F32 { flat, .. }, Query::F32(q)) => {
|
||||||
|
compute_distance(q, &flat[start..start + dim], metric)
|
||||||
|
}
|
||||||
|
(Vectors::Int8 { flat, scales, .. }, Query::Int8(q, q_scale)) => {
|
||||||
|
let dot = dot_i8(q, &flat[start..start + dim]);
|
||||||
|
1.0 - dot as f32 * q_scale * scales[i]
|
||||||
|
}
|
||||||
|
// Mixed forms cannot occur: `Query` is built from the same storage.
|
||||||
|
_ => f32::MAX,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Build a store from prepared rows.
|
||||||
|
fn from_rows(rows: &[Vec<f32>], storage: Storage, metric: DistanceMetric) -> Self {
|
||||||
|
let dim = rows.first().map_or(0, Vec::len);
|
||||||
|
let mut out = Vectors::new(dim, storage, metric);
|
||||||
|
for row in rows {
|
||||||
|
out.push(row);
|
||||||
|
}
|
||||||
|
out
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Prepare `query` for comparison against this store.
|
||||||
|
fn query(&self, query: Vec<f32>) -> Query {
|
||||||
|
match self {
|
||||||
|
Vectors::F32 { .. } => Query::F32(query),
|
||||||
|
Vectors::Int8 { .. } => {
|
||||||
|
let mut codes = Vec::with_capacity(query.len());
|
||||||
|
let scale = quantise_row(&query, &mut codes);
|
||||||
|
Query::Int8(codes, scale)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// What a layer search is measuring distance *to*: an incoming query, or a
|
||||||
|
/// node already in the index (which is what insertion compares against).
|
||||||
|
enum Target<'a> {
|
||||||
|
Query(&'a Query),
|
||||||
|
Node(usize),
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Vectors {
|
||||||
|
fn dist_to(&self, target: &Target<'_>, i: usize, metric: DistanceMetric) -> f32 {
|
||||||
|
match target {
|
||||||
|
Target::Query(q) => self.dist_query(q, i, metric),
|
||||||
|
Target::Node(n) => self.dist(*n, i, metric),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A search query in whichever form the store compares against.
|
||||||
|
enum Query {
|
||||||
|
F32(Vec<f32>),
|
||||||
|
/// Codes and the scale that inverts them, as in [`Vectors::Int8`].
|
||||||
|
Int8(Vec<i8>, f32),
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Sum of products, widened so it cannot overflow. Runtime-dispatched to the
|
||||||
|
/// same SIMD backend as the f32 kernels, so the two storages are compared on
|
||||||
|
/// equal terms.
|
||||||
|
#[inline]
|
||||||
|
fn dot_i8(a: &[i8], b: &[i8]) -> i32 {
|
||||||
|
clawhdf5_accel::dot_i8(a, b)
|
||||||
|
}
|
||||||
|
|
||||||
/// Magic for [`HnswIndex::graph_to_bytes`].
|
/// Magic for [`HnswIndex::graph_to_bytes`].
|
||||||
const GRAPH_MAGIC: &[u8; 4] = b"CHG1";
|
const GRAPH_MAGIC: &[u8; 4] = b"CHG1";
|
||||||
|
|
||||||
@@ -174,8 +386,8 @@ pub const HNSW_FORMAT_VERSION: i64 = 2;
|
|||||||
/// HDF5 format.
|
/// HDF5 format.
|
||||||
#[derive(Debug, Clone)]
|
#[derive(Debug, Clone)]
|
||||||
pub struct HnswIndex {
|
pub struct HnswIndex {
|
||||||
/// All vectors in the index.
|
/// All vectors in the index, flat and row-major.
|
||||||
vectors: Vec<Vec<f32>>,
|
vectors: Vectors,
|
||||||
/// Adjacency lists per layer. `graph[layer][node]` = list of neighbor IDs.
|
/// Adjacency lists per layer. `graph[layer][node]` = list of neighbor IDs.
|
||||||
graph: Vec<Vec<Vec<usize>>>,
|
graph: Vec<Vec<Vec<usize>>>,
|
||||||
/// Soft-deletion flags, one per node. Deleted nodes remain in the graph for
|
/// Soft-deletion flags, one per node. Deleted nodes remain in the graph for
|
||||||
@@ -214,6 +426,20 @@ impl HnswIndex {
|
|||||||
m: usize,
|
m: usize,
|
||||||
ef_construction: usize,
|
ef_construction: usize,
|
||||||
metric: DistanceMetric,
|
metric: DistanceMetric,
|
||||||
|
) -> Self {
|
||||||
|
Self::build_with(vectors, m, ef_construction, metric, Storage::default())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Build an index, choosing how the vectors are stored.
|
||||||
|
///
|
||||||
|
/// [`Storage::Int8`] keeps them at a quarter of the size; see its docs for
|
||||||
|
/// what that costs and when it applies.
|
||||||
|
pub fn build_with(
|
||||||
|
vectors: &[Vec<f32>],
|
||||||
|
m: usize,
|
||||||
|
ef_construction: usize,
|
||||||
|
metric: DistanceMetric,
|
||||||
|
storage: Storage,
|
||||||
) -> Self {
|
) -> Self {
|
||||||
assert!(!vectors.is_empty(), "cannot build index from empty vectors");
|
assert!(!vectors.is_empty(), "cannot build index from empty vectors");
|
||||||
assert!(m >= 2, "m must be at least 2");
|
assert!(m >= 2, "m must be at least 2");
|
||||||
@@ -224,8 +450,11 @@ impl HnswIndex {
|
|||||||
|
|
||||||
let m_max0 = m * 2;
|
let m_max0 = m * 2;
|
||||||
let n = vectors.len();
|
let n = vectors.len();
|
||||||
let prepared: Vec<Vec<f32>> = vectors.iter().map(|v| prepare(v.clone(), metric)).collect();
|
let mut prepared = Vectors::new(dim, storage, metric);
|
||||||
let vectors: &[Vec<f32>] = &prepared;
|
for v in vectors {
|
||||||
|
prepared.push(&prepare(v.clone(), metric));
|
||||||
|
}
|
||||||
|
let vectors = &prepared;
|
||||||
|
|
||||||
// Assign levels to all nodes
|
// Assign levels to all nodes
|
||||||
let mut node_levels = Vec::with_capacity(n);
|
let mut node_levels = Vec::with_capacity(n);
|
||||||
@@ -322,9 +551,20 @@ impl HnswIndex {
|
|||||||
/// point for incremental [`HnswIndex::insert`] and as the result of
|
/// point for incremental [`HnswIndex::insert`] and as the result of
|
||||||
/// [`HnswIndex::compact`] when every vector has been deleted.
|
/// [`HnswIndex::compact`] when every vector has been deleted.
|
||||||
pub fn new(m: usize, ef_construction: usize, metric: DistanceMetric) -> Self {
|
pub fn new(m: usize, ef_construction: usize, metric: DistanceMetric) -> Self {
|
||||||
|
Self::new_with(m, ef_construction, metric, Storage::default())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`HnswIndex::new`], choosing how the vectors are stored.
|
||||||
|
pub fn new_with(
|
||||||
|
m: usize,
|
||||||
|
ef_construction: usize,
|
||||||
|
metric: DistanceMetric,
|
||||||
|
storage: Storage,
|
||||||
|
) -> Self {
|
||||||
assert!(m >= 2, "m must be at least 2");
|
assert!(m >= 2, "m must be at least 2");
|
||||||
Self {
|
Self {
|
||||||
vectors: Vec::new(),
|
// The dimension is set by the first insert.
|
||||||
|
vectors: Vectors::new(0, storage, metric),
|
||||||
graph: Vec::new(),
|
graph: Vec::new(),
|
||||||
deleted: Vec::new(),
|
deleted: Vec::new(),
|
||||||
entry_point: 0,
|
entry_point: 0,
|
||||||
@@ -351,7 +591,8 @@ impl HnswIndex {
|
|||||||
// Seed an empty index.
|
// Seed an empty index.
|
||||||
if id == 0 {
|
if id == 0 {
|
||||||
let node_level = assign_level(0, self.m);
|
let node_level = assign_level(0, self.m);
|
||||||
self.vectors.push(vector);
|
self.vectors.set_dim(vector.len());
|
||||||
|
self.vectors.push(&vector);
|
||||||
self.deleted.push(false);
|
self.deleted.push(false);
|
||||||
self.node_levels.push(node_level);
|
self.node_levels.push(node_level);
|
||||||
self.graph = (0..=node_level).map(|_| vec![Vec::new(); 1]).collect();
|
self.graph = (0..=node_level).map(|_| vec![Vec::new(); 1]).collect();
|
||||||
@@ -361,12 +602,12 @@ impl HnswIndex {
|
|||||||
|
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
vector.len(),
|
vector.len(),
|
||||||
self.vectors[0].len(),
|
self.vectors.dim(),
|
||||||
"insert dimension mismatch"
|
"insert dimension mismatch"
|
||||||
);
|
);
|
||||||
|
|
||||||
let node_level = assign_level(id, self.m);
|
let node_level = assign_level(id, self.m);
|
||||||
self.vectors.push(vector);
|
self.vectors.push(&vector);
|
||||||
self.deleted.push(false);
|
self.deleted.push(false);
|
||||||
self.node_levels.push(node_level);
|
self.node_levels.push(node_level);
|
||||||
|
|
||||||
@@ -387,7 +628,7 @@ impl HnswIndex {
|
|||||||
ep = greedy_closest(
|
ep = greedy_closest(
|
||||||
&self.vectors,
|
&self.vectors,
|
||||||
&self.graph[layer],
|
&self.graph[layer],
|
||||||
&self.vectors[id],
|
&Target::Node(id),
|
||||||
ep,
|
ep,
|
||||||
self.metric,
|
self.metric,
|
||||||
);
|
);
|
||||||
@@ -400,7 +641,7 @@ impl HnswIndex {
|
|||||||
let neighbors = search_layer(
|
let neighbors = search_layer(
|
||||||
&self.vectors,
|
&self.vectors,
|
||||||
&self.graph[layer],
|
&self.graph[layer],
|
||||||
&self.vectors[id],
|
&Target::Node(id),
|
||||||
ep,
|
ep,
|
||||||
self.ef_construction,
|
self.ef_construction,
|
||||||
self.metric,
|
self.metric,
|
||||||
@@ -466,16 +707,25 @@ impl HnswIndex {
|
|||||||
pub fn compact(&mut self) -> Vec<Option<usize>> {
|
pub fn compact(&mut self) -> Vec<Option<usize>> {
|
||||||
let mut mapping = vec![None; self.vectors.len()];
|
let mut mapping = vec![None; self.vectors.len()];
|
||||||
let mut surviving: Vec<Vec<f32>> = Vec::with_capacity(self.active_len());
|
let mut surviving: Vec<Vec<f32>> = Vec::with_capacity(self.active_len());
|
||||||
for (old, v) in self.vectors.iter().enumerate() {
|
for (old, slot) in mapping.iter_mut().enumerate() {
|
||||||
if !self.deleted[old] {
|
if !self.deleted[old] {
|
||||||
mapping[old] = Some(surviving.len());
|
*slot = Some(surviving.len());
|
||||||
surviving.push(v.clone());
|
surviving.push(self.vectors.row(old));
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
// Rebuilding must keep the storage the caller chose; a compaction is
|
||||||
|
// not the place to silently quadruple the index's memory.
|
||||||
|
let storage = self.vectors.storage();
|
||||||
*self = if surviving.is_empty() {
|
*self = if surviving.is_empty() {
|
||||||
Self::new(self.m, self.ef_construction, self.metric)
|
Self::new_with(self.m, self.ef_construction, self.metric, storage)
|
||||||
} else {
|
} else {
|
||||||
Self::build_with_metric(&surviving, self.m, self.ef_construction, self.metric)
|
Self::build_with(
|
||||||
|
&surviving,
|
||||||
|
self.m,
|
||||||
|
self.ef_construction,
|
||||||
|
self.metric,
|
||||||
|
storage,
|
||||||
|
)
|
||||||
};
|
};
|
||||||
mapping
|
mapping
|
||||||
}
|
}
|
||||||
@@ -490,24 +740,22 @@ impl HnswIndex {
|
|||||||
/// # Returns
|
/// # Returns
|
||||||
/// A vector of `(id, distance)` pairs sorted by distance (closest first).
|
/// A vector of `(id, distance)` pairs sorted by distance (closest first).
|
||||||
pub fn search(&self, query: &[f32], k: usize, ef: usize) -> Vec<(usize, f32)> {
|
pub fn search(&self, query: &[f32], k: usize, ef: usize) -> Vec<(usize, f32)> {
|
||||||
if self.vectors.is_empty() {
|
if self.vectors.len() == 0 {
|
||||||
return Vec::new();
|
return Vec::new();
|
||||||
}
|
}
|
||||||
assert_eq!(
|
assert_eq!(query.len(), self.vectors.dim(), "query dimension mismatch");
|
||||||
query.len(),
|
|
||||||
self.vectors[0].len(),
|
|
||||||
"query dimension mismatch"
|
|
||||||
);
|
|
||||||
let ef = ef.max(k);
|
let ef = ef.max(k);
|
||||||
let prepared_query = prepare(query.to_vec(), self.metric);
|
// Prepared and, for a quantised store, quantised once per search
|
||||||
let query = prepared_query.as_slice();
|
// rather than once per comparison.
|
||||||
|
let prepared = self.vectors.query(prepare(query.to_vec(), self.metric));
|
||||||
|
let target = Target::Query(&prepared);
|
||||||
|
|
||||||
let mut ep = self.entry_point;
|
let mut ep = self.entry_point;
|
||||||
let top_layer = self.graph.len().saturating_sub(1);
|
let top_layer = self.graph.len().saturating_sub(1);
|
||||||
|
|
||||||
// Greedy search from top layer down to layer 1
|
// Greedy search from top layer down to layer 1
|
||||||
for layer in (1..=top_layer).rev() {
|
for layer in (1..=top_layer).rev() {
|
||||||
ep = greedy_closest(&self.vectors, &self.graph[layer], query, ep, self.metric);
|
ep = greedy_closest(&self.vectors, &self.graph[layer], &target, ep, self.metric);
|
||||||
}
|
}
|
||||||
|
|
||||||
// Search layer 0 for the ef nearest *live* nodes. Deleted nodes are
|
// Search layer 0 for the ef nearest *live* nodes. Deleted nodes are
|
||||||
@@ -516,7 +764,7 @@ impl HnswIndex {
|
|||||||
let candidates = search_layer(
|
let candidates = search_layer(
|
||||||
&self.vectors,
|
&self.vectors,
|
||||||
&self.graph[0],
|
&self.graph[0],
|
||||||
query,
|
&target,
|
||||||
ep,
|
ep,
|
||||||
ef,
|
ef,
|
||||||
self.metric,
|
self.metric,
|
||||||
@@ -543,14 +791,13 @@ impl HnswIndex {
|
|||||||
pub fn to_hdf5_bytes(&self) -> Result<Vec<u8>, FormatError> {
|
pub fn to_hdf5_bytes(&self) -> Result<Vec<u8>, FormatError> {
|
||||||
let mut fw = FmtWriter::new();
|
let mut fw = FmtWriter::new();
|
||||||
let n = self.vectors.len();
|
let n = self.vectors.len();
|
||||||
let dim = if n > 0 { self.vectors[0].len() } else { 0 };
|
let dim = self.vectors.dim();
|
||||||
|
|
||||||
// Flatten vectors into a 1D array for storage
|
// Flatten vectors into a 1D array for storage
|
||||||
let flat_vectors: Vec<f32> = self
|
let mut flat_vectors: Vec<f32> = Vec::with_capacity(n * dim);
|
||||||
.vectors
|
for i in 0..n {
|
||||||
.iter()
|
flat_vectors.extend_from_slice(&self.vectors.row(i));
|
||||||
.flat_map(|v| v.iter().copied())
|
}
|
||||||
.collect();
|
|
||||||
|
|
||||||
let mut group = fw.create_group("ann");
|
let mut group = fw.create_group("ann");
|
||||||
|
|
||||||
@@ -614,8 +861,9 @@ impl HnswIndex {
|
|||||||
/// The HDF5 data must contain the `/ann/vectors`, `/ann/graph_layer_*`,
|
/// The HDF5 data must contain the `/ann/vectors`, `/ann/graph_layer_*`,
|
||||||
/// and `/ann/config` datasets as produced by [`to_hdf5_bytes`].
|
/// and `/ann/config` datasets as produced by [`to_hdf5_bytes`].
|
||||||
pub fn load_from_hdf5(data: &[u8]) -> Result<Self, FormatError> {
|
pub fn load_from_hdf5(data: &[u8]) -> Result<Self, FormatError> {
|
||||||
let sig_offset = find_signature(data)?;
|
// Addresses are relative to the superblock: skip any user block.
|
||||||
let sb = Superblock::parse(data, sig_offset)?;
|
let (_, data) = split_user_block(data)?;
|
||||||
|
let sb = Superblock::parse(data, 0)?;
|
||||||
|
|
||||||
// Read config dataset and its attributes
|
// Read config dataset and its attributes
|
||||||
let config_attrs = read_dataset_attrs(data, &sb, "ann/config")?;
|
let config_attrs = read_dataset_attrs(data, &sb, "ann/config")?;
|
||||||
@@ -709,7 +957,9 @@ impl HnswIndex {
|
|||||||
};
|
};
|
||||||
|
|
||||||
Ok(Self {
|
Ok(Self {
|
||||||
vectors,
|
// Serialized files carry f32 vectors and no storage tag: a
|
||||||
|
// quantised index is rebuilt, not loaded.
|
||||||
|
vectors: Vectors::from_rows(&vectors, Storage::Float32, metric),
|
||||||
graph,
|
graph,
|
||||||
deleted,
|
deleted,
|
||||||
entry_point,
|
entry_point,
|
||||||
@@ -773,6 +1023,16 @@ impl HnswIndex {
|
|||||||
/// `bytes` is validated — a corrupt or mismatched graph is an error, never
|
/// `bytes` is validated — a corrupt or mismatched graph is an error, never
|
||||||
/// an index that panics or walks out of bounds during a search.
|
/// an index that panics or walks out of bounds during a search.
|
||||||
pub fn from_graph_bytes(bytes: &[u8], vectors: Vec<Vec<f32>>) -> Result<Self, FormatError> {
|
pub fn from_graph_bytes(bytes: &[u8], vectors: Vec<Vec<f32>>) -> Result<Self, FormatError> {
|
||||||
|
Self::from_graph_bytes_with(bytes, vectors, Storage::default())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// As [`from_graph_bytes`](Self::from_graph_bytes), choosing how the
|
||||||
|
/// rehydrated vectors are stored.
|
||||||
|
pub fn from_graph_bytes_with(
|
||||||
|
bytes: &[u8],
|
||||||
|
vectors: Vec<Vec<f32>>,
|
||||||
|
storage: Storage,
|
||||||
|
) -> Result<Self, FormatError> {
|
||||||
let bad = |what: &str| FormatError::SerializationError(format!("HNSW graph: {what}"));
|
let bad = |what: &str| FormatError::SerializationError(format!("HNSW graph: {what}"));
|
||||||
let body_len = bytes
|
let body_len = bytes
|
||||||
.len()
|
.len()
|
||||||
@@ -863,7 +1123,14 @@ impl HnswIndex {
|
|||||||
}
|
}
|
||||||
|
|
||||||
Ok(Self {
|
Ok(Self {
|
||||||
vectors: vectors.into_iter().map(|v| prepare(v, metric)).collect(),
|
vectors: Vectors::from_rows(
|
||||||
|
&vectors
|
||||||
|
.into_iter()
|
||||||
|
.map(|v| prepare(v, metric))
|
||||||
|
.collect::<Vec<_>>(),
|
||||||
|
storage,
|
||||||
|
metric,
|
||||||
|
),
|
||||||
graph,
|
graph,
|
||||||
deleted,
|
deleted,
|
||||||
entry_point,
|
entry_point,
|
||||||
@@ -882,16 +1149,17 @@ impl HnswIndex {
|
|||||||
|
|
||||||
/// Returns true if the index is empty.
|
/// Returns true if the index is empty.
|
||||||
pub fn is_empty(&self) -> bool {
|
pub fn is_empty(&self) -> bool {
|
||||||
self.vectors.is_empty()
|
self.vectors.len() == 0
|
||||||
|
}
|
||||||
|
|
||||||
|
/// How this index stores its copy of the vectors.
|
||||||
|
pub fn storage(&self) -> Storage {
|
||||||
|
self.vectors.storage()
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Returns the dimension of vectors in the index.
|
/// Returns the dimension of vectors in the index.
|
||||||
pub fn dimension(&self) -> usize {
|
pub fn dimension(&self) -> usize {
|
||||||
if self.vectors.is_empty() {
|
self.vectors.dim()
|
||||||
0
|
|
||||||
} else {
|
|
||||||
self.vectors[0].len()
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Returns the number of layers in the graph.
|
/// Returns the number of layers in the graph.
|
||||||
@@ -916,17 +1184,17 @@ impl HnswIndex {
|
|||||||
|
|
||||||
/// Greedy search: find the single closest node to `query` starting from `ep`.
|
/// Greedy search: find the single closest node to `query` starting from `ep`.
|
||||||
fn greedy_closest(
|
fn greedy_closest(
|
||||||
vectors: &[Vec<f32>],
|
vectors: &Vectors,
|
||||||
layer: &[Vec<usize>],
|
layer: &[Vec<usize>],
|
||||||
query: &[f32],
|
target: &Target<'_>,
|
||||||
mut ep: usize,
|
mut ep: usize,
|
||||||
metric: DistanceMetric,
|
metric: DistanceMetric,
|
||||||
) -> usize {
|
) -> usize {
|
||||||
let mut best_dist = compute_distance(query, &vectors[ep], metric);
|
let mut best_dist = vectors.dist_to(target, ep, metric);
|
||||||
loop {
|
loop {
|
||||||
let mut changed = false;
|
let mut changed = false;
|
||||||
for &neighbor in &layer[ep] {
|
for &neighbor in &layer[ep] {
|
||||||
let d = compute_distance(query, &vectors[neighbor], metric);
|
let d = vectors.dist_to(target, neighbor, metric);
|
||||||
if d < best_dist {
|
if d < best_dist {
|
||||||
best_dist = d;
|
best_dist = d;
|
||||||
ep = neighbor;
|
ep = neighbor;
|
||||||
@@ -950,15 +1218,15 @@ fn greedy_closest(
|
|||||||
/// instead meant a query whose neighbourhood had been deleted got back fewer
|
/// instead meant a query whose neighbourhood had been deleted got back fewer
|
||||||
/// than `k` results, or none, however many live records were nearby.
|
/// than `k` results, or none, however many live records were nearby.
|
||||||
fn search_layer(
|
fn search_layer(
|
||||||
vectors: &[Vec<f32>],
|
vectors: &Vectors,
|
||||||
layer: &[Vec<usize>],
|
layer: &[Vec<usize>],
|
||||||
query: &[f32],
|
target: &Target<'_>,
|
||||||
ep: usize,
|
ep: usize,
|
||||||
ef: usize,
|
ef: usize,
|
||||||
metric: DistanceMetric,
|
metric: DistanceMetric,
|
||||||
skip: Option<&[bool]>,
|
skip: Option<&[bool]>,
|
||||||
) -> Vec<Candidate> {
|
) -> Vec<Candidate> {
|
||||||
let ep_dist = compute_distance(query, &vectors[ep], metric);
|
let ep_dist = vectors.dist_to(target, ep, metric);
|
||||||
|
|
||||||
// Min-heap of candidates to explore
|
// Min-heap of candidates to explore
|
||||||
let mut candidates = BinaryHeap::new();
|
let mut candidates = BinaryHeap::new();
|
||||||
@@ -980,7 +1248,7 @@ fn search_layer(
|
|||||||
visited.begin(vectors.len());
|
visited.begin(vectors.len());
|
||||||
visited.insert(ep);
|
visited.insert(ep);
|
||||||
search_layer_visit(
|
search_layer_visit(
|
||||||
vectors, layer, query, ef, metric, skip, visited, candidates, results,
|
vectors, layer, target, ef, metric, skip, visited, candidates, results,
|
||||||
)
|
)
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
@@ -1023,9 +1291,9 @@ thread_local! {
|
|||||||
|
|
||||||
#[allow(clippy::too_many_arguments)]
|
#[allow(clippy::too_many_arguments)]
|
||||||
fn search_layer_visit(
|
fn search_layer_visit(
|
||||||
vectors: &[Vec<f32>],
|
vectors: &Vectors,
|
||||||
layer: &[Vec<usize>],
|
layer: &[Vec<usize>],
|
||||||
query: &[f32],
|
target: &Target<'_>,
|
||||||
ef: usize,
|
ef: usize,
|
||||||
metric: DistanceMetric,
|
metric: DistanceMetric,
|
||||||
skip: Option<&[bool]>,
|
skip: Option<&[bool]>,
|
||||||
@@ -1044,7 +1312,7 @@ fn search_layer_visit(
|
|||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
|
||||||
let d = compute_distance(query, &vectors[neighbor], metric);
|
let d = vectors.dist_to(target, neighbor, metric);
|
||||||
let furthest_dist = results.peek().map_or(f32::MAX, |f| f.distance);
|
let furthest_dist = results.peek().map_or(f32::MAX, |f| f.distance);
|
||||||
|
|
||||||
if d < furthest_dist || results.len() < ef {
|
if d < furthest_dist || results.len() < ef {
|
||||||
@@ -1095,7 +1363,7 @@ fn search_layer_visit(
|
|||||||
/// remaining slots are then filled with the closest rejected candidates, so a
|
/// remaining slots are then filled with the closest rejected candidates, so a
|
||||||
/// node is never left under-connected.
|
/// node is never left under-connected.
|
||||||
fn select_neighbors(
|
fn select_neighbors(
|
||||||
vectors: &[Vec<f32>],
|
vectors: &Vectors,
|
||||||
candidates: &[(usize, f32)],
|
candidates: &[(usize, f32)],
|
||||||
max_conn: usize,
|
max_conn: usize,
|
||||||
metric: DistanceMetric,
|
metric: DistanceMetric,
|
||||||
@@ -1111,7 +1379,7 @@ fn select_neighbors(
|
|||||||
}
|
}
|
||||||
let diverse = selected
|
let diverse = selected
|
||||||
.iter()
|
.iter()
|
||||||
.all(|&s| compute_distance(&vectors[id], &vectors[s], metric) > dist_to_node);
|
.all(|&s| vectors.dist(id, s, metric) > dist_to_node);
|
||||||
if diverse {
|
if diverse {
|
||||||
selected.push(id);
|
selected.push(id);
|
||||||
} else {
|
} else {
|
||||||
@@ -1136,7 +1404,7 @@ fn batch_len(linked: usize) -> usize {
|
|||||||
/// layers, found by searching the graph as it currently stands.
|
/// layers, found by searching the graph as it currently stands.
|
||||||
#[allow(clippy::too_many_arguments)]
|
#[allow(clippy::too_many_arguments)]
|
||||||
fn plan_batch(
|
fn plan_batch(
|
||||||
vectors: &[Vec<f32>],
|
vectors: &Vectors,
|
||||||
graph: &[Vec<Vec<usize>>],
|
graph: &[Vec<Vec<usize>>],
|
||||||
node_levels: &[usize],
|
node_levels: &[usize],
|
||||||
batch: std::ops::Range<usize>,
|
batch: std::ops::Range<usize>,
|
||||||
@@ -1150,7 +1418,7 @@ fn plan_batch(
|
|||||||
let mut ep = entry_point;
|
let mut ep = entry_point;
|
||||||
// Phase 1: greedy descent from the top layer down to node_level + 1.
|
// Phase 1: greedy descent from the top layer down to node_level + 1.
|
||||||
for layer in (node_level + 1..=ep_level).rev() {
|
for layer in (node_level + 1..=ep_level).rev() {
|
||||||
ep = greedy_closest(vectors, &graph[layer], &vectors[i], ep, metric);
|
ep = greedy_closest(vectors, &graph[layer], &Target::Node(i), ep, metric);
|
||||||
}
|
}
|
||||||
// Phase 2: search and select on every layer the node lives on.
|
// Phase 2: search and select on every layer the node lives on.
|
||||||
let mut plan = Vec::with_capacity(node_level.min(ep_level) + 1);
|
let mut plan = Vec::with_capacity(node_level.min(ep_level) + 1);
|
||||||
@@ -1159,7 +1427,7 @@ fn plan_batch(
|
|||||||
let neighbors = search_layer(
|
let neighbors = search_layer(
|
||||||
vectors,
|
vectors,
|
||||||
&graph[layer],
|
&graph[layer],
|
||||||
&vectors[i],
|
&Target::Node(i),
|
||||||
ep,
|
ep,
|
||||||
ef_construction,
|
ef_construction,
|
||||||
metric,
|
metric,
|
||||||
@@ -1186,7 +1454,7 @@ fn plan_batch(
|
|||||||
/// Prune every `(layer, node)` neighbour list in `overflowed` back to its
|
/// Prune every `(layer, node)` neighbour list in `overflowed` back to its
|
||||||
/// limit. Each list belongs to a different node, so they are independent.
|
/// limit. Each list belongs to a different node, so they are independent.
|
||||||
fn prune_overflowed(
|
fn prune_overflowed(
|
||||||
vectors: &[Vec<f32>],
|
vectors: &Vectors,
|
||||||
graph: &mut [Vec<Vec<usize>>],
|
graph: &mut [Vec<Vec<usize>>],
|
||||||
overflowed: Vec<(usize, usize)>,
|
overflowed: Vec<(usize, usize)>,
|
||||||
(m, m_max0): (usize, usize),
|
(m, m_max0): (usize, usize),
|
||||||
@@ -1226,7 +1494,7 @@ const PARALLEL_MIN: usize = 8;
|
|||||||
/// prunes is too fine-grained to parallelise profitably — measured 1.45x on 16
|
/// prunes is too fine-grained to parallelise profitably — measured 1.45x on 16
|
||||||
/// cores; bulk builds batch their pruning instead, see `prune_overflowed`.)
|
/// cores; bulk builds batch their pruning instead, see `prune_overflowed`.)
|
||||||
fn link_back(
|
fn link_back(
|
||||||
vectors: &[Vec<f32>],
|
vectors: &Vectors,
|
||||||
layer: &mut [Vec<usize>],
|
layer: &mut [Vec<usize>],
|
||||||
new_id: usize,
|
new_id: usize,
|
||||||
selected: &[usize],
|
selected: &[usize],
|
||||||
@@ -1248,7 +1516,7 @@ fn link_back(
|
|||||||
|
|
||||||
/// Trim `node`'s neighbour list back to `max_conn` with [`select_neighbors`].
|
/// Trim `node`'s neighbour list back to `max_conn` with [`select_neighbors`].
|
||||||
fn prune_connections(
|
fn prune_connections(
|
||||||
vectors: &[Vec<f32>],
|
vectors: &Vectors,
|
||||||
neighbors: &mut Vec<usize>,
|
neighbors: &mut Vec<usize>,
|
||||||
node: usize,
|
node: usize,
|
||||||
max_conn: usize,
|
max_conn: usize,
|
||||||
@@ -1259,7 +1527,7 @@ fn prune_connections(
|
|||||||
}
|
}
|
||||||
let mut scored: Vec<(usize, f32)> = neighbors
|
let mut scored: Vec<(usize, f32)> = neighbors
|
||||||
.iter()
|
.iter()
|
||||||
.map(|&n| (n, compute_distance(&vectors[node], &vectors[n], metric)))
|
.map(|&n| (n, vectors.dist(node, n, metric)))
|
||||||
.collect();
|
.collect();
|
||||||
scored.sort_by(|a, b| a.1.total_cmp(&b.1).then(a.0.cmp(&b.0)));
|
scored.sort_by(|a, b| a.1.total_cmp(&b.1).then(a.0.cmp(&b.0)));
|
||||||
*neighbors = select_neighbors(vectors, &scored, max_conn, metric);
|
*neighbors = select_neighbors(vectors, &scored, max_conn, metric);
|
||||||
@@ -1519,6 +1787,88 @@ mod tests {
|
|||||||
assert!(recall >= 0.95, "incremental recall@10 = {recall}");
|
assert!(recall >= 0.95, "incremental recall@10 = {recall}");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn int8_storage_needs_an_exact_re_score_to_match_f32() {
|
||||||
|
// Cosine only: rows are unit-length, so a quantised dot product
|
||||||
|
// reconstructs the similarity directly.
|
||||||
|
//
|
||||||
|
// Not the `clustered` generator: its clusters are far tighter than any
|
||||||
|
// real embedding, so neighbours sit closer together than the
|
||||||
|
// quantisation error and top-10 identity there is noise — that would
|
||||||
|
// measure the fixture, not the storage.
|
||||||
|
let mut vectors = make_random_vectors(3060, 128, 5);
|
||||||
|
let queries = vectors.split_off(3000);
|
||||||
|
let f32_index =
|
||||||
|
HnswIndex::build_with(&vectors, 8, 40, DistanceMetric::Cosine, Storage::Float32);
|
||||||
|
let quantised =
|
||||||
|
HnswIndex::build_with(&vectors, 8, 40, DistanceMetric::Cosine, Storage::Int8);
|
||||||
|
assert_eq!(quantised.storage(), Storage::Int8);
|
||||||
|
|
||||||
|
// Ground truth, not the f32 index's answers: re-scoring can beat that
|
||||||
|
// index, and measuring against it would score being right as drift.
|
||||||
|
let truth: Vec<Vec<usize>> = queries
|
||||||
|
.iter()
|
||||||
|
.map(|q| {
|
||||||
|
let mut d: Vec<(usize, f32)> = vectors
|
||||||
|
.iter()
|
||||||
|
.enumerate()
|
||||||
|
.map(|(i, v)| (i, compute_distance(q, v, DistanceMetric::Cosine)))
|
||||||
|
.collect();
|
||||||
|
d.sort_by(|a, b| a.1.total_cmp(&b.1));
|
||||||
|
d[..10].iter().map(|x| x.0).collect()
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
|
||||||
|
let recall = |got: &dyn Fn(&[f32]) -> Vec<usize>| -> f64 {
|
||||||
|
let mut hits = 0;
|
||||||
|
for (q, want) in queries.iter().zip(&truth) {
|
||||||
|
hits += got(q).iter().filter(|id| want.contains(id)).count();
|
||||||
|
}
|
||||||
|
hits as f64 / (10 * queries.len()) as f64
|
||||||
|
};
|
||||||
|
|
||||||
|
let exact_recall = recall(&|q| f32_index.search(q, 10, 64).iter().map(|r| r.0).collect());
|
||||||
|
let raw_recall = recall(&|q| quantised.search(q, 10, 64).iter().map(|r| r.0).collect());
|
||||||
|
// Quantised distances alone cost recall, and `ef` cannot buy it back:
|
||||||
|
// the loss is in the distances, not in the graph.
|
||||||
|
assert!(
|
||||||
|
raw_recall < exact_recall,
|
||||||
|
"int8 alone should cost recall: {raw_recall} vs {exact_recall}"
|
||||||
|
);
|
||||||
|
|
||||||
|
// Re-scoring a wider candidate pool against the exact vectors — what a
|
||||||
|
// caller holding them (the agent's embedding cache) does — puts it
|
||||||
|
// back, because only the *ordering* was approximate.
|
||||||
|
let rescored_recall = recall(&|q| {
|
||||||
|
let mut pool: Vec<(usize, f32)> = quantised
|
||||||
|
.search(q, 40, 64)
|
||||||
|
.into_iter()
|
||||||
|
.map(|(id, _)| {
|
||||||
|
(
|
||||||
|
id,
|
||||||
|
compute_distance(q, &vectors[id], DistanceMetric::Cosine),
|
||||||
|
)
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
pool.sort_by(|a, b| a.1.total_cmp(&b.1));
|
||||||
|
pool.truncate(10);
|
||||||
|
pool.into_iter().map(|p| p.0).collect()
|
||||||
|
});
|
||||||
|
assert!(
|
||||||
|
rescored_recall >= exact_recall - 0.01,
|
||||||
|
"int8 + exact re-score should match f32: {rescored_recall} vs {exact_recall} (raw {raw_recall})"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn int8_storage_falls_back_to_f32_for_non_cosine_metrics() {
|
||||||
|
// L2 distance is not recoverable from a quantised dot product, so the
|
||||||
|
// store silently stays f32 rather than returning wrong distances.
|
||||||
|
let vectors = clustered(100, 8, 5, 3);
|
||||||
|
let index = HnswIndex::build_with(&vectors, 8, 40, DistanceMetric::L2, Storage::Int8);
|
||||||
|
assert_eq!(index.storage(), Storage::Float32);
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn deletions_near_the_query_do_not_shrink_or_degrade_results() {
|
fn deletions_near_the_query_do_not_shrink_or_degrade_results() {
|
||||||
let mut vectors = clustered(2040, 16, 20, 11);
|
let mut vectors = clustered(2040, 16, 20, 11);
|
||||||
@@ -1650,6 +2000,7 @@ mod tests {
|
|||||||
vec![1.2, 0.0], // 3
|
vec![1.2, 0.0], // 3
|
||||||
vec![-2.0, 0.0], // 4
|
vec![-2.0, 0.0], // 4
|
||||||
];
|
];
|
||||||
|
let store = Vectors::from_rows(&vectors, Storage::Float32, DistanceMetric::L2);
|
||||||
let scored: Vec<(usize, f32)> = (1..5)
|
let scored: Vec<(usize, f32)> = (1..5)
|
||||||
.map(|i| {
|
.map(|i| {
|
||||||
(
|
(
|
||||||
@@ -1659,12 +2010,12 @@ mod tests {
|
|||||||
})
|
})
|
||||||
.collect();
|
.collect();
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
select_neighbors(&vectors, &scored, 2, DistanceMetric::L2),
|
select_neighbors(&store, &scored, 2, DistanceMetric::L2),
|
||||||
[1, 4]
|
[1, 4]
|
||||||
);
|
);
|
||||||
// Spare capacity is filled with the closest rejected candidates.
|
// Spare capacity is filled with the closest rejected candidates.
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
select_neighbors(&vectors, &scored, 3, DistanceMetric::L2),
|
select_neighbors(&store, &scored, 3, DistanceMetric::L2),
|
||||||
[1, 4, 2]
|
[1, 4, 2]
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
@@ -1790,7 +2141,7 @@ mod tests {
|
|||||||
|
|
||||||
// Verify vectors match
|
// Verify vectors match
|
||||||
for i in 0..loaded.len() {
|
for i in 0..loaded.len() {
|
||||||
assert_eq!(loaded.vectors[i], index.vectors[i]);
|
assert_eq!(loaded.vectors.row(i), index.vectors.row(i));
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -5,4 +5,4 @@
|
|||||||
|
|
||||||
mod hnsw;
|
mod hnsw;
|
||||||
|
|
||||||
pub use hnsw::{DistanceMetric, HnswIndex};
|
pub use hnsw::{DistanceMetric, HnswIndex, Storage};
|
||||||
|
|||||||
@@ -1,7 +1,8 @@
|
|||||||
[package]
|
[package]
|
||||||
name = "clawhdf5-bench"
|
name = "clawhdf5-bench"
|
||||||
version = "2.5.0"
|
version = "2.7.0"
|
||||||
edition = "2024"
|
edition = "2024"
|
||||||
|
rust-version.workspace = true
|
||||||
description = "Benchmark harnesses for clawhdf5-agent (Track 8)"
|
description = "Benchmark harnesses for clawhdf5-agent (Track 8)"
|
||||||
license = "MIT"
|
license = "MIT"
|
||||||
|
|
||||||
@@ -33,6 +34,10 @@ path = "src/bin/consolidation_efficiency.rs"
|
|||||||
name = "ephemeral_perf"
|
name = "ephemeral_perf"
|
||||||
path = "src/bin/ephemeral_perf.rs"
|
path = "src/bin/ephemeral_perf.rs"
|
||||||
|
|
||||||
|
[[bin]]
|
||||||
|
name = "concurrent_read"
|
||||||
|
path = "src/bin/concurrent_read.rs"
|
||||||
|
|
||||||
[[bin]]
|
[[bin]]
|
||||||
name = "mpi_io_bench"
|
name = "mpi_io_bench"
|
||||||
path = "src/bin/mpi_io_bench.rs"
|
path = "src/bin/mpi_io_bench.rs"
|
||||||
@@ -63,6 +68,10 @@ clawhdf5-io = { path = "../clawhdf5-io" }
|
|||||||
mpi = { version = "0.8", optional = true }
|
mpi = { version = "0.8", optional = true }
|
||||||
serde = { workspace = true }
|
serde = { workspace = true }
|
||||||
serde_json = "1"
|
serde_json = "1"
|
||||||
|
# concurrent_read: size the decode pool (--decode-threads) and evict files
|
||||||
|
# from the page cache (--cold, posix_fadvise). Both pure Rust / bindings only.
|
||||||
|
rayon = "1"
|
||||||
|
libc = "0.2"
|
||||||
tempfile = { workspace = true }
|
tempfile = { workspace = true }
|
||||||
# Optional: libhdf5 C wrapper for side-by-side comparison (requires system libhdf5).
|
# Optional: libhdf5 C wrapper for side-by-side comparison (requires system libhdf5).
|
||||||
# Enable with: cargo bench -p clawhdf5-bench --features libhdf5-compare
|
# Enable with: cargo bench -p clawhdf5-bench --features libhdf5-compare
|
||||||
|
|||||||
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,70 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Tabulate concurrent_read JSON results (clawhdf5, h5py threads/processes).
|
||||||
|
|
||||||
|
python compare_concurrent_read.py clawhdf5.json h5py-threads.json h5py-procs.json
|
||||||
|
|
||||||
|
Prints one Markdown table: for each layout, mode and thread count, every
|
||||||
|
tool's MB/s and scaling efficiency, and the first file's MB/s relative to each
|
||||||
|
of the others. Refuses to compare runs whose workload parameters differ.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import json
|
||||||
|
import sys
|
||||||
|
|
||||||
|
COMPARED = ("datasets", "rows", "cols", "chunk", "deflate_level", "slab", "slabs", "seed")
|
||||||
|
|
||||||
|
|
||||||
|
def main(paths):
|
||||||
|
if len(paths) < 2:
|
||||||
|
sys.exit(__doc__)
|
||||||
|
docs = []
|
||||||
|
for p in paths:
|
||||||
|
with open(p) as fh:
|
||||||
|
docs.append(json.load(fh))
|
||||||
|
ref = docs[0]
|
||||||
|
for d, p in zip(docs[1:], paths[1:]):
|
||||||
|
diff = [k for k in COMPARED if d["params"].get(k) != ref["params"].get(k)]
|
||||||
|
if diff:
|
||||||
|
sys.exit(f"{p}: workload differs from {paths[0]} in {', '.join(diff)}")
|
||||||
|
if d["cache"] != ref["cache"]:
|
||||||
|
print(f"warning: {p} ran {d['cache']!r}, {paths[0]} ran {ref['cache']!r}",
|
||||||
|
file=sys.stderr)
|
||||||
|
if d.get("host") != ref.get("host"):
|
||||||
|
print(f"warning: {p} ran on {d.get('host')}, {paths[0]} on {ref.get('host')}",
|
||||||
|
file=sys.stderr)
|
||||||
|
|
||||||
|
names = [d["tool"] for d in docs]
|
||||||
|
for d in docs:
|
||||||
|
extra = f", HDF5 {d['hdf5_version']}" if "hdf5_version" in d else ""
|
||||||
|
print(f"- {d['tool']} {d['version']}{extra}: host {d.get('host')}, "
|
||||||
|
f"{d.get('cpus')} CPUs, cache {d['cache']}, decode threads per read "
|
||||||
|
f"{d.get('decode_threads')}")
|
||||||
|
p = ref["params"]
|
||||||
|
print(f"\n{p['datasets']} datasets of {p['rows']} x {p['cols']} f32, chunks "
|
||||||
|
f"{p['chunk'][0]} x {p['chunk'][1]} (deflate {p['deflate_level']}); "
|
||||||
|
f"`same`: {p['slabs']} slabs of {p['slab']} x {p['slab']}\n")
|
||||||
|
|
||||||
|
index = [{(r["layout"], r["mode"], r["threads"]): r for r in d["results"]} for d in docs]
|
||||||
|
keys = [(r["layout"], r["mode"], r["threads"]) for r in ref["results"]]
|
||||||
|
|
||||||
|
head = ["layout", "mode", "threads"]
|
||||||
|
head += [f"{n} MB/s (eff)" for n in names]
|
||||||
|
head += [f"{names[0]} / {n}" for n in names[1:]]
|
||||||
|
print("| " + " | ".join(head) + " |")
|
||||||
|
print("|---|---|" + "---:|" * (len(head) - 2))
|
||||||
|
for key in keys:
|
||||||
|
cells = [key[0], key[1], str(key[2])]
|
||||||
|
rs = [ix.get(key) for ix in index]
|
||||||
|
for r in rs:
|
||||||
|
if r is None:
|
||||||
|
cells.append("-")
|
||||||
|
else:
|
||||||
|
eff = "-" if r["efficiency"] is None else f"{r['efficiency']:.2f}"
|
||||||
|
cells.append(f"{r['mb_s']:.0f} ({eff})")
|
||||||
|
for r in rs[1:]:
|
||||||
|
cells.append("-" if r is None else f"{rs[0]['mb_s'] / r['mb_s']:.2f}x")
|
||||||
|
print("| " + " | ".join(cells) + " |")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main(sys.argv[1:])
|
||||||
@@ -0,0 +1,265 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""The concurrent_read workload with h5py, on the files concurrent_read wrote.
|
||||||
|
|
||||||
|
libhdf5 serialises every API call under one global lock, and h5py holds its
|
||||||
|
own global lock around every call as well, so h5py *threads* cannot decode in
|
||||||
|
parallel. h5py users scale with *processes* instead; ``--executor processes``
|
||||||
|
measures that (each worker opens the file itself).
|
||||||
|
|
||||||
|
The workload mirrors ``crates/clawhdf5-bench/src/bin/concurrent_read.rs``:
|
||||||
|
|
||||||
|
* ``distinct``: every dataset read in full once per repetition; worker ``t``
|
||||||
|
of ``T`` reads datasets ``t, t + T, ...``.
|
||||||
|
* ``same``: ``--slabs`` random ``--slab`` x ``--slab`` hyperslabs of ``d00``
|
||||||
|
(slab ``j`` to worker ``j % T``), offsets from the same splitmix64 stream.
|
||||||
|
|
||||||
|
Each worker times itself from a start barrier; a repetition spans the earliest
|
||||||
|
start to the latest finish (CLOCK_MONOTONIC, comparable across processes).
|
||||||
|
Threads share one ``h5py.File`` per repetition; process workers open the file
|
||||||
|
inside the timed region (a few ms against reads of many MiB).
|
||||||
|
|
||||||
|
Generate the files first with the Rust harness (it writes ``manifest.json``),
|
||||||
|
then, for example::
|
||||||
|
|
||||||
|
python concurrent_read_h5py.py --dir DIR --executor threads --json h5py-threads.json
|
||||||
|
python concurrent_read_h5py.py --dir DIR --executor processes --json h5py-procs.json
|
||||||
|
"""
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import json
|
||||||
|
import multiprocessing as mp
|
||||||
|
import os
|
||||||
|
import platform
|
||||||
|
import socket
|
||||||
|
import sys
|
||||||
|
import threading
|
||||||
|
import time
|
||||||
|
|
||||||
|
import h5py
|
||||||
|
import numpy as np
|
||||||
|
|
||||||
|
M64 = (1 << 64) - 1
|
||||||
|
|
||||||
|
|
||||||
|
def splitmix64(state):
|
||||||
|
"""Return (new_state, value); the same stream as the Rust harness."""
|
||||||
|
state = (state + 0x9E3779B97F4A7C15) & M64
|
||||||
|
z = state
|
||||||
|
z = ((z ^ (z >> 30)) * 0xBF58476D1CE4E5B9) & M64
|
||||||
|
z = ((z ^ (z >> 27)) * 0x94D049BB133111EB) & M64
|
||||||
|
return state, z ^ (z >> 31)
|
||||||
|
|
||||||
|
|
||||||
|
def value(k, i):
|
||||||
|
"""Element i (row-major) of dataset k, exactly as concurrent_read writes it."""
|
||||||
|
_, noise = splitmix64(i ^ (k << 40))
|
||||||
|
return np.float32((((i >> 6) % 16384) + k) + (noise & 0xFF) / 256.0)
|
||||||
|
|
||||||
|
|
||||||
|
def slab_offsets(seed, count, rows, cols, slab):
|
||||||
|
s = seed
|
||||||
|
out = []
|
||||||
|
for _ in range(count):
|
||||||
|
s, r = splitmix64(s)
|
||||||
|
s, c = splitmix64(s)
|
||||||
|
out.append((r % (rows - slab + 1), c % (cols - slab + 1)))
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def now():
|
||||||
|
return time.clock_gettime(time.CLOCK_MONOTONIC)
|
||||||
|
|
||||||
|
|
||||||
|
def work(f, mode, t, threads, m, slabs, slab, verify):
|
||||||
|
"""Worker t's share of one repetition on an open h5py.File."""
|
||||||
|
n = m["rows"] * m["cols"]
|
||||||
|
if mode == "distinct":
|
||||||
|
for k in range(t, m["datasets"], threads):
|
||||||
|
got = f[f"d{k:02d}"][...]
|
||||||
|
assert got.size == n
|
||||||
|
if verify:
|
||||||
|
flat = got.reshape(-1)
|
||||||
|
for i in (0, n // 3, n - 1):
|
||||||
|
assert flat[i] == value(k, i), f"d{k:02d}[{i}]"
|
||||||
|
else:
|
||||||
|
ds = f["d00"]
|
||||||
|
cols = m["cols"]
|
||||||
|
for r, c in slabs[t::threads]:
|
||||||
|
got = ds[r : r + slab, c : c + slab]
|
||||||
|
assert got.shape == (slab, slab)
|
||||||
|
if verify:
|
||||||
|
assert got[0, 0] == value(0, r * cols + c)
|
||||||
|
last = (r + slab - 1) * cols + c + slab - 1
|
||||||
|
assert got[-1, -1] == value(0, last)
|
||||||
|
|
||||||
|
|
||||||
|
# ----- process workers ------------------------------------------------------
|
||||||
|
|
||||||
|
_barrier = None
|
||||||
|
|
||||||
|
|
||||||
|
def _init(barrier):
|
||||||
|
global _barrier
|
||||||
|
_barrier = barrier
|
||||||
|
|
||||||
|
|
||||||
|
def _proc_task(task):
|
||||||
|
path, mode, t, threads, m, slabs, slab = task
|
||||||
|
_barrier.wait()
|
||||||
|
start = now()
|
||||||
|
with h5py.File(path, "r") as f:
|
||||||
|
work(f, mode, t, threads, m, slabs, slab, False)
|
||||||
|
return start, now()
|
||||||
|
|
||||||
|
|
||||||
|
def _noop(_):
|
||||||
|
return os.getpid()
|
||||||
|
|
||||||
|
|
||||||
|
def run_threads(path, mode, threads, m, slabs, slab):
|
||||||
|
spans = [None] * threads
|
||||||
|
barrier = threading.Barrier(threads)
|
||||||
|
with h5py.File(path, "r") as f:
|
||||||
|
|
||||||
|
def body(t):
|
||||||
|
barrier.wait()
|
||||||
|
start = now()
|
||||||
|
work(f, mode, t, threads, m, slabs, slab, False)
|
||||||
|
spans[t] = (start, now())
|
||||||
|
|
||||||
|
ts = [threading.Thread(target=body, args=(t,)) for t in range(threads)]
|
||||||
|
for th in ts:
|
||||||
|
th.start()
|
||||||
|
for th in ts:
|
||||||
|
th.join()
|
||||||
|
return max(e for _, e in spans) - min(s for s, _ in spans)
|
||||||
|
|
||||||
|
|
||||||
|
def run_processes(pool, path, mode, threads, m, slabs, slab):
|
||||||
|
tasks = [(path, mode, t, threads, m, slabs, slab) for t in range(threads)]
|
||||||
|
# One task per worker: each blocks in the barrier until all T have
|
||||||
|
# started, so no worker can take a second task.
|
||||||
|
spans = pool.map(_proc_task, tasks, chunksize=1)
|
||||||
|
return max(e for _, e in spans) - min(s for s, _ in spans)
|
||||||
|
|
||||||
|
|
||||||
|
def warm(path):
|
||||||
|
with open(path, "rb") as fh:
|
||||||
|
while fh.read(1 << 24):
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
def evict(path):
|
||||||
|
fd = os.open(path, os.O_RDONLY)
|
||||||
|
try:
|
||||||
|
os.posix_fadvise(fd, 0, 0, os.POSIX_FADV_DONTNEED)
|
||||||
|
finally:
|
||||||
|
os.close(fd)
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
ap = argparse.ArgumentParser(description=__doc__.split("\n\n")[0])
|
||||||
|
ap.add_argument("--dir", default="concurrent-read-data")
|
||||||
|
ap.add_argument("--executor", choices=["threads", "processes"], default="threads")
|
||||||
|
ap.add_argument("--threads", default="1,2,4,8,16")
|
||||||
|
ap.add_argument("--reps", type=int, default=3)
|
||||||
|
ap.add_argument("--slab", type=int, default=256)
|
||||||
|
ap.add_argument("--slabs", type=int, default=1024)
|
||||||
|
ap.add_argument("--seed", type=int, default=42)
|
||||||
|
ap.add_argument("--cold", action="store_true")
|
||||||
|
ap.add_argument("--modes", default="distinct,same")
|
||||||
|
ap.add_argument("--layouts", default="deflate,contiguous")
|
||||||
|
ap.add_argument("--json")
|
||||||
|
a = ap.parse_args()
|
||||||
|
|
||||||
|
# The Rust harness pins this value (splitmix64_reference).
|
||||||
|
assert splitmix64(42)[1] == 0xBDD732262FEB6E95, "splitmix64 port is wrong"
|
||||||
|
|
||||||
|
try:
|
||||||
|
with open(os.path.join(a.dir, "manifest.json")) as fh:
|
||||||
|
m = json.load(fh)
|
||||||
|
except FileNotFoundError:
|
||||||
|
sys.exit(f"{a.dir}/manifest.json not found: generate the files with "
|
||||||
|
"`cargo run --release -p clawhdf5-bench --bin concurrent_read -- --dir ...` first")
|
||||||
|
threads_list = [int(x) for x in a.threads.split(",")]
|
||||||
|
modes = a.modes.split(",")
|
||||||
|
layouts = a.layouts.split(",")
|
||||||
|
if a.slab < 1 or a.slab > min(m["rows"], m["cols"]):
|
||||||
|
sys.exit(f"--slab must be 1..={min(m['rows'], m['cols'])}")
|
||||||
|
files = dict(m["files"])
|
||||||
|
slabs = slab_offsets(a.seed, a.slabs, m["rows"], m["cols"], a.slab)
|
||||||
|
dataset_bytes = m["rows"] * m["cols"] * 4
|
||||||
|
tool = f"h5py-{a.executor}"
|
||||||
|
|
||||||
|
ctx = mp.get_context("spawn") # never fork a process holding HDF5 state
|
||||||
|
pools = {}
|
||||||
|
if a.executor == "processes":
|
||||||
|
for t in threads_list:
|
||||||
|
pool = ctx.Pool(t, initializer=_init, initargs=(ctx.Barrier(t),))
|
||||||
|
pool.map(_noop, range(t)) # start the workers outside the timing
|
||||||
|
pools[t] = pool
|
||||||
|
|
||||||
|
rows = []
|
||||||
|
print("| layout | mode | threads | MB/s | efficiency | median s |")
|
||||||
|
print("|---|---|---:|---:|---:|---:|")
|
||||||
|
try:
|
||||||
|
for layout in layouts:
|
||||||
|
path = os.path.join(a.dir, files[layout])
|
||||||
|
if not a.cold:
|
||||||
|
warm(path)
|
||||||
|
for mode in modes:
|
||||||
|
with h5py.File(path, "r") as f: # untimed, checked pass
|
||||||
|
work(f, mode, 0, 1, m, slabs, a.slab, True)
|
||||||
|
nbytes = (dataset_bytes * m["datasets"] if mode == "distinct"
|
||||||
|
else a.slab * a.slab * 4 * a.slabs)
|
||||||
|
base = None
|
||||||
|
for t in threads_list:
|
||||||
|
times = []
|
||||||
|
for _ in range(a.reps):
|
||||||
|
if a.cold:
|
||||||
|
evict(path)
|
||||||
|
if a.executor == "threads":
|
||||||
|
times.append(run_threads(path, mode, t, m, slabs, a.slab))
|
||||||
|
else:
|
||||||
|
times.append(run_processes(pools[t], path, mode, t, m, slabs, a.slab))
|
||||||
|
med = sorted(times)[len(times) // 2]
|
||||||
|
mb_s = nbytes / (1 << 20) / med
|
||||||
|
if t == 1:
|
||||||
|
base = mb_s
|
||||||
|
eff = mb_s / (t * base) if base else None
|
||||||
|
print(f"| {layout} | {mode} | {t} | {mb_s:.0f} | "
|
||||||
|
f"{'-' if eff is None else f'{eff:.2f}'} | {med:.4f} |")
|
||||||
|
rows.append({
|
||||||
|
"layout": layout, "mode": mode, "threads": t, "bytes": nbytes,
|
||||||
|
"times_s": times, "median_s": med, "mb_s": mb_s, "efficiency": eff,
|
||||||
|
})
|
||||||
|
finally:
|
||||||
|
for pool in pools.values():
|
||||||
|
pool.terminate()
|
||||||
|
|
||||||
|
if a.json:
|
||||||
|
doc = {
|
||||||
|
"tool": tool,
|
||||||
|
"version": h5py.__version__,
|
||||||
|
"hdf5_version": h5py.version.hdf5_version,
|
||||||
|
"python": platform.python_version(),
|
||||||
|
"host": socket.gethostname(),
|
||||||
|
"cpus": os.cpu_count(),
|
||||||
|
"unix_time": int(time.time()),
|
||||||
|
"cache": ("cold (posix_fadvise DONTNEED before each repetition)"
|
||||||
|
if a.cold else "warm"),
|
||||||
|
"decode_threads": 1,
|
||||||
|
"params": {
|
||||||
|
"datasets": m["datasets"], "rows": m["rows"], "cols": m["cols"],
|
||||||
|
"chunk": m["chunk"], "deflate_level": m["deflate_level"],
|
||||||
|
"mib": dataset_bytes // (1 << 20), "slab": a.slab, "slabs": a.slabs,
|
||||||
|
"seed": a.seed, "reps": a.reps, "dir": a.dir,
|
||||||
|
},
|
||||||
|
"results": rows,
|
||||||
|
}
|
||||||
|
with open(a.json, "w") as fh:
|
||||||
|
json.dump(doc, fh, indent=2)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
@@ -0,0 +1,523 @@
|
|||||||
|
//! Concurrent-read harness: how does decoded read throughput scale with the
|
||||||
|
//! number of threads reading one open file?
|
||||||
|
//!
|
||||||
|
//! libhdf5 (threadsafe build) serialises every API call under one global
|
||||||
|
//! mutex, and h5py holds it too, so threads cannot decode in parallel there.
|
||||||
|
//! A clawhdf5 [`File`] is `Send + Sync`; this harness measures what that buys.
|
||||||
|
//! `crates/clawhdf5-bench/scripts/concurrent_read_h5py.py` runs the same
|
||||||
|
//! workload on the same files with h5py (threads, and processes), and
|
||||||
|
//! `compare_concurrent_read.py` tabulates the JSON both write.
|
||||||
|
//!
|
||||||
|
//! Files (generated on first use, reused while `manifest.json` matches):
|
||||||
|
//!
|
||||||
|
//! * `<dir>/deflate.h5`: `--datasets` datasets `d00`, `d01`, ... of `f32`,
|
||||||
|
//! `--mib` MiB decoded each, shape `[mib * 256, 1024]`, chunks `256 x 256`,
|
||||||
|
//! deflate level 4.
|
||||||
|
//! * `<dir>/contiguous.h5`: the same datasets, contiguous.
|
||||||
|
//!
|
||||||
|
//! Modes, for each layout and each thread count `T` (strong scaling: the total
|
||||||
|
//! work per repetition is fixed, split among the threads):
|
||||||
|
//!
|
||||||
|
//! * `distinct`: every dataset is read in full once; thread `t` reads datasets
|
||||||
|
//! `t, t + T, t + 2T, ...`.
|
||||||
|
//! * `same`: all threads read `d00`, `--slabs` random `--slab` x `--slab`
|
||||||
|
//! hyperslabs in total (slab `j` goes to thread `j % T`). The offsets come
|
||||||
|
//! from a splitmix64 stream seeded with `--seed`, identical in the h5py
|
||||||
|
//! script.
|
||||||
|
//!
|
||||||
|
//! One `File` per layout per repetition is shared by all threads (opened
|
||||||
|
//! fresh each repetition, so no chunk cache carries over). Page cache:
|
||||||
|
//! `warm` (default) reads every file once before timing; `--cold` evicts the
|
||||||
|
//! files from the page cache with `posix_fadvise(POSIX_FADV_DONTNEED)` before
|
||||||
|
//! every repetition (no root needed; it only evicts clean, unmapped pages, so
|
||||||
|
//! it is best effort — the JSON says which was used).
|
||||||
|
//!
|
||||||
|
//! Decode inside one read is itself parallel when clawhdf5-format's `parallel`
|
||||||
|
//! feature is on (it is in this binary, via clawhdf5-agent). `--decode-threads
|
||||||
|
//! N` sizes that rayon pool; `--decode-threads 1` measures the API's own
|
||||||
|
//! thread scaling, comparable with h5py where each call decodes on the
|
||||||
|
//! calling thread.
|
||||||
|
//!
|
||||||
|
//! ```text
|
||||||
|
//! cargo run --release -p clawhdf5-bench --bin concurrent_read -- \
|
||||||
|
//! --dir /data/concurrent-read --json clawhdf5.json
|
||||||
|
//! cargo run --release -p clawhdf5-bench --bin concurrent_read -- \
|
||||||
|
//! --dir /tmp/cr --datasets 4 --mib 1 --threads 1,2 --slabs 16 --reps 1 # smoke
|
||||||
|
//! ```
|
||||||
|
|
||||||
|
use std::path::{Path, PathBuf};
|
||||||
|
use std::sync::Barrier;
|
||||||
|
use std::time::Instant;
|
||||||
|
|
||||||
|
use clawhdf5::{File, FileBuilder, Selection};
|
||||||
|
use serde::{Deserialize, Serialize};
|
||||||
|
|
||||||
|
const COLS: u64 = 1024;
|
||||||
|
const ROWS_PER_MIB: u64 = 256; // 256 rows x 1024 cols x 4 bytes = 1 MiB
|
||||||
|
const CHUNK: u64 = 256;
|
||||||
|
const DEFLATE_LEVEL: u32 = 4;
|
||||||
|
const LAYOUTS: [&str; 2] = ["deflate", "contiguous"];
|
||||||
|
const MANIFEST_VERSION: u32 = 1;
|
||||||
|
|
||||||
|
/// splitmix64 — shared with the h5py script, which must produce the same
|
||||||
|
/// stream (both the data and the hyperslab offsets depend on it).
|
||||||
|
fn splitmix64(state: &mut u64) -> u64 {
|
||||||
|
*state = state.wrapping_add(0x9E37_79B9_7F4A_7C15);
|
||||||
|
let mut z = *state;
|
||||||
|
z = (z ^ (z >> 30)).wrapping_mul(0xBF58_476D_1CE4_E5B9);
|
||||||
|
z = (z ^ (z >> 27)).wrapping_mul(0x94D0_49BB_1331_11EB);
|
||||||
|
z ^ (z >> 31)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Element `i` (row-major) of dataset `k`: a slowly varying integer part plus
|
||||||
|
/// 8 bits of noise, so deflate has real work to do (about 3.1x) and every value
|
||||||
|
/// is exact in `f32` (< 2^15 with 8 fraction bits), which lets both harnesses
|
||||||
|
/// check what they read against this formula.
|
||||||
|
fn value(k: u64, i: u64) -> f32 {
|
||||||
|
let mut s = i ^ (k << 40);
|
||||||
|
let noise = splitmix64(&mut s) & 0xff;
|
||||||
|
(((i >> 6) % 16384) + k) as f32 + noise as f32 / 256.0
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Serialize, Deserialize, PartialEq, Debug, Clone)]
|
||||||
|
struct Manifest {
|
||||||
|
version: u32,
|
||||||
|
datasets: u64,
|
||||||
|
rows: u64,
|
||||||
|
cols: u64,
|
||||||
|
chunk: [u64; 2],
|
||||||
|
deflate_level: u32,
|
||||||
|
files: Vec<(String, String)>, // (layout, file name)
|
||||||
|
writer: String,
|
||||||
|
}
|
||||||
|
|
||||||
|
fn manifest_for(datasets: u64, mib: u64) -> Manifest {
|
||||||
|
Manifest {
|
||||||
|
version: MANIFEST_VERSION,
|
||||||
|
datasets,
|
||||||
|
rows: mib * ROWS_PER_MIB,
|
||||||
|
cols: COLS,
|
||||||
|
chunk: [CHUNK, CHUNK],
|
||||||
|
deflate_level: DEFLATE_LEVEL,
|
||||||
|
files: LAYOUTS
|
||||||
|
.iter()
|
||||||
|
.map(|l| (l.to_string(), format!("{l}.h5")))
|
||||||
|
.collect(),
|
||||||
|
writer: format!("clawhdf5 {}", env!("CARGO_PKG_VERSION")),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn dataset_values(k: u64, n: u64) -> Vec<f32> {
|
||||||
|
(0..n).map(|i| value(k, i)).collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write the files unless `dir` already holds ones matching `want`.
|
||||||
|
fn ensure_files(dir: &Path, want: &Manifest) -> std::io::Result<bool> {
|
||||||
|
let manifest_path = dir.join("manifest.json");
|
||||||
|
if let Ok(text) = std::fs::read_to_string(&manifest_path)
|
||||||
|
&& let Ok(have) = serde_json::from_str::<Manifest>(&text)
|
||||||
|
&& have.version == want.version
|
||||||
|
&& have.datasets == want.datasets
|
||||||
|
&& have.rows == want.rows
|
||||||
|
&& have.cols == want.cols
|
||||||
|
&& have.chunk == want.chunk
|
||||||
|
&& have.deflate_level == want.deflate_level
|
||||||
|
&& have.files == want.files
|
||||||
|
&& want.files.iter().all(|(_, f)| dir.join(f).exists())
|
||||||
|
{
|
||||||
|
return Ok(false);
|
||||||
|
}
|
||||||
|
std::fs::create_dir_all(dir)?;
|
||||||
|
// A stale manifest must not survive a half-written regeneration.
|
||||||
|
let _ = std::fs::remove_file(&manifest_path);
|
||||||
|
let n = want.rows * want.cols;
|
||||||
|
for (layout, file) in &want.files {
|
||||||
|
// One layout at a time keeps the peak memory to about twice one
|
||||||
|
// file's decoded size.
|
||||||
|
let mut b = FileBuilder::new();
|
||||||
|
for k in 0..want.datasets {
|
||||||
|
let ds = b.create_dataset(&format!("d{k:02}"));
|
||||||
|
ds.with_f32_data(&dataset_values(k, n))
|
||||||
|
.with_shape(&[want.rows, want.cols]);
|
||||||
|
if layout == "deflate" {
|
||||||
|
ds.with_chunks(&[CHUNK.min(want.rows), CHUNK])
|
||||||
|
.with_deflate(DEFLATE_LEVEL);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
b.write(dir.join(file)).map_err(std::io::Error::other)?;
|
||||||
|
}
|
||||||
|
std::fs::write(
|
||||||
|
&manifest_path,
|
||||||
|
serde_json::to_string_pretty(want).map_err(std::io::Error::other)?,
|
||||||
|
)?;
|
||||||
|
Ok(true)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn slab_offsets(seed: u64, count: usize, rows: u64, cols: u64, slab: u64) -> Vec<(u64, u64)> {
|
||||||
|
let mut s = seed;
|
||||||
|
(0..count)
|
||||||
|
.map(|_| {
|
||||||
|
let r = splitmix64(&mut s) % (rows - slab + 1);
|
||||||
|
let c = splitmix64(&mut s) % (cols - slab + 1);
|
||||||
|
(r, c)
|
||||||
|
})
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Warm the page cache by reading every byte of `path`.
|
||||||
|
fn warm(path: &Path) -> std::io::Result<()> {
|
||||||
|
let mut f = std::fs::File::open(path)?;
|
||||||
|
std::io::copy(&mut f, &mut std::io::sink())?;
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Ask the kernel to drop `path`'s pages from the page cache.
|
||||||
|
fn evict(path: &Path) -> std::io::Result<()> {
|
||||||
|
use std::os::fd::AsRawFd;
|
||||||
|
let f = std::fs::File::open(path)?;
|
||||||
|
// SAFETY: plain syscall on a valid, open file descriptor.
|
||||||
|
let rc = unsafe { libc::posix_fadvise(f.as_raw_fd(), 0, 0, libc::POSIX_FADV_DONTNEED) };
|
||||||
|
if rc != 0 {
|
||||||
|
return Err(std::io::Error::from_raw_os_error(rc));
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Serialize)]
|
||||||
|
struct Row {
|
||||||
|
layout: String,
|
||||||
|
mode: String,
|
||||||
|
threads: usize,
|
||||||
|
/// Decoded (selected) bytes read per repetition.
|
||||||
|
bytes: u64,
|
||||||
|
times_s: Vec<f64>,
|
||||||
|
median_s: f64,
|
||||||
|
mb_s: f64,
|
||||||
|
/// `mb_s / (threads * mb_s at threads = 1)`; null without a 1-thread row.
|
||||||
|
efficiency: Option<f64>,
|
||||||
|
}
|
||||||
|
|
||||||
|
struct Args {
|
||||||
|
dir: PathBuf,
|
||||||
|
datasets: u64,
|
||||||
|
mib: u64,
|
||||||
|
threads: Vec<usize>,
|
||||||
|
reps: usize,
|
||||||
|
slab: u64,
|
||||||
|
slabs: usize,
|
||||||
|
seed: u64,
|
||||||
|
cold: bool,
|
||||||
|
decode_threads: usize,
|
||||||
|
modes: Vec<String>,
|
||||||
|
layouts: Vec<String>,
|
||||||
|
json: Option<PathBuf>,
|
||||||
|
}
|
||||||
|
|
||||||
|
const USAGE: &str = "\
|
||||||
|
usage: concurrent_read [--dir DIR] [--datasets N] [--mib N] [--threads 1,2,4,8,16]
|
||||||
|
[--reps N] [--slab N] [--slabs N] [--seed N] [--cold]
|
||||||
|
[--decode-threads N] [--modes distinct,same]
|
||||||
|
[--layouts deflate,contiguous] [--json FILE]";
|
||||||
|
|
||||||
|
fn parse_list<T: std::str::FromStr>(s: &str) -> Result<Vec<T>, String> {
|
||||||
|
s.split(',')
|
||||||
|
.map(|x| x.trim().parse().map_err(|_| format!("bad list item {x:?}")))
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parse_args() -> Result<Args, String> {
|
||||||
|
let mut a = Args {
|
||||||
|
dir: PathBuf::from("concurrent-read-data"),
|
||||||
|
datasets: 64,
|
||||||
|
mib: 64,
|
||||||
|
threads: vec![1, 2, 4, 8, 16],
|
||||||
|
reps: 3,
|
||||||
|
slab: 256,
|
||||||
|
slabs: 1024,
|
||||||
|
seed: 42,
|
||||||
|
cold: false,
|
||||||
|
decode_threads: 0,
|
||||||
|
modes: vec!["distinct".into(), "same".into()],
|
||||||
|
layouts: LAYOUTS.iter().map(|s| s.to_string()).collect(),
|
||||||
|
json: None,
|
||||||
|
};
|
||||||
|
let mut it = std::env::args().skip(1);
|
||||||
|
while let Some(flag) = it.next() {
|
||||||
|
if flag == "--cold" {
|
||||||
|
a.cold = true;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if flag == "-h" || flag == "--help" {
|
||||||
|
return Err(USAGE.into());
|
||||||
|
}
|
||||||
|
let v = it.next().ok_or(format!("{flag} needs a value\n{USAGE}"))?;
|
||||||
|
let num = |v: &str| {
|
||||||
|
v.parse::<u64>()
|
||||||
|
.map_err(|_| format!("{flag}: bad number {v:?}"))
|
||||||
|
};
|
||||||
|
match flag.as_str() {
|
||||||
|
"--dir" => a.dir = v.into(),
|
||||||
|
"--datasets" => a.datasets = num(&v)?,
|
||||||
|
"--mib" => a.mib = num(&v)?,
|
||||||
|
"--threads" => a.threads = parse_list(&v)?,
|
||||||
|
"--reps" => a.reps = num(&v)? as usize,
|
||||||
|
"--slab" => a.slab = num(&v)?,
|
||||||
|
"--slabs" => a.slabs = num(&v)? as usize,
|
||||||
|
"--seed" => a.seed = num(&v)?,
|
||||||
|
"--decode-threads" => a.decode_threads = num(&v)? as usize,
|
||||||
|
"--modes" => a.modes = parse_list(&v)?,
|
||||||
|
"--layouts" => a.layouts = parse_list(&v)?,
|
||||||
|
"--json" => a.json = Some(v.into()),
|
||||||
|
_ => return Err(format!("unknown flag {flag}\n{USAGE}")),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if a.datasets == 0 || a.datasets > 100 {
|
||||||
|
return Err("--datasets must be 1..=100".into());
|
||||||
|
}
|
||||||
|
if a.mib == 0 || a.reps == 0 || a.slabs == 0 || a.threads.contains(&0) {
|
||||||
|
return Err("--mib, --reps, --slabs and every --threads value must be > 0".into());
|
||||||
|
}
|
||||||
|
if a.slab == 0 || a.slab > COLS || a.slab > a.mib * ROWS_PER_MIB {
|
||||||
|
return Err(format!(
|
||||||
|
"--slab must be 1..={}",
|
||||||
|
COLS.min(a.mib * ROWS_PER_MIB)
|
||||||
|
));
|
||||||
|
}
|
||||||
|
for m in &a.modes {
|
||||||
|
if m != "distinct" && m != "same" {
|
||||||
|
return Err(format!("unknown mode {m:?}"));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for l in &a.layouts {
|
||||||
|
if !LAYOUTS.contains(&l.as_str()) {
|
||||||
|
return Err(format!("unknown layout {l:?}"));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(a)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One timed repetition: `T` threads on one shared `File`. Returns seconds.
|
||||||
|
fn run_once(
|
||||||
|
path: &Path,
|
||||||
|
mode: &str,
|
||||||
|
threads: usize,
|
||||||
|
m: &Manifest,
|
||||||
|
slabs: &[(u64, u64)],
|
||||||
|
slab: u64,
|
||||||
|
verify: bool,
|
||||||
|
) -> f64 {
|
||||||
|
let file = File::open(path).expect("open");
|
||||||
|
let barrier = Barrier::new(threads + 1); // + the spawning thread
|
||||||
|
let n = m.rows * m.cols;
|
||||||
|
// Each thread times itself from the barrier; the repetition spans the
|
||||||
|
// earliest start to the latest finish (timing on the spawning thread
|
||||||
|
// instead undercounts whenever it is scheduled after the workers ran).
|
||||||
|
let spans: Vec<(Instant, Instant)> = std::thread::scope(|s| {
|
||||||
|
let handles: Vec<_> = (0..threads)
|
||||||
|
.map(|t| {
|
||||||
|
let (file, barrier) = (&file, &barrier);
|
||||||
|
s.spawn(move || {
|
||||||
|
barrier.wait();
|
||||||
|
let start = Instant::now();
|
||||||
|
match mode {
|
||||||
|
"distinct" => {
|
||||||
|
for k in (t as u64..m.datasets).step_by(threads) {
|
||||||
|
let got = file.dataset(&format!("d{k:02}")).unwrap().read_f32();
|
||||||
|
let got = got.unwrap();
|
||||||
|
assert_eq!(got.len() as u64, n);
|
||||||
|
if verify {
|
||||||
|
for i in [0, n / 3, n - 1] {
|
||||||
|
assert_eq!(got[i as usize], value(k, i), "d{k:02}[{i}]");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
std::hint::black_box(got);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
_ => {
|
||||||
|
let ds = file.dataset("d00").unwrap();
|
||||||
|
for &(r, c) in slabs.iter().skip(t).step_by(threads) {
|
||||||
|
let sel = Selection::Hyperslab {
|
||||||
|
start: vec![r, c],
|
||||||
|
stride: vec![1, 1],
|
||||||
|
count: vec![slab, slab],
|
||||||
|
block: vec![1, 1],
|
||||||
|
};
|
||||||
|
let got = ds.read_f32_selection(&sel).unwrap();
|
||||||
|
assert_eq!(got.len() as u64, slab * slab);
|
||||||
|
if verify {
|
||||||
|
let last = (r + slab - 1) * m.cols + c + slab - 1;
|
||||||
|
assert_eq!(got[0], value(0, r * m.cols + c));
|
||||||
|
assert_eq!(*got.last().unwrap(), value(0, last));
|
||||||
|
}
|
||||||
|
std::hint::black_box(got);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
(start, Instant::now())
|
||||||
|
})
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
barrier.wait();
|
||||||
|
handles.into_iter().map(|h| h.join().unwrap()).collect()
|
||||||
|
});
|
||||||
|
let start = spans.iter().map(|s| s.0).min().unwrap();
|
||||||
|
let end = spans.iter().map(|s| s.1).max().unwrap();
|
||||||
|
(end - start).as_secs_f64()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn median(v: &[f64]) -> f64 {
|
||||||
|
let mut s = v.to_vec();
|
||||||
|
s.sort_by(f64::total_cmp);
|
||||||
|
s[s.len() / 2]
|
||||||
|
}
|
||||||
|
|
||||||
|
fn hostname() -> String {
|
||||||
|
std::fs::read_to_string("/proc/sys/kernel/hostname")
|
||||||
|
.map(|s| s.trim().to_string())
|
||||||
|
.unwrap_or_else(|_| "unknown".into())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn main() {
|
||||||
|
let args = match parse_args() {
|
||||||
|
Ok(a) => a,
|
||||||
|
Err(e) => {
|
||||||
|
eprintln!("{e}");
|
||||||
|
std::process::exit(2);
|
||||||
|
}
|
||||||
|
};
|
||||||
|
if cfg!(debug_assertions) {
|
||||||
|
eprintln!("warning: debug build — numbers are meaningless. Use --release.");
|
||||||
|
}
|
||||||
|
if args.decode_threads > 0 {
|
||||||
|
rayon::ThreadPoolBuilder::new()
|
||||||
|
.num_threads(args.decode_threads)
|
||||||
|
.build_global()
|
||||||
|
.expect("configure rayon pool");
|
||||||
|
}
|
||||||
|
|
||||||
|
let manifest = manifest_for(args.datasets, args.mib);
|
||||||
|
let t = Instant::now();
|
||||||
|
match ensure_files(&args.dir, &manifest) {
|
||||||
|
Ok(true) => eprintln!(
|
||||||
|
"generated {} in {:.1} s",
|
||||||
|
args.dir.display(),
|
||||||
|
t.elapsed().as_secs_f64()
|
||||||
|
),
|
||||||
|
Ok(false) => eprintln!("reusing {}", args.dir.display()),
|
||||||
|
Err(e) => {
|
||||||
|
eprintln!("cannot write test files in {}: {e}", args.dir.display());
|
||||||
|
std::process::exit(1);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let path_of = |layout: &str| args.dir.join(format!("{layout}.h5"));
|
||||||
|
let slabs = slab_offsets(
|
||||||
|
args.seed,
|
||||||
|
args.slabs,
|
||||||
|
manifest.rows,
|
||||||
|
manifest.cols,
|
||||||
|
args.slab,
|
||||||
|
);
|
||||||
|
let dataset_bytes = manifest.rows * manifest.cols * 4;
|
||||||
|
|
||||||
|
let mut rows: Vec<Row> = Vec::new();
|
||||||
|
println!("| layout | mode | threads | MB/s | efficiency | median s |");
|
||||||
|
println!("|---|---|---:|---:|---:|---:|");
|
||||||
|
for layout in &args.layouts {
|
||||||
|
let path = path_of(layout);
|
||||||
|
// Untimed pass: page cache warm (unless --cold), results checked.
|
||||||
|
if !args.cold {
|
||||||
|
warm(&path).expect("warm page cache");
|
||||||
|
}
|
||||||
|
for mode in &args.modes {
|
||||||
|
run_once(&path, mode, 1, &manifest, &slabs, args.slab, true);
|
||||||
|
let bytes = match mode.as_str() {
|
||||||
|
"distinct" => dataset_bytes * manifest.datasets,
|
||||||
|
_ => args.slab * args.slab * 4 * args.slabs as u64,
|
||||||
|
};
|
||||||
|
let mut base: Option<f64> = None;
|
||||||
|
for &threads in &args.threads {
|
||||||
|
let times: Vec<f64> = (0..args.reps)
|
||||||
|
.map(|_| {
|
||||||
|
if args.cold {
|
||||||
|
evict(&path).expect("posix_fadvise");
|
||||||
|
}
|
||||||
|
run_once(&path, mode, threads, &manifest, &slabs, args.slab, false)
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
let med = median(×);
|
||||||
|
let mb_s = bytes as f64 / (1 << 20) as f64 / med;
|
||||||
|
if threads == 1 {
|
||||||
|
base = Some(mb_s);
|
||||||
|
}
|
||||||
|
let efficiency = base.map(|b| mb_s / (threads as f64 * b));
|
||||||
|
println!(
|
||||||
|
"| {layout} | {mode} | {threads} | {mb_s:.0} | {} | {med:.4} |",
|
||||||
|
efficiency.map_or("-".into(), |e| format!("{e:.2}"))
|
||||||
|
);
|
||||||
|
rows.push(Row {
|
||||||
|
layout: layout.clone(),
|
||||||
|
mode: mode.clone(),
|
||||||
|
threads,
|
||||||
|
bytes,
|
||||||
|
times_s: times,
|
||||||
|
median_s: med,
|
||||||
|
mb_s,
|
||||||
|
efficiency,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if let Some(out) = &args.json {
|
||||||
|
let doc = serde_json::json!({
|
||||||
|
"tool": "clawhdf5",
|
||||||
|
"version": env!("CARGO_PKG_VERSION"),
|
||||||
|
"host": hostname(),
|
||||||
|
"cpus": std::thread::available_parallelism().map_or(0, |n| n.get()),
|
||||||
|
"unix_time": std::time::SystemTime::now()
|
||||||
|
.duration_since(std::time::UNIX_EPOCH)
|
||||||
|
.map_or(0, |d| d.as_secs()),
|
||||||
|
"cache": if args.cold { "cold (posix_fadvise DONTNEED before each repetition)" } else { "warm" },
|
||||||
|
"decode_threads": rayon::current_num_threads(),
|
||||||
|
"params": {
|
||||||
|
"datasets": manifest.datasets,
|
||||||
|
"mib": args.mib,
|
||||||
|
"rows": manifest.rows,
|
||||||
|
"cols": manifest.cols,
|
||||||
|
"chunk": manifest.chunk,
|
||||||
|
"deflate_level": manifest.deflate_level,
|
||||||
|
"slab": args.slab,
|
||||||
|
"slabs": args.slabs,
|
||||||
|
"seed": args.seed,
|
||||||
|
"reps": args.reps,
|
||||||
|
"dir": args.dir,
|
||||||
|
},
|
||||||
|
"results": rows,
|
||||||
|
});
|
||||||
|
std::fs::write(out, serde_json::to_string_pretty(&doc).unwrap()).expect("write json");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn values_are_exact_in_f32() {
|
||||||
|
for k in [0, 7, 63] {
|
||||||
|
for i in [0u64, 1, 4095, 1 << 20, (1 << 24) - 1] {
|
||||||
|
let v = value(k, i);
|
||||||
|
assert_eq!(v, (v as f64) as f32);
|
||||||
|
assert!(v < 32768.0);
|
||||||
|
assert_eq!((v * 256.0).fract(), 0.0);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The h5py script hard-codes this vector to check its splitmix64 port.
|
||||||
|
#[test]
|
||||||
|
fn splitmix64_reference() {
|
||||||
|
let mut s = 42;
|
||||||
|
assert_eq!(splitmix64(&mut s), 0xBDD7_3226_2FEB_6E95);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -396,7 +396,7 @@ fn run_memory_reduction_benchmark() {
|
|||||||
println!();
|
println!();
|
||||||
println!(
|
println!(
|
||||||
"{:>8} {:>10} {:>10} {:>10} {:>12}",
|
"{:>8} {:>10} {:>10} {:>10} {:>12}",
|
||||||
"Initial", "Remaining", "Eviction%", "Signal OK?", "BM25 Speedup"
|
"Initial", "Remaining", "Eviction%", "Signal OK?", "Records ÷"
|
||||||
);
|
);
|
||||||
println!("{}", "-".repeat(58));
|
println!("{}", "-".repeat(58));
|
||||||
|
|
||||||
@@ -440,7 +440,8 @@ fn run_memory_reduction_benchmark() {
|
|||||||
// Check all signal records survived
|
// Check all signal records survived
|
||||||
let signal_survived = signal_ids.iter().all(|&id| engine.get_by_id(id).is_some());
|
let signal_survived = signal_ids.iter().all(|&id| engine.get_by_id(id).is_some());
|
||||||
|
|
||||||
// Rough speedup: BM25 scales roughly linearly with record count
|
// How many times fewer records there are. Not a measured speedup —
|
||||||
|
// Part 1 measures search latency before and after.
|
||||||
let speedup = before_count as f64 / after_count.max(1) as f64;
|
let speedup = before_count as f64 / after_count.max(1) as f64;
|
||||||
|
|
||||||
println!(
|
println!(
|
||||||
@@ -480,7 +481,7 @@ fn main() {
|
|||||||
println!(" 3. Reducing search latency proportional to record reduction");
|
println!(" 3. Reducing search latency proportional to record reduction");
|
||||||
println!();
|
println!();
|
||||||
println!(
|
println!(
|
||||||
"Cycle time scales sub-linearly: 100 records ~microseconds, 100K records ~tens of ms."
|
"Cycle time grows a little faster than linearly: 100 records ~microseconds, 100K records ~tens of ms."
|
||||||
);
|
);
|
||||||
println!("Signal records with Correction source + high access_count survive eviction.");
|
println!("Signal records with Correction source + high access_count survive eviction.");
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -11,12 +11,14 @@
|
|||||||
//!
|
//!
|
||||||
//! Configuration matrix:
|
//! Configuration matrix:
|
||||||
//! - Text lengths: short (50 chars), medium (200 chars), long (1000 chars)
|
//! - Text lengths: short (50 chars), medium (200 chars), long (1000 chars)
|
||||||
//! - Embedding: 384-dim f32 (1536 bytes raw per record)
|
//! - Embedding: 384-dim, stored as float16 (the default for new stores) or
|
||||||
|
//! f32 with `--f32`; "raw" bytes are counted as f32 input either way
|
||||||
//! - WAL: enabled and disabled
|
//! - WAL: enabled and disabled
|
||||||
//!
|
//!
|
||||||
//! # Usage
|
//! # Usage
|
||||||
//! ```
|
//! ```
|
||||||
//! cargo run --release --bin footprint_bench
|
//! cargo run --release --bin footprint_bench # float16 stores
|
||||||
|
//! cargo run --release --bin footprint_bench -- --f32 # f32 stores
|
||||||
//! ```
|
//! ```
|
||||||
|
|
||||||
use std::time::Instant;
|
use std::time::Instant;
|
||||||
@@ -24,6 +26,9 @@ use std::time::Instant;
|
|||||||
use clawhdf5_agent::{AgentMemory, HDF5Memory, MemoryConfig, MemoryEntry};
|
use clawhdf5_agent::{AgentMemory, HDF5Memory, MemoryConfig, MemoryEntry};
|
||||||
use tempfile::TempDir;
|
use tempfile::TempDir;
|
||||||
|
|
||||||
|
/// `--f32`: build f32 stores instead of the library's float16 default.
|
||||||
|
static F32: std::sync::atomic::AtomicBool = std::sync::atomic::AtomicBool::new(false);
|
||||||
|
|
||||||
const EMBEDDING_DIM: usize = 384;
|
const EMBEDDING_DIM: usize = 384;
|
||||||
|
|
||||||
// Raw bytes per record: 384 f32 embeddings + median text + overhead
|
// Raw bytes per record: 384 f32 embeddings + median text + overhead
|
||||||
@@ -152,6 +157,9 @@ fn measure_footprint(
|
|||||||
config.compression = compression;
|
config.compression = compression;
|
||||||
config.compression_level = if compression { 6 } else { 0 };
|
config.compression_level = if compression { 6 } else { 0 };
|
||||||
config.compact_threshold = 0.0;
|
config.compact_threshold = 0.0;
|
||||||
|
if F32.load(std::sync::atomic::Ordering::Relaxed) {
|
||||||
|
config.float16 = false;
|
||||||
|
}
|
||||||
|
|
||||||
let mut memory = HDF5Memory::create(config).expect("HDF5Memory::create failed");
|
let mut memory = HDF5Memory::create(config).expect("HDF5Memory::create failed");
|
||||||
|
|
||||||
@@ -241,11 +249,19 @@ fn fmt_n(n: usize) -> String {
|
|||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
fn main() {
|
fn main() {
|
||||||
|
if std::env::args().skip(1).any(|a| a == "--f32") {
|
||||||
|
F32.store(true, std::sync::atomic::Ordering::Relaxed);
|
||||||
|
}
|
||||||
|
let stored = if F32.load(std::sync::atomic::Ordering::Relaxed) {
|
||||||
|
"f32 (1,536 bytes per record)"
|
||||||
|
} else {
|
||||||
|
"float16 (768 bytes per record; the default for new stores)"
|
||||||
|
};
|
||||||
println!("=================================================================");
|
println!("=================================================================");
|
||||||
println!(" ClawhDF5 Memory Footprint Benchmark");
|
println!(" ClawhDF5 Memory Footprint Benchmark");
|
||||||
println!("=================================================================");
|
println!("=================================================================");
|
||||||
println!();
|
println!();
|
||||||
println!("Embedding: 384-dim f32 = 1,536 bytes raw per record");
|
println!("Embedding: 384-dim, stored as {stored}; raw input counted as f32");
|
||||||
println!("Text lengths: short=50 chars, medium=200 chars, long=1000 chars");
|
println!("Text lengths: short=50 chars, medium=200 chars, long=1000 chars");
|
||||||
println!();
|
println!();
|
||||||
|
|
||||||
|
|||||||
@@ -57,21 +57,35 @@ mod embedder;
|
|||||||
|
|
||||||
use clawhdf5_agent::bm25::TokenFilter;
|
use clawhdf5_agent::bm25::TokenFilter;
|
||||||
use clawhdf5_agent::hybrid::Fusion;
|
use clawhdf5_agent::hybrid::Fusion;
|
||||||
use clawhdf5_agent::{AgentMemory, HDF5Memory, MemoryConfig, MemoryEntry};
|
use clawhdf5_agent::reranker::{ReRankConfig, RerankInput, rerank};
|
||||||
|
use clawhdf5_agent::{AgentMemory, HDF5Memory, MemoryConfig, MemoryEntry, SearchResult};
|
||||||
use serde::Deserialize;
|
use serde::Deserialize;
|
||||||
use tempfile::TempDir;
|
use tempfile::TempDir;
|
||||||
|
|
||||||
const EMBEDDING_DIM: usize = 384;
|
const EMBEDDING_DIM: usize = 384;
|
||||||
|
|
||||||
|
/// `--float16`: build every per-question store with `MemoryConfig::float16`,
|
||||||
|
/// so embeddings are rounded to half precision as they are saved — exactly
|
||||||
|
/// what such a store searches over.
|
||||||
|
static FLOAT16: std::sync::atomic::AtomicBool = std::sync::atomic::AtomicBool::new(false);
|
||||||
|
|
||||||
/// A mode's fusion, as one short string for the reports.
|
/// A mode's fusion, as one short string for the reports.
|
||||||
fn describe(mode: Mode) -> String {
|
fn describe(mode: Mode) -> String {
|
||||||
let fusion = match mode.fusion {
|
let fusion = match mode.fusion {
|
||||||
Fusion::Weighted { vector, keyword } => format!("vector_{vector:.1}_keyword_{keyword:.1}"),
|
Fusion::Weighted { vector, keyword } => format!("vector_{vector:.1}_keyword_{keyword:.1}"),
|
||||||
Fusion::Rrf { k } => format!("rrf_k{k:.0}"),
|
Fusion::Rrf { k } => format!("rrf_k{k:.0}"),
|
||||||
};
|
};
|
||||||
match mode.tokens {
|
let tokens = match mode.tokens {
|
||||||
TokenFilter::Plain => fusion,
|
TokenFilter::Plain => fusion,
|
||||||
TokenFilter::Stemmed => format!("{fusion}_stemmed"),
|
TokenFilter::Stemmed => format!("{fusion}_stemmed"),
|
||||||
|
};
|
||||||
|
match mode.rerank {
|
||||||
|
None => tokens,
|
||||||
|
Some(cfg) if cfg.relevance_weight == 0.0 => format!("{tokens}_rerank_metadata"),
|
||||||
|
Some(cfg) => format!(
|
||||||
|
"{tokens}_rerank_blended_hl{:.0}d",
|
||||||
|
cfg.temporal_half_life_secs / 86_400.0
|
||||||
|
),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -83,6 +97,9 @@ struct Mode {
|
|||||||
fusion: Fusion,
|
fusion: Fusion,
|
||||||
/// How keyword tokens are normalised before indexing and querying.
|
/// How keyword tokens are normalised before indexing and querying.
|
||||||
tokens: TokenFilter,
|
tokens: TokenFilter,
|
||||||
|
/// Re-rank the retrieved candidates with recency and friends, relative to
|
||||||
|
/// the question's own date.
|
||||||
|
rerank: Option<ReRankConfig>,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl Mode {
|
impl Mode {
|
||||||
@@ -91,9 +108,17 @@ impl Mode {
|
|||||||
label,
|
label,
|
||||||
fusion: Fusion::Weighted { vector, keyword },
|
fusion: Fusion::Weighted { vector, keyword },
|
||||||
tokens: TokenFilter::Plain,
|
tokens: TokenFilter::Plain,
|
||||||
|
rerank: None,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[cfg_attr(not(feature = "embeddings"), allow(dead_code))]
|
||||||
|
fn reranked(mut self, label: &'static str, rerank: ReRankConfig) -> Self {
|
||||||
|
self.label = label;
|
||||||
|
self.rerank = Some(rerank);
|
||||||
|
self
|
||||||
|
}
|
||||||
|
|
||||||
const fn stemmed(mut self, label: &'static str) -> Self {
|
const fn stemmed(mut self, label: &'static str) -> Self {
|
||||||
self.label = label;
|
self.label = label;
|
||||||
self.tokens = TokenFilter::Stemmed;
|
self.tokens = TokenFilter::Stemmed;
|
||||||
@@ -121,11 +146,61 @@ const RRF: Mode = Mode {
|
|||||||
label: "Hybrid (reciprocal rank fusion, k=60)",
|
label: "Hybrid (reciprocal rank fusion, k=60)",
|
||||||
fusion: Fusion::Rrf { k: 60.0 },
|
fusion: Fusion::Rrf { k: 60.0 },
|
||||||
tokens: TokenFilter::Plain,
|
tokens: TokenFilter::Plain,
|
||||||
|
rerank: None,
|
||||||
};
|
};
|
||||||
|
|
||||||
/// The same two configurations with stemmed keyword tokens, so the tokenizer's
|
/// The same two configurations with stemmed keyword tokens, so the tokenizer's
|
||||||
/// effect is isolated from everything else.
|
/// effect is isolated from everything else.
|
||||||
const BM25_STEMMED: Mode = BM25_ONLY.stemmed("BM25 only, stemmed tokens");
|
const BM25_STEMMED: Mode = BM25_ONLY.stemmed("BM25 only, stemmed tokens");
|
||||||
|
|
||||||
|
/// Re-ranking as it behaved before `relevance` was an input: the combined
|
||||||
|
/// score was recency + authority + activation only, so the retriever's own
|
||||||
|
/// ordering was discarded.
|
||||||
|
#[cfg(feature = "embeddings")]
|
||||||
|
fn hybrid_rerank_metadata_only() -> Mode {
|
||||||
|
HYBRID.reranked(
|
||||||
|
"Hybrid + rerank (metadata only, pre-fix)",
|
||||||
|
ReRankConfig {
|
||||||
|
relevance_weight: 0.0,
|
||||||
|
..ReRankConfig::default()
|
||||||
|
},
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Re-ranking as it behaves now: relevance leads, recency nudges.
|
||||||
|
#[cfg(feature = "embeddings")]
|
||||||
|
fn hybrid_rerank_blended() -> Mode {
|
||||||
|
HYBRID.reranked(
|
||||||
|
"Hybrid + rerank (relevance + recency)",
|
||||||
|
ReRankConfig::default(),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The same blend at several half-lives. Decay is `2^(-age / half_life)`, so a
|
||||||
|
/// half-life far shorter than the gaps between memories sends every score to
|
||||||
|
/// zero and the signal vanishes; far longer and everything scores ~1 and it
|
||||||
|
/// vanishes the other way. The right value tracks how far apart the memories
|
||||||
|
/// actually are.
|
||||||
|
#[cfg(feature = "embeddings")]
|
||||||
|
fn hybrid_rerank_half_lives() -> Vec<Mode> {
|
||||||
|
[
|
||||||
|
("1 day", 86_400.0),
|
||||||
|
("7 days", 7.0 * 86_400.0),
|
||||||
|
("30 days", 30.0 * 86_400.0),
|
||||||
|
("90 days", 90.0 * 86_400.0),
|
||||||
|
]
|
||||||
|
.into_iter()
|
||||||
|
.map(|(label, half_life)| {
|
||||||
|
HYBRID.reranked(
|
||||||
|
Box::leak(format!("Hybrid + rerank, half-life {label}").into_boxed_str()),
|
||||||
|
ReRankConfig {
|
||||||
|
temporal_half_life_secs: half_life,
|
||||||
|
..ReRankConfig::default()
|
||||||
|
},
|
||||||
|
)
|
||||||
|
})
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
#[cfg(feature = "embeddings")]
|
#[cfg(feature = "embeddings")]
|
||||||
const HYBRID_STEMMED: Mode = HYBRID.stemmed("Hybrid 0.4/0.6, stemmed tokens");
|
const HYBRID_STEMMED: Mode = HYBRID.stemmed("Hybrid 0.4/0.6, stemmed tokens");
|
||||||
|
|
||||||
@@ -217,6 +292,37 @@ struct Question {
|
|||||||
haystack_session_ids: Vec<String>,
|
haystack_session_ids: Vec<String>,
|
||||||
haystack_sessions: Vec<Vec<Turn>>,
|
haystack_sessions: Vec<Vec<Turn>>,
|
||||||
answer_session_ids: Vec<String>,
|
answer_session_ids: Vec<String>,
|
||||||
|
/// One timestamp per haystack session, e.g. "2023/05/25 (Thu) 20:21".
|
||||||
|
#[serde(default)]
|
||||||
|
haystack_dates: Vec<String>,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Seconds since the epoch for a LongMemEval session date, which looks like
|
||||||
|
/// `2023/05/25 (Thu) 20:21`. Sessions are stored in chronological order, so a
|
||||||
|
/// date that cannot be parsed falls back to its position — order is preserved
|
||||||
|
/// even if the interval is not.
|
||||||
|
fn session_time(date: &str, position: usize) -> f64 {
|
||||||
|
let stamp = |y: i64, mo: i64, d: i64, h: i64, mi: i64| -> f64 {
|
||||||
|
// Days since 1970-01-01 via the civil-from-days algorithm.
|
||||||
|
let (y, mo) = if mo <= 2 { (y - 1, mo + 12) } else { (y, mo) };
|
||||||
|
let era = y.div_euclid(400);
|
||||||
|
let yoe = y - era * 400;
|
||||||
|
let doy = (153 * (mo - 3) + 2) / 5 + d - 1;
|
||||||
|
let doe = yoe * 365 + yoe / 4 - yoe / 100 + doy;
|
||||||
|
let days = era * 146_097 + doe - 719_468;
|
||||||
|
(days * 86_400 + h * 3_600 + mi * 60) as f64
|
||||||
|
};
|
||||||
|
let parse = || -> Option<f64> {
|
||||||
|
let (ymd, rest) = date.split_once(' ')?;
|
||||||
|
let mut ymd = ymd.split('/');
|
||||||
|
let y = ymd.next()?.parse().ok()?;
|
||||||
|
let mo = ymd.next()?.parse().ok()?;
|
||||||
|
let d = ymd.next()?.parse().ok()?;
|
||||||
|
let hm = rest.rsplit(' ').next()?;
|
||||||
|
let (h, mi) = hm.split_once(':')?;
|
||||||
|
Some(stamp(y, mo, d, h.parse().ok()?, mi.parse().ok()?))
|
||||||
|
};
|
||||||
|
parse().unwrap_or(1_000_000.0 + position as f64 * 86_400.0)
|
||||||
}
|
}
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
@@ -235,11 +341,21 @@ struct Metrics {
|
|||||||
rr_turn: f64,
|
rr_turn: f64,
|
||||||
abstention_correct: u32,
|
abstention_correct: u32,
|
||||||
abstention_total: u32,
|
abstention_total: u32,
|
||||||
|
/// Questions where the newest gold session outranked the older ones, out
|
||||||
|
/// of those with more than one gold session and at least one retrieved.
|
||||||
|
newest_gold_first: u32,
|
||||||
|
newest_gold_total: u32,
|
||||||
latency_ns: Vec<u64>,
|
latency_ns: Vec<u64>,
|
||||||
count: u32,
|
count: u32,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl Metrics {
|
impl Metrics {
|
||||||
|
/// `None` when no question in this bucket had multiple gold sessions.
|
||||||
|
fn newest_gold_first_pct(&self) -> Option<f64> {
|
||||||
|
(self.newest_gold_total > 0)
|
||||||
|
.then(|| self.newest_gold_first as f64 / self.newest_gold_total as f64 * 100.0)
|
||||||
|
}
|
||||||
|
|
||||||
fn hit1_session_pct(&self) -> f64 {
|
fn hit1_session_pct(&self) -> f64 {
|
||||||
self.hit1_session as f64 / self.count.max(1) as f64 * 100.0
|
self.hit1_session as f64 / self.count.max(1) as f64 * 100.0
|
||||||
}
|
}
|
||||||
@@ -297,6 +413,16 @@ struct EvalResult {
|
|||||||
hit5_turn: bool,
|
hit5_turn: bool,
|
||||||
hit10_turn: bool,
|
hit10_turn: bool,
|
||||||
rr_turn: Option<f64>,
|
rr_turn: Option<f64>,
|
||||||
|
/// For a question whose evidence spans several dated sessions (a
|
||||||
|
/// `knowledge-update`, where an earlier fact is superseded by a later
|
||||||
|
/// one): did the *newest* gold session outrank every older gold session
|
||||||
|
/// that was returned? `None` when the question has one gold session, or
|
||||||
|
/// when none were retrieved, so there is nothing to discriminate.
|
||||||
|
///
|
||||||
|
/// Plain recall cannot see this. LongMemEval labels *both* the stale and
|
||||||
|
/// the updated session as gold, so returning either counts as a hit — yet
|
||||||
|
/// only one of them answers the question correctly.
|
||||||
|
newest_gold_first: Option<bool>,
|
||||||
latency: Duration,
|
latency: Duration,
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -310,6 +436,7 @@ fn evaluate_question(
|
|||||||
let mut config = MemoryConfig::new(dir.path().join("lme.h5"), "lme-bench", EMBEDDING_DIM);
|
let mut config = MemoryConfig::new(dir.path().join("lme.h5"), "lme-bench", EMBEDDING_DIM);
|
||||||
config.wal_enabled = false;
|
config.wal_enabled = false;
|
||||||
config.compact_threshold = 0.0;
|
config.compact_threshold = 0.0;
|
||||||
|
config.float16 = FLOAT16.load(std::sync::atomic::Ordering::Relaxed);
|
||||||
|
|
||||||
let mut memory = HDF5Memory::create(config).expect("failed to create HDF5Memory");
|
let mut memory = HDF5Memory::create(config).expect("failed to create HDF5Memory");
|
||||||
memory.set_token_filter(mode.tokens);
|
memory.set_token_filter(mode.tokens);
|
||||||
@@ -317,15 +444,21 @@ fn evaluate_question(
|
|||||||
// Build MemoryEntry list from all haystack sessions
|
// Build MemoryEntry list from all haystack sessions
|
||||||
let mut entries: Vec<MemoryEntry> = Vec::new();
|
let mut entries: Vec<MemoryEntry> = Vec::new();
|
||||||
let mut turn_has_answer: Vec<bool> = Vec::new();
|
let mut turn_has_answer: Vec<bool> = Vec::new();
|
||||||
let mut ts = 1_000_000.0f64;
|
|
||||||
|
|
||||||
for (sess_idx, session) in q.haystack_sessions.iter().enumerate() {
|
for (sess_idx, session) in q.haystack_sessions.iter().enumerate() {
|
||||||
let sess_id = q
|
let sess_id = q
|
||||||
.haystack_session_ids
|
.haystack_session_ids
|
||||||
.get(sess_idx)
|
.get(sess_idx)
|
||||||
.map(String::as_str)
|
.map(String::as_str)
|
||||||
.unwrap_or("unknown");
|
.unwrap_or("unknown");
|
||||||
for turn in session {
|
// Real session dates, not a synthetic counter: anything that decays
|
||||||
|
// with age needs true intervals, not just the right order.
|
||||||
|
let session_start = q
|
||||||
|
.haystack_dates
|
||||||
|
.get(sess_idx)
|
||||||
|
.map_or(sess_idx as f64 * 86_400.0, |d| session_time(d, sess_idx));
|
||||||
|
for (turn_idx, turn) in session.iter().enumerate() {
|
||||||
|
// Spread a session's turns over the minutes following its start.
|
||||||
|
let ts = session_start + turn_idx as f64 * 60.0;
|
||||||
entries.push(MemoryEntry {
|
entries.push(MemoryEntry {
|
||||||
chunk: turn.content.clone(),
|
chunk: turn.content.clone(),
|
||||||
embedding: embedding_for(embeddings, &turn.content),
|
embedding: embedding_for(embeddings, &turn.content),
|
||||||
@@ -339,7 +472,6 @@ fn evaluate_question(
|
|||||||
},
|
},
|
||||||
});
|
});
|
||||||
turn_has_answer.push(turn.has_answer);
|
turn_has_answer.push(turn.has_answer);
|
||||||
ts += 1.0;
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -356,11 +488,87 @@ fn evaluate_question(
|
|||||||
// Set of session IDs that contain the answer
|
// Set of session IDs that contain the answer
|
||||||
let answer_sess_set: HashSet<&str> = q.answer_session_ids.iter().map(String::as_str).collect();
|
let answer_sess_set: HashSet<&str> = q.answer_session_ids.iter().map(String::as_str).collect();
|
||||||
|
|
||||||
|
// When each gold session was recorded, so "newest" is by date rather than
|
||||||
|
// by position (the two agree in this dataset, but the metric should not
|
||||||
|
// depend on that).
|
||||||
|
let gold_times: HashMap<&str, f64> = q
|
||||||
|
.haystack_session_ids
|
||||||
|
.iter()
|
||||||
|
.enumerate()
|
||||||
|
.filter(|(_, sid)| answer_sess_set.contains(sid.as_str()))
|
||||||
|
.map(|(i, sid)| {
|
||||||
|
let t = q
|
||||||
|
.haystack_dates
|
||||||
|
.get(i)
|
||||||
|
.map_or(i as f64 * 86_400.0, |d| session_time(d, i));
|
||||||
|
(sid.as_str(), t)
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
|
||||||
let query_emb = embedding_for(embeddings, &q.question);
|
let query_emb = embedding_for(embeddings, &q.question);
|
||||||
let t0 = Instant::now();
|
let t0 = Instant::now();
|
||||||
let results = memory.hybrid_search_with(&query_emb, &q.question, mode.fusion, top_k);
|
// Re-ranking only reorders; it needs a candidate pool larger than `top_k`
|
||||||
|
// to have anything to promote.
|
||||||
|
let pool = if mode.rerank.is_some() {
|
||||||
|
top_k * 4
|
||||||
|
} else {
|
||||||
|
top_k
|
||||||
|
};
|
||||||
|
let mut results = memory.hybrid_search_with(&query_emb, &q.question, mode.fusion, pool);
|
||||||
|
if let Some(config) = mode.rerank {
|
||||||
|
// "Now" is the moment the question was asked, so decay measures how
|
||||||
|
// stale each memory was at that point.
|
||||||
|
let now = session_time(&q.question_date, q.haystack_sessions.len());
|
||||||
|
let inputs: Vec<RerankInput> = results
|
||||||
|
.iter()
|
||||||
|
.map(|r| RerankInput {
|
||||||
|
index: r.index,
|
||||||
|
timestamp: r.timestamp,
|
||||||
|
source_channel: r.source_channel.clone(),
|
||||||
|
raw_activation: r.activation,
|
||||||
|
relevance: r.score,
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
let order: Vec<usize> = rerank(&inputs, &config, now)
|
||||||
|
.into_iter()
|
||||||
|
.map(|r| r.index)
|
||||||
|
.collect();
|
||||||
|
let by_index: HashMap<usize, SearchResult> =
|
||||||
|
results.into_iter().map(|r| (r.index, r)).collect();
|
||||||
|
results = order
|
||||||
|
.into_iter()
|
||||||
|
.filter_map(|i| by_index.get(&i).cloned())
|
||||||
|
.collect();
|
||||||
|
}
|
||||||
|
results.truncate(top_k);
|
||||||
let latency = t0.elapsed();
|
let latency = t0.elapsed();
|
||||||
|
|
||||||
|
// Rank of the best-placed result from each gold session.
|
||||||
|
let mut first_rank: HashMap<&str, usize> = HashMap::new();
|
||||||
|
for (rank, result) in results.iter().enumerate() {
|
||||||
|
let sid = memory.cache.session_ids[result.index].as_str();
|
||||||
|
if let Some((gold_sid, _)) = gold_times.get_key_value(sid) {
|
||||||
|
first_rank.entry(gold_sid).or_insert(rank);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let newest_gold_first = if gold_times.len() < 2 || first_rank.is_empty() {
|
||||||
|
None
|
||||||
|
} else {
|
||||||
|
// The newest gold session must be retrieved, and no older gold session
|
||||||
|
// may outrank it.
|
||||||
|
let newest = gold_times
|
||||||
|
.iter()
|
||||||
|
.max_by(|a, b| a.1.total_cmp(b.1))
|
||||||
|
.map(|(sid, _)| *sid)
|
||||||
|
.expect("at least two gold sessions");
|
||||||
|
Some(match first_rank.get(newest) {
|
||||||
|
Some(&newest_rank) => first_rank
|
||||||
|
.iter()
|
||||||
|
.all(|(sid, &rank)| *sid == newest || rank > newest_rank),
|
||||||
|
None => false,
|
||||||
|
})
|
||||||
|
};
|
||||||
|
|
||||||
// Session-level recall
|
// Session-level recall
|
||||||
let mut hit1_session = false;
|
let mut hit1_session = false;
|
||||||
let mut hit5_session = false;
|
let mut hit5_session = false;
|
||||||
@@ -415,6 +623,7 @@ fn evaluate_question(
|
|||||||
hit5_turn,
|
hit5_turn,
|
||||||
hit10_turn,
|
hit10_turn,
|
||||||
rr_turn,
|
rr_turn,
|
||||||
|
newest_gold_first,
|
||||||
latency,
|
latency,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -566,6 +775,24 @@ fn print_report(
|
|||||||
);
|
);
|
||||||
println!();
|
println!();
|
||||||
|
|
||||||
|
if let Some(pct) = overall.newest_gold_first_pct() {
|
||||||
|
println!(
|
||||||
|
"## Recency Discrimination (n={})",
|
||||||
|
overall.newest_gold_total
|
||||||
|
);
|
||||||
|
println!(
|
||||||
|
" Newest gold session ranked first: {}/{} ({pct:.1}%)",
|
||||||
|
overall.newest_gold_first, overall.newest_gold_total
|
||||||
|
);
|
||||||
|
println!(
|
||||||
|
" Questions whose evidence spans several dated sessions — a fact and\n \
|
||||||
|
its later correction. Both sessions are labelled gold, so recall\n \
|
||||||
|
scores either as a hit; this asks whether the *current* one came\n \
|
||||||
|
first. A retriever with no sense of time scores near chance."
|
||||||
|
);
|
||||||
|
println!();
|
||||||
|
}
|
||||||
|
|
||||||
if overall.abstention_total > 0 {
|
if overall.abstention_total > 0 {
|
||||||
println!("## Abstention Accuracy");
|
println!("## Abstention Accuracy");
|
||||||
println!(
|
println!(
|
||||||
@@ -679,6 +906,14 @@ fn print_report(
|
|||||||
} else {
|
} else {
|
||||||
println!(" \"abstention_accuracy\": null,");
|
println!(" \"abstention_accuracy\": null,");
|
||||||
}
|
}
|
||||||
|
match overall.newest_gold_first_pct() {
|
||||||
|
Some(pct) => println!(
|
||||||
|
" \"newest_gold_first\": {:.4}, \"newest_gold_n\": {},",
|
||||||
|
pct / 100.0,
|
||||||
|
overall.newest_gold_total
|
||||||
|
),
|
||||||
|
None => println!(" \"newest_gold_first\": null,"),
|
||||||
|
}
|
||||||
println!(" \"latency_us\": {{");
|
println!(" \"latency_us\": {{");
|
||||||
println!(
|
println!(
|
||||||
" \"avg\": {:.1}, \"p50\": {:.1}, \"p95\": {:.1}, \"p99\": {:.1}",
|
" \"avg\": {:.1}, \"p50\": {:.1}, \"p95\": {:.1}, \"p99\": {:.1}",
|
||||||
@@ -701,6 +936,8 @@ fn main() {
|
|||||||
let mut limit: Option<usize> = None;
|
let mut limit: Option<usize> = None;
|
||||||
let mut weights_dir: Option<String> = None;
|
let mut weights_dir: Option<String> = None;
|
||||||
let mut sweep = false;
|
let mut sweep = false;
|
||||||
|
#[cfg_attr(not(feature = "embeddings"), allow(unused_mut, unused_variables))]
|
||||||
|
let mut rerank_sweep = false;
|
||||||
let mut args = std::env::args().skip(1);
|
let mut args = std::env::args().skip(1);
|
||||||
while let Some(arg) = args.next() {
|
while let Some(arg) = args.next() {
|
||||||
match arg.as_str() {
|
match arg.as_str() {
|
||||||
@@ -709,6 +946,20 @@ fn main() {
|
|||||||
limit = Some(v.parse().expect("--limit must be a positive integer"));
|
limit = Some(v.parse().expect("--limit must be a positive integer"));
|
||||||
}
|
}
|
||||||
"--sweep" => sweep = true,
|
"--sweep" => sweep = true,
|
||||||
|
"--float16" => {
|
||||||
|
FLOAT16.store(true, std::sync::atomic::Ordering::Relaxed);
|
||||||
|
eprintln!("Stores use MemoryConfig::float16 (half-precision embeddings)");
|
||||||
|
}
|
||||||
|
"--rerank-sweep" => {
|
||||||
|
// Re-ranking needs the vector stage to have candidates worth
|
||||||
|
// reordering, so this is an embeddings-only comparison.
|
||||||
|
#[cfg(feature = "embeddings")]
|
||||||
|
{
|
||||||
|
rerank_sweep = true;
|
||||||
|
}
|
||||||
|
#[cfg(not(feature = "embeddings"))]
|
||||||
|
eprintln!("warning: --rerank-sweep needs --features embeddings; ignoring");
|
||||||
|
}
|
||||||
"--embeddings" => {
|
"--embeddings" => {
|
||||||
weights_dir = Some(args.next().expect("--embeddings needs a directory"));
|
weights_dir = Some(args.next().expect("--embeddings needs a directory"));
|
||||||
}
|
}
|
||||||
@@ -727,6 +978,12 @@ fn main() {
|
|||||||
BM25-only, vector-only, and hybrid separately. Requires\n\
|
BM25-only, vector-only, and hybrid separately. Requires\n\
|
||||||
--features embeddings; without it the vector stage is\n\
|
--features embeddings; without it the vector stage is\n\
|
||||||
inert and only the BM25 row is produced.\n\
|
inert and only the BM25 row is produced.\n\
|
||||||
|
--rerank-sweep\n\
|
||||||
|
compare re-ranking off, metadata-only (the old\n\
|
||||||
|
behaviour) and blended at several half-lives.\n\
|
||||||
|
--float16\n\
|
||||||
|
build each store with MemoryConfig::float16, to\n\
|
||||||
|
compare retrieval on half-precision embeddings.\n\
|
||||||
--sweep instead of the three named modes, sweep vector_weight\n\
|
--sweep instead of the three named modes, sweep vector_weight\n\
|
||||||
from 0.0 to 1.0 in 0.1 steps. The 0.7/0.3 default was\n\
|
from 0.0 to 1.0 in 0.1 steps. The 0.7/0.3 default was\n\
|
||||||
never searched; this is what searches it."
|
never searched; this is what searches it."
|
||||||
@@ -792,6 +1049,10 @@ fn main() {
|
|||||||
{
|
{
|
||||||
if sweep {
|
if sweep {
|
||||||
sweep_modes()
|
sweep_modes()
|
||||||
|
} else if rerank_sweep {
|
||||||
|
let mut modes = vec![HYBRID, hybrid_rerank_metadata_only()];
|
||||||
|
modes.extend(hybrid_rerank_half_lives());
|
||||||
|
modes
|
||||||
} else {
|
} else {
|
||||||
vec![
|
vec![
|
||||||
BM25_ONLY,
|
BM25_ONLY,
|
||||||
@@ -800,6 +1061,8 @@ fn main() {
|
|||||||
RRF,
|
RRF,
|
||||||
BM25_STEMMED,
|
BM25_STEMMED,
|
||||||
HYBRID_STEMMED,
|
HYBRID_STEMMED,
|
||||||
|
hybrid_rerank_metadata_only(),
|
||||||
|
hybrid_rerank_blended(),
|
||||||
]
|
]
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -916,6 +1179,14 @@ fn run_mode(
|
|||||||
entry.rr_turn += rr;
|
entry.rr_turn += rr;
|
||||||
overall.rr_turn += rr;
|
overall.rr_turn += rr;
|
||||||
}
|
}
|
||||||
|
if let Some(newest_first) = result.newest_gold_first {
|
||||||
|
entry.newest_gold_total += 1;
|
||||||
|
overall.newest_gold_total += 1;
|
||||||
|
if newest_first {
|
||||||
|
entry.newest_gold_first += 1;
|
||||||
|
overall.newest_gold_first += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
let ns = result.latency.as_nanos() as u64;
|
let ns = result.latency.as_nanos() as u64;
|
||||||
entry.latency_ns.push(ns);
|
entry.latency_ns.push(ns);
|
||||||
@@ -927,3 +1198,30 @@ fn run_mode(
|
|||||||
eprintln!();
|
eprintln!();
|
||||||
print_report(&overall, &by_type, profile, mode);
|
print_report(&overall, &by_type, profile, mode);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::session_time;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn session_dates_parse_to_the_right_instant() {
|
||||||
|
// Reference values from Python's datetime, UTC.
|
||||||
|
for (date, expected) in [
|
||||||
|
("2023/05/25 (Thu) 20:21", 1_685_046_060.0),
|
||||||
|
("1970/01/01 (Thu) 00:00", 0.0),
|
||||||
|
("2000/02/29 (Tue) 12:00", 951_825_600.0),
|
||||||
|
("2023/12/31 (Sun) 23:59", 1_704_067_140.0),
|
||||||
|
("2024/03/01 (Fri) 00:00", 1_709_251_200.0),
|
||||||
|
] {
|
||||||
|
assert_eq!(session_time(date, 0), expected, "{date}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn unparseable_dates_fall_back_to_position_order() {
|
||||||
|
let a = session_time("not a date", 0);
|
||||||
|
let b = session_time("", 1);
|
||||||
|
let c = session_time("2023/13/99 (???) 99:99", 2);
|
||||||
|
assert!(a < b && b < c, "fallback must preserve session order");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -19,12 +19,15 @@
|
|||||||
//! cargo run --release -p clawhdf5-bench --bin search_harness -- --full # + 100K
|
//! cargo run --release -p clawhdf5-bench --bin search_harness -- --full # + 100K
|
||||||
//! cargo run --release -p clawhdf5-bench --bin search_harness -- --json out.json
|
//! cargo run --release -p clawhdf5-bench --bin search_harness -- --json out.json
|
||||||
//! cargo run --release -p clawhdf5-bench --bin search_harness -- --ann-only --uniform
|
//! cargo run --release -p clawhdf5-bench --bin search_harness -- --ann-only --uniform
|
||||||
|
//! cargo run --release -p clawhdf5-bench --bin search_harness -- --float16-study --full
|
||||||
|
//! cargo run --release -p clawhdf5-bench --bin search_harness -- --options-study --full
|
||||||
|
//! cargo run --release -p clawhdf5-bench --bin search_harness -- --signing-study --full
|
||||||
//! ```
|
//! ```
|
||||||
|
|
||||||
use std::time::{Duration, Instant};
|
use std::time::{Duration, Instant};
|
||||||
|
|
||||||
use clawhdf5_agent::{AgentMemory, HDF5Memory, MemoryConfig, MemoryEntry};
|
use clawhdf5_agent::{AgentMemory, HDF5Memory, MemoryConfig, MemoryEntry};
|
||||||
use clawhdf5_ann::{DistanceMetric, HnswIndex};
|
use clawhdf5_ann::{DistanceMetric, HnswIndex, Storage};
|
||||||
|
|
||||||
const DIM: usize = 384;
|
const DIM: usize = 384;
|
||||||
const K: usize = 10;
|
const K: usize = 10;
|
||||||
@@ -84,6 +87,25 @@ struct Dataset {
|
|||||||
/// that appears only on clustered data points at graph connectivity.
|
/// that appears only on clustered data points at graph connectivity.
|
||||||
static UNIFORM: std::sync::atomic::AtomicBool = std::sync::atomic::AtomicBool::new(false);
|
static UNIFORM: std::sync::atomic::AtomicBool = std::sync::atomic::AtomicBool::new(false);
|
||||||
|
|
||||||
|
/// `--int8`: build the HNSW index over int8-quantised vectors (a quarter of
|
||||||
|
/// the memory) instead of f32, to price the recall it costs.
|
||||||
|
static INT8: std::sync::atomic::AtomicBool = std::sync::atomic::AtomicBool::new(false);
|
||||||
|
|
||||||
|
/// `--f16-first`: in `--float16-study`, run the float16 store first.
|
||||||
|
static F16_FIRST: std::sync::atomic::AtomicBool = std::sync::atomic::AtomicBool::new(false);
|
||||||
|
|
||||||
|
/// `--rerank`: re-score the candidate pool against the exact vectors before
|
||||||
|
/// taking the top K.
|
||||||
|
static RERANK: std::sync::atomic::AtomicBool = std::sync::atomic::AtomicBool::new(false);
|
||||||
|
|
||||||
|
fn storage() -> Storage {
|
||||||
|
if INT8.load(std::sync::atomic::Ordering::Relaxed) {
|
||||||
|
Storage::Int8
|
||||||
|
} else {
|
||||||
|
Storage::Float32
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
fn make_dataset(n: usize, seed: u64) -> Dataset {
|
fn make_dataset(n: usize, seed: u64) -> Dataset {
|
||||||
let mut rng = Rng(seed);
|
let mut rng = Rng(seed);
|
||||||
if UNIFORM.load(std::sync::atomic::Ordering::Relaxed) {
|
if UNIFORM.load(std::sync::atomic::Ordering::Relaxed) {
|
||||||
@@ -169,6 +191,11 @@ fn text_for(cluster: usize, i: usize, rng: &mut Rng) -> String {
|
|||||||
// Measurement helpers
|
// Measurement helpers
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// Exact cosine distance between unit-length vectors.
|
||||||
|
fn exact_dist(a: &[f32], b: &[f32]) -> f32 {
|
||||||
|
1.0 - a.iter().zip(b).map(|(x, y)| x * y).sum::<f32>()
|
||||||
|
}
|
||||||
|
|
||||||
fn exact_top_k(vectors: &[Vec<f32>], query: &[f32], k: usize) -> Vec<usize> {
|
fn exact_top_k(vectors: &[Vec<f32>], query: &[f32], k: usize) -> Vec<usize> {
|
||||||
// Vectors are unit length, so cosine order == dot-product order.
|
// Vectors are unit length, so cosine order == dot-product order.
|
||||||
let mut scored: Vec<(usize, f32)> = vectors
|
let mut scored: Vec<(usize, f32)> = vectors
|
||||||
@@ -198,6 +225,83 @@ fn summarize(mut samples: Vec<Duration>) -> Latency {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Counts live heap bytes, so a structure's cost can be measured by
|
||||||
|
/// difference.
|
||||||
|
///
|
||||||
|
/// RSS cannot do this from inside one process: freeing a large structure
|
||||||
|
/// returns its pages to the allocator's pool rather than to the OS, so
|
||||||
|
/// allocating the next one shows no change. Measured that way, a store that
|
||||||
|
/// holds the corpus twice and one that holds it once look identical.
|
||||||
|
struct CountingAllocator;
|
||||||
|
|
||||||
|
static LIVE_BYTES: std::sync::atomic::AtomicI64 = std::sync::atomic::AtomicI64::new(0);
|
||||||
|
|
||||||
|
/// High-water mark of [`LIVE_BYTES`] since it was last reset.
|
||||||
|
///
|
||||||
|
/// Live bytes at a checkpoint cannot see a buffer that was allocated and
|
||||||
|
/// freed in between, and that is exactly the shape of a transient copy —
|
||||||
|
/// which still has to fit in memory while it exists.
|
||||||
|
static PEAK_BYTES: std::sync::atomic::AtomicI64 = std::sync::atomic::AtomicI64::new(0);
|
||||||
|
|
||||||
|
fn note_peak(live: i64) {
|
||||||
|
PEAK_BYTES.fetch_max(live, std::sync::atomic::Ordering::Relaxed);
|
||||||
|
}
|
||||||
|
|
||||||
|
// SAFETY: every method forwards to the system allocator with the same layout
|
||||||
|
// it was given, and only adds bookkeeping around it.
|
||||||
|
unsafe impl std::alloc::GlobalAlloc for CountingAllocator {
|
||||||
|
unsafe fn alloc(&self, layout: std::alloc::Layout) -> *mut u8 {
|
||||||
|
let ptr = unsafe { std::alloc::System.alloc(layout) };
|
||||||
|
if !ptr.is_null() {
|
||||||
|
let live = LIVE_BYTES
|
||||||
|
.fetch_add(layout.size() as i64, std::sync::atomic::Ordering::Relaxed)
|
||||||
|
+ layout.size() as i64;
|
||||||
|
note_peak(live);
|
||||||
|
}
|
||||||
|
ptr
|
||||||
|
}
|
||||||
|
|
||||||
|
unsafe fn dealloc(&self, ptr: *mut u8, layout: std::alloc::Layout) {
|
||||||
|
LIVE_BYTES.fetch_sub(layout.size() as i64, std::sync::atomic::Ordering::Relaxed);
|
||||||
|
unsafe { std::alloc::System.dealloc(ptr, layout) }
|
||||||
|
}
|
||||||
|
|
||||||
|
unsafe fn realloc(&self, ptr: *mut u8, layout: std::alloc::Layout, new_size: usize) -> *mut u8 {
|
||||||
|
let new_ptr = unsafe { std::alloc::System.realloc(ptr, layout, new_size) };
|
||||||
|
if !new_ptr.is_null() {
|
||||||
|
let delta = new_size as i64 - layout.size() as i64;
|
||||||
|
let live = LIVE_BYTES.fetch_add(delta, std::sync::atomic::Ordering::Relaxed) + delta;
|
||||||
|
note_peak(live);
|
||||||
|
}
|
||||||
|
new_ptr
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[global_allocator]
|
||||||
|
static ALLOCATOR: CountingAllocator = CountingAllocator;
|
||||||
|
|
||||||
|
/// Live heap bytes right now.
|
||||||
|
fn heap_bytes() -> u64 {
|
||||||
|
LIVE_BYTES.load(std::sync::atomic::Ordering::Relaxed).max(0) as u64
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Start watching for a new high-water mark from the current live total.
|
||||||
|
fn reset_peak() {
|
||||||
|
PEAK_BYTES.store(
|
||||||
|
LIVE_BYTES.load(std::sync::atomic::Ordering::Relaxed),
|
||||||
|
std::sync::atomic::Ordering::Relaxed,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The highest live total seen since [`reset_peak`].
|
||||||
|
fn peak_bytes() -> u64 {
|
||||||
|
PEAK_BYTES.load(std::sync::atomic::Ordering::Relaxed).max(0) as u64
|
||||||
|
}
|
||||||
|
|
||||||
|
fn mib(bytes: u64) -> f64 {
|
||||||
|
bytes as f64 / (1 << 20) as f64
|
||||||
|
}
|
||||||
|
|
||||||
fn micros(d: Duration) -> f64 {
|
fn micros(d: Duration) -> f64 {
|
||||||
d.as_secs_f64() * 1e6
|
d.as_secs_f64() * 1e6
|
||||||
}
|
}
|
||||||
@@ -219,11 +323,12 @@ fn bench_ann(n: usize, json: &mut Vec<serde_json::Value>) {
|
|||||||
.collect();
|
.collect();
|
||||||
|
|
||||||
let started = Instant::now();
|
let started = Instant::now();
|
||||||
let index = HnswIndex::build_with_metric(
|
let index = HnswIndex::build_with(
|
||||||
&data.vectors,
|
&data.vectors,
|
||||||
HNSW_M,
|
HNSW_M,
|
||||||
HNSW_EF_CONSTRUCTION,
|
HNSW_EF_CONSTRUCTION,
|
||||||
DistanceMetric::Cosine,
|
DistanceMetric::Cosine,
|
||||||
|
storage(),
|
||||||
);
|
);
|
||||||
let build = started.elapsed();
|
let build = started.elapsed();
|
||||||
|
|
||||||
@@ -240,7 +345,8 @@ fn bench_ann(n: usize, json: &mut Vec<serde_json::Value>) {
|
|||||||
);
|
);
|
||||||
|
|
||||||
println!(
|
println!(
|
||||||
"\n### HNSW, N = {n}, dim = {DIM}, M = {HNSW_M}, ef_construction = {HNSW_EF_CONSTRUCTION}\n"
|
"\n### HNSW, N = {n}, dim = {DIM}, M = {HNSW_M}, ef_construction = {HNSW_EF_CONSTRUCTION}, storage = {:?}\n",
|
||||||
|
index.storage()
|
||||||
);
|
);
|
||||||
println!(
|
println!(
|
||||||
"build: {:.1} ms ({:.0} vectors/s) · exact scan: {:.0} QPS, p50 {:.0} µs\n",
|
"build: {:.1} ms ({:.0} vectors/s) · exact scan: {:.0} QPS, p50 {:.0} µs\n",
|
||||||
@@ -251,12 +357,26 @@ fn bench_ann(n: usize, json: &mut Vec<serde_json::Value>) {
|
|||||||
);
|
);
|
||||||
println!("| ef | recall@{K} | QPS | p50 µs | p99 µs |");
|
println!("| ef | recall@{K} | QPS | p50 µs | p99 µs |");
|
||||||
println!("|---:|---:|---:|---:|---:|");
|
println!("|---:|---:|---:|---:|---:|");
|
||||||
|
// With a quantised index the distances it returns are approximate, so
|
||||||
|
// the candidates are re-scored against the exact vectors the caller
|
||||||
|
// already holds (in the agent, the embedding cache) before taking the
|
||||||
|
// top K. `--rerank` prices that: it costs one exact distance per
|
||||||
|
// candidate and is what decides whether int8 is usable.
|
||||||
|
let rerank = RERANK.load(std::sync::atomic::Ordering::Relaxed);
|
||||||
|
let pool = if rerank { K * 4 } else { K };
|
||||||
for ef in EF_VALUES {
|
for ef in EF_VALUES {
|
||||||
let mut hits = 0usize;
|
let mut hits = 0usize;
|
||||||
let mut samples = Vec::with_capacity(data.queries.len());
|
let mut samples = Vec::with_capacity(data.queries.len());
|
||||||
for (q, want) in data.queries.iter().zip(&truth) {
|
for (q, want) in data.queries.iter().zip(&truth) {
|
||||||
let t = Instant::now();
|
let t = Instant::now();
|
||||||
let got = index.search(q, K, ef);
|
let mut got = index.search(q, pool, ef.max(pool));
|
||||||
|
if rerank {
|
||||||
|
for cand in &mut got {
|
||||||
|
cand.1 = exact_dist(&data.vectors[cand.0], q);
|
||||||
|
}
|
||||||
|
got.select_nth_unstable_by(K - 1, |a, b| a.1.total_cmp(&b.1));
|
||||||
|
got.truncate(K);
|
||||||
|
}
|
||||||
samples.push(t.elapsed());
|
samples.push(t.elapsed());
|
||||||
hits += got.iter().filter(|(id, _)| want.contains(id)).count();
|
hits += got.iter().filter(|(id, _)| want.contains(id)).count();
|
||||||
}
|
}
|
||||||
@@ -369,6 +489,394 @@ fn bench_end_to_end(n: usize, json: &mut Vec<serde_json::Value>) {
|
|||||||
}));
|
}));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Signing study: what does an Ed25519-signed checkpoint cost?
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// `--signing-study`: checkpoint time unsigned vs signed, `verify` time, and
|
||||||
|
/// the file-size cost of the stored per-record hashes. Default store
|
||||||
|
/// settings (float16, int8 index). Medians of five checkpoints / three
|
||||||
|
/// verifies.
|
||||||
|
fn signing_study(n: usize) {
|
||||||
|
use clawhdf5_agent::signing::SigningKey;
|
||||||
|
let data = make_dataset(n, 0x516 ^ n as u64);
|
||||||
|
let mut rng = Rng(9);
|
||||||
|
let entries: Vec<MemoryEntry> = data
|
||||||
|
.vectors
|
||||||
|
.iter()
|
||||||
|
.enumerate()
|
||||||
|
.map(|(i, v)| MemoryEntry {
|
||||||
|
chunk: text_for(data.cluster_of[i], i, &mut rng),
|
||||||
|
embedding: v.clone(),
|
||||||
|
source_channel: "bench".into(),
|
||||||
|
timestamp: i as f64,
|
||||||
|
session_id: format!("s{}", i % 50),
|
||||||
|
tags: format!("t{i}"),
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
let dir = tempfile::tempdir().unwrap();
|
||||||
|
let path = dir.path().join("sign.h5");
|
||||||
|
let mut mem = HDF5Memory::create(MemoryConfig::new(path.clone(), "bench", DIM)).unwrap();
|
||||||
|
mem.save_batch(entries).unwrap();
|
||||||
|
std::hint::black_box(mem.hybrid_search(&data.queries[0], "", 1.0, 0.0, K));
|
||||||
|
|
||||||
|
let median = |mut v: Vec<Duration>| {
|
||||||
|
v.sort();
|
||||||
|
v[v.len() / 2]
|
||||||
|
};
|
||||||
|
let checkpoint = |mem: &mut HDF5Memory| {
|
||||||
|
median(
|
||||||
|
(0..5)
|
||||||
|
.map(|_| {
|
||||||
|
let t = Instant::now();
|
||||||
|
mem.flush_wal().unwrap();
|
||||||
|
t.elapsed()
|
||||||
|
})
|
||||||
|
.collect(),
|
||||||
|
)
|
||||||
|
};
|
||||||
|
let unsigned = checkpoint(&mut mem);
|
||||||
|
let unsigned_bytes = std::fs::metadata(&path).unwrap().len();
|
||||||
|
let key = SigningKey::from_bytes(&[7; 32]);
|
||||||
|
mem.set_signing_key(key.clone());
|
||||||
|
let signed = checkpoint(&mut mem);
|
||||||
|
let signed_bytes = std::fs::metadata(&path).unwrap().len();
|
||||||
|
drop(mem);
|
||||||
|
let vk = key.verifying_key();
|
||||||
|
let verify = median(
|
||||||
|
(0..3)
|
||||||
|
.map(|_| {
|
||||||
|
let t = Instant::now();
|
||||||
|
let r = HDF5Memory::verify(&path, &vk).unwrap();
|
||||||
|
let d = t.elapsed();
|
||||||
|
assert!(r.is_valid());
|
||||||
|
d
|
||||||
|
})
|
||||||
|
.collect(),
|
||||||
|
);
|
||||||
|
println!(
|
||||||
|
"| {n} | {:.1} | {:.1} | {:+.1} | {:.1} | {:+.2} |",
|
||||||
|
millis(unsigned),
|
||||||
|
millis(signed),
|
||||||
|
millis(signed) - millis(unsigned),
|
||||||
|
millis(verify),
|
||||||
|
(signed_bytes as f64 - unsigned_bytes as f64) / (1024.0 * 1024.0),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Search options study: source filters, re-ranking, confidence rejection
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// `--options-study`: what `HDF5Memory::search`'s options cost and whether a
|
||||||
|
/// filtered search finds the right records. Filters keep 50%, 10% or 1% of
|
||||||
|
/// the store at random, or two whole clusters away from the query (the case
|
||||||
|
/// the index cannot serve, which falls back to an exact scan). Recall is
|
||||||
|
/// vector-only against an exact scan of the allowed records; latency is full
|
||||||
|
/// hybrid search. Hebbian boosting is off.
|
||||||
|
fn options_study(n: usize) {
|
||||||
|
use clawhdf5_agent::SearchOptions;
|
||||||
|
use clawhdf5_agent::confidence::ConfidenceConfig;
|
||||||
|
use clawhdf5_agent::hybrid::Fusion;
|
||||||
|
use clawhdf5_agent::reranker::ReRankConfig;
|
||||||
|
|
||||||
|
let data = make_dataset(n, 0x0B7 ^ n as u64);
|
||||||
|
let n_clusters = data.cluster_of.iter().max().map_or(1, |m| m + 1);
|
||||||
|
let mut rng = Rng(5);
|
||||||
|
let bucket_of: Vec<usize> = (0..n).map(|_| rng.below(100)).collect();
|
||||||
|
let bucket = &bucket_of;
|
||||||
|
let query_texts: Vec<String> = data
|
||||||
|
.query_cluster
|
||||||
|
.iter()
|
||||||
|
.enumerate()
|
||||||
|
.map(|(i, c)| text_for(*c, i, &mut rng))
|
||||||
|
.collect();
|
||||||
|
let exact_top = |q: &[f32], allowed: &dyn Fn(usize) -> bool| -> Vec<usize> {
|
||||||
|
let mut s: Vec<(usize, f32)> = (0..n)
|
||||||
|
.filter(|&i| allowed(i))
|
||||||
|
.map(|i| (i, data.vectors[i].iter().zip(q).map(|(a, b)| a * b).sum()))
|
||||||
|
.collect();
|
||||||
|
s.sort_by(|a, b| b.1.total_cmp(&a.1).then(a.0.cmp(&b.0)));
|
||||||
|
s.into_iter().take(K).map(|(i, _)| i).collect()
|
||||||
|
};
|
||||||
|
|
||||||
|
// Two stores: channel = random bucket, and channel = cluster.
|
||||||
|
let dir = tempfile::tempdir().unwrap();
|
||||||
|
let mut stores = Vec::new();
|
||||||
|
for by_cluster in [false, true] {
|
||||||
|
let mut rng = Rng(3);
|
||||||
|
let entries: Vec<MemoryEntry> = data
|
||||||
|
.vectors
|
||||||
|
.iter()
|
||||||
|
.enumerate()
|
||||||
|
.map(|(i, v)| MemoryEntry {
|
||||||
|
chunk: text_for(data.cluster_of[i], i, &mut rng),
|
||||||
|
embedding: v.clone(),
|
||||||
|
source_channel: if by_cluster {
|
||||||
|
format!("c{}", data.cluster_of[i])
|
||||||
|
} else {
|
||||||
|
format!("b{}", bucket[i])
|
||||||
|
},
|
||||||
|
timestamp: i as f64,
|
||||||
|
session_id: format!("s{}", i % 50),
|
||||||
|
tags: format!("t{i}"),
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
let mut config = MemoryConfig::new(
|
||||||
|
dir.path().join(format!("opt_{by_cluster}.h5")),
|
||||||
|
"bench",
|
||||||
|
DIM,
|
||||||
|
);
|
||||||
|
config.hebbian_boost = 0.0;
|
||||||
|
let mut mem = HDF5Memory::create(config).unwrap();
|
||||||
|
mem.save_batch(entries).unwrap();
|
||||||
|
std::hint::black_box(mem.search(&data.queries[0], "", &SearchOptions::new(K)));
|
||||||
|
stores.push(mem);
|
||||||
|
}
|
||||||
|
|
||||||
|
let vector_only = SearchOptions::new(K).with_fusion(Fusion::Weighted {
|
||||||
|
vector: 1.0,
|
||||||
|
keyword: 0.0,
|
||||||
|
});
|
||||||
|
// (label, store, channels for query i, allowed(i, record))
|
||||||
|
type Case<'a> = (
|
||||||
|
String,
|
||||||
|
usize,
|
||||||
|
Box<dyn Fn(usize) -> Option<Vec<String>> + 'a>,
|
||||||
|
Box<dyn Fn(usize, usize) -> bool + 'a>,
|
||||||
|
);
|
||||||
|
let mut cases: Vec<Case> = vec![(
|
||||||
|
"no filter".into(),
|
||||||
|
0,
|
||||||
|
Box::new(|_| None),
|
||||||
|
Box::new(|_, _| true),
|
||||||
|
)];
|
||||||
|
for pct in [50usize, 10, 1] {
|
||||||
|
cases.push((
|
||||||
|
format!("random {pct}%"),
|
||||||
|
0,
|
||||||
|
Box::new(move |_| Some((0..pct).map(|b| format!("b{b}")).collect())),
|
||||||
|
Box::new(move |_, i| bucket[i] < pct),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
let d = &data;
|
||||||
|
let away = move |qi: usize| {
|
||||||
|
let qc = d.query_cluster[qi];
|
||||||
|
[
|
||||||
|
(qc + n_clusters / 3) % n_clusters,
|
||||||
|
(qc + 2 * n_clusters / 3) % n_clusters,
|
||||||
|
]
|
||||||
|
};
|
||||||
|
cases.push((
|
||||||
|
"2 clusters away from the query".into(),
|
||||||
|
1,
|
||||||
|
Box::new(move |qi| Some(away(qi).iter().map(|c| format!("c{c}")).collect())),
|
||||||
|
Box::new(move |qi, i| away(qi).contains(&d.cluster_of[i])),
|
||||||
|
));
|
||||||
|
|
||||||
|
for (label, store, channels, allowed) in &cases {
|
||||||
|
let mem = &mut stores[*store];
|
||||||
|
let mut hits = 0;
|
||||||
|
let mut kept = 0;
|
||||||
|
for (qi, q) in data.queries.iter().enumerate() {
|
||||||
|
let mut opts = vector_only.clone();
|
||||||
|
opts.source_channels = channels(qi);
|
||||||
|
let got = mem.search(q, "", &opts);
|
||||||
|
let want = exact_top(q, &|i| allowed(qi, i));
|
||||||
|
kept += want.len();
|
||||||
|
hits += got.iter().filter(|r| want.contains(&r.index)).count();
|
||||||
|
}
|
||||||
|
let latency = summarize(
|
||||||
|
(0..N_QUERIES)
|
||||||
|
.map(|qi| {
|
||||||
|
let mut opts = SearchOptions::new(K);
|
||||||
|
opts.source_channels = channels(qi);
|
||||||
|
let t = Instant::now();
|
||||||
|
std::hint::black_box(mem.search(&data.queries[qi], &query_texts[qi], &opts));
|
||||||
|
t.elapsed()
|
||||||
|
})
|
||||||
|
.collect(),
|
||||||
|
);
|
||||||
|
println!(
|
||||||
|
"| {n} | {label} | {:.4} | {:.3} | {:.3} |",
|
||||||
|
hits as f64 / kept.max(1) as f64,
|
||||||
|
millis(latency.p50),
|
||||||
|
millis(latency.p99),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
let mem = &mut stores[0];
|
||||||
|
for (label, opts) in [
|
||||||
|
(
|
||||||
|
"re-rank",
|
||||||
|
SearchOptions::new(K).with_rerank(ReRankConfig::default()),
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"re-rank + confidence",
|
||||||
|
SearchOptions::new(K)
|
||||||
|
.with_rerank(ReRankConfig::default())
|
||||||
|
.with_confidence(ConfidenceConfig::default()),
|
||||||
|
),
|
||||||
|
] {
|
||||||
|
let latency = summarize(
|
||||||
|
(0..N_QUERIES)
|
||||||
|
.map(|qi| {
|
||||||
|
let t = Instant::now();
|
||||||
|
std::hint::black_box(mem.search(&data.queries[qi], &query_texts[qi], &opts));
|
||||||
|
t.elapsed()
|
||||||
|
})
|
||||||
|
.collect(),
|
||||||
|
);
|
||||||
|
println!(
|
||||||
|
"| {n} | {label} | — | {:.3} | {:.3} |",
|
||||||
|
millis(latency.p50),
|
||||||
|
millis(latency.p99)
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// float16 study: what does half-precision embedding storage cost?
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// `--float16-study`: the same data in an `f32` store and a `float16` store.
|
||||||
|
/// Reports file size, checkpoint and open time, vector-search recall@10
|
||||||
|
/// against an exact scan of the *original* f32 vectors, how often the two
|
||||||
|
/// stores return the same top 10, and `hybrid_search` latency. Hebbian
|
||||||
|
/// boosting is off, so every query sees the same store.
|
||||||
|
fn float16_study(n: usize) {
|
||||||
|
let data = make_dataset(n, 0xF16 ^ n as u64);
|
||||||
|
let mut rng = Rng(11);
|
||||||
|
let query_texts: Vec<String> = data
|
||||||
|
.query_cluster
|
||||||
|
.iter()
|
||||||
|
.enumerate()
|
||||||
|
.map(|(i, c)| text_for(*c, i, &mut rng))
|
||||||
|
.collect();
|
||||||
|
|
||||||
|
// Exact top K by cosine (the vectors are unit length) on the f32 inputs.
|
||||||
|
let exact: Vec<Vec<usize>> = data
|
||||||
|
.queries
|
||||||
|
.iter()
|
||||||
|
.map(|q| {
|
||||||
|
let mut scored: Vec<(usize, f32)> = data
|
||||||
|
.vectors
|
||||||
|
.iter()
|
||||||
|
.enumerate()
|
||||||
|
.map(|(i, v)| (i, v.iter().zip(q).map(|(a, b)| a * b).sum()))
|
||||||
|
.collect();
|
||||||
|
scored.sort_by(|a, b| b.1.total_cmp(&a.1).then(a.0.cmp(&b.0)));
|
||||||
|
scored.into_iter().take(K).map(|(i, _)| i).collect()
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
|
||||||
|
let dir = tempfile::tempdir().unwrap();
|
||||||
|
let mut per_variant: Vec<(bool, Vec<Vec<usize>>)> = Vec::new();
|
||||||
|
// `--f16-first` swaps the order, to check the numbers do not depend on
|
||||||
|
// which store runs first (page cache, allocator, CPU frequency).
|
||||||
|
let order = if F16_FIRST.load(std::sync::atomic::Ordering::Relaxed) {
|
||||||
|
[true, false]
|
||||||
|
} else {
|
||||||
|
[false, true]
|
||||||
|
};
|
||||||
|
for float16 in order {
|
||||||
|
let path = dir.path().join(format!("f16study_{float16}.h5"));
|
||||||
|
let mut rng = Rng(3);
|
||||||
|
let entries: Vec<MemoryEntry> = data
|
||||||
|
.vectors
|
||||||
|
.iter()
|
||||||
|
.enumerate()
|
||||||
|
.map(|(i, v)| MemoryEntry {
|
||||||
|
chunk: text_for(data.cluster_of[i], i, &mut rng),
|
||||||
|
embedding: v.clone(),
|
||||||
|
source_channel: "bench".into(),
|
||||||
|
timestamp: i as f64,
|
||||||
|
session_id: format!("s{}", i % 50),
|
||||||
|
tags: format!("t{i}"),
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
let mut config = MemoryConfig::new(path.clone(), "bench", DIM);
|
||||||
|
config.float16 = float16;
|
||||||
|
config.hebbian_boost = 0.0;
|
||||||
|
let mut mem = HDF5Memory::create(config).unwrap();
|
||||||
|
mem.save_batch(entries).unwrap();
|
||||||
|
// Build the indexes, then time a checkpoint that writes everything.
|
||||||
|
std::hint::black_box(mem.hybrid_search(&data.queries[0], "", 1.0, 0.0, K));
|
||||||
|
let t = Instant::now();
|
||||||
|
mem.flush_wal().unwrap();
|
||||||
|
let checkpoint = t.elapsed();
|
||||||
|
drop(mem);
|
||||||
|
let file_bytes = std::fs::metadata(&path).unwrap().len();
|
||||||
|
|
||||||
|
// Median of three opens.
|
||||||
|
let mut opens: Vec<Duration> = (0..3)
|
||||||
|
.map(|_| {
|
||||||
|
let t = Instant::now();
|
||||||
|
let m = HDF5Memory::open(&path).unwrap();
|
||||||
|
let d = t.elapsed();
|
||||||
|
drop(m);
|
||||||
|
d
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
opens.sort();
|
||||||
|
let mut mem = HDF5Memory::open(&path).unwrap();
|
||||||
|
|
||||||
|
// Vector-only search: empty text, all weight on the vector stage.
|
||||||
|
let results: Vec<Vec<usize>> = data
|
||||||
|
.queries
|
||||||
|
.iter()
|
||||||
|
.map(|q| {
|
||||||
|
mem.hybrid_search(q, "", 1.0, 0.0, K)
|
||||||
|
.iter()
|
||||||
|
.map(|r| r.index)
|
||||||
|
.collect()
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
let hits: usize = results
|
||||||
|
.iter()
|
||||||
|
.zip(&exact)
|
||||||
|
.map(|(got, want)| got.iter().filter(|i| want.contains(i)).count())
|
||||||
|
.sum();
|
||||||
|
let recall = hits as f64 / (K * data.queries.len()) as f64;
|
||||||
|
|
||||||
|
let latency = summarize(
|
||||||
|
(0..N_QUERIES)
|
||||||
|
.map(|i| {
|
||||||
|
let t = Instant::now();
|
||||||
|
std::hint::black_box(mem.hybrid_search(
|
||||||
|
&data.queries[i],
|
||||||
|
&query_texts[i],
|
||||||
|
0.4,
|
||||||
|
0.6,
|
||||||
|
K,
|
||||||
|
));
|
||||||
|
t.elapsed()
|
||||||
|
})
|
||||||
|
.collect(),
|
||||||
|
);
|
||||||
|
let overlap = match per_variant.first() {
|
||||||
|
Some((_, other)) => {
|
||||||
|
let same: usize = results
|
||||||
|
.iter()
|
||||||
|
.zip(other)
|
||||||
|
.map(|(a, b)| a.iter().filter(|i| b.contains(i)).count())
|
||||||
|
.sum();
|
||||||
|
format!("{:.4}", same as f64 / (K * data.queries.len()) as f64)
|
||||||
|
}
|
||||||
|
None => "—".into(),
|
||||||
|
};
|
||||||
|
println!(
|
||||||
|
"| {n} | {} | {:.1} | {:.0} | {:.1} | {recall:.4} | {overlap} | {:.3} |",
|
||||||
|
if float16 { "float16" } else { "f32" },
|
||||||
|
mib(file_bytes),
|
||||||
|
millis(checkpoint),
|
||||||
|
millis(opens[1]),
|
||||||
|
millis(latency.p50),
|
||||||
|
);
|
||||||
|
per_variant.push((float16, results));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// Fusion study: does capping the keyword candidate pool change the ranking?
|
// Fusion study: does capping the keyword candidate pool change the ranking?
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
@@ -394,11 +902,12 @@ fn fusion_study(n: usize) {
|
|||||||
.map(|(i, c)| text_for(*c, i, &mut rng))
|
.map(|(i, c)| text_for(*c, i, &mut rng))
|
||||||
.collect();
|
.collect();
|
||||||
let bm25 = BM25Index::build(&texts, &vec![0u8; n]);
|
let bm25 = BM25Index::build(&texts, &vec![0u8; n]);
|
||||||
let index = HnswIndex::build_with_metric(
|
let index = HnswIndex::build_with(
|
||||||
&data.vectors,
|
&data.vectors,
|
||||||
HNSW_M,
|
HNSW_M,
|
||||||
HNSW_EF_CONSTRUCTION,
|
HNSW_EF_CONSTRUCTION,
|
||||||
DistanceMetric::Cosine,
|
DistanceMetric::Cosine,
|
||||||
|
storage(),
|
||||||
);
|
);
|
||||||
|
|
||||||
let vec_pool = (K * 8).max(64);
|
let vec_pool = (K * 8).max(64);
|
||||||
@@ -450,6 +959,69 @@ fn fusion_study(n: usize) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// What an in-memory store costs, stage by stage. The vectors are the floor:
|
||||||
|
/// everything above it is bookkeeping that could in principle be shared.
|
||||||
|
fn bench_footprint(n: usize) {
|
||||||
|
let data = make_dataset(n, 0xF007 ^ n as u64);
|
||||||
|
let mut rng = Rng(11);
|
||||||
|
let dir = tempfile::TempDir::new().unwrap();
|
||||||
|
let path = dir.path().join("footprint.h5");
|
||||||
|
|
||||||
|
let base = heap_bytes();
|
||||||
|
let entries: Vec<MemoryEntry> = data
|
||||||
|
.vectors
|
||||||
|
.iter()
|
||||||
|
.enumerate()
|
||||||
|
.map(|(i, v)| MemoryEntry {
|
||||||
|
chunk: text_for(data.cluster_of[i], i, &mut rng),
|
||||||
|
embedding: v.clone(),
|
||||||
|
source_channel: "bench".into(),
|
||||||
|
timestamp: i as f64,
|
||||||
|
session_id: format!("s{}", i % 50),
|
||||||
|
tags: format!("t{i}"),
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
let after_entries = heap_bytes();
|
||||||
|
|
||||||
|
let mut config = MemoryConfig::new(path, "bench", DIM);
|
||||||
|
config.quantized_index = INT8.load(std::sync::atomic::Ordering::Relaxed);
|
||||||
|
let mut mem = HDF5Memory::create(config).unwrap();
|
||||||
|
mem.save_batch(entries).unwrap();
|
||||||
|
let after_store = heap_bytes();
|
||||||
|
|
||||||
|
// First query builds the vector and keyword indexes.
|
||||||
|
std::hint::black_box(mem.hybrid_search(&data.queries[0], "record", 0.7, 0.3, K));
|
||||||
|
let after_indexes = heap_bytes();
|
||||||
|
|
||||||
|
// Reopening is the figure that matters for a long-lived process, and the
|
||||||
|
// only one RSS reports honestly: memory freed when the ingest buffers went
|
||||||
|
// away stays in the allocator's pool, so the stage deltas above understate
|
||||||
|
// what was given back.
|
||||||
|
let path = mem.config().path.clone();
|
||||||
|
drop(mem);
|
||||||
|
let before_open = heap_bytes();
|
||||||
|
reset_peak();
|
||||||
|
let reopened = HDF5Memory::open(&path).unwrap();
|
||||||
|
let after_open = heap_bytes();
|
||||||
|
let loaded = after_open.saturating_sub(before_open);
|
||||||
|
// Peak over the open, not just what it leaves behind: a buffer allocated
|
||||||
|
// and freed during the parse never shows up in the live total.
|
||||||
|
let peak = peak_bytes().saturating_sub(before_open);
|
||||||
|
drop(reopened);
|
||||||
|
|
||||||
|
let raw = (n * DIM * 4) as u64;
|
||||||
|
println!(
|
||||||
|
"| {n} | {:.0} | {:.0} | {:.0} | {:.0} | {:.0} | {:.0} | {:.2}x |",
|
||||||
|
mib(raw),
|
||||||
|
mib(after_entries.saturating_sub(base)),
|
||||||
|
mib(after_store.saturating_sub(after_entries)),
|
||||||
|
mib(after_indexes.saturating_sub(after_store)),
|
||||||
|
mib(loaded),
|
||||||
|
mib(peak),
|
||||||
|
loaded as f64 / raw as f64,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
fn main() {
|
fn main() {
|
||||||
let args: Vec<String> = std::env::args().skip(1).collect();
|
let args: Vec<String> = std::env::args().skip(1).collect();
|
||||||
let full = args.iter().any(|a| a == "--full");
|
let full = args.iter().any(|a| a == "--full");
|
||||||
@@ -464,6 +1036,60 @@ fn main() {
|
|||||||
}
|
}
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
if args.iter().any(|a| a == "--signing-study") {
|
||||||
|
println!("## Signed checkpoints ({DIM}-dim, float16, int8 index)\n");
|
||||||
|
println!(
|
||||||
|
"| N | checkpoint ms, unsigned | checkpoint ms, signed | signing adds ms | verify ms | file MiB added |"
|
||||||
|
);
|
||||||
|
println!("|---:|---:|---:|---:|---:|---:|");
|
||||||
|
for &n in if full {
|
||||||
|
&[1_000, 10_000, 100_000][..]
|
||||||
|
} else {
|
||||||
|
&[1_000, 10_000][..]
|
||||||
|
} {
|
||||||
|
signing_study(n);
|
||||||
|
}
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if args.iter().any(|a| a == "--options-study") {
|
||||||
|
println!("## Search options ({DIM}-dim, k = {K}, Hebbian boost off)\n");
|
||||||
|
println!("| N | options | filtered recall@10 | p50 ms | p99 ms |");
|
||||||
|
println!("|---:|---|---:|---:|---:|");
|
||||||
|
for &n in if full {
|
||||||
|
&[10_000, 100_000][..]
|
||||||
|
} else {
|
||||||
|
&[10_000][..]
|
||||||
|
} {
|
||||||
|
options_study(n);
|
||||||
|
}
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if args.iter().any(|a| a == "--f16-first") {
|
||||||
|
F16_FIRST.store(true, std::sync::atomic::Ordering::Relaxed);
|
||||||
|
}
|
||||||
|
if args.iter().any(|a| a == "--float16-study") {
|
||||||
|
println!("## float16 embedding storage ({DIM}-dim, int8 index, Hebbian boost off)\n");
|
||||||
|
println!(
|
||||||
|
"| N | embeddings | file MiB | checkpoint ms | open ms | recall@10 | top-10 overlap with the other | hybrid p50 ms |"
|
||||||
|
);
|
||||||
|
println!("|---:|---|---:|---:|---:|---:|---:|---:|");
|
||||||
|
for &n in if full {
|
||||||
|
&[1_000, 10_000, 100_000][..]
|
||||||
|
} else {
|
||||||
|
&[1_000, 10_000][..]
|
||||||
|
} {
|
||||||
|
float16_study(n);
|
||||||
|
}
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if args.iter().any(|a| a == "--int8") {
|
||||||
|
INT8.store(true, std::sync::atomic::Ordering::Relaxed);
|
||||||
|
println!("(int8-quantised index vectors)");
|
||||||
|
}
|
||||||
|
if args.iter().any(|a| a == "--rerank") {
|
||||||
|
RERANK.store(true, std::sync::atomic::Ordering::Relaxed);
|
||||||
|
println!("(candidates re-scored against exact vectors)");
|
||||||
|
}
|
||||||
if args.iter().any(|a| a == "--uniform") {
|
if args.iter().any(|a| a == "--uniform") {
|
||||||
UNIFORM.store(true, std::sync::atomic::Ordering::Relaxed);
|
UNIFORM.store(true, std::sync::atomic::Ordering::Relaxed);
|
||||||
println!("(uniform random data)");
|
println!("(uniform random data)");
|
||||||
@@ -485,6 +1111,18 @@ fn main() {
|
|||||||
|
|
||||||
let mut json = Vec::new();
|
let mut json = Vec::new();
|
||||||
println!("## Search harness");
|
println!("## Search harness");
|
||||||
|
|
||||||
|
if args.iter().any(|a| a == "--footprint") {
|
||||||
|
println!("\n### Resident memory, {DIM}-dim f32\n");
|
||||||
|
println!(
|
||||||
|
"| N | vectors (raw) | entries MiB | store MiB | indexes MiB | reopened MiB | peak during open MiB | reopened / raw |"
|
||||||
|
);
|
||||||
|
println!("|---:|---:|---:|---:|---:|---:|---:|---:|");
|
||||||
|
for &n in sizes {
|
||||||
|
bench_footprint(n);
|
||||||
|
}
|
||||||
|
return;
|
||||||
|
}
|
||||||
// `--e2e-only` skips the index benchmarks, so the end-to-end section runs
|
// `--e2e-only` skips the index benchmarks, so the end-to-end section runs
|
||||||
// in a process that has not already spun up a thread pool.
|
// in a process that has not already spun up a thread pool.
|
||||||
if !args.iter().any(|a| a == "--e2e-only") {
|
if !args.iter().any(|a| a == "--e2e-only") {
|
||||||
|
|||||||
@@ -0,0 +1,148 @@
|
|||||||
|
//! Keeps the concurrent-read harnesses working: runs `concurrent_read`, the
|
||||||
|
//! h5py script (threads and processes) and the comparison script end to end
|
||||||
|
//! on tiny files. h5py reading the files also checks, element by element at
|
||||||
|
//! spot positions, that both harnesses generate the same data and slabs.
|
||||||
|
//!
|
||||||
|
//! The h5py half is skipped when python3 with h5py is unavailable, unless
|
||||||
|
//! `CLAWHDF5_REQUIRE_INTEROP=1`; `CLAWHDF5_PYTHON` picks the interpreter.
|
||||||
|
|
||||||
|
use std::path::{Path, PathBuf};
|
||||||
|
use std::process::Command;
|
||||||
|
|
||||||
|
fn python() -> String {
|
||||||
|
std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn interop_required() -> bool {
|
||||||
|
std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1")
|
||||||
|
}
|
||||||
|
|
||||||
|
fn python_available() -> bool {
|
||||||
|
Command::new(python())
|
||||||
|
.args(["-c", "import h5py, numpy"])
|
||||||
|
.output()
|
||||||
|
.map(|o| o.status.success())
|
||||||
|
.unwrap_or(false)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn scripts() -> PathBuf {
|
||||||
|
Path::new(env!("CARGO_MANIFEST_DIR")).join("scripts")
|
||||||
|
}
|
||||||
|
|
||||||
|
fn run(cmd: &mut Command) -> String {
|
||||||
|
let out = cmd.output().expect("spawn");
|
||||||
|
assert!(
|
||||||
|
out.status.success(),
|
||||||
|
"{cmd:?} failed\nSTDOUT:\n{}\nSTDERR:\n{}",
|
||||||
|
String::from_utf8_lossy(&out.stdout),
|
||||||
|
String::from_utf8_lossy(&out.stderr)
|
||||||
|
);
|
||||||
|
String::from_utf8_lossy(&out.stdout).into_owned()
|
||||||
|
}
|
||||||
|
|
||||||
|
const SMALL: [&str; 8] = [
|
||||||
|
"--threads",
|
||||||
|
"1,2",
|
||||||
|
"--slabs",
|
||||||
|
"8",
|
||||||
|
"--reps",
|
||||||
|
"1",
|
||||||
|
"--slab",
|
||||||
|
"64",
|
||||||
|
];
|
||||||
|
|
||||||
|
fn results(path: &Path) -> serde_json::Value {
|
||||||
|
serde_json::from_str(&std::fs::read_to_string(path).unwrap()).unwrap()
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn harnesses_run_end_to_end_on_tiny_files() {
|
||||||
|
let dir = tempfile::TempDir::new().unwrap();
|
||||||
|
let data = dir.path().join("data");
|
||||||
|
let claw = dir.path().join("claw.json");
|
||||||
|
|
||||||
|
let bin = env!("CARGO_BIN_EXE_concurrent_read");
|
||||||
|
run(Command::new(bin)
|
||||||
|
.arg("--dir")
|
||||||
|
.arg(&data)
|
||||||
|
.args(["--datasets", "3", "--mib", "1"])
|
||||||
|
.args(SMALL)
|
||||||
|
.arg("--json")
|
||||||
|
.arg(&claw));
|
||||||
|
// Second run reuses the files (and exercises --cold).
|
||||||
|
let out = Command::new(bin)
|
||||||
|
.arg("--dir")
|
||||||
|
.arg(&data)
|
||||||
|
.args(["--datasets", "3", "--mib", "1", "--cold"])
|
||||||
|
.args(SMALL)
|
||||||
|
.output()
|
||||||
|
.unwrap();
|
||||||
|
assert!(out.status.success());
|
||||||
|
assert!(String::from_utf8_lossy(&out.stderr).contains("reusing"));
|
||||||
|
|
||||||
|
let doc = results(&claw);
|
||||||
|
assert_eq!(doc["tool"], "clawhdf5");
|
||||||
|
// 2 layouts x 2 modes x 2 thread counts.
|
||||||
|
assert_eq!(doc["results"].as_array().unwrap().len(), 8);
|
||||||
|
for r in doc["results"].as_array().unwrap() {
|
||||||
|
assert!(r["mb_s"].as_f64().unwrap() > 0.0, "{r}");
|
||||||
|
}
|
||||||
|
|
||||||
|
if !python_available() {
|
||||||
|
assert!(
|
||||||
|
!interop_required(),
|
||||||
|
"CLAWHDF5_REQUIRE_INTEROP=1 but {} has no h5py",
|
||||||
|
python()
|
||||||
|
);
|
||||||
|
eprintln!("skipping the h5py half: no h5py in {}", python());
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
let mut jsons = vec![claw];
|
||||||
|
for executor in ["threads", "processes"] {
|
||||||
|
let out = dir.path().join(format!("h5py-{executor}.json"));
|
||||||
|
run(Command::new(python())
|
||||||
|
.arg(scripts().join("concurrent_read_h5py.py"))
|
||||||
|
.arg("--dir")
|
||||||
|
.arg(&data)
|
||||||
|
.args(["--executor", executor])
|
||||||
|
.args(SMALL)
|
||||||
|
.arg("--json")
|
||||||
|
.arg(&out));
|
||||||
|
let doc = results(&out);
|
||||||
|
assert_eq!(doc["tool"], format!("h5py-{executor}"));
|
||||||
|
assert_eq!(doc["results"].as_array().unwrap().len(), 8);
|
||||||
|
jsons.push(out);
|
||||||
|
}
|
||||||
|
let table = run(Command::new(python())
|
||||||
|
.arg(scripts().join("compare_concurrent_read.py"))
|
||||||
|
.args(&jsons));
|
||||||
|
assert!(table.contains("| deflate | same | 2 |"), "{table}");
|
||||||
|
assert!(table.contains("clawhdf5 / h5py-processes"), "{table}");
|
||||||
|
|
||||||
|
// A different workload must not be compared.
|
||||||
|
let other = dir.path().join("other.json");
|
||||||
|
run(Command::new(python())
|
||||||
|
.arg(scripts().join("concurrent_read_h5py.py"))
|
||||||
|
.arg("--dir")
|
||||||
|
.arg(&data)
|
||||||
|
.args([
|
||||||
|
"--threads",
|
||||||
|
"1",
|
||||||
|
"--slabs",
|
||||||
|
"4",
|
||||||
|
"--reps",
|
||||||
|
"1",
|
||||||
|
"--slab",
|
||||||
|
"64",
|
||||||
|
])
|
||||||
|
.arg("--json")
|
||||||
|
.arg(&other));
|
||||||
|
let out = Command::new(python())
|
||||||
|
.arg(scripts().join("compare_concurrent_read.py"))
|
||||||
|
.arg(&jsons[0])
|
||||||
|
.arg(&other)
|
||||||
|
.output()
|
||||||
|
.unwrap();
|
||||||
|
assert!(!out.status.success());
|
||||||
|
assert!(String::from_utf8_lossy(&out.stderr).contains("slabs"));
|
||||||
|
}
|
||||||
@@ -1,7 +1,8 @@
|
|||||||
[package]
|
[package]
|
||||||
name = "clawhdf5-cli"
|
name = "clawhdf5-cli"
|
||||||
version = "2.5.0"
|
version = "2.7.0"
|
||||||
edition = "2024"
|
edition = "2024"
|
||||||
|
rust-version.workspace = true
|
||||||
license = "MIT"
|
license = "MIT"
|
||||||
description = "CLI for clawhdf5 agent memory — create, save, search, recall, stats"
|
description = "CLI for clawhdf5 agent memory — create, save, search, recall, stats"
|
||||||
repository = "https://git.redclaw.dev/quantumclaw/clawhdf5"
|
repository = "https://git.redclaw.dev/quantumclaw/clawhdf5"
|
||||||
@@ -14,7 +15,7 @@ name = "clawhdf5"
|
|||||||
path = "src/main.rs"
|
path = "src/main.rs"
|
||||||
|
|
||||||
[dependencies]
|
[dependencies]
|
||||||
clawhdf5-agent = { path = "../clawhdf5-agent", version = "2.5.0" }
|
clawhdf5-agent = { path = "../clawhdf5-agent", version = "2.7.0" }
|
||||||
clap = { version = "4", features = ["derive", "env"] }
|
clap = { version = "4", features = ["derive", "env"] }
|
||||||
serde_json = "1"
|
serde_json = "1"
|
||||||
serde = { workspace = true }
|
serde = { workspace = true }
|
||||||
|
|||||||
+165
-17
@@ -1,15 +1,22 @@
|
|||||||
use std::path::PathBuf;
|
use std::path::{Path, PathBuf};
|
||||||
|
|
||||||
use clap::{Parser, Subcommand};
|
use clap::{Parser, Subcommand};
|
||||||
|
use clawhdf5_agent::signing::{self, SigningKey, VerifyingKey};
|
||||||
use clawhdf5_agent::{AgentMemory, HDF5Memory, MemoryConfig, MemoryEntry};
|
use clawhdf5_agent::{AgentMemory, HDF5Memory, MemoryConfig, MemoryEntry};
|
||||||
|
|
||||||
/// ClawhDF5 — HDF5-backed cognitive memory for AI agents
|
/// ClawhDF5 — HDF5-backed cognitive memory for AI agents
|
||||||
#[derive(Parser)]
|
#[derive(Parser)]
|
||||||
#[command(name = "clawhdf5", version, about)]
|
#[command(name = "clawhdf5", version, about)]
|
||||||
struct Cli {
|
struct Cli {
|
||||||
/// Path to the .h5 memory file
|
/// Path to the .h5 memory file (not needed for `keygen`)
|
||||||
#[arg(short, long, env = "CLAWHDF5_PATH")]
|
#[arg(short, long, env = "CLAWHDF5_PATH")]
|
||||||
path: PathBuf,
|
path: Option<PathBuf>,
|
||||||
|
|
||||||
|
/// File holding an Ed25519 signing key (64 hex characters, from
|
||||||
|
/// `keygen`). Every checkpoint this command makes is then signed; a
|
||||||
|
/// signed store refuses to checkpoint without it.
|
||||||
|
#[arg(long, env = "CLAWHDF5_SIGNING_KEY", global = true)]
|
||||||
|
signing_key: Option<PathBuf>,
|
||||||
|
|
||||||
#[command(subcommand)]
|
#[command(subcommand)]
|
||||||
command: Commands,
|
command: Commands,
|
||||||
@@ -28,6 +35,22 @@ enum Commands {
|
|||||||
/// Enable write-ahead log
|
/// Enable write-ahead log
|
||||||
#[arg(long)]
|
#[arg(long)]
|
||||||
wal: bool,
|
wal: bool,
|
||||||
|
/// Hold the vector index's copy of the embeddings as f32 instead of
|
||||||
|
/// the default int8 (which uses a quarter of the memory and is faster
|
||||||
|
/// at equal recall)
|
||||||
|
#[arg(long)]
|
||||||
|
f32_index: bool,
|
||||||
|
/// Accepted for compatibility; int8 is now the default
|
||||||
|
#[arg(long, hide = true, conflicts_with = "f32_index")]
|
||||||
|
quantized_index: bool,
|
||||||
|
/// Store embeddings as full-precision f32 instead of the default
|
||||||
|
/// half precision (float16: half the bytes, about three significant
|
||||||
|
/// digits, values within ±65504)
|
||||||
|
#[arg(long)]
|
||||||
|
f32: bool,
|
||||||
|
/// Accepted for compatibility; float16 is now the default
|
||||||
|
#[arg(long, hide = true, conflicts_with = "f32")]
|
||||||
|
float16: bool,
|
||||||
},
|
},
|
||||||
/// Save a memory entry (reads JSON from stdin or --json)
|
/// Save a memory entry (reads JSON from stdin or --json)
|
||||||
Save {
|
Save {
|
||||||
@@ -75,6 +98,38 @@ enum Commands {
|
|||||||
/// Destination path
|
/// Destination path
|
||||||
dest: PathBuf,
|
dest: PathBuf,
|
||||||
},
|
},
|
||||||
|
/// Generate an Ed25519 signing key for signed checkpoints
|
||||||
|
Keygen {
|
||||||
|
/// Where to write the secret key (created new, owner-only on Unix)
|
||||||
|
#[arg(long)]
|
||||||
|
out: PathBuf,
|
||||||
|
},
|
||||||
|
/// Verify a signed store against a public key; exit status 2 if not valid
|
||||||
|
Verify {
|
||||||
|
/// The trusted public key: 64 hex characters, or a file holding them
|
||||||
|
#[arg(long)]
|
||||||
|
public_key: String,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
fn read_signing_key(path: &Path) -> Result<SigningKey, Box<dyn std::error::Error>> {
|
||||||
|
let text = std::fs::read_to_string(path)
|
||||||
|
.map_err(|e| format!("cannot read signing key {}: {e}", path.display()))?;
|
||||||
|
let bytes = signing::from_hex::<32>(&text)
|
||||||
|
.ok_or_else(|| format!("{} is not a 64-hex-character key", path.display()))?;
|
||||||
|
Ok(SigningKey::from_bytes(&bytes))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Open for writing, with the signing key applied if one was given.
|
||||||
|
fn open_writable(
|
||||||
|
path: &Path,
|
||||||
|
key: &Option<SigningKey>,
|
||||||
|
) -> Result<HDF5Memory, Box<dyn std::error::Error>> {
|
||||||
|
let mut mem = HDF5Memory::open(path)?;
|
||||||
|
if let Some(k) = key {
|
||||||
|
mem.set_signing_key(k.clone());
|
||||||
|
}
|
||||||
|
Ok(mem)
|
||||||
}
|
}
|
||||||
|
|
||||||
fn main() {
|
fn main() {
|
||||||
@@ -87,17 +142,76 @@ fn main() {
|
|||||||
}
|
}
|
||||||
|
|
||||||
fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
||||||
|
if let Commands::Keygen { out } = &cli.command {
|
||||||
|
let key = signing::generate_key();
|
||||||
|
let mut opts = std::fs::OpenOptions::new();
|
||||||
|
opts.write(true).create_new(true);
|
||||||
|
#[cfg(unix)]
|
||||||
|
{
|
||||||
|
use std::os::unix::fs::OpenOptionsExt;
|
||||||
|
opts.mode(0o600);
|
||||||
|
}
|
||||||
|
use std::io::Write;
|
||||||
|
let mut f = opts
|
||||||
|
.open(out)
|
||||||
|
.map_err(|e| format!("cannot create {}: {e}", out.display()))?;
|
||||||
|
writeln!(f, "{}", signing::to_hex(&key.to_bytes()))?;
|
||||||
|
let j = serde_json::json!({
|
||||||
|
"status": "generated",
|
||||||
|
"secret_key_file": out.display().to_string(),
|
||||||
|
"public_key": signing::to_hex(&key.verifying_key().to_bytes()),
|
||||||
|
});
|
||||||
|
println!("{}", serde_json::to_string_pretty(&j)?);
|
||||||
|
return Ok(());
|
||||||
|
}
|
||||||
|
let path = cli
|
||||||
|
.path
|
||||||
|
.clone()
|
||||||
|
.ok_or("--path (or CLAWHDF5_PATH) is required")?;
|
||||||
|
let key = cli
|
||||||
|
.signing_key
|
||||||
|
.as_deref()
|
||||||
|
.map(read_signing_key)
|
||||||
|
.transpose()?;
|
||||||
match cli.command {
|
match cli.command {
|
||||||
Commands::Create { agent_id, dim, wal } => {
|
Commands::Create {
|
||||||
let mut config = MemoryConfig::new(cli.path.clone(), &agent_id, dim);
|
agent_id,
|
||||||
|
dim,
|
||||||
|
wal,
|
||||||
|
f32_index,
|
||||||
|
quantized_index: _,
|
||||||
|
f32,
|
||||||
|
float16: _,
|
||||||
|
} => {
|
||||||
|
let mut config = MemoryConfig::new(path.clone(), &agent_id, dim);
|
||||||
config.wal_enabled = wal;
|
config.wal_enabled = wal;
|
||||||
let mem = HDF5Memory::create(config)?;
|
// As with --f32-index: only ever switch the library default off.
|
||||||
|
if f32 {
|
||||||
|
config.float16 = false;
|
||||||
|
}
|
||||||
|
let config_float16 = config.float16;
|
||||||
|
// Only ever switch *off* the library default: assigning the flag
|
||||||
|
// outright would force every CLI-created store back to f32 unless
|
||||||
|
// the caller knew to ask for int8.
|
||||||
|
if f32_index {
|
||||||
|
config.quantized_index = false;
|
||||||
|
}
|
||||||
|
let config_quantized = config.quantized_index;
|
||||||
|
let mut mem = HDF5Memory::create(config)?;
|
||||||
|
// Sign straight away, so the store is never on disk unsigned.
|
||||||
|
if let Some(k) = &key {
|
||||||
|
mem.set_signing_key(k.clone());
|
||||||
|
mem.flush_wal()?;
|
||||||
|
}
|
||||||
let j = serde_json::json!({
|
let j = serde_json::json!({
|
||||||
"status": "created",
|
"status": "created",
|
||||||
"path": cli.path.display().to_string(),
|
"path": path.display().to_string(),
|
||||||
"agent_id": agent_id,
|
"agent_id": agent_id,
|
||||||
"embedding_dim": dim,
|
"embedding_dim": dim,
|
||||||
"wal_enabled": wal,
|
"wal_enabled": wal,
|
||||||
|
"quantized_index": config_quantized,
|
||||||
|
"float16": config_float16,
|
||||||
|
"signed": mem.is_signed(),
|
||||||
"count": mem.count(),
|
"count": mem.count(),
|
||||||
});
|
});
|
||||||
println!("{}", serde_json::to_string_pretty(&j)?);
|
println!("{}", serde_json::to_string_pretty(&j)?);
|
||||||
@@ -114,7 +228,7 @@ fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
let entry: MemoryEntry = serde_json::from_str(&input)?;
|
let entry: MemoryEntry = serde_json::from_str(&input)?;
|
||||||
let mut mem = HDF5Memory::open(&cli.path)?;
|
let mut mem = open_writable(&path, &key)?;
|
||||||
let idx = mem.save(entry)?;
|
let idx = mem.save(entry)?;
|
||||||
let j = serde_json::json!({ "status": "saved", "index": idx, "count": mem.count() });
|
let j = serde_json::json!({ "status": "saved", "index": idx, "count": mem.count() });
|
||||||
println!("{}", serde_json::to_string(&j)?);
|
println!("{}", serde_json::to_string(&j)?);
|
||||||
@@ -128,7 +242,7 @@ fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
|||||||
keyword_weight,
|
keyword_weight,
|
||||||
} => {
|
} => {
|
||||||
let emb: Vec<f32> = serde_json::from_str(&embedding)?;
|
let emb: Vec<f32> = serde_json::from_str(&embedding)?;
|
||||||
let mut mem = HDF5Memory::open(&cli.path)?;
|
let mut mem = open_writable(&path, &key)?;
|
||||||
let results = mem.hybrid_search(&emb, &query, vector_weight, keyword_weight, top_k);
|
let results = mem.hybrid_search(&emb, &query, vector_weight, keyword_weight, top_k);
|
||||||
let j: Vec<serde_json::Value> = results
|
let j: Vec<serde_json::Value> = results
|
||||||
.iter()
|
.iter()
|
||||||
@@ -146,7 +260,7 @@ fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
|||||||
}
|
}
|
||||||
|
|
||||||
Commands::Recall { index } => {
|
Commands::Recall { index } => {
|
||||||
let mem = HDF5Memory::open_read_only(&cli.path)?;
|
let mem = HDF5Memory::open_read_only(&path)?;
|
||||||
match mem.get_chunk(index) {
|
match mem.get_chunk(index) {
|
||||||
Some(content) => {
|
Some(content) => {
|
||||||
let j = serde_json::json!({ "index": index, "chunk": content });
|
let j = serde_json::json!({ "index": index, "chunk": content });
|
||||||
@@ -160,22 +274,23 @@ fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
|||||||
}
|
}
|
||||||
|
|
||||||
Commands::Stats => {
|
Commands::Stats => {
|
||||||
let mem = HDF5Memory::open_read_only(&cli.path)?;
|
let mem = HDF5Memory::open_read_only(&path)?;
|
||||||
let cfg = mem.config();
|
let cfg = mem.config();
|
||||||
let j = serde_json::json!({
|
let j = serde_json::json!({
|
||||||
"path": cli.path.display().to_string(),
|
"path": path.display().to_string(),
|
||||||
"agent_id": cfg.agent_id,
|
"agent_id": cfg.agent_id,
|
||||||
"embedding_dim": cfg.embedding_dim,
|
"embedding_dim": cfg.embedding_dim,
|
||||||
"count": mem.count(),
|
"count": mem.count(),
|
||||||
"active": mem.count_active(),
|
"active": mem.count_active(),
|
||||||
"wal_enabled": cfg.wal_enabled,
|
"wal_enabled": cfg.wal_enabled,
|
||||||
"wal_pending": mem.wal_pending_count(),
|
"wal_pending": mem.wal_pending_count(),
|
||||||
|
"signed": mem.is_signed(),
|
||||||
});
|
});
|
||||||
println!("{}", serde_json::to_string_pretty(&j)?);
|
println!("{}", serde_json::to_string_pretty(&j)?);
|
||||||
}
|
}
|
||||||
|
|
||||||
Commands::FlushWal => {
|
Commands::FlushWal => {
|
||||||
let mut mem = HDF5Memory::open(&cli.path)?;
|
let mut mem = open_writable(&path, &key)?;
|
||||||
let before = mem.wal_pending_count();
|
let before = mem.wal_pending_count();
|
||||||
mem.flush_wal()?;
|
mem.flush_wal()?;
|
||||||
let j = serde_json::json!({
|
let j = serde_json::json!({
|
||||||
@@ -187,7 +302,7 @@ fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
|||||||
}
|
}
|
||||||
|
|
||||||
Commands::AgentsMd { output } => {
|
Commands::AgentsMd { output } => {
|
||||||
let mem = HDF5Memory::open_read_only(&cli.path)?;
|
let mem = HDF5Memory::open_read_only(&path)?;
|
||||||
let md = mem.generate_agents_md();
|
let md = mem.generate_agents_md();
|
||||||
match output {
|
match output {
|
||||||
Some(p) => {
|
Some(p) => {
|
||||||
@@ -199,7 +314,7 @@ fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
|||||||
}
|
}
|
||||||
|
|
||||||
Commands::Export => {
|
Commands::Export => {
|
||||||
let mem = HDF5Memory::open_read_only(&cli.path)?;
|
let mem = HDF5Memory::open_read_only(&path)?;
|
||||||
for i in 0..mem.count() {
|
for i in 0..mem.count() {
|
||||||
if let Some(chunk) = mem.get_chunk(i) {
|
if let Some(chunk) = mem.get_chunk(i) {
|
||||||
let j = serde_json::json!({ "index": i, "chunk": chunk });
|
let j = serde_json::json!({ "index": i, "chunk": chunk });
|
||||||
@@ -208,11 +323,44 @@ fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
Commands::Keygen { .. } => unreachable!("handled before opening a store"),
|
||||||
|
|
||||||
|
Commands::Verify { public_key } => {
|
||||||
|
let text = if Path::new(&public_key).is_file() {
|
||||||
|
std::fs::read_to_string(&public_key)?
|
||||||
|
} else {
|
||||||
|
public_key
|
||||||
|
};
|
||||||
|
let bytes = signing::from_hex::<32>(&text)
|
||||||
|
.ok_or("--public-key must be 64 hex characters or a file holding them")?;
|
||||||
|
let trusted = VerifyingKey::from_bytes(&bytes)?;
|
||||||
|
let r = HDF5Memory::verify(&path, &trusted)?;
|
||||||
|
let j = serde_json::json!({
|
||||||
|
"valid": r.is_valid(),
|
||||||
|
"signed": r.signed,
|
||||||
|
"key_matches": r.key_matches,
|
||||||
|
"signature_valid": r.signature_valid,
|
||||||
|
"records_match": r.records_match,
|
||||||
|
"settings_match": r.settings_match,
|
||||||
|
"sessions_match": r.sessions_match,
|
||||||
|
"graph_match": r.graph_match,
|
||||||
|
"changed_records": r.changed_records,
|
||||||
|
"record_count": r.record_count,
|
||||||
|
"signed_record_count": r.signed_record_count,
|
||||||
|
"signed_by": r.public_key.map(|k| signing::to_hex(&k)),
|
||||||
|
"wal_entries_unsigned": r.wal_entries_unsigned,
|
||||||
|
});
|
||||||
|
println!("{}", serde_json::to_string_pretty(&j)?);
|
||||||
|
if !r.is_valid() {
|
||||||
|
std::process::exit(2);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
Commands::Snapshot { dest } => {
|
Commands::Snapshot { dest } => {
|
||||||
let _result = clawhdf5_agent::storage::snapshot_file(&cli.path, &dest)?;
|
let _result = clawhdf5_agent::storage::snapshot_file(&path, &dest)?;
|
||||||
let j = serde_json::json!({
|
let j = serde_json::json!({
|
||||||
"status": "snapshot_created",
|
"status": "snapshot_created",
|
||||||
"source": cli.path.display().to_string(),
|
"source": path.display().to_string(),
|
||||||
"dest": dest.display().to_string(),
|
"dest": dest.display().to_string(),
|
||||||
});
|
});
|
||||||
println!("{}", serde_json::to_string(&j)?);
|
println!("{}", serde_json::to_string(&j)?);
|
||||||
|
|||||||
@@ -1,7 +1,8 @@
|
|||||||
[package]
|
[package]
|
||||||
name = "clawhdf5-derive"
|
name = "clawhdf5-derive"
|
||||||
version = "2.5.0"
|
version = "2.7.0"
|
||||||
edition = "2024"
|
edition = "2024"
|
||||||
|
rust-version.workspace = true
|
||||||
description = "Derive macros for rustyhdf5 HDF5 traits"
|
description = "Derive macros for rustyhdf5 HDF5 traits"
|
||||||
license = "MIT"
|
license = "MIT"
|
||||||
repository = "https://git.redclaw.dev/quantumclaw/clawhdf5"
|
repository = "https://git.redclaw.dev/quantumclaw/clawhdf5"
|
||||||
|
|||||||
@@ -1,7 +1,8 @@
|
|||||||
[package]
|
[package]
|
||||||
name = "clawhdf5-filters"
|
name = "clawhdf5-filters"
|
||||||
version = "2.5.0"
|
version = "2.7.0"
|
||||||
edition = "2024"
|
edition = "2024"
|
||||||
|
rust-version.workspace = true
|
||||||
description = "Filter and compression pipeline for clawhdf5"
|
description = "Filter and compression pipeline for clawhdf5"
|
||||||
license = "MIT"
|
license = "MIT"
|
||||||
repository = "https://git.redclaw.dev/quantumclaw/clawhdf5"
|
repository = "https://git.redclaw.dev/quantumclaw/clawhdf5"
|
||||||
@@ -25,8 +26,12 @@ name = "compression_bench"
|
|||||||
harness = false
|
harness = false
|
||||||
|
|
||||||
[features]
|
[features]
|
||||||
default = ["fast-deflate"]
|
# Pure-Rust zlib-rs by default; `fast-deflate` (zlib-ng, C) overrides it.
|
||||||
|
default = ["zlib-rs"]
|
||||||
fast-deflate = ["flate2/zlib-ng"]
|
fast-deflate = ["flate2/zlib-ng"]
|
||||||
system-zlib = ["flate2/zlib-default"]
|
system-zlib = ["flate2/zlib-default"]
|
||||||
zlib-rs = ["flate2/zlib-rs"]
|
# `runtime_detection` gives zlib-rs `std`, which it needs to detect and use
|
||||||
|
# SIMD at runtime. flate2 enables it by default, but we build flate2 with
|
||||||
|
# default-features = false, and without it zlib-rs inflates 3.5x slower.
|
||||||
|
zlib-rs = ["flate2/zlib-rs", "flate2/runtime_detection"]
|
||||||
apple-compression = []
|
apple-compression = []
|
||||||
|
|||||||
@@ -8,16 +8,18 @@ Filter and compression pipeline for clawhdf5.
|
|||||||
## Features
|
## Features
|
||||||
|
|
||||||
- DEFLATE compression/decompression
|
- DEFLATE compression/decompression
|
||||||
- Fast deflate via zlib-ng (`fast-deflate` feature)
|
- Pure-Rust deflate via zlib-rs (default, `zlib-rs` feature)
|
||||||
|
- zlib-ng instead, if you want it (`fast-deflate` feature; C, needs cmake)
|
||||||
- Apple Compression framework support (`apple-compression` feature)
|
- Apple Compression framework support (`apple-compression` feature)
|
||||||
|
|
||||||
## Usage
|
## Usage
|
||||||
|
|
||||||
```rust
|
```rust
|
||||||
use clawhdf5_filters::{deflate_decode, deflate_encode};
|
use clawhdf5_filters::{deflate_compress, deflate_decompress};
|
||||||
|
|
||||||
let compressed = deflate_encode(&data, 6).unwrap();
|
let compressed = deflate_compress(&data, 6).unwrap();
|
||||||
let decompressed = deflate_decode(&compressed).unwrap();
|
// The second argument bounds the output: the expected decompressed size.
|
||||||
|
let decompressed = deflate_decompress(&compressed, data.len()).unwrap();
|
||||||
```
|
```
|
||||||
|
|
||||||
## License
|
## License
|
||||||
|
|||||||
@@ -1,12 +1,13 @@
|
|||||||
//! Fast deflate backends: Apple Compression Framework and zlib-ng.
|
//! Deflate backends: Apple Compression Framework, zlib-ng and zlib-rs.
|
||||||
//!
|
//!
|
||||||
//! Backend selection priority (decompression & compression):
|
//! Backend selection priority (decompression & compression):
|
||||||
//! 1. Apple Compression Framework (macOS only, `apple-compression` feature)
|
//! 1. Apple Compression Framework (macOS only, `apple-compression` feature)
|
||||||
//! 2. flate2 with zlib-ng backend (`fast-deflate` feature) or miniz_oxide (default)
|
//! 2. flate2 with zlib-ng (`fast-deflate`), else zlib-rs (`zlib-rs`, the
|
||||||
|
//! default), else miniz_oxide
|
||||||
//!
|
//!
|
||||||
//! The Apple Compression Framework uses hardware-accelerated zlib on Apple Silicon
|
//! The Apple Compression Framework uses hardware-accelerated zlib on Apple Silicon
|
||||||
//! and is typically the fastest option on macOS. zlib-ng is the fastest portable
|
//! and is typically the fastest option on macOS. zlib-rs is a pure-Rust port of
|
||||||
//! option and what C HDF5 uses internally.
|
//! zlib-ng; see `BENCHMARKS.md` for how the two compare.
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// Apple Compression Framework FFI (macOS only)
|
// Apple Compression Framework FFI (macOS only)
|
||||||
@@ -243,65 +244,117 @@ mod apple {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// Streaming decompression via flate2 (uses zlib-ng when fast-deflate enabled)
|
// One-shot (de)compression via flate2 (whichever backend flate2 was built with)
|
||||||
|
//
|
||||||
|
// The whole input goes to the codec in one call, into an output buffer sized
|
||||||
|
// up front. `flate2::read::ZlibDecoder` / `write::ZlibEncoder` stream through a
|
||||||
|
// 32 KiB buffer instead, which cost zlib-rs up to 3.7x against zlib-ng on a
|
||||||
|
// 1 MB chunk. clawhdf5-format's deflate filter does the same; see
|
||||||
|
// `BENCHMARKS.md`, "Deflate backend".
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
/// Streaming decompress with pre-allocated output buffer.
|
/// Decompress into a buffer pre-sized to `output_size`, the expected
|
||||||
///
|
/// decompressed length (known for HDF5 chunks). Output longer than that is an
|
||||||
/// When the output size is known (typical for HDF5 chunks), this avoids
|
/// error, as is a stream that ends early.
|
||||||
/// dynamic reallocation by writing directly into a pre-sized buffer.
|
|
||||||
pub(crate) fn flate2_decompress_preallocated(
|
pub(crate) fn flate2_decompress_preallocated(
|
||||||
data: &[u8],
|
data: &[u8],
|
||||||
output_size: usize,
|
output_size: usize,
|
||||||
) -> Result<Vec<u8>, String> {
|
) -> Result<Vec<u8>, String> {
|
||||||
use std::io::Read;
|
inflate_bounded(data, output_size, output_size)
|
||||||
let mut decoder = flate2::read::ZlibDecoder::new(data);
|
|
||||||
let mut output = vec![0u8; output_size];
|
|
||||||
let mut total_read = 0;
|
|
||||||
|
|
||||||
loop {
|
|
||||||
match decoder.read(&mut output[total_read..]) {
|
|
||||||
Ok(0) => break,
|
|
||||||
Ok(n) => total_read += n,
|
|
||||||
Err(e) => return Err(e.to_string()),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
output.truncate(total_read);
|
|
||||||
Ok(output)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Absolute ceiling on decompressed output when the caller has no size hint,
|
/// Absolute ceiling on decompressed output when the caller has no size hint,
|
||||||
/// preventing unbounded allocation from a hostile/corrupted zlib stream.
|
/// preventing unbounded allocation from a hostile/corrupted zlib stream.
|
||||||
const MAX_DECOMPRESS_SIZE: usize = 256 * 1024 * 1024;
|
const MAX_DECOMPRESS_SIZE: usize = 256 * 1024 * 1024;
|
||||||
|
|
||||||
/// Streaming decompress with dynamic sizing (when output size is unknown).
|
/// Decompress with no size hint, bounded by [`MAX_DECOMPRESS_SIZE`] so a
|
||||||
///
|
/// hostile zlib stream cannot force arbitrarily large allocation (a "zlib
|
||||||
/// Bounded by [`MAX_DECOMPRESS_SIZE`] since there is no chunk-size hint to
|
/// bomb").
|
||||||
/// validate against here — an unbounded `read_to_end` would let a hostile
|
|
||||||
/// zlib stream force arbitrarily large allocation (a "zlib bomb").
|
|
||||||
pub(crate) fn flate2_decompress_streaming(data: &[u8]) -> Result<Vec<u8>, String> {
|
pub(crate) fn flate2_decompress_streaming(data: &[u8]) -> Result<Vec<u8>, String> {
|
||||||
use std::io::Read;
|
let hint = data.len().saturating_mul(4).min(1 << 20);
|
||||||
let decoder = flate2::read::ZlibDecoder::new(data);
|
inflate_bounded(data, hint, MAX_DECOMPRESS_SIZE).map_err(|e| {
|
||||||
let mut result = Vec::new();
|
if e.ends_with("exceeds size limit") {
|
||||||
decoder
|
format!(
|
||||||
.take(MAX_DECOMPRESS_SIZE as u64 + 1)
|
"decompressed output exceeds {} MiB limit",
|
||||||
.read_to_end(&mut result)
|
MAX_DECOMPRESS_SIZE / 1024 / 1024
|
||||||
.map_err(|e| e.to_string())?;
|
)
|
||||||
if result.len() > MAX_DECOMPRESS_SIZE {
|
} else {
|
||||||
return Err(format!(
|
e
|
||||||
"decompressed output exceeds {} MiB limit",
|
}
|
||||||
MAX_DECOMPRESS_SIZE / 1024 / 1024
|
})
|
||||||
));
|
|
||||||
}
|
|
||||||
Ok(result)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Compress data using flate2 (zlib-ng when fast-deflate enabled, else miniz_oxide).
|
/// Inflate a zlib stream, starting from `size_hint` bytes of output and
|
||||||
|
/// failing past `limit`.
|
||||||
|
fn inflate_bounded(data: &[u8], size_hint: usize, limit: usize) -> Result<Vec<u8>, String> {
|
||||||
|
use flate2::{Decompress, FlushDecompress, Status};
|
||||||
|
|
||||||
|
// One byte of headroom past the limit distinguishes an over-size stream
|
||||||
|
// from one that legitimately ends exactly at the limit.
|
||||||
|
let max_capacity = limit.saturating_add(1);
|
||||||
|
let mut out = Vec::new();
|
||||||
|
out.try_reserve_exact(size_hint.clamp(1, max_capacity))
|
||||||
|
.map_err(|e| format!("deflate: cannot allocate output: {e}"))?;
|
||||||
|
|
||||||
|
let mut inflater = Decompress::new(true);
|
||||||
|
loop {
|
||||||
|
let (in_before, out_before) = (inflater.total_in(), inflater.total_out());
|
||||||
|
let status = inflater
|
||||||
|
.decompress_vec(
|
||||||
|
&data[in_before as usize..],
|
||||||
|
&mut out,
|
||||||
|
FlushDecompress::Finish,
|
||||||
|
)
|
||||||
|
.map_err(|e| format!("deflate: {e}"))?;
|
||||||
|
if out.len() > limit {
|
||||||
|
return Err("deflate: output exceeds size limit".into());
|
||||||
|
}
|
||||||
|
match status {
|
||||||
|
Status::StreamEnd => return Ok(out),
|
||||||
|
Status::Ok | Status::BufError if out.len() == out.capacity() => {
|
||||||
|
let grow = out.capacity().min(max_capacity - out.capacity()).max(1);
|
||||||
|
out.try_reserve_exact(grow)
|
||||||
|
.map_err(|e| format!("deflate: cannot allocate output: {e}"))?;
|
||||||
|
}
|
||||||
|
Status::Ok | Status::BufError => {
|
||||||
|
if inflater.total_in() as usize >= data.len()
|
||||||
|
|| (inflater.total_in(), inflater.total_out()) == (in_before, out_before)
|
||||||
|
{
|
||||||
|
return Err("deflate: truncated stream".into());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Compress data using flate2 (zlib-ng, zlib-rs or miniz_oxide; see module docs).
|
||||||
pub(crate) fn flate2_compress(data: &[u8], level: u32) -> Result<Vec<u8>, String> {
|
pub(crate) fn flate2_compress(data: &[u8], level: u32) -> Result<Vec<u8>, String> {
|
||||||
use std::io::Write;
|
use flate2::{Compress, Compression, FlushCompress, Status};
|
||||||
let mut encoder = flate2::write::ZlibEncoder::new(Vec::new(), flate2::Compression::new(level));
|
|
||||||
encoder.write_all(data).map_err(|e| e.to_string())?;
|
// zlib's compressBound, plus the zlib header and trailer.
|
||||||
encoder.finish().map_err(|e| e.to_string())
|
let bound = data.len() + (data.len() >> 12) + (data.len() >> 14) + (data.len() >> 25) + 13 + 6;
|
||||||
|
let mut out = Vec::new();
|
||||||
|
out.try_reserve_exact(bound)
|
||||||
|
.map_err(|e| format!("deflate: cannot allocate output: {e}"))?;
|
||||||
|
|
||||||
|
let mut deflater = Compress::new(Compression::new(level), true);
|
||||||
|
loop {
|
||||||
|
let (in_before, out_before) = (deflater.total_in(), deflater.total_out());
|
||||||
|
let status = deflater
|
||||||
|
.compress_vec(&data[in_before as usize..], &mut out, FlushCompress::Finish)
|
||||||
|
.map_err(|e| format!("deflate: {e}"))?;
|
||||||
|
match status {
|
||||||
|
Status::StreamEnd => return Ok(out),
|
||||||
|
Status::Ok | Status::BufError if out.len() == out.capacity() => out
|
||||||
|
.try_reserve(out.capacity().max(4096))
|
||||||
|
.map_err(|e| format!("deflate: cannot allocate output: {e}"))?,
|
||||||
|
Status::Ok | Status::BufError => {
|
||||||
|
if (deflater.total_in(), deflater.total_out()) == (in_before, out_before) {
|
||||||
|
return Err("deflate: encoder made no progress".into());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
@@ -312,7 +365,7 @@ pub(crate) fn flate2_compress(data: &[u8], level: u32) -> Result<Vec<u8>, String
|
|||||||
///
|
///
|
||||||
/// Selection order:
|
/// Selection order:
|
||||||
/// 1. Apple Compression Framework (macOS + `apple-compression` feature)
|
/// 1. Apple Compression Framework (macOS + `apple-compression` feature)
|
||||||
/// 2. flate2 (zlib-ng with `fast-deflate`, otherwise miniz_oxide)
|
/// 2. flate2 (zlib-ng with `fast-deflate`, else zlib-rs, else miniz_oxide)
|
||||||
///
|
///
|
||||||
/// When `output_hint` > 0, pre-allocates the output buffer for zero-copy
|
/// When `output_hint` > 0, pre-allocates the output buffer for zero-copy
|
||||||
/// decompression (avoids reallocation).
|
/// decompression (avoids reallocation).
|
||||||
@@ -344,7 +397,7 @@ pub fn decompress(data: &[u8], output_hint: usize) -> Result<Vec<u8>, String> {
|
|||||||
///
|
///
|
||||||
/// Selection order:
|
/// Selection order:
|
||||||
/// 1. Apple Compression Framework (macOS + `apple-compression` feature)
|
/// 1. Apple Compression Framework (macOS + `apple-compression` feature)
|
||||||
/// 2. flate2 (zlib-ng with `fast-deflate`, otherwise miniz_oxide)
|
/// 2. flate2 (zlib-ng with `fast-deflate`, else zlib-rs, else miniz_oxide)
|
||||||
pub fn compress(data: &[u8], level: u32) -> Result<Vec<u8>, String> {
|
pub fn compress(data: &[u8], level: u32) -> Result<Vec<u8>, String> {
|
||||||
#[cfg(all(target_os = "macos", feature = "apple-compression"))]
|
#[cfg(all(target_os = "macos", feature = "apple-compression"))]
|
||||||
{
|
{
|
||||||
@@ -377,9 +430,19 @@ pub fn active_backend() -> &'static str {
|
|||||||
{
|
{
|
||||||
"zlib-ng"
|
"zlib-ng"
|
||||||
}
|
}
|
||||||
|
// flate2 prefers a C zlib over zlib-rs when both are enabled.
|
||||||
|
#[cfg(all(
|
||||||
|
not(all(target_os = "macos", feature = "apple-compression")),
|
||||||
|
not(feature = "fast-deflate"),
|
||||||
|
feature = "zlib-rs"
|
||||||
|
))]
|
||||||
|
{
|
||||||
|
"zlib-rs"
|
||||||
|
}
|
||||||
#[cfg(not(any(
|
#[cfg(not(any(
|
||||||
all(target_os = "macos", feature = "apple-compression"),
|
all(target_os = "macos", feature = "apple-compression"),
|
||||||
feature = "fast-deflate"
|
feature = "fast-deflate",
|
||||||
|
feature = "zlib-rs"
|
||||||
)))]
|
)))]
|
||||||
{
|
{
|
||||||
"miniz_oxide"
|
"miniz_oxide"
|
||||||
@@ -436,7 +499,7 @@ mod tests {
|
|||||||
fn backend_name_is_set() {
|
fn backend_name_is_set() {
|
||||||
let name = active_backend();
|
let name = active_backend();
|
||||||
assert!(
|
assert!(
|
||||||
["miniz_oxide", "zlib-ng", "apple-compression"].contains(&name),
|
["miniz_oxide", "zlib-rs", "zlib-ng", "apple-compression"].contains(&name),
|
||||||
"unexpected backend: {name}"
|
"unexpected backend: {name}"
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -2,12 +2,14 @@
|
|||||||
//!
|
//!
|
||||||
//! Provides deflate (zlib) decompression/compression with multiple backend options:
|
//! Provides deflate (zlib) decompression/compression with multiple backend options:
|
||||||
//!
|
//!
|
||||||
//! - **Default**: `miniz_oxide` (pure Rust, no C dependencies)
|
//! - **Default (`zlib-rs` feature)**: `zlib-rs` via flate2 (pure Rust, no C
|
||||||
//! - **`fast-deflate` feature**: `zlib-ng` via flate2 (~2-3x faster, matches C HDF5)
|
//! dependencies)
|
||||||
|
//! - **`fast-deflate` feature**: `zlib-ng` via flate2 (C, built with cmake)
|
||||||
//! - **`apple-compression` feature**: Apple Compression Framework on macOS
|
//! - **`apple-compression` feature**: Apple Compression Framework on macOS
|
||||||
//! (hardware-accelerated on Apple Silicon)
|
//! (hardware-accelerated on Apple Silicon)
|
||||||
|
//! - With none of the above: `miniz_oxide` (pure Rust, slower)
|
||||||
//!
|
//!
|
||||||
//! Backend priority: apple-compression > zlib-ng > miniz_oxide.
|
//! Backend priority: apple-compression > zlib-ng > zlib-rs > miniz_oxide.
|
||||||
|
|
||||||
pub mod fast_deflate;
|
pub mod fast_deflate;
|
||||||
|
|
||||||
@@ -115,7 +117,7 @@ mod tests {
|
|||||||
fn backend_reports_name() {
|
fn backend_reports_name() {
|
||||||
let name = deflate_backend();
|
let name = deflate_backend();
|
||||||
assert!(
|
assert!(
|
||||||
["miniz_oxide", "zlib-ng", "apple-compression"].contains(&name),
|
["miniz_oxide", "zlib-rs", "zlib-ng", "apple-compression"].contains(&name),
|
||||||
"unexpected backend: {name}"
|
"unexpected backend: {name}"
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,7 +1,8 @@
|
|||||||
[package]
|
[package]
|
||||||
name = "clawhdf5-format"
|
name = "clawhdf5-format"
|
||||||
version = "2.5.0"
|
version = "2.7.0"
|
||||||
edition = "2024"
|
edition = "2024"
|
||||||
|
rust-version.workspace = true
|
||||||
description = "Pure-Rust HDF5 binary format parsing and writing — no C dependencies"
|
description = "Pure-Rust HDF5 binary format parsing and writing — no C dependencies"
|
||||||
license = "MIT"
|
license = "MIT"
|
||||||
repository = "https://git.redclaw.dev/quantumclaw/clawhdf5"
|
repository = "https://git.redclaw.dev/quantumclaw/clawhdf5"
|
||||||
@@ -21,18 +22,33 @@ zstd = { version = "0.13", optional = true }
|
|||||||
blake3 = { version = "1", optional = true }
|
blake3 = { version = "1", optional = true }
|
||||||
libaec-sys = { path = "../libaec-sys", version = "0.1", optional = true }
|
libaec-sys = { path = "../libaec-sys", version = "0.1", optional = true }
|
||||||
pco = { version = "1.0", optional = true }
|
pco = { version = "1.0", optional = true }
|
||||||
|
# Pure-Rust Zstandard, for the plugin filters that embed zstd (bitshuffle,
|
||||||
|
# blosc). The `zstd` feature (filter 32015) links libzstd instead.
|
||||||
|
ruzstd = { version = "0.9", optional = true }
|
||||||
|
# bzip2 with its default backend, libbz2-rs-sys: a pure-Rust port of
|
||||||
|
# libbzip2 (no C is compiled, despite the -sys name).
|
||||||
|
bzip2 = { version = "0.6", optional = true }
|
||||||
|
snap = { version = "1", optional = true }
|
||||||
|
|
||||||
|
[target.'cfg(target_os = "linux")'.dependencies]
|
||||||
|
# madvise(MADV_HUGEPAGE) for large read buffers (see src/bulk_alloc.rs).
|
||||||
|
libc = { version = "0.2", default-features = false }
|
||||||
|
|
||||||
[dev-dependencies]
|
[dev-dependencies]
|
||||||
|
half = { workspace = true }
|
||||||
serde_json = "1"
|
serde_json = "1"
|
||||||
criterion = { workspace = true }
|
criterion = { workspace = true }
|
||||||
clawhdf5-derive = { path = "../clawhdf5-derive", version = "2.5.0" }
|
clawhdf5-derive = { path = "../clawhdf5-derive", version = "2.7.0" }
|
||||||
|
|
||||||
[[bench]]
|
[[bench]]
|
||||||
name = "bench"
|
name = "bench"
|
||||||
harness = false
|
harness = false
|
||||||
|
|
||||||
[features]
|
[features]
|
||||||
default = ["std", "checksum", "deflate", "provenance", "fast-deflate", "system-zlib-decompress"]
|
# Deflate backend: `zlib-rs` (pure Rust) by default. `fast-deflate` selects
|
||||||
|
# zlib-ng instead (C, built with cmake); flate2 prefers a C zlib whenever one
|
||||||
|
# is enabled, so turning it on anywhere in the build overrides the default.
|
||||||
|
default = ["std", "checksum", "deflate", "provenance", "zlib-rs", "system-zlib-decompress", "lzf"]
|
||||||
std = []
|
std = []
|
||||||
checksum = []
|
checksum = []
|
||||||
deflate = ["flate2"]
|
deflate = ["flate2"]
|
||||||
@@ -42,12 +58,34 @@ fast-checksum = ["crc32fast"]
|
|||||||
fast-deflate = ["flate2/zlib-ng"]
|
fast-deflate = ["flate2/zlib-ng"]
|
||||||
system-zlib = ["flate2/zlib-default"]
|
system-zlib = ["flate2/zlib-default"]
|
||||||
system-zlib-decompress = []
|
system-zlib-decompress = []
|
||||||
zlib-rs = ["flate2/zlib-rs"]
|
# `runtime_detection` gives zlib-rs `std`, which it needs to detect and use
|
||||||
|
# SIMD at runtime. flate2 enables it by default, but we build flate2 with
|
||||||
|
# default-features = false, and without it zlib-rs inflates 3.5x slower.
|
||||||
|
zlib-rs = ["flate2/zlib-rs", "flate2/runtime_detection"]
|
||||||
lz4 = ["lz4_flex"]
|
lz4 = ["lz4_flex"]
|
||||||
zstd = ["dep:zstd"]
|
zstd = ["dep:zstd"]
|
||||||
blake3_hash = ["blake3"]
|
blake3_hash = ["blake3"]
|
||||||
szip = ["libaec-sys"]
|
szip = ["libaec-sys"]
|
||||||
pcodec = ["dep:pco"]
|
pcodec = ["dep:pco"]
|
||||||
|
# Plugin filters, pure Rust. LZF (32000) is h5py's built-in compression; it
|
||||||
|
# has no dependencies, so it is on by default.
|
||||||
|
lzf = []
|
||||||
|
# Bitshuffle (32008), with its LZ4 and Zstandard modes.
|
||||||
|
bitshuffle = ["lz4_flex", "ruzstd"]
|
||||||
|
# bzip2 (307).
|
||||||
|
bzip2 = ["dep:bzip2", "std"]
|
||||||
|
# Blosc 1 (32001) with its BloscLZ, LZ4, Snappy, Zlib and Zstandard codecs.
|
||||||
|
blosc = ["lz4_flex", "ruzstd", "snap", "deflate", "std"]
|
||||||
|
# Blosc2 (32026), read-only: frames, B2ND arrays, and the Blosc codecs above.
|
||||||
|
blosc2 = ["blosc"]
|
||||||
|
# ZFP (32013, H5Z-ZFP), read-only: every mode, for int32, int64, float and
|
||||||
|
# double fields of 1 to 4 dimensions.
|
||||||
|
zfp = []
|
||||||
|
# Every plugin filter above.
|
||||||
|
plugin-filters = ["lzf", "bitshuffle", "bzip2", "blosc", "blosc2", "zfp"]
|
||||||
|
# Test instrumentation: per-thread counts of heap objects read (see
|
||||||
|
# `lookup_stats`), so tests can bound the cost of a name lookup.
|
||||||
|
lookup-stats = ["std"]
|
||||||
|
|
||||||
[[bench]]
|
[[bench]]
|
||||||
name = "parallel_decompress_bench"
|
name = "parallel_decompress_bench"
|
||||||
|
|||||||
@@ -1 +1,4 @@
|
|||||||
target/
|
target/
|
||||||
|
corpus/
|
||||||
|
artifacts/
|
||||||
|
coverage/
|
||||||
|
|||||||
Binary file not shown.
@@ -1,15 +1,36 @@
|
|||||||
#![no_main]
|
#![no_main]
|
||||||
|
use clawhdf5_format::btree_v2::{BTreeV2Header, collect_btree_v2_records};
|
||||||
use libfuzzer_sys::fuzz_target;
|
use libfuzzer_sys::fuzz_target;
|
||||||
|
|
||||||
fuzz_target!(|data: &[u8]| {
|
fuzz_target!(|data: &[u8]| {
|
||||||
for &offset_size in &[4u8, 8] {
|
for &offset_size in &[4u8, 8] {
|
||||||
for &length_size in &[4u8, 8] {
|
for &length_size in &[4u8, 8] {
|
||||||
let _ = clawhdf5_format::btree_v2::BTreeV2Header::parse(
|
if let Ok(header) = BTreeV2Header::parse(data, 0, offset_size, length_size) {
|
||||||
data,
|
let _ = collect_btree_v2_records(data, &header, offset_size, length_size);
|
||||||
0,
|
}
|
||||||
offset_size,
|
|
||||||
length_size,
|
|
||||||
);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Parsing a header requires a valid checksum, which random input almost
|
||||||
|
// never has, so the traversal behind it went unfuzzed — and that is where
|
||||||
|
// a node listing itself as its own child overflowed the stack. Take the
|
||||||
|
// header fields straight from the input instead and walk the rest.
|
||||||
|
let Some((fields, file)) = data.split_first_chunk::<20>() else {
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
let header = BTreeV2Header {
|
||||||
|
tree_type: fields[0],
|
||||||
|
node_size: u32::from_le_bytes([fields[1], fields[2], fields[3], fields[4]]),
|
||||||
|
record_size: u16::from_le_bytes([fields[5], fields[6]]),
|
||||||
|
depth: u16::from_le_bytes([fields[7], fields[8]]),
|
||||||
|
root_node_address: u64::from(u32::from_le_bytes([
|
||||||
|
fields[9], fields[10], fields[11], fields[12],
|
||||||
|
])),
|
||||||
|
num_records_in_root: u16::from_le_bytes([fields[13], fields[14]]),
|
||||||
|
total_records: u64::from(u32::from_le_bytes([
|
||||||
|
fields[15], fields[16], fields[17], fields[18],
|
||||||
|
])),
|
||||||
|
};
|
||||||
|
let offset_size = if fields[19] & 1 == 0 { 4 } else { 8 };
|
||||||
|
let _ = collect_btree_v2_records(file, &header, offset_size, 8);
|
||||||
});
|
});
|
||||||
|
|||||||
@@ -0,0 +1,121 @@
|
|||||||
|
//! File address and length → in-memory index conversion.
|
||||||
|
//!
|
||||||
|
//! HDF5 addresses and lengths are 64-bit; the file is parsed through a
|
||||||
|
//! `&[u8]` indexed by `usize`. On a 64-bit target every `u64` fits, but on a
|
||||||
|
//! 32-bit one (`wasm32`, `i686`, `thumbv7em`) an address past `usize::MAX`
|
||||||
|
//! used to be truncated by an `as usize` cast — silently pointing at another
|
||||||
|
//! part of the file — or to panic. [`to_usize`] is the one conversion the
|
||||||
|
//! parsers use instead: such an address is a clean
|
||||||
|
//! [`FormatError::Overflow`]. It cannot be inside the data anyway: no slice
|
||||||
|
//! is longer than `isize::MAX` bytes.
|
||||||
|
|
||||||
|
#[cfg(not(feature = "std"))]
|
||||||
|
use alloc::format;
|
||||||
|
|
||||||
|
use crate::error::FormatError;
|
||||||
|
|
||||||
|
/// A file address, offset or length from the file as a `usize` index.
|
||||||
|
///
|
||||||
|
/// Fails with [`FormatError::Overflow`] when the value does not fit this
|
||||||
|
/// platform's `usize` (only possible on targets narrower than 64 bits).
|
||||||
|
#[inline]
|
||||||
|
pub fn to_usize(value: u64) -> Result<usize, FormatError> {
|
||||||
|
to_index::<usize>(value)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A file address for a [`crate::storage::Storage`] read, checked as
|
||||||
|
/// [`to_usize`] checks it: the parsers read through 64-bit offsets, but an
|
||||||
|
/// address that could not index an in-memory file on this platform is the
|
||||||
|
/// same [`FormatError::Overflow`] the slice parsers gave for it.
|
||||||
|
#[inline]
|
||||||
|
pub fn checked_addr(value: u64) -> Result<u64, FormatError> {
|
||||||
|
to_usize(value).map(|_| value)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`to_usize`] for an index type of any width. `usize` is 64 bits wide on
|
||||||
|
/// the hosts CI tests on, where the error path cannot be reached through
|
||||||
|
/// `usize`; tests run the same code with `u32` in its place, as on a 32-bit
|
||||||
|
/// target.
|
||||||
|
#[inline]
|
||||||
|
fn to_index<T: TryFrom<u64>>(value: u64) -> Result<T, FormatError> {
|
||||||
|
T::try_from(value).map_err(|_| too_large(value))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A count or offset into an in-memory buffer (a codec's progress counter,
|
||||||
|
/// a size the writer computed from data it holds) as a `usize`, saturating
|
||||||
|
/// at `usize::MAX` instead of truncating.
|
||||||
|
///
|
||||||
|
/// For values that are bounded by the length of something in memory, so
|
||||||
|
/// always fit; if one ever did not, a saturated index fails its bounds check
|
||||||
|
/// or allocation instead of silently addressing the wrong bytes. A value
|
||||||
|
/// read from the file uses [`to_usize`].
|
||||||
|
#[inline]
|
||||||
|
pub fn saturating_usize(value: u64) -> usize {
|
||||||
|
saturating_index(value, usize::MAX)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`saturating_usize`] for an index type of any width, whose largest
|
||||||
|
/// value is `max` (see [`to_index`]).
|
||||||
|
#[inline]
|
||||||
|
fn saturating_index<T: TryFrom<u64>>(value: u64, max: T) -> T {
|
||||||
|
T::try_from(value).unwrap_or(max)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cold]
|
||||||
|
#[inline(never)]
|
||||||
|
fn too_large(value: u64) -> FormatError {
|
||||||
|
FormatError::Overflow(format!(
|
||||||
|
"file address or length {value:#x} exceeds this platform's address space"
|
||||||
|
))
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn values_that_fit_convert_exactly() {
|
||||||
|
assert_eq!(to_usize(0), Ok(0));
|
||||||
|
assert_eq!(to_usize(0x1234), Ok(0x1234));
|
||||||
|
assert_eq!(to_usize(usize::MAX as u64), Ok(usize::MAX));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn saturating_conversion_never_wraps() {
|
||||||
|
assert_eq!(saturating_usize(0), 0);
|
||||||
|
assert_eq!(saturating_usize(0x1234), 0x1234);
|
||||||
|
assert_eq!(saturating_usize(usize::MAX as u64), usize::MAX);
|
||||||
|
// Past usize::MAX (32-bit targets) or at u64::MAX: saturates.
|
||||||
|
assert_eq!(saturating_usize(u64::MAX), usize::MAX);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn values_past_usize_max_are_an_error_not_truncated() {
|
||||||
|
// Reachable through `usize` only where it is narrower than u64 (no
|
||||||
|
// such target runs tests in CI), so the same conversion is run with
|
||||||
|
// u32 standing in for a 32-bit usize.
|
||||||
|
let max = u64::from(u32::MAX);
|
||||||
|
assert_eq!(to_index::<u32>(max), Ok(u32::MAX));
|
||||||
|
for past in [max + 1, max + 0x10, 0x1_0000_1234, u64::MAX] {
|
||||||
|
let err = to_index::<u32>(past).unwrap_err();
|
||||||
|
assert!(
|
||||||
|
matches!(err, FormatError::Overflow(_)),
|
||||||
|
"{past:#x}: {err:?}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
// Where an `as` cast would have wrapped to a small, valid-looking
|
||||||
|
// index, it is not returned.
|
||||||
|
assert_eq!(0x1_0000_1234_u64 as u32, 0x1234);
|
||||||
|
assert!(to_index::<u32>(0x1_0000_1234).is_err());
|
||||||
|
|
||||||
|
assert_eq!(saturating_index(max + 1, u32::MAX), u32::MAX);
|
||||||
|
assert_eq!(saturating_index(0x1_0000_1234, u32::MAX), u32::MAX);
|
||||||
|
assert_eq!(saturating_index(0x1234, u32::MAX), 0x1234);
|
||||||
|
|
||||||
|
// And through `usize` itself, whichever width it has here.
|
||||||
|
match (usize::MAX as u64).checked_add(1) {
|
||||||
|
Some(past) => assert!(matches!(to_usize(past), Err(FormatError::Overflow(_)))),
|
||||||
|
None => assert_eq!(to_usize(u64::MAX), Ok(usize::MAX)),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -5,8 +5,10 @@ use alloc::{borrow::Cow, string::String, vec::Vec};
|
|||||||
#[cfg(feature = "std")]
|
#[cfg(feature = "std")]
|
||||||
use std::borrow::Cow;
|
use std::borrow::Cow;
|
||||||
|
|
||||||
|
use crate::addr::to_usize;
|
||||||
use crate::attribute_info::AttributeInfoMessage;
|
use crate::attribute_info::AttributeInfoMessage;
|
||||||
use crate::btree_v2::{BTreeV2Header, collect_btree_v2_records};
|
use crate::btree_v2::{BTreeV2Header, collect_btree_v2_records_in, find_btree_v2_records_in};
|
||||||
|
use crate::checksum::jenkins_lookup3;
|
||||||
use crate::data_read;
|
use crate::data_read;
|
||||||
use crate::dataspace::Dataspace;
|
use crate::dataspace::Dataspace;
|
||||||
use crate::datatype::Datatype;
|
use crate::datatype::Datatype;
|
||||||
@@ -15,6 +17,7 @@ use crate::fractal_heap::FractalHeapHeader;
|
|||||||
use crate::message_type::MessageType;
|
use crate::message_type::MessageType;
|
||||||
use crate::object_header::ObjectHeader;
|
use crate::object_header::ObjectHeader;
|
||||||
use crate::shared_message;
|
use crate::shared_message;
|
||||||
|
use crate::storage::Storage;
|
||||||
use crate::vl_data;
|
use crate::vl_data;
|
||||||
|
|
||||||
/// A parsed HDF5 attribute message.
|
/// A parsed HDF5 attribute message.
|
||||||
@@ -50,7 +53,7 @@ impl AttributeMessage {
|
|||||||
///
|
///
|
||||||
/// `length_size` is needed for dataspace dimension parsing.
|
/// `length_size` is needed for dataspace dimension parsing.
|
||||||
pub fn parse(data: &[u8], length_size: u8) -> Result<AttributeMessage, FormatError> {
|
pub fn parse(data: &[u8], length_size: u8) -> Result<AttributeMessage, FormatError> {
|
||||||
Self::parse_impl(data, length_size, None)
|
Self::parse_impl(data, length_size, None::<(&[u8], u8)>)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// [`AttributeMessage::parse`] with access to the rest of the file, which
|
/// [`AttributeMessage::parse`] with access to the rest of the file, which
|
||||||
@@ -65,13 +68,24 @@ impl AttributeMessage {
|
|||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
) -> Result<AttributeMessage, FormatError> {
|
) -> Result<AttributeMessage, FormatError> {
|
||||||
Self::parse_impl(data, length_size, Some((file_data, offset_size)))
|
Self::parse_in_storage(data, file_data, offset_size, length_size)
|
||||||
}
|
}
|
||||||
|
|
||||||
fn parse_impl(
|
/// [`AttributeMessage::parse_in_file`] with the file behind any
|
||||||
|
/// [`Storage`].
|
||||||
|
pub fn parse_in_storage<S: Storage + ?Sized>(
|
||||||
|
data: &[u8],
|
||||||
|
file: &S,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<AttributeMessage, FormatError> {
|
||||||
|
Self::parse_impl(data, length_size, Some((file, offset_size)))
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parse_impl<S: Storage + ?Sized>(
|
||||||
data: &[u8],
|
data: &[u8],
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
file: Option<(&[u8], u8)>,
|
file: Option<(&S, u8)>,
|
||||||
) -> Result<AttributeMessage, FormatError> {
|
) -> Result<AttributeMessage, FormatError> {
|
||||||
ensure_len(data, 0, 2)?;
|
ensure_len(data, 0, 2)?;
|
||||||
let version = data[0];
|
let version = data[0];
|
||||||
@@ -86,19 +100,19 @@ impl AttributeMessage {
|
|||||||
|
|
||||||
/// The bytes of an embedded datatype/dataspace message, following the
|
/// The bytes of an embedded datatype/dataspace message, following the
|
||||||
/// shared-message reference when `shared` is set.
|
/// shared-message reference when `shared` is set.
|
||||||
fn embedded_message<'a>(
|
fn embedded_message<'a, S: Storage + ?Sized>(
|
||||||
bytes: &'a [u8],
|
bytes: &'a [u8],
|
||||||
shared: bool,
|
shared: bool,
|
||||||
msg_type: MessageType,
|
msg_type: MessageType,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
file: Option<(&[u8], u8)>,
|
file: Option<(&S, u8)>,
|
||||||
) -> Result<Cow<'a, [u8]>, FormatError> {
|
) -> Result<Cow<'a, [u8]>, FormatError> {
|
||||||
if !shared {
|
if !shared {
|
||||||
return Ok(Cow::Borrowed(bytes));
|
return Ok(Cow::Borrowed(bytes));
|
||||||
}
|
}
|
||||||
let (file_data, offset_size) = file.ok_or(FormatError::UnresolvedSharedMessage)?;
|
let (file_data, offset_size) = file.ok_or(FormatError::UnresolvedSharedMessage)?;
|
||||||
let shared_ref = shared_message::parse_shared_ref(bytes, offset_size)?;
|
let shared_ref = shared_message::parse_shared_ref_sized(bytes, offset_size, length_size)?;
|
||||||
shared_message::resolve_shared_message(
|
shared_message::resolve_shared_message_in(
|
||||||
file_data,
|
file_data,
|
||||||
&shared_ref,
|
&shared_ref,
|
||||||
msg_type,
|
msg_type,
|
||||||
@@ -143,10 +157,10 @@ impl AttributeMessage {
|
|||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
fn parse_v2(
|
fn parse_v2<S: Storage + ?Sized>(
|
||||||
data: &[u8],
|
data: &[u8],
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
file: Option<(&[u8], u8)>,
|
file: Option<(&S, u8)>,
|
||||||
) -> Result<AttributeMessage, FormatError> {
|
) -> Result<AttributeMessage, FormatError> {
|
||||||
// Flags: bit 0 = datatype is shared, bit 1 = dataspace is shared.
|
// Flags: bit 0 = datatype is shared, bit 1 = dataspace is shared.
|
||||||
let flags = data.get(1).copied().unwrap_or(0);
|
let flags = data.get(1).copied().unwrap_or(0);
|
||||||
@@ -197,10 +211,10 @@ impl AttributeMessage {
|
|||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
fn parse_v3(
|
fn parse_v3<S: Storage + ?Sized>(
|
||||||
data: &[u8],
|
data: &[u8],
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
file: Option<(&[u8], u8)>,
|
file: Option<(&S, u8)>,
|
||||||
) -> Result<AttributeMessage, FormatError> {
|
) -> Result<AttributeMessage, FormatError> {
|
||||||
// Flags: bit 0 = datatype is shared, bit 1 = dataspace is shared.
|
// Flags: bit 0 = datatype is shared, bit 1 = dataspace is shared.
|
||||||
let flags = data.get(1).copied().unwrap_or(0);
|
let flags = data.get(1).copied().unwrap_or(0);
|
||||||
@@ -322,9 +336,19 @@ impl AttributeMessage {
|
|||||||
file_data: &[u8],
|
file_data: &[u8],
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
|
) -> Result<Vec<String>, FormatError> {
|
||||||
|
self.read_vl_strings_in(file_data, offset_size, length_size)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`Self::read_vl_strings`] over any [`Storage`].
|
||||||
|
pub fn read_vl_strings_in<S: Storage + ?Sized>(
|
||||||
|
&self,
|
||||||
|
file_data: &S,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
) -> Result<Vec<String>, FormatError> {
|
) -> Result<Vec<String>, FormatError> {
|
||||||
let num_elements = self.dataspace.num_elements();
|
let num_elements = self.dataspace.num_elements();
|
||||||
vl_data::read_vl_strings(
|
vl_data::read_vl_strings_in(
|
||||||
file_data,
|
file_data,
|
||||||
&self.raw_data,
|
&self.raw_data,
|
||||||
num_elements,
|
num_elements,
|
||||||
@@ -341,7 +365,8 @@ fn compute_raw_data(
|
|||||||
dataspace: &Dataspace,
|
dataspace: &Dataspace,
|
||||||
datatype: &Datatype,
|
datatype: &Datatype,
|
||||||
) -> Vec<u8> {
|
) -> Vec<u8> {
|
||||||
let num_elements = dataspace.num_elements() as usize;
|
// Saturating, like the product: the size is capped at what is there.
|
||||||
|
let num_elements = usize::try_from(dataspace.num_elements()).unwrap_or(usize::MAX);
|
||||||
let elem_size = datatype.type_size() as usize;
|
let elem_size = datatype.type_size() as usize;
|
||||||
let expected_size = num_elements.saturating_mul(elem_size);
|
let expected_size = num_elements.saturating_mul(elem_size);
|
||||||
let available = data.len().saturating_sub(pos);
|
let available = data.len().saturating_sub(pos);
|
||||||
@@ -362,6 +387,18 @@ fn extract_name(bytes: &[u8]) -> String {
|
|||||||
String::from_utf8_lossy(&bytes[..end]).into_owned()
|
String::from_utf8_lossy(&bytes[..end]).into_owned()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// An attribute's datatype gets libhdf5's extra check for a header without
|
||||||
|
/// a checksum (see [`Datatype::check_unused_bits`]).
|
||||||
|
fn check_in_header(
|
||||||
|
attr: AttributeMessage,
|
||||||
|
header: &ObjectHeader,
|
||||||
|
) -> Result<AttributeMessage, FormatError> {
|
||||||
|
if header.version == 1 {
|
||||||
|
attr.datatype.check_unused_bits()?;
|
||||||
|
}
|
||||||
|
Ok(attr)
|
||||||
|
}
|
||||||
|
|
||||||
/// Extract all attribute messages from an object header.
|
/// Extract all attribute messages from an object header.
|
||||||
pub fn extract_attributes(
|
pub fn extract_attributes(
|
||||||
header: &ObjectHeader,
|
header: &ObjectHeader,
|
||||||
@@ -371,7 +408,7 @@ pub fn extract_attributes(
|
|||||||
for msg in &header.messages {
|
for msg in &header.messages {
|
||||||
if msg.msg_type == MessageType::Attribute {
|
if msg.msg_type == MessageType::Attribute {
|
||||||
let attr = AttributeMessage::parse(&msg.data, length_size)?;
|
let attr = AttributeMessage::parse(&msg.data, length_size)?;
|
||||||
attrs.push(attr);
|
attrs.push(check_in_header(attr, header)?);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
Ok(attrs)
|
Ok(attrs)
|
||||||
@@ -394,59 +431,304 @@ pub fn find_attribute<'a>(
|
|||||||
///
|
///
|
||||||
/// Use this instead of `extract_attributes` when reading files that may use dense storage
|
/// Use this instead of `extract_attributes` when reading files that may use dense storage
|
||||||
/// (e.g., objects with many attributes, typically >8).
|
/// (e.g., objects with many attributes, typically >8).
|
||||||
|
///
|
||||||
|
/// Fails if any attribute cannot be read; see [`extract_attributes_tolerant`]
|
||||||
|
/// to read the others.
|
||||||
pub fn extract_attributes_full(
|
pub fn extract_attributes_full(
|
||||||
file_data: &[u8],
|
file_data: &[u8],
|
||||||
header: &ObjectHeader,
|
header: &ObjectHeader,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
) -> Result<Vec<AttributeMessage>, FormatError> {
|
) -> Result<Vec<AttributeMessage>, FormatError> {
|
||||||
let mut attrs = Vec::new();
|
extract_attributes_full_in(file_data, header, offset_size, length_size)
|
||||||
|
}
|
||||||
|
|
||||||
// Collect compact attributes (inline in OH)
|
/// [`extract_attributes_full`] over any [`Storage`]. Dense attribute
|
||||||
for msg in &header.messages {
|
/// storage is indexed by a v2 B-tree, which is not read over [`Storage`]
|
||||||
if msg.msg_type == MessageType::Attribute {
|
/// yet: on a backend without the whole file in memory an object with dense
|
||||||
if shared_message::is_shared(msg.flags) {
|
/// attributes is [`FormatError::ContiguousStorageRequired`].
|
||||||
// Shared attribute: resolve the reference to get actual attribute data
|
pub fn extract_attributes_full_in<S: Storage + ?Sized>(
|
||||||
let shared_ref = shared_message::parse_shared_ref(&msg.data, offset_size)?;
|
file: &S,
|
||||||
let resolved_data = shared_message::resolve_shared_message(
|
header: &ObjectHeader,
|
||||||
file_data,
|
offset_size: u8,
|
||||||
&shared_ref,
|
length_size: u8,
|
||||||
MessageType::Attribute,
|
) -> Result<Vec<AttributeMessage>, FormatError> {
|
||||||
offset_size,
|
extract_attributes_with(file, header, offset_size, length_size, &mut Err)
|
||||||
length_size,
|
}
|
||||||
)?;
|
|
||||||
let attr = AttributeMessage::parse_in_file(
|
/// Like [`extract_attributes_full`], but an attribute that cannot be read
|
||||||
&resolved_data,
|
/// (a corrupt or unsupported attribute message, or a heap object that cannot
|
||||||
file_data,
|
/// be located) is left out and its error returned alongside the attributes
|
||||||
offset_size,
|
/// that could be read, instead of failing them all.
|
||||||
length_size,
|
///
|
||||||
)?;
|
/// Errors in the structures that index the attributes (the Attribute Info
|
||||||
attrs.push(attr);
|
/// message, the dense-storage heap header or B-tree) still fail the call:
|
||||||
} else {
|
/// then it is unknown which attributes exist at all.
|
||||||
let attr = AttributeMessage::parse_in_file(
|
pub fn extract_attributes_tolerant(
|
||||||
&msg.data,
|
file_data: &[u8],
|
||||||
file_data,
|
header: &ObjectHeader,
|
||||||
offset_size,
|
offset_size: u8,
|
||||||
length_size,
|
length_size: u8,
|
||||||
)?;
|
) -> Result<(Vec<AttributeMessage>, Vec<FormatError>), FormatError> {
|
||||||
attrs.push(attr);
|
extract_attributes_tolerant_core(file_data, header, offset_size, length_size)
|
||||||
}
|
}
|
||||||
}
|
|
||||||
|
/// [`extract_attributes_tolerant`] over any [`Storage`] (see
|
||||||
|
/// [`extract_attributes_full_in`] for dense storage). One with the whole
|
||||||
|
/// file in memory is read as the slice, by code compiled in this crate (see
|
||||||
|
/// [`crate::storage`], "Slice entry points").
|
||||||
|
#[inline]
|
||||||
|
pub fn extract_attributes_tolerant_in<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
header: &ObjectHeader,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<(Vec<AttributeMessage>, Vec<FormatError>), FormatError> {
|
||||||
|
match file_data.as_contiguous() {
|
||||||
|
Some(all) => extract_attributes_tolerant(all, header, offset_size, length_size),
|
||||||
|
None => extract_attributes_tolerant_core(file_data, header, offset_size, length_size),
|
||||||
}
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn extract_attributes_tolerant_core<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
header: &ObjectHeader,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<(Vec<AttributeMessage>, Vec<FormatError>), FormatError> {
|
||||||
|
let mut errors = Vec::new();
|
||||||
|
let attrs = extract_attributes_with(file_data, header, offset_size, length_size, &mut |e| {
|
||||||
|
errors.push(e);
|
||||||
|
Ok(())
|
||||||
|
})?;
|
||||||
|
Ok((attrs, errors))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read every attribute; each one that fails goes to `on_error`, which
|
||||||
|
/// either stops the read (returns the error) or skips that attribute.
|
||||||
|
fn extract_attributes_with<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
header: &ObjectHeader,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
on_error: &mut dyn FnMut(FormatError) -> Result<(), FormatError>,
|
||||||
|
) -> Result<Vec<AttributeMessage>, FormatError> {
|
||||||
|
let mut attrs = Vec::new();
|
||||||
|
// Each attribute's creation order, where the file records one.
|
||||||
|
let mut orders: Vec<u32> = Vec::new();
|
||||||
|
|
||||||
|
extract_compact_attributes(
|
||||||
|
file_data,
|
||||||
|
header,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
&mut attrs,
|
||||||
|
&mut orders,
|
||||||
|
on_error,
|
||||||
|
)?;
|
||||||
|
|
||||||
// Check for dense attributes via AttributeInfo message
|
// Check for dense attributes via AttributeInfo message
|
||||||
let attr_info = find_attribute_info(header, offset_size)?;
|
let attr_info = find_attribute_info(header, offset_size)?;
|
||||||
if let Some(info) = attr_info
|
if let Some(info) = &attr_info
|
||||||
&& let Some(fh_addr) = info.fractal_heap_address
|
&& let Some(fh_addr) = info.fractal_heap_address
|
||||||
{
|
{
|
||||||
let dense_attrs =
|
extract_dense_attributes(
|
||||||
extract_dense_attributes(file_data, &info, fh_addr, offset_size, length_size)?;
|
file_data,
|
||||||
attrs.extend(dense_attrs);
|
info,
|
||||||
|
fh_addr,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
&mut attrs,
|
||||||
|
&mut orders,
|
||||||
|
on_error,
|
||||||
|
)?;
|
||||||
|
}
|
||||||
|
|
||||||
|
// An object that tracks attribute creation order lists its attributes
|
||||||
|
// in that order (h5py's `track_order=True`), as libhdf5 does; otherwise
|
||||||
|
// they come in storage order.
|
||||||
|
if attr_info.is_some_and(|i| i.max_creation_index.is_some()) {
|
||||||
|
let mut paired: Vec<(u32, AttributeMessage)> = orders.into_iter().zip(attrs).collect();
|
||||||
|
paired.sort_by_key(|(o, _)| *o);
|
||||||
|
attrs = paired.into_iter().map(|(_, a)| a).collect();
|
||||||
}
|
}
|
||||||
|
|
||||||
Ok(attrs)
|
Ok(attrs)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// B-tree v2 record type of dense attribute storage's name index.
|
||||||
|
const ATTRIBUTE_NAME_INDEX: u8 = 8;
|
||||||
|
|
||||||
|
/// The attribute called `name` on the object with header `header`: the
|
||||||
|
/// first one [`extract_attributes_tolerant`] returns under that name, or
|
||||||
|
/// `None` if it returns none (an attribute that cannot be read is not
|
||||||
|
/// returned there either).
|
||||||
|
///
|
||||||
|
/// Compact attributes are in the header and are scanned. Dense attributes
|
||||||
|
/// are found through the name index (a v2 B-tree of lookup3 name hashes,
|
||||||
|
/// record type 8): only the attributes whose names hash like `name` are read
|
||||||
|
/// from the heap, O(log n) instead of all of them. Errors in the structures
|
||||||
|
/// that index the attributes fail the call, as they fail a listing.
|
||||||
|
pub fn find_attribute_in_file(
|
||||||
|
file_data: &[u8],
|
||||||
|
header: &ObjectHeader,
|
||||||
|
name: &str,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<Option<AttributeMessage>, FormatError> {
|
||||||
|
find_attribute_core(file_data, header, name, offset_size, length_size)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`find_attribute_in_file`] over any [`Storage`] (see
|
||||||
|
/// [`extract_attributes_full_in`] for dense storage, whose name index still
|
||||||
|
/// needs the whole file in memory). One with the whole file in memory is
|
||||||
|
/// read as the slice, by code compiled in this crate (see
|
||||||
|
/// [`crate::storage`], "Slice entry points").
|
||||||
|
#[inline]
|
||||||
|
pub fn find_attribute_in<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
header: &ObjectHeader,
|
||||||
|
name: &str,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<Option<AttributeMessage>, FormatError> {
|
||||||
|
match file_data.as_contiguous() {
|
||||||
|
Some(all) => find_attribute_in_file(all, header, name, offset_size, length_size),
|
||||||
|
None => find_attribute_core(file_data, header, name, offset_size, length_size),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn find_attribute_core<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
header: &ObjectHeader,
|
||||||
|
name: &str,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<Option<AttributeMessage>, FormatError> {
|
||||||
|
let attr_info = find_attribute_info(header, offset_size)?;
|
||||||
|
let dense = attr_info
|
||||||
|
.as_ref()
|
||||||
|
.and_then(|i| Some((i.fractal_heap_address?, i.btree_name_index_address?)));
|
||||||
|
let Some((fh_addr, btree_addr)) = dense else {
|
||||||
|
// Compact only (or dense storage without a name index, which a
|
||||||
|
// listing reports): as a listing finds it.
|
||||||
|
return Ok(
|
||||||
|
extract_attributes_tolerant_in(file_data, header, offset_size, length_size)?
|
||||||
|
.0
|
||||||
|
.into_iter()
|
||||||
|
.find(|a| a.name == name),
|
||||||
|
);
|
||||||
|
};
|
||||||
|
let btree_hdr = BTreeV2Header::parse_in(
|
||||||
|
file_data,
|
||||||
|
to_usize(btree_addr)? as u64,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
)?;
|
||||||
|
let fh = FractalHeapHeader::parse_in(file_data, fh_addr, offset_size, length_size)?;
|
||||||
|
if btree_hdr.tree_type != ATTRIBUTE_NAME_INDEX || btree_hdr.record_size < 4 {
|
||||||
|
return Ok(
|
||||||
|
extract_attributes_tolerant_in(file_data, header, offset_size, length_size)?
|
||||||
|
.0
|
||||||
|
.into_iter()
|
||||||
|
.find(|a| a.name == name),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
// A listing has the compact attributes first.
|
||||||
|
let mut compact = Vec::new();
|
||||||
|
extract_compact_attributes(
|
||||||
|
file_data,
|
||||||
|
header,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
&mut compact,
|
||||||
|
&mut Vec::new(),
|
||||||
|
&mut |_| Ok(()),
|
||||||
|
)?;
|
||||||
|
if let Some(a) = compact.into_iter().find(|a| a.name == name) {
|
||||||
|
return Ok(Some(a));
|
||||||
|
}
|
||||||
|
|
||||||
|
// Record: heap ID + message flags(1) + creation order(4) + hash(4); the
|
||||||
|
// hash is the last field.
|
||||||
|
let hash = jenkins_lookup3(name.as_bytes());
|
||||||
|
let hash_at = usize::from(btree_hdr.record_size) - 4;
|
||||||
|
let records = find_btree_v2_records_in(file_data, &btree_hdr, offset_size, &mut |r| match r
|
||||||
|
.get(hash_at..hash_at + 4)
|
||||||
|
{
|
||||||
|
Some(h) => u32::from_le_bytes([h[0], h[1], h[2], h[3]]).cmp(&hash),
|
||||||
|
None => core::cmp::Ordering::Less,
|
||||||
|
})?;
|
||||||
|
let id_len = usize::from(fh.heap_id_length);
|
||||||
|
for record in &records {
|
||||||
|
let Some(id_bytes) = record.data.get(..id_len) else {
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
let attr = fh
|
||||||
|
.read_managed_object_in(file_data, id_bytes, offset_size)
|
||||||
|
.and_then(|d| {
|
||||||
|
AttributeMessage::parse_in_storage(&d, file_data, offset_size, length_size)
|
||||||
|
});
|
||||||
|
// One that cannot be read is left out, as from a listing.
|
||||||
|
if let Ok(attr) = attr
|
||||||
|
&& attr.name == name
|
||||||
|
{
|
||||||
|
return Ok(Some(attr));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(None)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The attributes stored in the object header itself (compact storage), and
|
||||||
|
/// each one's creation order into `orders`.
|
||||||
|
fn extract_compact_attributes<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
header: &ObjectHeader,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
attrs: &mut Vec<AttributeMessage>,
|
||||||
|
orders: &mut Vec<u32>,
|
||||||
|
on_error: &mut dyn FnMut(FormatError) -> Result<(), FormatError>,
|
||||||
|
) -> Result<(), FormatError> {
|
||||||
|
for msg in &header.messages {
|
||||||
|
if msg.msg_type == MessageType::Attribute {
|
||||||
|
let attr = if shared_message::is_shared(msg.flags) {
|
||||||
|
// Shared attribute: resolve the reference to get actual attribute data
|
||||||
|
shared_message::parse_shared_ref_sized(&msg.data, offset_size, length_size)
|
||||||
|
.and_then(|shared_ref| {
|
||||||
|
shared_message::resolve_shared_message_in(
|
||||||
|
file_data,
|
||||||
|
&shared_ref,
|
||||||
|
MessageType::Attribute,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
)
|
||||||
|
})
|
||||||
|
.and_then(|resolved| {
|
||||||
|
AttributeMessage::parse_in_storage(
|
||||||
|
&resolved,
|
||||||
|
file_data,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
)
|
||||||
|
})
|
||||||
|
} else {
|
||||||
|
AttributeMessage::parse_in_storage(&msg.data, file_data, offset_size, length_size)
|
||||||
|
};
|
||||||
|
let attr = attr.and_then(|a| check_in_header(a, header));
|
||||||
|
match attr {
|
||||||
|
Ok(attr) => {
|
||||||
|
attrs.push(attr);
|
||||||
|
orders.push(msg.creation_order.map_or(0, u32::from));
|
||||||
|
}
|
||||||
|
Err(e) => on_error(e)?,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
/// Find and parse the Attribute Info message from an object header.
|
/// Find and parse the Attribute Info message from an object header.
|
||||||
fn find_attribute_info(
|
fn find_attribute_info(
|
||||||
header: &ObjectHeader,
|
header: &ObjectHeader,
|
||||||
@@ -461,16 +743,21 @@ fn find_attribute_info(
|
|||||||
Ok(None)
|
Ok(None)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Extract attributes from dense storage (fractal heap + B-tree v2).
|
/// Extract attributes from dense storage (fractal heap + B-tree v2), and
|
||||||
fn extract_dense_attributes(
|
/// each one's creation order into `orders`.
|
||||||
file_data: &[u8],
|
#[allow(clippy::too_many_arguments)]
|
||||||
|
fn extract_dense_attributes<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
attr_info: &AttributeInfoMessage,
|
attr_info: &AttributeInfoMessage,
|
||||||
fh_addr: u64,
|
fh_addr: u64,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
) -> Result<Vec<AttributeMessage>, FormatError> {
|
attrs: &mut Vec<AttributeMessage>,
|
||||||
|
orders: &mut Vec<u32>,
|
||||||
|
on_error: &mut dyn FnMut(FormatError) -> Result<(), FormatError>,
|
||||||
|
) -> Result<(), FormatError> {
|
||||||
// Parse fractal heap
|
// Parse fractal heap
|
||||||
let fh = FractalHeapHeader::parse(file_data, fh_addr as usize, offset_size, length_size)?;
|
let fh = FractalHeapHeader::parse_in(file_data, fh_addr, offset_size, length_size)?;
|
||||||
|
|
||||||
// Parse B-tree v2 for name index (type 8)
|
// Parse B-tree v2 for name index (type 8)
|
||||||
let btree_addr = attr_info
|
let btree_addr = attr_info
|
||||||
@@ -479,31 +766,47 @@ fn extract_dense_attributes(
|
|||||||
expected: 1,
|
expected: 1,
|
||||||
available: 0,
|
available: 0,
|
||||||
})?;
|
})?;
|
||||||
let btree_hdr = BTreeV2Header::parse(file_data, btree_addr as usize, offset_size, length_size)?;
|
let btree_hdr = BTreeV2Header::parse_in(
|
||||||
let records = collect_btree_v2_records(file_data, &btree_hdr, offset_size, length_size)?;
|
file_data,
|
||||||
|
to_usize(btree_addr)? as u64,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
)?;
|
||||||
|
let records = collect_btree_v2_records_in(file_data, &btree_hdr, offset_size, length_size)?;
|
||||||
|
|
||||||
let mut attrs = Vec::new();
|
|
||||||
for record in &records {
|
for record in &records {
|
||||||
// Per HDF5 spec, both type 8 and type 9 records start with heap_id:
|
// Per HDF5 spec, both type 8 and type 9 records start with heap_id:
|
||||||
// Type 8: heap_id(8) + msg_flags(1) + creation_order(4) + hash(4)
|
// Type 8: heap_id(8) + msg_flags(1) + creation_order(4) + hash(4)
|
||||||
// Type 9: heap_id(8) + msg_flags(1) + creation_order(4)
|
// Type 9: heap_id(8) + msg_flags(1) + creation_order(4)
|
||||||
let id_offset = 0;
|
let id_len = fh.heap_id_length as usize;
|
||||||
|
let Some(id_bytes) = record.data.get(..id_len) else {
|
||||||
if record.data.len() < id_offset + fh.heap_id_length as usize {
|
on_error(FormatError::UnexpectedEof {
|
||||||
|
expected: id_len,
|
||||||
|
available: record.data.len(),
|
||||||
|
})?;
|
||||||
continue;
|
continue;
|
||||||
}
|
};
|
||||||
let id_bytes = &record.data[id_offset..id_offset + fh.heap_id_length as usize];
|
|
||||||
|
|
||||||
// Read attribute message from fractal heap
|
|
||||||
let attr_data = fh.read_managed_object(file_data, id_bytes, offset_size)?;
|
|
||||||
|
|
||||||
// The data in the heap is a complete attribute message
|
// The data in the heap is a complete attribute message
|
||||||
let attr =
|
let attr = fh
|
||||||
AttributeMessage::parse_in_file(&attr_data, file_data, offset_size, length_size)?;
|
.read_managed_object_in(file_data, id_bytes, offset_size)
|
||||||
attrs.push(attr);
|
.and_then(|attr_data| {
|
||||||
|
AttributeMessage::parse_in_storage(&attr_data, file_data, offset_size, length_size)
|
||||||
|
});
|
||||||
|
match attr {
|
||||||
|
Ok(attr) => {
|
||||||
|
attrs.push(attr);
|
||||||
|
let order = record
|
||||||
|
.data
|
||||||
|
.get(id_len + 1..id_len + 5)
|
||||||
|
.map_or(0, |b| u32::from_le_bytes([b[0], b[1], b[2], b[3]]));
|
||||||
|
orders.push(order);
|
||||||
|
}
|
||||||
|
Err(e) => on_error(e)?,
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
Ok(attrs)
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
@@ -523,7 +826,8 @@ mod tests {
|
|||||||
|
|
||||||
/// Build an f64 LE datatype message.
|
/// Build an f64 LE datatype message.
|
||||||
fn build_f64_dt() -> Vec<u8> {
|
fn build_f64_dt() -> Vec<u8> {
|
||||||
let mut buf = build_dt_header(1, 1, [0x00, 0x00, 0x02], 8);
|
// Sign bit 63 (bits 8-15 of the class bits).
|
||||||
|
let mut buf = build_dt_header(1, 1, [0x20, 63, 0x00], 8);
|
||||||
let mut props = [0u8; 12];
|
let mut props = [0u8; 12];
|
||||||
props[2..4].copy_from_slice(&64u16.to_le_bytes()); // bit_precision
|
props[2..4].copy_from_slice(&64u16.to_le_bytes()); // bit_precision
|
||||||
props[4] = 52; // exp_location
|
props[4] = 52; // exp_location
|
||||||
@@ -897,4 +1201,73 @@ mod tests {
|
|||||||
let strs = attr.read_as_strings().unwrap();
|
let strs = attr.read_as_strings().unwrap();
|
||||||
assert_eq!(strs, vec!["abcd", "EFGH"]);
|
assert_eq!(strs, vec!["abcd", "EFGH"]);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Every object's attributes in h5py-written files read identically
|
||||||
|
/// through a read_at-only CountingStorage — compact ones, shared ones,
|
||||||
|
/// those behind an Attribute Info message and dense storage (its v2
|
||||||
|
/// B-tree name index included) — and through a slice as Storage.
|
||||||
|
#[test]
|
||||||
|
fn storage_reads_match_slice_reads() {
|
||||||
|
use crate::storage::CountingStorage;
|
||||||
|
let files: [(&str, &[u8]); 5] = [
|
||||||
|
("attrs", include_bytes!("../tests/fixtures/attrs.h5")),
|
||||||
|
(
|
||||||
|
"mixed_attrs",
|
||||||
|
include_bytes!("../tests/fixtures/mixed_attrs.h5"),
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"dense_attrs",
|
||||||
|
include_bytes!("../tests/fixtures/dense_attrs.h5"),
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"dense_attrs_root",
|
||||||
|
include_bytes!("../tests/fixtures/dense_attrs_root.h5"),
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"shared_fill_value",
|
||||||
|
include_bytes!("../tests/fixtures/shared_fill_value.h5"),
|
||||||
|
),
|
||||||
|
];
|
||||||
|
let (mut same, mut dense, mut attrs) = (0, 0, 0);
|
||||||
|
for (name, file) in files {
|
||||||
|
let sb = crate::superblock::Superblock::parse(file, 0).unwrap();
|
||||||
|
let (os, ls) = (sb.offset_size, sb.length_size);
|
||||||
|
let mut addrs = vec![sb.root_group_address];
|
||||||
|
addrs.extend(
|
||||||
|
crate::group_v2::resolve_group_children(file, &sb, sb.root_group_address)
|
||||||
|
.unwrap()
|
||||||
|
.iter()
|
||||||
|
.map(|e| e.object_header_address),
|
||||||
|
);
|
||||||
|
let storage = CountingStorage::new(file.to_vec());
|
||||||
|
for addr in addrs {
|
||||||
|
let header = ObjectHeader::parse(file, addr as usize, os, ls).unwrap();
|
||||||
|
let want = extract_attributes_full(file, &header, os, ls);
|
||||||
|
let slice_storage = extract_attributes_full_in(&file, &header, os, ls);
|
||||||
|
assert_eq!(format!("{slice_storage:?}"), format!("{want:?}"));
|
||||||
|
let got = extract_attributes_full_in(&storage, &header, os, ls);
|
||||||
|
let got_t = extract_attributes_tolerant_in(&storage, &header, os, ls);
|
||||||
|
let is_dense = find_attribute_info(&header, os)
|
||||||
|
.unwrap()
|
||||||
|
.is_some_and(|i| i.fractal_heap_address.is_some());
|
||||||
|
if is_dense {
|
||||||
|
dense += 1;
|
||||||
|
}
|
||||||
|
attrs += want.as_ref().map_or(0, Vec::len);
|
||||||
|
assert_eq!(format!("{got:?}"), format!("{want:?}"), "{name}");
|
||||||
|
let want_t = extract_attributes_tolerant(file, &header, os, ls);
|
||||||
|
assert_eq!(format!("{got_t:?}"), format!("{want_t:?}"), "{name}");
|
||||||
|
same += 1;
|
||||||
|
for a in want.iter().flatten() {
|
||||||
|
let one = find_attribute_in(&storage, &header, &a.name, os, ls);
|
||||||
|
let want_one = find_attribute_in_file(file, &header, &a.name, os, ls);
|
||||||
|
assert_eq!(format!("{one:?}"), format!("{want_one:?}"), "{name}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
assert!(
|
||||||
|
same >= 5 && dense >= 2 && attrs >= 5,
|
||||||
|
"{same} {dense} {attrs}"
|
||||||
|
);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -4,6 +4,7 @@
|
|||||||
use alloc::vec::Vec;
|
use alloc::vec::Vec;
|
||||||
|
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
|
use crate::storage::{Storage, read_exact_at};
|
||||||
|
|
||||||
/// A parsed B-tree v1 node.
|
/// A parsed B-tree v1 node.
|
||||||
#[derive(Debug, Clone)]
|
#[derive(Debug, Clone)]
|
||||||
@@ -74,13 +75,28 @@ impl BTreeV1Node {
|
|||||||
file_data: &[u8],
|
file_data: &[u8],
|
||||||
offset: usize,
|
offset: usize,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<BTreeV1Node, FormatError> {
|
||||||
|
Self::parse_in(file_data, offset as u64, offset_size, length_size)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`Self::parse`] over any [`Storage`]: one read of the node's header,
|
||||||
|
/// one of its keys and children.
|
||||||
|
pub fn parse_in<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
offset: u64,
|
||||||
|
offset_size: u8,
|
||||||
_length_size: u8,
|
_length_size: u8,
|
||||||
) -> Result<BTreeV1Node, FormatError> {
|
) -> Result<BTreeV1Node, FormatError> {
|
||||||
// signature(4) + node_type(1) + node_level(1) + entries_used(2) = 8
|
// signature(4) + node_type(1) + node_level(1) + entries_used(2) = 8
|
||||||
// + left_sibling(offset_size) + right_sibling(offset_size)
|
// + left_sibling(offset_size) + right_sibling(offset_size)
|
||||||
let os = offset_size as usize;
|
let os = offset_size as usize;
|
||||||
let header_size = 8 + os * 2;
|
let header_size = 8 + os * 2;
|
||||||
ensure_len(file_data, offset, header_size)?;
|
let header = read_exact_at(file, offset, header_size)?;
|
||||||
|
let file_data: &[u8] = &header;
|
||||||
|
// The header's read checked that `offset + header_size` fits.
|
||||||
|
let body_start = offset + header_size as u64;
|
||||||
|
let offset = 0usize;
|
||||||
|
|
||||||
if &file_data[offset..offset + 4] != b"TREE" {
|
if &file_data[offset..offset + 4] != b"TREE" {
|
||||||
return Err(FormatError::InvalidBTreeSignature);
|
return Err(FormatError::InvalidBTreeSignature);
|
||||||
@@ -102,31 +118,30 @@ impl BTreeV1Node {
|
|||||||
} else {
|
} else {
|
||||||
Some(read_offset(file_data, pos, offset_size)?)
|
Some(read_offset(file_data, pos, offset_size)?)
|
||||||
};
|
};
|
||||||
pos += os;
|
|
||||||
|
|
||||||
// For type 0: keys are offset_size bytes, children are offset_size bytes
|
// For type 0: keys are offset_size bytes, children are offset_size bytes
|
||||||
// Layout: key[0], child[0], key[1], child[1], ..., key[N-1], child[N-1], key[N]
|
// Layout: key[0], child[0], key[1], child[1], ..., key[N-1], child[N-1], key[N]
|
||||||
let eu = entries_used as usize;
|
let eu = entries_used as usize;
|
||||||
let key_size = os; // For type 0, key = offset_size
|
let key_size = os; // For type 0, key = offset_size
|
||||||
let needed = eu * (key_size + os) + key_size; // eu children + (eu+1) keys
|
let needed = eu * (key_size + os) + key_size; // eu children + (eu+1) keys
|
||||||
ensure_len(file_data, pos, needed)?;
|
let body = read_exact_at(file, body_start, needed)?;
|
||||||
|
let file_data: &[u8] = &body;
|
||||||
|
|
||||||
let mut keys = Vec::with_capacity(eu + 1);
|
let mut keys = Vec::with_capacity(eu + 1);
|
||||||
let mut children = Vec::with_capacity(eu);
|
let mut children = Vec::with_capacity(eu);
|
||||||
|
|
||||||
for _i in 0..eu {
|
if os == 0 {
|
||||||
// key[i]
|
// What reading the first key reports (and keeps `chunks_exact`
|
||||||
let key = read_offset(file_data, pos, offset_size)?;
|
// below from being given a zero size).
|
||||||
keys.push(key);
|
return Err(FormatError::InvalidOffsetSize(offset_size));
|
||||||
pos += key_size;
|
|
||||||
// child[i]
|
|
||||||
let child = read_offset(file_data, pos, offset_size)?;
|
|
||||||
children.push(child);
|
|
||||||
pos += os;
|
|
||||||
}
|
}
|
||||||
// final key
|
// `needed` bytes: key[0], child[0], ..., child[eu - 1], key[eu].
|
||||||
let key = read_offset(file_data, pos, offset_size)?;
|
let (pairs, last) = file_data.split_at(eu * (key_size + os));
|
||||||
keys.push(key);
|
for pair in pairs.chunks_exact(key_size + os) {
|
||||||
|
keys.push(read_offset(pair, 0, offset_size)?);
|
||||||
|
children.push(read_offset(pair, key_size, offset_size)?);
|
||||||
|
}
|
||||||
|
keys.push(read_offset(last, 0, offset_size)?);
|
||||||
|
|
||||||
Ok(BTreeV1Node {
|
Ok(BTreeV1Node {
|
||||||
node_type,
|
node_type,
|
||||||
@@ -150,11 +165,21 @@ pub fn collect_symbol_table_nodes(
|
|||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
) -> Result<Vec<u64>, FormatError> {
|
) -> Result<Vec<u64>, FormatError> {
|
||||||
collect_symbol_table_nodes_inner(file_data, btree_address, offset_size, length_size, 0)
|
collect_symbol_table_nodes_in(file_data, btree_address, offset_size, length_size)
|
||||||
}
|
}
|
||||||
|
|
||||||
fn collect_symbol_table_nodes_inner(
|
/// [`collect_symbol_table_nodes`] over any [`Storage`]: two reads per node.
|
||||||
file_data: &[u8],
|
pub fn collect_symbol_table_nodes_in<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
btree_address: u64,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<Vec<u64>, FormatError> {
|
||||||
|
collect_symbol_table_nodes_inner(file, btree_address, offset_size, length_size, 0)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn collect_symbol_table_nodes_inner<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
btree_address: u64,
|
btree_address: u64,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
@@ -164,7 +189,7 @@ fn collect_symbol_table_nodes_inner(
|
|||||||
return Err(FormatError::NestingDepthExceeded);
|
return Err(FormatError::NestingDepthExceeded);
|
||||||
}
|
}
|
||||||
|
|
||||||
let node = BTreeV1Node::parse(file_data, btree_address as usize, offset_size, length_size)?;
|
let node = BTreeV1Node::parse_in(file, btree_address, offset_size, length_size)?;
|
||||||
|
|
||||||
if node.node_type != 0 {
|
if node.node_type != 0 {
|
||||||
return Err(FormatError::InvalidBTreeNodeType(node.node_type));
|
return Err(FormatError::InvalidBTreeNodeType(node.node_type));
|
||||||
@@ -178,7 +203,7 @@ fn collect_symbol_table_nodes_inner(
|
|||||||
let mut result = Vec::new();
|
let mut result = Vec::new();
|
||||||
for &child_addr in &node.children {
|
for &child_addr in &node.children {
|
||||||
let child_snods = collect_symbol_table_nodes_inner(
|
let child_snods = collect_symbol_table_nodes_inner(
|
||||||
file_data,
|
file,
|
||||||
child_addr,
|
child_addr,
|
||||||
offset_size,
|
offset_size,
|
||||||
length_size,
|
length_size,
|
||||||
@@ -317,4 +342,48 @@ mod tests {
|
|||||||
assert_eq!(node.entries_used, 1);
|
assert_eq!(node.entries_used, 1);
|
||||||
assert_eq!(node.children, vec![0x50]);
|
assert_eq!(node.children, vec![0x50]);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Nodes and trees, cut at every length, parse identically through a
|
||||||
|
/// `read_at`-only storage.
|
||||||
|
#[test]
|
||||||
|
fn storage_parse_matches_slice_parse() {
|
||||||
|
use crate::storage::CountingStorage;
|
||||||
|
let nodes = [
|
||||||
|
build_btree_node(0, 0, &[0, 5, 10], &[0x100, 0x200], None, None, 8),
|
||||||
|
build_btree_node(0, 0, &[0, 5], &[0x100], Some(0x40), Some(0x80), 4),
|
||||||
|
build_btree_node(1, 2, &[0, 5], &[0x100], None, Some(0x80), 8),
|
||||||
|
];
|
||||||
|
for (n, node) in nodes.iter().enumerate() {
|
||||||
|
let os = if n == 1 { 4 } else { 8 };
|
||||||
|
for cut in 0..=node.len() {
|
||||||
|
let f = &node[..cut];
|
||||||
|
let storage = CountingStorage::new(f.to_vec());
|
||||||
|
let want = BTreeV1Node::parse(f, 0, os, 8);
|
||||||
|
let got = BTreeV1Node::parse_in(&storage, 0, os, 8);
|
||||||
|
assert_eq!(format!("{got:?}"), format!("{want:?}"));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let leaf1 = build_btree_node(0, 0, &[0, 5], &[0xA00], None, None, 8);
|
||||||
|
let leaf2 = build_btree_node(0, 0, &[5, 10], &[0xB00], None, None, 8);
|
||||||
|
let internal = build_btree_node(0, 1, &[0, 5, 10], &[0, 256], None, None, 8);
|
||||||
|
let mut file = vec![0u8; 512 + internal.len()];
|
||||||
|
file[..leaf1.len()].copy_from_slice(&leaf1);
|
||||||
|
file[256..256 + leaf2.len()].copy_from_slice(&leaf2);
|
||||||
|
file[512..].copy_from_slice(&internal);
|
||||||
|
for cut in [file.len(), 300, 260, 100, 10] {
|
||||||
|
let mut f = file.clone();
|
||||||
|
if cut < 512 {
|
||||||
|
// Truncate the leaves, keep the root.
|
||||||
|
f[cut..512].fill(0);
|
||||||
|
}
|
||||||
|
let storage = CountingStorage::new(f.clone());
|
||||||
|
assert_eq!(
|
||||||
|
collect_symbol_table_nodes_in(&storage, 512, 8, 8),
|
||||||
|
collect_symbol_table_nodes(&f, 512, 8, 8)
|
||||||
|
);
|
||||||
|
}
|
||||||
|
let storage = CountingStorage::new(file);
|
||||||
|
collect_symbol_table_nodes_in(&storage, 512, 8, 8).unwrap();
|
||||||
|
assert_eq!(storage.reads(), 6);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -2,11 +2,14 @@
|
|||||||
|
|
||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
use alloc::vec::Vec;
|
use alloc::vec::Vec;
|
||||||
|
use core::cmp::Ordering;
|
||||||
|
|
||||||
#[cfg(feature = "checksum")]
|
#[cfg(feature = "checksum")]
|
||||||
use byteorder::{ByteOrder, LittleEndian};
|
use byteorder::{ByteOrder, LittleEndian};
|
||||||
|
|
||||||
|
use crate::addr::to_usize;
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
|
use crate::storage::{Storage, Window, len_usize};
|
||||||
|
|
||||||
/// Parsed B-tree v2 header (signature "BTHD").
|
/// Parsed B-tree v2 header (signature "BTHD").
|
||||||
#[derive(Debug, Clone)]
|
#[derive(Debug, Clone)]
|
||||||
@@ -71,7 +74,7 @@ fn ensure_len(data: &[u8], pos: usize, needed: usize) -> Result<(), FormatError>
|
|||||||
|
|
||||||
/// Compute the number of bytes needed to represent a count, using variable-width encoding.
|
/// Compute the number of bytes needed to represent a count, using variable-width encoding.
|
||||||
/// B-tree v2 uses this for the number of records fields in internal nodes.
|
/// B-tree v2 uses this for the number of records fields in internal nodes.
|
||||||
fn bytes_for_max_records(max_nrec: u64) -> usize {
|
pub(crate) fn bytes_for_max_records(max_nrec: u64) -> usize {
|
||||||
if max_nrec == 0 {
|
if max_nrec == 0 {
|
||||||
return 1;
|
return 1;
|
||||||
}
|
}
|
||||||
@@ -97,38 +100,52 @@ impl BTreeV2Header {
|
|||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
) -> Result<BTreeV2Header, FormatError> {
|
) -> Result<BTreeV2Header, FormatError> {
|
||||||
ensure_len(file_data, offset, 4)?;
|
Self::parse_in(file_data, offset as u64, offset_size, length_size)
|
||||||
if &file_data[offset..offset + 4] != b"BTHD" {
|
}
|
||||||
|
|
||||||
|
/// [`Self::parse`] over any [`Storage`]: one bounded read of the
|
||||||
|
/// header.
|
||||||
|
pub fn parse_in<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
offset: u64,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<BTreeV2Header, FormatError> {
|
||||||
|
// Every field and the checksum; the window holds all of it or ends
|
||||||
|
// at the end of the file, so its bounds checks are the whole-file
|
||||||
|
// ones.
|
||||||
|
let full = 16 + usize::from(offset_size) + 2 + usize::from(length_size) + 4;
|
||||||
|
let w = Window::read(file, offset, full)?;
|
||||||
|
let d = &w.bytes;
|
||||||
|
w.ensure(0, 4)?;
|
||||||
|
if &d[..4] != b"BTHD" {
|
||||||
return Err(FormatError::InvalidBTreeV2Signature);
|
return Err(FormatError::InvalidBTreeV2Signature);
|
||||||
}
|
}
|
||||||
|
|
||||||
ensure_len(file_data, offset, 4 + 1 + 1 + 4 + 2 + 2 + 1 + 1)?;
|
w.ensure(0, 4 + 1 + 1 + 4 + 2 + 2 + 1 + 1)?;
|
||||||
let version = file_data[offset + 4];
|
let version = d[4];
|
||||||
if version != 0 {
|
if version != 0 {
|
||||||
return Err(FormatError::InvalidBTreeV2Version(version));
|
return Err(FormatError::InvalidBTreeV2Version(version));
|
||||||
}
|
}
|
||||||
|
|
||||||
let tree_type = file_data[offset + 5];
|
let tree_type = d[5];
|
||||||
let node_size = u32::from_le_bytes([
|
let node_size = u32::from_le_bytes([d[6], d[7], d[8], d[9]]);
|
||||||
file_data[offset + 6],
|
let record_size = u16::from_le_bytes([d[10], d[11]]);
|
||||||
file_data[offset + 7],
|
let depth = u16::from_le_bytes([d[12], d[13]]);
|
||||||
file_data[offset + 8],
|
let _split_percent = d[14];
|
||||||
file_data[offset + 9],
|
let _merge_percent = d[15];
|
||||||
]);
|
|
||||||
let record_size = u16::from_le_bytes([file_data[offset + 10], file_data[offset + 11]]);
|
|
||||||
let depth = u16::from_le_bytes([file_data[offset + 12], file_data[offset + 13]]);
|
|
||||||
let _split_percent = file_data[offset + 14];
|
|
||||||
let _merge_percent = file_data[offset + 15];
|
|
||||||
|
|
||||||
let mut pos = offset + 16;
|
let mut pos = 16;
|
||||||
let root_node_address = read_offset(file_data, pos, offset_size)?;
|
w.ensure(pos, usize::from(offset_size))?;
|
||||||
|
let root_node_address = read_offset(d, pos, offset_size)?;
|
||||||
pos += offset_size as usize;
|
pos += offset_size as usize;
|
||||||
|
|
||||||
ensure_len(file_data, pos, 2)?;
|
w.ensure(pos, 2)?;
|
||||||
let num_records_in_root = u16::from_le_bytes([file_data[pos], file_data[pos + 1]]);
|
let num_records_in_root = u16::from_le_bytes([d[pos], d[pos + 1]]);
|
||||||
pos += 2;
|
pos += 2;
|
||||||
|
|
||||||
let total_records = read_offset(file_data, pos, length_size)?;
|
w.ensure(pos, usize::from(length_size))?;
|
||||||
|
let total_records = read_offset(d, pos, length_size)?;
|
||||||
#[allow(unused_assignments)]
|
#[allow(unused_assignments)]
|
||||||
{
|
{
|
||||||
pos += length_size as usize;
|
pos += length_size as usize;
|
||||||
@@ -137,9 +154,9 @@ impl BTreeV2Header {
|
|||||||
// Validate header checksum
|
// Validate header checksum
|
||||||
#[cfg(feature = "checksum")]
|
#[cfg(feature = "checksum")]
|
||||||
{
|
{
|
||||||
ensure_len(file_data, pos, 4)?;
|
w.ensure(pos, 4)?;
|
||||||
let stored = LittleEndian::read_u32(&file_data[pos..pos + 4]);
|
let stored = LittleEndian::read_u32(&d[pos..pos + 4]);
|
||||||
let computed = crate::checksum::jenkins_lookup3(&file_data[offset..pos]);
|
let computed = crate::checksum::jenkins_lookup3(&d[..pos]);
|
||||||
if computed != stored {
|
if computed != stored {
|
||||||
return Err(FormatError::ChecksumMismatch {
|
return Err(FormatError::ChecksumMismatch {
|
||||||
expected: stored,
|
expected: stored,
|
||||||
@@ -163,7 +180,7 @@ impl BTreeV2Header {
|
|||||||
/// Compute maximum records per node for a given depth level.
|
/// Compute maximum records per node for a given depth level.
|
||||||
/// leaf: (node_size - overhead) / record_size
|
/// leaf: (node_size - overhead) / record_size
|
||||||
/// internal: depends on pointers
|
/// internal: depends on pointers
|
||||||
fn max_records_leaf(node_size: u32, record_size: u16) -> u64 {
|
pub(crate) fn max_records_leaf(node_size: u32, record_size: u16) -> u64 {
|
||||||
// Leaf overhead: signature(4) + version(1) + type(1) + checksum(4) = 10
|
// Leaf overhead: signature(4) + version(1) + type(1) + checksum(4) = 10
|
||||||
let overhead = 10u32;
|
let overhead = 10u32;
|
||||||
if node_size <= overhead || record_size == 0 {
|
if node_size <= overhead || record_size == 0 {
|
||||||
@@ -172,33 +189,72 @@ fn max_records_leaf(node_size: u32, record_size: u16) -> u64 {
|
|||||||
((node_size - overhead) / record_size as u32) as u64
|
((node_size - overhead) / record_size as u32) as u64
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Deepest B-tree v2 accepted. See [`collect_btree_v2_records`].
|
||||||
|
const MAX_DEPTH: u16 = 64;
|
||||||
|
|
||||||
|
/// Take `n` records from the traversal's budget, or refuse the tree.
|
||||||
|
fn spend(budget: &mut usize, n: usize) -> Result<(), FormatError> {
|
||||||
|
*budget = budget
|
||||||
|
.checked_sub(n)
|
||||||
|
.ok_or(FormatError::NestingDepthExceeded)?;
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
/// Collect all records from a B-tree v2 by traversing from the root.
|
/// Collect all records from a B-tree v2 by traversing from the root.
|
||||||
pub fn collect_btree_v2_records(
|
pub fn collect_btree_v2_records(
|
||||||
file_data: &[u8],
|
file_data: &[u8],
|
||||||
header: &BTreeV2Header,
|
header: &BTreeV2Header,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
|
) -> Result<Vec<BTreeV2Record>, FormatError> {
|
||||||
|
collect_btree_v2_records_in(file_data, header, offset_size, length_size)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`collect_btree_v2_records`] over any [`Storage`]: one bounded read per
|
||||||
|
/// node.
|
||||||
|
pub fn collect_btree_v2_records_in<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
header: &BTreeV2Header,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
) -> Result<Vec<BTreeV2Record>, FormatError> {
|
) -> Result<Vec<BTreeV2Record>, FormatError> {
|
||||||
if header.total_records == 0 || header.num_records_in_root == 0 {
|
if header.total_records == 0 || header.num_records_in_root == 0 {
|
||||||
return Ok(Vec::new());
|
return Ok(Vec::new());
|
||||||
}
|
}
|
||||||
|
// Recursion is one frame per level, and the depth is read from the file:
|
||||||
|
// a crafted header claiming 65 535 levels over a node that is its own
|
||||||
|
// child overflowed the stack. 64 matches the fractal heap's guard, and no
|
||||||
|
// real tree comes close — even at the minimum fan-out of two it would
|
||||||
|
// hold more than 2^64 records.
|
||||||
|
if header.depth > MAX_DEPTH {
|
||||||
|
return Err(FormatError::NestingDepthExceeded);
|
||||||
|
}
|
||||||
|
// A valid tree stores each record once, in its own bytes, so it cannot
|
||||||
|
// hold more records than the file has room for. Children are addresses,
|
||||||
|
// though, and nothing makes them distinct: levels whose children all
|
||||||
|
// point at one shared node below reach it fan-out^depth times, which is
|
||||||
|
// millions of records from a few kilobytes. Counting against what the
|
||||||
|
// file could physically contain bounds that without trusting the
|
||||||
|
// header's own `total_records`.
|
||||||
|
let mut budget = len_usize(file) / usize::from(header.record_size.max(1));
|
||||||
|
|
||||||
let max_leaf_nrec = max_records_leaf(header.node_size, header.record_size);
|
let max_leaf_nrec = max_records_leaf(header.node_size, header.record_size);
|
||||||
|
|
||||||
if header.depth == 0 {
|
if header.depth == 0 {
|
||||||
// Root is a leaf
|
// Root is a leaf
|
||||||
parse_leaf_records(
|
parse_leaf_records(
|
||||||
file_data,
|
file,
|
||||||
header.root_node_address as usize,
|
to_usize(header.root_node_address)?,
|
||||||
header.num_records_in_root,
|
header.num_records_in_root,
|
||||||
header.record_size,
|
header.record_size,
|
||||||
|
header.node_size,
|
||||||
)
|
)
|
||||||
} else {
|
} else {
|
||||||
// Root is internal; traverse recursively
|
// Root is internal; traverse recursively
|
||||||
let mut records = Vec::new();
|
let mut records = Vec::new();
|
||||||
collect_internal_records(
|
collect_internal_records(
|
||||||
file_data,
|
file,
|
||||||
header.root_node_address as usize,
|
to_usize(header.root_node_address)?,
|
||||||
header.num_records_in_root,
|
header.num_records_in_root,
|
||||||
header.depth,
|
header.depth,
|
||||||
header.record_size,
|
header.record_size,
|
||||||
@@ -206,42 +262,79 @@ pub fn collect_btree_v2_records(
|
|||||||
offset_size,
|
offset_size,
|
||||||
length_size,
|
length_size,
|
||||||
max_leaf_nrec,
|
max_leaf_nrec,
|
||||||
|
&mut budget,
|
||||||
&mut records,
|
&mut records,
|
||||||
)?;
|
)?;
|
||||||
Ok(records)
|
Ok(records)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// A node's bytes: `want` bytes at `offset` (fewer only at the end of the
|
||||||
|
/// file), after checking its 4-byte signature. A node is read in one piece
|
||||||
|
/// when it fits in `node_size` (every valid node does); a larger claimed
|
||||||
|
/// extent — record counts from a damaged parent — is first checked against
|
||||||
|
/// the end of the file, so it costs a read only of bytes the file has.
|
||||||
|
/// Bounds errors are the whole-file ones: the signature check needs the
|
||||||
|
/// first 6 bytes, then `checks` — `(position, length)` pairs relative to
|
||||||
|
/// the node, in the order the parser checks them — must lie in the file.
|
||||||
|
fn read_node<'a, S: Storage + ?Sized>(
|
||||||
|
file: &'a S,
|
||||||
|
offset: usize,
|
||||||
|
want: usize,
|
||||||
|
node_size: u32,
|
||||||
|
signature: &[u8; 4],
|
||||||
|
checks: &[(usize, usize)],
|
||||||
|
) -> Result<Window<'a>, FormatError> {
|
||||||
|
let one_read = usize::try_from(node_size).unwrap_or(usize::MAX).max(6);
|
||||||
|
let w = Window::read(file, offset as u64, want.min(one_read))?;
|
||||||
|
w.ensure(0, 6)?;
|
||||||
|
if &w.bytes[..4] != signature {
|
||||||
|
return Err(FormatError::InvalidBTreeV2Signature);
|
||||||
|
}
|
||||||
|
if want <= one_read {
|
||||||
|
return Ok(w);
|
||||||
|
}
|
||||||
|
for &(rel, len) in checks {
|
||||||
|
Window::check_extent(file, offset as u64, rel, len)?;
|
||||||
|
}
|
||||||
|
Window::read(file, offset as u64, want)
|
||||||
|
}
|
||||||
|
|
||||||
/// Parse records from a leaf node (signature "BTLF").
|
/// Parse records from a leaf node (signature "BTLF").
|
||||||
fn parse_leaf_records(
|
fn parse_leaf_records<S: Storage + ?Sized>(
|
||||||
file_data: &[u8],
|
file: &S,
|
||||||
offset: usize,
|
offset: usize,
|
||||||
num_records: u16,
|
num_records: u16,
|
||||||
record_size: u16,
|
record_size: u16,
|
||||||
|
node_size: u32,
|
||||||
) -> Result<Vec<BTreeV2Record>, FormatError> {
|
) -> Result<Vec<BTreeV2Record>, FormatError> {
|
||||||
// signature(4) + version(1) + type(1) = 6 bytes header
|
// signature(4) + version(1) + type(1) = 6 bytes header
|
||||||
ensure_len(file_data, offset, 6)?;
|
let pos = 6;
|
||||||
if &file_data[offset..offset + 4] != b"BTLF" {
|
|
||||||
return Err(FormatError::InvalidBTreeV2Signature);
|
|
||||||
}
|
|
||||||
|
|
||||||
let pos = offset + 6;
|
|
||||||
let rs = record_size as usize;
|
let rs = record_size as usize;
|
||||||
let total = (num_records as usize)
|
let total = (num_records as usize)
|
||||||
.checked_mul(rs)
|
.checked_mul(rs)
|
||||||
.ok_or(FormatError::UnexpectedEof {
|
.ok_or(FormatError::UnexpectedEof {
|
||||||
expected: usize::MAX,
|
expected: usize::MAX,
|
||||||
available: file_data.len(),
|
available: len_usize(file),
|
||||||
})?;
|
})?;
|
||||||
ensure_len(file_data, pos, total)?;
|
let w = read_node(
|
||||||
|
file,
|
||||||
|
offset,
|
||||||
|
pos + total + 4,
|
||||||
|
node_size,
|
||||||
|
b"BTLF",
|
||||||
|
&[(pos, total)],
|
||||||
|
)?;
|
||||||
|
let d = &w.bytes;
|
||||||
|
w.ensure(pos, total)?;
|
||||||
|
|
||||||
// Validate checksum: 4 bytes after records + padding
|
// Validate checksum: 4 bytes after records + padding
|
||||||
#[cfg(feature = "checksum")]
|
#[cfg(feature = "checksum")]
|
||||||
{
|
{
|
||||||
let checksum_pos = pos + total;
|
let checksum_pos = pos + total;
|
||||||
if file_data.len() >= checksum_pos + 4 {
|
if d.len() >= checksum_pos + 4 {
|
||||||
let stored = LittleEndian::read_u32(&file_data[checksum_pos..checksum_pos + 4]);
|
let stored = LittleEndian::read_u32(&d[checksum_pos..checksum_pos + 4]);
|
||||||
let computed = crate::checksum::jenkins_lookup3(&file_data[offset..checksum_pos]);
|
let computed = crate::checksum::jenkins_lookup3(&d[..checksum_pos]);
|
||||||
if computed != stored {
|
if computed != stored {
|
||||||
return Err(FormatError::ChecksumMismatch {
|
return Err(FormatError::ChecksumMismatch {
|
||||||
expected: stored,
|
expected: stored,
|
||||||
@@ -255,16 +348,135 @@ fn parse_leaf_records(
|
|||||||
for i in 0..num_records as usize {
|
for i in 0..num_records as usize {
|
||||||
let start = pos + i * rs;
|
let start = pos + i * rs;
|
||||||
records.push(BTreeV2Record {
|
records.push(BTreeV2Record {
|
||||||
data: file_data[start..start + rs].to_vec(),
|
data: d[start..start + rs].to_vec(),
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
Ok(records)
|
Ok(records)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// An internal node read from the file: its bytes (from the signature on),
|
||||||
|
/// where its records start, and its children as `(address, record count)`.
|
||||||
|
struct InternalNode<'a> {
|
||||||
|
node: Window<'a>,
|
||||||
|
records_start: usize,
|
||||||
|
children: Vec<(u64, u16)>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl InternalNode<'_> {
|
||||||
|
/// Record `i`, `rs` bytes long.
|
||||||
|
fn record(&self, i: usize, rs: usize) -> Result<&[u8], FormatError> {
|
||||||
|
let overflow = || FormatError::UnexpectedEof {
|
||||||
|
expected: usize::MAX,
|
||||||
|
available: usize::MAX,
|
||||||
|
};
|
||||||
|
let rec_start = i
|
||||||
|
.checked_mul(rs)
|
||||||
|
.and_then(|o| self.records_start.checked_add(o))
|
||||||
|
.ok_or_else(overflow)?;
|
||||||
|
self.node.ensure(rec_start, rs)?;
|
||||||
|
Ok(&self.node.bytes[rec_start..rec_start + rs])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// An internal node's layout: where its records start, and its children as
|
||||||
|
/// `(address, record count)`.
|
||||||
|
#[allow(clippy::too_many_arguments)]
|
||||||
|
fn read_internal_node<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
offset: usize,
|
||||||
|
num_records: u16,
|
||||||
|
depth: u16,
|
||||||
|
record_size: u16,
|
||||||
|
node_size: u32,
|
||||||
|
offset_size: u8,
|
||||||
|
max_leaf_nrec: u64,
|
||||||
|
) -> Result<InternalNode<'_>, FormatError> {
|
||||||
|
let nr = num_records as usize;
|
||||||
|
let rs = record_size as usize;
|
||||||
|
|
||||||
|
// Records first
|
||||||
|
let records_total = nr.checked_mul(rs).ok_or(FormatError::UnexpectedEof {
|
||||||
|
expected: usize::MAX,
|
||||||
|
available: len_usize(file),
|
||||||
|
})?;
|
||||||
|
|
||||||
|
// Child pointer layout, as libhdf5 computes it (H5B2__hdr_init): the
|
||||||
|
// child's record count is always encoded in the width needed for a
|
||||||
|
// *leaf's* maximum, and — below the first internal level — the child
|
||||||
|
// subtree's total record count in the width needed for the most records
|
||||||
|
// a subtree of that depth can hold.
|
||||||
|
let child_depth = depth - 1;
|
||||||
|
let nrec_width = bytes_for_max_records(max_leaf_nrec);
|
||||||
|
let total_nrec_width = if depth > 1 {
|
||||||
|
bytes_for_max_records(cum_max_records(
|
||||||
|
node_size,
|
||||||
|
record_size,
|
||||||
|
offset_size,
|
||||||
|
max_leaf_nrec,
|
||||||
|
child_depth,
|
||||||
|
))
|
||||||
|
} else {
|
||||||
|
0
|
||||||
|
};
|
||||||
|
|
||||||
|
let num_children = nr + 1;
|
||||||
|
let child_ptr_size = offset_size as usize + nrec_width + total_nrec_width;
|
||||||
|
let pointers = num_children * child_ptr_size;
|
||||||
|
|
||||||
|
// signature(4) + version(1) + type(1) = 6, records, pointers, checksum.
|
||||||
|
let w = read_node(
|
||||||
|
file,
|
||||||
|
offset,
|
||||||
|
6 + records_total + pointers + 4,
|
||||||
|
node_size,
|
||||||
|
b"BTIN",
|
||||||
|
&[(6, records_total), (6 + records_total, pointers)],
|
||||||
|
)?;
|
||||||
|
let d = &w.bytes;
|
||||||
|
let mut pos = 6;
|
||||||
|
w.ensure(pos, records_total)?;
|
||||||
|
let records_start = pos;
|
||||||
|
pos += records_total;
|
||||||
|
|
||||||
|
w.ensure(pos, pointers)?;
|
||||||
|
|
||||||
|
let mut children = Vec::with_capacity(num_children);
|
||||||
|
for _ in 0..num_children {
|
||||||
|
let addr = read_offset(d, pos, offset_size)?;
|
||||||
|
pos += offset_size as usize;
|
||||||
|
let child_nrec = read_var_uint(d, pos, nrec_width)? as u16;
|
||||||
|
pos += nrec_width;
|
||||||
|
pos += total_nrec_width; // skip total records in subtree
|
||||||
|
children.push((addr, child_nrec));
|
||||||
|
}
|
||||||
|
|
||||||
|
// The checksum follows the child pointers and covers the node up to it.
|
||||||
|
// Lookups prune children by the keys in this node, so an unverified
|
||||||
|
// internal node could hide a record without any error: libhdf5 refuses
|
||||||
|
// a mismatch here, and so does this.
|
||||||
|
#[cfg(feature = "checksum")]
|
||||||
|
{
|
||||||
|
w.ensure(pos, 4)?;
|
||||||
|
let stored = LittleEndian::read_u32(&d[pos..pos + 4]);
|
||||||
|
let computed = crate::checksum::jenkins_lookup3(&d[..pos]);
|
||||||
|
if computed != stored {
|
||||||
|
return Err(FormatError::ChecksumMismatch {
|
||||||
|
expected: stored,
|
||||||
|
computed,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(InternalNode {
|
||||||
|
node: w,
|
||||||
|
records_start,
|
||||||
|
children,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
/// Recursively collect records from an internal node.
|
/// Recursively collect records from an internal node.
|
||||||
#[allow(clippy::too_many_arguments, clippy::only_used_in_recursion)]
|
#[allow(clippy::too_many_arguments, clippy::only_used_in_recursion)]
|
||||||
fn collect_internal_records(
|
fn collect_internal_records<S: Storage + ?Sized>(
|
||||||
file_data: &[u8],
|
file: &S,
|
||||||
offset: usize,
|
offset: usize,
|
||||||
num_records: u16,
|
num_records: u16,
|
||||||
depth: u16,
|
depth: u16,
|
||||||
@@ -273,90 +485,41 @@ fn collect_internal_records(
|
|||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
max_leaf_nrec: u64,
|
max_leaf_nrec: u64,
|
||||||
|
budget: &mut usize,
|
||||||
out: &mut Vec<BTreeV2Record>,
|
out: &mut Vec<BTreeV2Record>,
|
||||||
) -> Result<(), FormatError> {
|
) -> Result<(), FormatError> {
|
||||||
// signature(4) + version(1) + type(1) = 6
|
|
||||||
ensure_len(file_data, offset, 6)?;
|
|
||||||
if &file_data[offset..offset + 4] != b"BTIN" {
|
|
||||||
return Err(FormatError::InvalidBTreeV2Signature);
|
|
||||||
}
|
|
||||||
|
|
||||||
let nr = num_records as usize;
|
let nr = num_records as usize;
|
||||||
let rs = record_size as usize;
|
let rs = record_size as usize;
|
||||||
let mut pos = offset + 6;
|
let node = read_internal_node(
|
||||||
|
file,
|
||||||
// Read all records first
|
offset,
|
||||||
let records_total = nr.checked_mul(rs).ok_or(FormatError::UnexpectedEof {
|
num_records,
|
||||||
expected: usize::MAX,
|
depth,
|
||||||
available: file_data.len(),
|
record_size,
|
||||||
})?;
|
node_size,
|
||||||
ensure_len(file_data, pos, records_total)?;
|
offset_size,
|
||||||
let records_start = pos;
|
max_leaf_nrec,
|
||||||
pos += records_total;
|
)?;
|
||||||
|
|
||||||
// Compute sizes for child pointers
|
|
||||||
// max_records at child depth - for variable-width nrec encoding
|
|
||||||
let child_depth = depth - 1;
|
let child_depth = depth - 1;
|
||||||
let max_nrec_child = if child_depth == 0 {
|
|
||||||
max_leaf_nrec
|
|
||||||
} else {
|
|
||||||
// For internal nodes at child_depth, the true max_nrec depends on the
|
|
||||||
// node size, record size, and the recursive width of child pointer
|
|
||||||
// entries (which themselves depend on max_nrec at deeper levels).
|
|
||||||
// Computing the exact value requires iterating from the leaf level
|
|
||||||
// upward, as described in the HDF5 spec (III.A.2 "Computing the Size
|
|
||||||
// of B-tree Nodes").
|
|
||||||
//
|
|
||||||
// We use `max_leaf_nrec * 2` as a conservative upper bound. This
|
|
||||||
// over-estimates the nrec encoding width, which means we may read
|
|
||||||
// slightly more bytes per child pointer than strictly necessary, but
|
|
||||||
// never fewer. The over-read bytes are harmless because we only
|
|
||||||
// decode `num_records` entries (the actual count from the node header).
|
|
||||||
//
|
|
||||||
// Known limitation: for very deep trees (depth > 3) with small record
|
|
||||||
// sizes, the true max could exceed this estimate, causing us to
|
|
||||||
// under-allocate the nrec encoding width and misparse child pointers.
|
|
||||||
// In practice, HDF5 B-tree v2 depths rarely exceed 2-3.
|
|
||||||
max_leaf_nrec * 2
|
|
||||||
};
|
|
||||||
let nrec_width = bytes_for_max_records(max_nrec_child);
|
|
||||||
|
|
||||||
// Total records in subtree width (only if depth > 1)
|
|
||||||
let total_nrec_width = if depth > 1 {
|
|
||||||
// Width to hold total records in a subtree
|
|
||||||
// We compute max possible total records at this subtree depth
|
|
||||||
let max_total = header_max_total_records(max_leaf_nrec, depth - 1);
|
|
||||||
bytes_for_max_records(max_total)
|
|
||||||
} else {
|
|
||||||
0
|
|
||||||
};
|
|
||||||
|
|
||||||
let num_children = nr + 1;
|
|
||||||
let child_ptr_size = offset_size as usize + nrec_width + total_nrec_width;
|
|
||||||
ensure_len(file_data, pos, num_children * child_ptr_size)?;
|
|
||||||
|
|
||||||
// Read child pointers
|
|
||||||
let mut children = Vec::with_capacity(num_children);
|
|
||||||
for _ in 0..num_children {
|
|
||||||
let addr = read_offset(file_data, pos, offset_size)?;
|
|
||||||
pos += offset_size as usize;
|
|
||||||
let child_nrec = read_var_uint(file_data, pos, nrec_width)? as u16;
|
|
||||||
pos += nrec_width;
|
|
||||||
pos += total_nrec_width; // skip total records in subtree
|
|
||||||
children.push((addr, child_nrec));
|
|
||||||
}
|
|
||||||
|
|
||||||
// Interleave: child[0], record[0], child[1], record[1], ..., child[nr]
|
// Interleave: child[0], record[0], child[1], record[1], ..., child[nr]
|
||||||
// We collect child[0] records, then record[0], then child[1], etc.
|
// We collect child[0] records, then record[0], then child[1], etc.
|
||||||
for (i, &(child_addr, child_nrec)) in children.iter().enumerate() {
|
for (i, &(child_addr, child_nrec)) in node.children.iter().enumerate() {
|
||||||
if child_depth == 0 {
|
if child_depth == 0 {
|
||||||
let leaf_recs =
|
// Before parsing, so a refused tree is not also a large allocation.
|
||||||
parse_leaf_records(file_data, child_addr as usize, child_nrec, record_size)?;
|
spend(budget, usize::from(child_nrec))?;
|
||||||
|
let leaf_recs = parse_leaf_records(
|
||||||
|
file,
|
||||||
|
to_usize(child_addr)?,
|
||||||
|
child_nrec,
|
||||||
|
record_size,
|
||||||
|
node_size,
|
||||||
|
)?;
|
||||||
out.extend(leaf_recs);
|
out.extend(leaf_recs);
|
||||||
} else {
|
} else {
|
||||||
collect_internal_records(
|
collect_internal_records(
|
||||||
file_data,
|
file,
|
||||||
child_addr as usize,
|
to_usize(child_addr)?,
|
||||||
child_nrec,
|
child_nrec,
|
||||||
child_depth,
|
child_depth,
|
||||||
record_size,
|
record_size,
|
||||||
@@ -364,37 +527,17 @@ fn collect_internal_records(
|
|||||||
offset_size,
|
offset_size,
|
||||||
length_size,
|
length_size,
|
||||||
max_leaf_nrec,
|
max_leaf_nrec,
|
||||||
|
budget,
|
||||||
out,
|
out,
|
||||||
)?;
|
)?;
|
||||||
}
|
}
|
||||||
|
|
||||||
// Add record[i] (except after the last child)
|
// Add record[i] (except after the last child)
|
||||||
if i < nr {
|
if i < nr {
|
||||||
let rec_offset = i.checked_mul(rs).ok_or(FormatError::UnexpectedEof {
|
let data = node.record(i, rs)?;
|
||||||
expected: usize::MAX,
|
spend(budget, 1)?;
|
||||||
available: file_data.len(),
|
|
||||||
})?;
|
|
||||||
let rec_start =
|
|
||||||
records_start
|
|
||||||
.checked_add(rec_offset)
|
|
||||||
.ok_or(FormatError::UnexpectedEof {
|
|
||||||
expected: usize::MAX,
|
|
||||||
available: file_data.len(),
|
|
||||||
})?;
|
|
||||||
let rec_end = rec_start
|
|
||||||
.checked_add(rs)
|
|
||||||
.ok_or(FormatError::UnexpectedEof {
|
|
||||||
expected: usize::MAX,
|
|
||||||
available: file_data.len(),
|
|
||||||
})?;
|
|
||||||
if rec_end > file_data.len() {
|
|
||||||
return Err(FormatError::UnexpectedEof {
|
|
||||||
expected: rec_end,
|
|
||||||
available: file_data.len(),
|
|
||||||
});
|
|
||||||
}
|
|
||||||
out.push(BTreeV2Record {
|
out.push(BTreeV2Record {
|
||||||
data: file_data[rec_start..rec_end].to_vec(),
|
data: data.to_vec(),
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -402,14 +545,218 @@ fn collect_internal_records(
|
|||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Estimate maximum total records at a given depth (for variable-width encoding).
|
/// The records of a B-tree v2 that fall in one key range, found by
|
||||||
fn header_max_total_records(max_leaf_nrec: u64, depth: u16) -> u64 {
|
/// descending the tree instead of reading all of it.
|
||||||
// Conservative: branching factor * max_leaf at each level
|
///
|
||||||
let mut total = max_leaf_nrec;
|
/// `cmp` places a record relative to the range: `Less` if the record sorts
|
||||||
for _ in 0..depth {
|
/// before it, `Greater` if after, `Equal` if the record is in it. The tree
|
||||||
total = total.saturating_mul(max_leaf_nrec.max(2));
|
/// must be ordered consistently with `cmp`, as libhdf5 orders it (a link or
|
||||||
|
/// attribute name index by name hash, so all records with one hash form a
|
||||||
|
/// range whatever order their names are in). Only the nodes whose key
|
||||||
|
/// interval overlaps the range are read: O(depth) nodes plus those holding
|
||||||
|
/// the matches. Matches come in tree order.
|
||||||
|
pub fn find_btree_v2_records(
|
||||||
|
file_data: &[u8],
|
||||||
|
header: &BTreeV2Header,
|
||||||
|
offset_size: u8,
|
||||||
|
cmp: &mut dyn FnMut(&[u8]) -> Ordering,
|
||||||
|
) -> Result<Vec<BTreeV2Record>, FormatError> {
|
||||||
|
find_btree_v2_records_in(file_data, header, offset_size, cmp)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`find_btree_v2_records`] over any [`Storage`]: one bounded read per
|
||||||
|
/// node visited.
|
||||||
|
pub fn find_btree_v2_records_in<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
header: &BTreeV2Header,
|
||||||
|
offset_size: u8,
|
||||||
|
cmp: &mut dyn FnMut(&[u8]) -> Ordering,
|
||||||
|
) -> Result<Vec<BTreeV2Record>, FormatError> {
|
||||||
|
if header.total_records == 0 || header.num_records_in_root == 0 {
|
||||||
|
return Ok(Vec::new());
|
||||||
}
|
}
|
||||||
total
|
if header.depth > MAX_DEPTH {
|
||||||
|
return Err(FormatError::NestingDepthExceeded);
|
||||||
|
}
|
||||||
|
// As in `collect_btree_v2_records`: a valid tree cannot hold more
|
||||||
|
// records than the file has room for, however its children are shared.
|
||||||
|
let mut budget = len_usize(file) / usize::from(header.record_size.max(1));
|
||||||
|
let max_leaf_nrec = max_records_leaf(header.node_size, header.record_size);
|
||||||
|
let mut out = Vec::new();
|
||||||
|
find_in_node(
|
||||||
|
file,
|
||||||
|
header,
|
||||||
|
to_usize(header.root_node_address)?,
|
||||||
|
header.num_records_in_root,
|
||||||
|
header.depth,
|
||||||
|
offset_size,
|
||||||
|
max_leaf_nrec,
|
||||||
|
cmp,
|
||||||
|
&mut budget,
|
||||||
|
&mut out,
|
||||||
|
)?;
|
||||||
|
Ok(out)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[allow(clippy::too_many_arguments)]
|
||||||
|
fn find_in_node<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
header: &BTreeV2Header,
|
||||||
|
offset: usize,
|
||||||
|
num_records: u16,
|
||||||
|
depth: u16,
|
||||||
|
offset_size: u8,
|
||||||
|
max_leaf_nrec: u64,
|
||||||
|
cmp: &mut dyn FnMut(&[u8]) -> Ordering,
|
||||||
|
budget: &mut usize,
|
||||||
|
out: &mut Vec<BTreeV2Record>,
|
||||||
|
) -> Result<(), FormatError> {
|
||||||
|
spend(budget, usize::from(num_records))?;
|
||||||
|
if depth == 0 {
|
||||||
|
let records = parse_leaf_records(
|
||||||
|
file,
|
||||||
|
offset,
|
||||||
|
num_records,
|
||||||
|
header.record_size,
|
||||||
|
header.node_size,
|
||||||
|
)?;
|
||||||
|
out.extend(
|
||||||
|
records
|
||||||
|
.into_iter()
|
||||||
|
.filter(|r| cmp(&r.data) == Ordering::Equal),
|
||||||
|
);
|
||||||
|
return Ok(());
|
||||||
|
}
|
||||||
|
let rs = usize::from(header.record_size);
|
||||||
|
let node = read_internal_node(
|
||||||
|
file,
|
||||||
|
offset,
|
||||||
|
num_records,
|
||||||
|
depth,
|
||||||
|
header.record_size,
|
||||||
|
header.node_size,
|
||||||
|
offset_size,
|
||||||
|
max_leaf_nrec,
|
||||||
|
)?;
|
||||||
|
let nr = usize::from(num_records);
|
||||||
|
let mut order = Vec::with_capacity(nr);
|
||||||
|
for i in 0..nr {
|
||||||
|
order.push(cmp(node.record(i, rs)?));
|
||||||
|
}
|
||||||
|
// Child `i` holds the keys between record `i - 1` and record `i`: it can
|
||||||
|
// hold a match unless the record before it is already past the range or
|
||||||
|
// the record after it is still before it.
|
||||||
|
for (i, &(child_addr, child_nrec)) in node.children.iter().enumerate() {
|
||||||
|
let after_left = i == 0 || order[i - 1] != Ordering::Greater;
|
||||||
|
let before_right = i == nr || order[i] != Ordering::Less;
|
||||||
|
if after_left && before_right {
|
||||||
|
find_in_node(
|
||||||
|
file,
|
||||||
|
header,
|
||||||
|
to_usize(child_addr)?,
|
||||||
|
child_nrec,
|
||||||
|
depth - 1,
|
||||||
|
offset_size,
|
||||||
|
max_leaf_nrec,
|
||||||
|
cmp,
|
||||||
|
budget,
|
||||||
|
out,
|
||||||
|
)?;
|
||||||
|
}
|
||||||
|
if i < nr && order[i] == Ordering::Equal {
|
||||||
|
out.push(BTreeV2Record {
|
||||||
|
data: node.record(i, rs)?.to_vec(),
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Most records a subtree whose root is at `depth` can hold (libhdf5's
|
||||||
|
/// `cum_max_nrec`). See [`node_info`].
|
||||||
|
fn cum_max_records(
|
||||||
|
node_size: u32,
|
||||||
|
record_size: u16,
|
||||||
|
offset_size: u8,
|
||||||
|
max_leaf_nrec: u64,
|
||||||
|
depth: u16,
|
||||||
|
) -> u64 {
|
||||||
|
node_info_from_leaf(node_size, record_size, offset_size, max_leaf_nrec, depth)
|
||||||
|
.last()
|
||||||
|
.map_or(max_leaf_nrec, |n| n.cum_max_nrec)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Capacity of a B-tree v2 node at one depth, as libhdf5 computes it
|
||||||
|
/// (`H5B2__hdr_init`'s `node_info`).
|
||||||
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||||
|
pub(crate) struct NodeInfo {
|
||||||
|
/// Most records one node at this depth holds.
|
||||||
|
pub(crate) max_nrec: u64,
|
||||||
|
/// Most records a subtree rooted at this depth holds.
|
||||||
|
pub(crate) cum_max_nrec: u64,
|
||||||
|
/// Bytes a subtree's total record count takes in a pointer to a node
|
||||||
|
/// at this depth (0 for a leaf, whose count is its own).
|
||||||
|
pub(crate) cum_max_nrec_size: usize,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Node capacities for depths `0..=depth` (entry `d` for depth `d`): a leaf
|
||||||
|
/// holds `max_nrec(0)` records; an internal node at depth `d` holds
|
||||||
|
/// `max_nrec(d)` records and `max_nrec(d) + 1` subtrees of depth `d - 1`,
|
||||||
|
/// where `max_nrec(d)` is what fits in a node once each record is paired
|
||||||
|
/// with a child pointer of the width depth `d` needs (address, the child's
|
||||||
|
/// record count in the width a *leaf's* maximum needs, and below the first
|
||||||
|
/// internal level the child subtree's total in the width its maximum
|
||||||
|
/// needs), with one pointer more than records.
|
||||||
|
pub(crate) fn node_info(
|
||||||
|
node_size: u32,
|
||||||
|
record_size: u16,
|
||||||
|
offset_size: u8,
|
||||||
|
depth: u16,
|
||||||
|
) -> Vec<NodeInfo> {
|
||||||
|
let max_leaf = max_records_leaf(node_size, record_size);
|
||||||
|
node_info_from_leaf(node_size, record_size, offset_size, max_leaf, depth)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn node_info_from_leaf(
|
||||||
|
node_size: u32,
|
||||||
|
record_size: u16,
|
||||||
|
offset_size: u8,
|
||||||
|
max_leaf_nrec: u64,
|
||||||
|
depth: u16,
|
||||||
|
) -> Vec<NodeInfo> {
|
||||||
|
// Internal node overhead: signature(4) + version(1) + type(1) + checksum(4).
|
||||||
|
const PREFIX: u64 = 10;
|
||||||
|
let nrec_width = bytes_for_max_records(max_leaf_nrec) as u64;
|
||||||
|
let mut info = Vec::with_capacity(usize::from(depth) + 1);
|
||||||
|
info.push(NodeInfo {
|
||||||
|
max_nrec: max_leaf_nrec,
|
||||||
|
cum_max_nrec: max_leaf_nrec,
|
||||||
|
cum_max_nrec_size: 0,
|
||||||
|
});
|
||||||
|
for d in 1..=depth {
|
||||||
|
let below = info[usize::from(d) - 1];
|
||||||
|
let ptr = u64::from(offset_size)
|
||||||
|
+ nrec_width
|
||||||
|
+ if d > 1 {
|
||||||
|
below.cum_max_nrec_size as u64
|
||||||
|
} else {
|
||||||
|
0
|
||||||
|
};
|
||||||
|
let max_nrec = u64::from(node_size)
|
||||||
|
.saturating_sub(PREFIX)
|
||||||
|
.saturating_sub(ptr)
|
||||||
|
/ (u64::from(record_size) + ptr).max(1);
|
||||||
|
let cum = max_nrec
|
||||||
|
.saturating_add(1)
|
||||||
|
.saturating_mul(below.cum_max_nrec)
|
||||||
|
.saturating_add(max_nrec);
|
||||||
|
info.push(NodeInfo {
|
||||||
|
max_nrec,
|
||||||
|
cum_max_nrec: cum,
|
||||||
|
cum_max_nrec_size: bytes_for_max_records(cum),
|
||||||
|
});
|
||||||
|
}
|
||||||
|
info
|
||||||
}
|
}
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
@@ -466,6 +813,132 @@ mod tests {
|
|||||||
buf
|
buf
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// An internal node laid out exactly as `collect_internal_records` will
|
||||||
|
/// read it at `depth`: `records` zeroed records, then `children` pointers,
|
||||||
|
/// all to `child_addr` claiming `child_nrec` records.
|
||||||
|
fn internal_node(
|
||||||
|
depth: u16,
|
||||||
|
node_size: u32,
|
||||||
|
record_size: u16,
|
||||||
|
records: usize,
|
||||||
|
children: usize,
|
||||||
|
child_addr: u64,
|
||||||
|
child_nrec: u64,
|
||||||
|
) -> Vec<u8> {
|
||||||
|
let max_leaf = max_records_leaf(node_size, record_size);
|
||||||
|
let nrec_width = bytes_for_max_records(max_leaf);
|
||||||
|
let total_width = if depth > 1 {
|
||||||
|
bytes_for_max_records(cum_max_records(
|
||||||
|
node_size,
|
||||||
|
record_size,
|
||||||
|
8,
|
||||||
|
max_leaf,
|
||||||
|
depth - 1,
|
||||||
|
))
|
||||||
|
} else {
|
||||||
|
0
|
||||||
|
};
|
||||||
|
let mut buf = b"BTIN".to_vec();
|
||||||
|
buf.extend_from_slice(&[0, 5]);
|
||||||
|
buf.resize(buf.len() + records * record_size as usize, 0);
|
||||||
|
for _ in 0..children {
|
||||||
|
buf.extend_from_slice(&child_addr.to_le_bytes());
|
||||||
|
buf.extend_from_slice(&child_nrec.to_le_bytes()[..nrec_width]);
|
||||||
|
buf.resize(buf.len() + total_width, 0);
|
||||||
|
}
|
||||||
|
let sum = crate::checksum::jenkins_lookup3(&buf);
|
||||||
|
buf.extend_from_slice(&sum.to_le_bytes());
|
||||||
|
buf
|
||||||
|
}
|
||||||
|
|
||||||
|
fn header(depth: u16, root: u64, root_nrec: u16, total: u64) -> BTreeV2Header {
|
||||||
|
BTreeV2Header {
|
||||||
|
tree_type: 5,
|
||||||
|
node_size: 512,
|
||||||
|
record_size: 8,
|
||||||
|
depth,
|
||||||
|
root_node_address: root,
|
||||||
|
num_records_in_root: root_nrec,
|
||||||
|
total_records: total,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn a_node_that_is_its_own_child_is_rejected_not_recursed() {
|
||||||
|
// One internal node whose two children are itself, under a header
|
||||||
|
// claiming the deepest tree a u16 allows. The layout stops depending
|
||||||
|
// on depth once the subtree-total width saturates, so every level
|
||||||
|
// parses cleanly and recursion runs ~65 000 frames deep: before the
|
||||||
|
// cap this overflowed the stack and aborted the process, from a file
|
||||||
|
// of under 100 bytes.
|
||||||
|
let mut data = internal_node(u16::MAX, 512, 8, 1, 2, 0, 1);
|
||||||
|
data.resize(4096, 0);
|
||||||
|
let result = collect_btree_v2_records(&data, &header(u16::MAX, 0, 1, 1), 8, 8);
|
||||||
|
assert!(result.is_err(), "{result:?}");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn a_shared_subtree_cannot_multiply_the_work() {
|
||||||
|
// A chain of distinct levels, each node's children all pointing at the
|
||||||
|
// single node below, ending in a real leaf. Every node parses and
|
||||||
|
// nothing is cyclic, yet the leaf is reached fan-out^depth times: 62
|
||||||
|
// children over 4 levels is ~15 million leaf visits from a few
|
||||||
|
// kilobytes. A valid tree cannot hold more records than the file has
|
||||||
|
// room for, so that bounds the traversal instead.
|
||||||
|
let (node_size, record_size) = (512u32, 8u16);
|
||||||
|
let fanout = 62usize;
|
||||||
|
let depth = 4u16;
|
||||||
|
let leaf = build_leaf_node(5, &[&[0u8; 8][..]]);
|
||||||
|
|
||||||
|
// Lay out root first, then each lower level, then the leaf.
|
||||||
|
let mut nodes: Vec<Vec<u8>> = Vec::new();
|
||||||
|
let mut addrs = Vec::new();
|
||||||
|
let mut at = 0u64;
|
||||||
|
let mut sizes = Vec::new();
|
||||||
|
for d in (1..=depth).rev() {
|
||||||
|
let n = internal_node(d, node_size, record_size, fanout - 1, fanout, 0, 0);
|
||||||
|
sizes.push(n.len());
|
||||||
|
}
|
||||||
|
for size in &sizes {
|
||||||
|
addrs.push(at);
|
||||||
|
at += *size as u64;
|
||||||
|
}
|
||||||
|
let leaf_addr = at;
|
||||||
|
for (i, d) in (1..=depth).rev().enumerate() {
|
||||||
|
let (child, child_nrec) = if d == 1 {
|
||||||
|
(leaf_addr, 1)
|
||||||
|
} else {
|
||||||
|
(addrs[i + 1], fanout as u64 - 1)
|
||||||
|
};
|
||||||
|
nodes.push(internal_node(
|
||||||
|
d,
|
||||||
|
node_size,
|
||||||
|
record_size,
|
||||||
|
fanout - 1,
|
||||||
|
fanout,
|
||||||
|
child,
|
||||||
|
child_nrec,
|
||||||
|
));
|
||||||
|
}
|
||||||
|
let mut data: Vec<u8> = nodes.concat();
|
||||||
|
data.extend_from_slice(&leaf);
|
||||||
|
data.resize(data.len() + 64, 0);
|
||||||
|
|
||||||
|
let started = std::time::Instant::now();
|
||||||
|
let result =
|
||||||
|
collect_btree_v2_records(&data, &header(depth, 0, fanout as u16 - 1, u64::MAX), 8, 8);
|
||||||
|
assert!(
|
||||||
|
result.is_err(),
|
||||||
|
"expected a refusal, got {} records",
|
||||||
|
result.map_or(0, |r| r.len())
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
started.elapsed() < std::time::Duration::from_secs(2),
|
||||||
|
"took {:?}",
|
||||||
|
started.elapsed()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn parse_header() {
|
fn parse_header() {
|
||||||
let data = build_btree_v2_header(5, 512, 11, 0, 0x1000, 3, 3, 8, 8);
|
let data = build_btree_v2_header(5, 512, 11, 0, 0x1000, 3, 3, 8, 8);
|
||||||
@@ -522,4 +995,18 @@ mod tests {
|
|||||||
let records = collect_btree_v2_records(&header, &hdr, 8, 8).unwrap();
|
let records = collect_btree_v2_records(&header, &hdr, 8, 8).unwrap();
|
||||||
assert!(records.is_empty());
|
assert!(records.is_empty());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn subtree_capacity_matches_libhdf5() {
|
||||||
|
// A link-name index (11-byte records, 512-byte nodes, 8-byte
|
||||||
|
// addresses): libhdf5's H5B2__hdr_init gives 45 records per leaf,
|
||||||
|
// then cum_max_nrec 1 149 at depth 1 and 26 449 at depth 2 — two
|
||||||
|
// bytes of subtree count in a depth-3 root's child pointers, where
|
||||||
|
// leaf_max^3 = 91 125 would need three.
|
||||||
|
let leaf = max_records_leaf(512, 11);
|
||||||
|
assert_eq!(leaf, 45);
|
||||||
|
assert_eq!(cum_max_records(512, 11, 8, leaf, 0), 45);
|
||||||
|
assert_eq!(cum_max_records(512, 11, 8, leaf, 1), 1_149);
|
||||||
|
assert_eq!(cum_max_records(512, 11, 8, leaf, 2), 26_449);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,503 @@
|
|||||||
|
//! Writing version-2 B-trees: a header (`BTHD`) and its nodes, leaves
|
||||||
|
//! (`BTLF`) and, for more records than one leaf holds, internal nodes
|
||||||
|
//! (`BTIN`) to any depth.
|
||||||
|
//!
|
||||||
|
//! Node capacities come from [`crate::btree_v2::node_info`], the arithmetic
|
||||||
|
//! libhdf5 uses (`H5B2__hdr_init`) and the reader decodes pointers with, so
|
||||||
|
//! the pointer widths the writer encodes are the ones every reader expects.
|
||||||
|
|
||||||
|
use crate::addr::saturating_usize;
|
||||||
|
#[cfg(not(feature = "std"))]
|
||||||
|
use alloc::{format, vec, vec::Vec};
|
||||||
|
|
||||||
|
use crate::btree_v2::{NodeInfo, bytes_for_max_records, node_info};
|
||||||
|
use crate::checksum::jenkins_lookup3;
|
||||||
|
use crate::error::FormatError;
|
||||||
|
|
||||||
|
/// How a B-tree is laid out: its record type and node geometry, as the
|
||||||
|
/// header records them.
|
||||||
|
#[derive(Debug, Clone, Copy)]
|
||||||
|
pub(crate) struct BTreeV2Params {
|
||||||
|
/// Record type (5: link names, 6: link creation order, 8: attribute
|
||||||
|
/// names, 9: attribute creation order, 10/11: chunks).
|
||||||
|
pub(crate) tree_type: u8,
|
||||||
|
/// Bytes per node.
|
||||||
|
pub(crate) node_size: u32,
|
||||||
|
/// Bytes per record.
|
||||||
|
pub(crate) record_size: u16,
|
||||||
|
/// Split and merge percentages. The writer fills nodes itself; these
|
||||||
|
/// only tell libhdf5 when to split and merge as it modifies the tree.
|
||||||
|
pub(crate) split_percent: u8,
|
||||||
|
pub(crate) merge_percent: u8,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Size of a B-tree v2 header.
|
||||||
|
pub(crate) fn header_size(offset_size: u8, length_size: u8) -> usize {
|
||||||
|
4 + 1 + 1 + 4 + 2 + 2 + 1 + 1 + offset_size as usize + 2 + length_size as usize + 4
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Deepest tree the writer builds. Even at the smallest fan-out libhdf5's
|
||||||
|
/// arithmetic allows, a few levels hold more records than any file could.
|
||||||
|
const MAX_WRITE_DEPTH: u16 = 32;
|
||||||
|
|
||||||
|
/// Write a B-tree v2 holding `records` (`record_size` bytes each,
|
||||||
|
/// concatenated, already in the tree's key order) at `addr`: the header,
|
||||||
|
/// then its nodes, each `node_size` bytes. No records gives a header with
|
||||||
|
/// an undefined root.
|
||||||
|
///
|
||||||
|
/// The tree is as shallow as the node size allows: a single leaf when the
|
||||||
|
/// records fit one, otherwise internal nodes above leaves. Records are
|
||||||
|
/// spread evenly over each node's children, so every node but the root is
|
||||||
|
/// at least about half full (above libhdf5's merge threshold, which is below
|
||||||
|
/// half), and each node holds at most its depth's maximum.
|
||||||
|
pub(crate) fn build_btree_v2(
|
||||||
|
p: BTreeV2Params,
|
||||||
|
records: &[u8],
|
||||||
|
addr: u64,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<Vec<u8>, FormatError> {
|
||||||
|
let rs = usize::from(p.record_size);
|
||||||
|
if rs == 0 || !records.len().is_multiple_of(rs) {
|
||||||
|
return Err(FormatError::SerializationError(format!(
|
||||||
|
"B-tree v2 records are {} bytes, not a multiple of the record size {rs}",
|
||||||
|
records.len()
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
let n = (records.len() / rs) as u64;
|
||||||
|
let hdr_len = header_size(offset_size, length_size);
|
||||||
|
|
||||||
|
// The shallowest depth whose subtree can hold every record.
|
||||||
|
let mut info = node_info(p.node_size, p.record_size, offset_size, 0);
|
||||||
|
let max_leaf = info[0].max_nrec;
|
||||||
|
if max_leaf == 0 || max_leaf > u64::from(u16::MAX) {
|
||||||
|
return Err(FormatError::SerializationError(format!(
|
||||||
|
"a {}-byte B-tree v2 node holds {max_leaf} {}-byte records; \
|
||||||
|
a node holds 1 to 65535",
|
||||||
|
p.node_size, p.record_size
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
let mut depth = 0u16;
|
||||||
|
while info[usize::from(depth)].cum_max_nrec < n {
|
||||||
|
depth += 1;
|
||||||
|
if depth > MAX_WRITE_DEPTH {
|
||||||
|
return Err(FormatError::SerializationError(format!(
|
||||||
|
"{n} records do not fit a B-tree v2 of {}-byte nodes",
|
||||||
|
p.node_size
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
info = node_info(p.node_size, p.record_size, offset_size, depth);
|
||||||
|
let max = info[usize::from(depth)].max_nrec;
|
||||||
|
if max == 0 || max > u64::from(u16::MAX) {
|
||||||
|
return Err(FormatError::SerializationError(format!(
|
||||||
|
"a {}-byte B-tree v2 internal node holds {max} records; \
|
||||||
|
a node holds 1 to 65535",
|
||||||
|
p.node_size
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
let mut w = TreeWriter {
|
||||||
|
p,
|
||||||
|
records,
|
||||||
|
info: &info,
|
||||||
|
nrec_width: bytes_for_max_records(max_leaf),
|
||||||
|
offset_size,
|
||||||
|
first_node: addr + hdr_len as u64,
|
||||||
|
nodes: Vec::new(),
|
||||||
|
};
|
||||||
|
let root = (n > 0)
|
||||||
|
.then(|| w.node(depth, 0, saturating_usize(n)))
|
||||||
|
.transpose()?;
|
||||||
|
|
||||||
|
let mut out = Vec::with_capacity(hdr_len + w.nodes.len() * p.node_size as usize);
|
||||||
|
out.extend_from_slice(b"BTHD");
|
||||||
|
out.push(0); // version
|
||||||
|
out.push(p.tree_type);
|
||||||
|
out.extend_from_slice(&p.node_size.to_le_bytes());
|
||||||
|
out.extend_from_slice(&p.record_size.to_le_bytes());
|
||||||
|
out.extend_from_slice(&depth.to_le_bytes());
|
||||||
|
out.push(p.split_percent);
|
||||||
|
out.push(p.merge_percent);
|
||||||
|
match root {
|
||||||
|
Some(r) => push_uint(&mut out, r.addr, offset_size as usize),
|
||||||
|
None => out.extend(core::iter::repeat_n(0xFF, offset_size as usize)),
|
||||||
|
}
|
||||||
|
let root_nrec = root.map_or(0, |r| r.nrec);
|
||||||
|
out.extend_from_slice(&(root_nrec as u16).to_le_bytes());
|
||||||
|
push_uint(&mut out, n, length_size as usize);
|
||||||
|
let sum = jenkins_lookup3(&out);
|
||||||
|
out.extend_from_slice(&sum.to_le_bytes());
|
||||||
|
debug_assert_eq!(out.len(), hdr_len);
|
||||||
|
for node in &w.nodes {
|
||||||
|
out.extend_from_slice(node);
|
||||||
|
}
|
||||||
|
Ok(out)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A written node, as its parent points at it.
|
||||||
|
#[derive(Debug, Clone, Copy)]
|
||||||
|
struct NodeRef {
|
||||||
|
addr: u64,
|
||||||
|
/// Records in the node itself.
|
||||||
|
nrec: u64,
|
||||||
|
/// Records in the subtree it roots.
|
||||||
|
all_nrec: u64,
|
||||||
|
}
|
||||||
|
|
||||||
|
struct TreeWriter<'a> {
|
||||||
|
p: BTreeV2Params,
|
||||||
|
records: &'a [u8],
|
||||||
|
info: &'a [NodeInfo],
|
||||||
|
/// Width of a child's record count: what a leaf's maximum needs.
|
||||||
|
nrec_width: usize,
|
||||||
|
offset_size: u8,
|
||||||
|
/// Address of the first node (right after the header).
|
||||||
|
first_node: u64,
|
||||||
|
/// Nodes in file order (children before their parent).
|
||||||
|
nodes: Vec<Vec<u8>>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl TreeWriter<'_> {
|
||||||
|
fn record(&self, i: usize) -> &[u8] {
|
||||||
|
let rs = usize::from(self.p.record_size);
|
||||||
|
&self.records[i * rs..(i + 1) * rs]
|
||||||
|
}
|
||||||
|
|
||||||
|
fn push_node(&mut self, mut node: Vec<u8>) -> u64 {
|
||||||
|
// The checksum covers the node up to it, not the padding after.
|
||||||
|
let sum = jenkins_lookup3(&node);
|
||||||
|
node.extend_from_slice(&sum.to_le_bytes());
|
||||||
|
debug_assert!(node.len() <= self.p.node_size as usize);
|
||||||
|
node.resize(self.p.node_size as usize, 0);
|
||||||
|
let addr = self.first_node + self.nodes.len() as u64 * u64::from(self.p.node_size);
|
||||||
|
self.nodes.push(node);
|
||||||
|
addr
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write the subtree of `depth` holding records `first..first + n`.
|
||||||
|
fn node(&mut self, depth: u16, first: usize, n: usize) -> Result<NodeRef, FormatError> {
|
||||||
|
let rs = usize::from(self.p.record_size);
|
||||||
|
let mut node = Vec::with_capacity(self.p.node_size as usize);
|
||||||
|
if depth == 0 {
|
||||||
|
debug_assert!(n as u64 <= self.info[0].max_nrec);
|
||||||
|
node.extend_from_slice(b"BTLF");
|
||||||
|
node.push(0); // version
|
||||||
|
node.push(self.p.tree_type);
|
||||||
|
node.extend_from_slice(&self.records[first * rs..(first + n) * rs]);
|
||||||
|
let addr = self.push_node(node);
|
||||||
|
return Ok(NodeRef {
|
||||||
|
addr,
|
||||||
|
nrec: n as u64,
|
||||||
|
all_nrec: n as u64,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
// As few children as hold the records, at least two, with the
|
||||||
|
// records spread evenly: `k` children and `k - 1` records between
|
||||||
|
// them.
|
||||||
|
let below = self.info[usize::from(depth) - 1].cum_max_nrec;
|
||||||
|
let k = (n as u64 + 1).div_ceil(below + 1).max(2);
|
||||||
|
let max = self.info[usize::from(depth)].max_nrec;
|
||||||
|
if k - 1 > max || (n as u64) < k - 1 + k {
|
||||||
|
return Err(FormatError::SerializationError(format!(
|
||||||
|
"cannot spread {n} B-tree v2 records over {k} children at depth {depth}"
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
let k = saturating_usize(k);
|
||||||
|
let in_children = n - (k - 1);
|
||||||
|
let (base, extra) = (in_children / k, in_children % k);
|
||||||
|
|
||||||
|
let mut children = Vec::with_capacity(k);
|
||||||
|
let mut separators = Vec::with_capacity(k - 1);
|
||||||
|
let mut next = first;
|
||||||
|
for c in 0..k {
|
||||||
|
let m = base + usize::from(c < extra);
|
||||||
|
children.push(self.node(depth - 1, next, m)?);
|
||||||
|
next += m;
|
||||||
|
if c + 1 < k {
|
||||||
|
separators.push(next);
|
||||||
|
next += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
debug_assert_eq!(next, first + n);
|
||||||
|
|
||||||
|
node.extend_from_slice(b"BTIN");
|
||||||
|
node.push(0); // version
|
||||||
|
node.push(self.p.tree_type);
|
||||||
|
for &s in &separators {
|
||||||
|
node.extend_from_slice(self.record(s));
|
||||||
|
}
|
||||||
|
let total_width = if depth > 1 {
|
||||||
|
self.info[usize::from(depth) - 1].cum_max_nrec_size
|
||||||
|
} else {
|
||||||
|
0
|
||||||
|
};
|
||||||
|
for c in &children {
|
||||||
|
push_uint(&mut node, c.addr, self.offset_size as usize);
|
||||||
|
push_uint(&mut node, c.nrec, self.nrec_width);
|
||||||
|
if depth > 1 {
|
||||||
|
push_uint(&mut node, c.all_nrec, total_width);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let addr = self.push_node(node);
|
||||||
|
Ok(NodeRef {
|
||||||
|
addr,
|
||||||
|
nrec: (k - 1) as u64,
|
||||||
|
all_nrec: n as u64,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Append `v` as a `width`-byte little-endian integer.
|
||||||
|
fn push_uint(buf: &mut Vec<u8>, v: u64, width: usize) {
|
||||||
|
let bytes = v.to_le_bytes();
|
||||||
|
buf.extend_from_slice(&bytes[..width.min(8)]);
|
||||||
|
buf.extend(vec![0u8; width.saturating_sub(8)]);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
use crate::btree_v2::{BTreeV2Header, collect_btree_v2_records};
|
||||||
|
|
||||||
|
fn params(node_size: u32, record_size: u16) -> BTreeV2Params {
|
||||||
|
BTreeV2Params {
|
||||||
|
tree_type: 5,
|
||||||
|
node_size,
|
||||||
|
record_size,
|
||||||
|
split_percent: 100,
|
||||||
|
merge_percent: 40,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `n` 11-byte records: a big-endian counter, so byte order is key order.
|
||||||
|
fn records(n: usize, rs: usize) -> Vec<u8> {
|
||||||
|
let mut out = Vec::with_capacity(n * rs);
|
||||||
|
for i in 0..n {
|
||||||
|
let mut r = vec![0u8; rs];
|
||||||
|
r[..8].copy_from_slice(&(i as u64).to_be_bytes());
|
||||||
|
out.extend_from_slice(&r);
|
||||||
|
}
|
||||||
|
out
|
||||||
|
}
|
||||||
|
|
||||||
|
fn roundtrip(node_size: u32, rs: u16, n: usize, os: u8, ls: u8) -> BTreeV2Header {
|
||||||
|
let recs = records(n, usize::from(rs));
|
||||||
|
let base = 4096u64;
|
||||||
|
let tree = build_btree_v2(params(node_size, rs), &recs, base, os, ls).unwrap();
|
||||||
|
let mut file = vec![0u8; base as usize];
|
||||||
|
file.extend_from_slice(&tree);
|
||||||
|
let hdr = BTreeV2Header::parse(&file, base as usize, os, ls).unwrap();
|
||||||
|
assert_eq!(hdr.total_records, n as u64);
|
||||||
|
let got = collect_btree_v2_records(&file, &hdr, os, ls).unwrap();
|
||||||
|
assert_eq!(got.len(), n);
|
||||||
|
let flat: Vec<u8> = got.into_iter().flat_map(|r| r.data).collect();
|
||||||
|
assert_eq!(flat, recs, "node {node_size} rs {rs} n {n}");
|
||||||
|
hdr
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn one_leaf_then_deeper_trees_read_back_in_order() {
|
||||||
|
// 512-byte nodes of 11-byte records: 45 per leaf, 1149 at depth 1,
|
||||||
|
// 26 449 at depth 2.
|
||||||
|
let info = node_info(512, 11, 8, 3);
|
||||||
|
assert_eq!(
|
||||||
|
info.iter().map(|i| i.cum_max_nrec).collect::<Vec<_>>(),
|
||||||
|
[45, 1149, 26_449, 608_349]
|
||||||
|
);
|
||||||
|
for (n, depth) in [
|
||||||
|
(0, 0),
|
||||||
|
(1, 0),
|
||||||
|
(45, 0),
|
||||||
|
(46, 1),
|
||||||
|
(1149, 1),
|
||||||
|
(1150, 2),
|
||||||
|
(26_449, 2),
|
||||||
|
(26_450, 3),
|
||||||
|
(100_000, 3),
|
||||||
|
] {
|
||||||
|
let hdr = roundtrip(512, 11, n, 8, 8);
|
||||||
|
assert_eq!(hdr.depth, depth, "{n} records");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn pointer_widths_follow_the_offset_and_length_sizes() {
|
||||||
|
for (os, ls) in [(4, 4), (8, 4), (4, 8), (2, 2)] {
|
||||||
|
roundtrip(512, 11, 5000, os, ls);
|
||||||
|
}
|
||||||
|
// Wide counts: a leaf of 2048 bytes / 9-byte records (226, one byte)
|
||||||
|
// and deeper subtree totals of three bytes.
|
||||||
|
roundtrip(2048, 9, 300_000, 8, 8);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn every_node_is_within_its_capacity_and_above_the_merge_threshold() {
|
||||||
|
let rs = 17u16;
|
||||||
|
let n = 70_000usize;
|
||||||
|
let info = node_info(512, rs, 8, 3);
|
||||||
|
let recs = records(n, usize::from(rs));
|
||||||
|
let tree = build_btree_v2(params(512, rs), &recs, 0, 8, 8).unwrap();
|
||||||
|
let hdr_len = header_size(8, 8);
|
||||||
|
let nodes = (tree.len() - hdr_len) / 512;
|
||||||
|
for i in 0..nodes {
|
||||||
|
let node = &tree[hdr_len + i * 512..hdr_len + (i + 1) * 512];
|
||||||
|
let sig = &node[..4];
|
||||||
|
if sig == b"BTLF" {
|
||||||
|
continue; // counts checked through the parents below
|
||||||
|
}
|
||||||
|
assert_eq!(sig, b"BTIN");
|
||||||
|
}
|
||||||
|
// Walk from the header: each child's count within [40%, 100%].
|
||||||
|
let hdr = BTreeV2Header::parse(&tree, 0, 8, 8).unwrap();
|
||||||
|
assert_eq!(hdr.depth, 3);
|
||||||
|
assert!(u64::from(hdr.num_records_in_root) <= info[3].max_nrec);
|
||||||
|
fn walk(tree: &[u8], addr: usize, nrec: usize, depth: usize, info: &[NodeInfo], rs: usize) {
|
||||||
|
if depth == 0 {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
let nrec_w = bytes_for_max_records(info[0].max_nrec);
|
||||||
|
let tot_w = if depth > 1 {
|
||||||
|
info[depth - 1].cum_max_nrec_size
|
||||||
|
} else {
|
||||||
|
0
|
||||||
|
};
|
||||||
|
let mut pos = addr + 6 + nrec * rs;
|
||||||
|
for _ in 0..=nrec {
|
||||||
|
let a = u64::from_le_bytes(tree[pos..pos + 8].try_into().unwrap()) as usize;
|
||||||
|
pos += 8;
|
||||||
|
let mut c = 0usize;
|
||||||
|
for b in 0..nrec_w {
|
||||||
|
c |= usize::from(tree[pos + b]) << (8 * b);
|
||||||
|
}
|
||||||
|
pos += nrec_w + tot_w;
|
||||||
|
let max = info[depth - 1].max_nrec as usize;
|
||||||
|
assert!(c <= max && c * 100 > max * 40, "{c} of {max}");
|
||||||
|
walk(tree, a, c, depth - 1, info, rs);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
walk(
|
||||||
|
&tree,
|
||||||
|
hdr.root_node_address as usize,
|
||||||
|
usize::from(hdr.num_records_in_root),
|
||||||
|
3,
|
||||||
|
&info,
|
||||||
|
usize::from(rs),
|
||||||
|
);
|
||||||
|
assert!(nodes > 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Descending to a key range finds exactly the records a full read
|
||||||
|
/// holds in it — runs of equal keys that straddle node boundaries
|
||||||
|
/// included — at every depth, and nothing for keys not in the tree.
|
||||||
|
#[test]
|
||||||
|
fn a_key_range_search_matches_a_full_scan() {
|
||||||
|
use crate::btree_v2::find_btree_v2_records;
|
||||||
|
use core::cmp::Ordering;
|
||||||
|
let rs = 11usize;
|
||||||
|
// Keys 0, 0, 0, 2, 2, 2, 4, ...: runs of three, odd keys missing.
|
||||||
|
for n in [1usize, 45, 46, 1150, 30_000] {
|
||||||
|
let mut recs = Vec::with_capacity(n * rs);
|
||||||
|
for i in 0..n {
|
||||||
|
let mut r = vec![0u8; rs];
|
||||||
|
r[..8].copy_from_slice(&((i / 3 * 2) as u64).to_be_bytes());
|
||||||
|
r[8..].copy_from_slice(&[(i % 3) as u8, 0, 0]);
|
||||||
|
recs.extend_from_slice(&r);
|
||||||
|
}
|
||||||
|
let base = 4096u64;
|
||||||
|
let tree = build_btree_v2(params(512, 11), &recs, base, 8, 8).unwrap();
|
||||||
|
let mut file = vec![0u8; base as usize];
|
||||||
|
file.extend_from_slice(&tree);
|
||||||
|
let hdr = BTreeV2Header::parse(&file, base as usize, 8, 8).unwrap();
|
||||||
|
let all = collect_btree_v2_records(&file, &hdr, 8, 8).unwrap();
|
||||||
|
let key = |r: &[u8]| u64::from_be_bytes(r[..8].try_into().unwrap());
|
||||||
|
let last = key(&all[n - 1].data);
|
||||||
|
let probes = (0..=last + 1).step_by(if n > 1000 { 37 } else { 1 });
|
||||||
|
for k in probes.chain([last, last + 1, u64::MAX]) {
|
||||||
|
let found =
|
||||||
|
find_btree_v2_records(&file, &hdr, 8, &mut |r: &[u8]| key(r).cmp(&k)).unwrap();
|
||||||
|
let want: Vec<&[u8]> = all
|
||||||
|
.iter()
|
||||||
|
.map(|r| r.data.as_slice())
|
||||||
|
.filter(|r| key(r) == k)
|
||||||
|
.collect();
|
||||||
|
let got: Vec<&[u8]> = found.iter().map(|r| r.data.as_slice()).collect();
|
||||||
|
assert_eq!(got, want, "n {n} key {k}");
|
||||||
|
assert_eq!(
|
||||||
|
got.len(),
|
||||||
|
if k % 2 == 0 && k <= last {
|
||||||
|
want.len()
|
||||||
|
} else {
|
||||||
|
0
|
||||||
|
}
|
||||||
|
);
|
||||||
|
}
|
||||||
|
// Every record, or none, when the whole tree is in or out of range.
|
||||||
|
let every = find_btree_v2_records(&file, &hdr, 8, &mut |_| Ordering::Equal).unwrap();
|
||||||
|
assert_eq!(every.len(), n);
|
||||||
|
let none = find_btree_v2_records(&file, &hdr, 8, &mut |_| Ordering::Less).unwrap();
|
||||||
|
assert!(none.is_empty());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A two-level tree read through a `read_at`-only storage gives what
|
||||||
|
/// the slice gives — records, descents and errors — whole, truncated
|
||||||
|
/// at every length, and with each byte of its nodes flipped, and each
|
||||||
|
/// node costs one read.
|
||||||
|
#[test]
|
||||||
|
fn storage_reads_match_slice_reads() {
|
||||||
|
use crate::btree_v2::{
|
||||||
|
collect_btree_v2_records_in, find_btree_v2_records, find_btree_v2_records_in,
|
||||||
|
};
|
||||||
|
use crate::storage::CountingStorage;
|
||||||
|
let (rs, n, base) = (11usize, 120usize, 64usize);
|
||||||
|
let recs = records(n, rs);
|
||||||
|
let tree = build_btree_v2(params(128, 11), &recs, base as u64, 8, 8).unwrap();
|
||||||
|
let mut whole = vec![0u8; base];
|
||||||
|
whole.extend_from_slice(&tree);
|
||||||
|
let hdr = BTreeV2Header::parse(&whole, base, 8, 8).unwrap();
|
||||||
|
assert!(hdr.depth >= 1, "{hdr:?}");
|
||||||
|
let key = |r: &[u8]| u64::from_be_bytes(r[..8].try_into().unwrap());
|
||||||
|
let mut files = Vec::new();
|
||||||
|
for cut in base..=whole.len() {
|
||||||
|
files.push(whole[..cut].to_vec());
|
||||||
|
}
|
||||||
|
for at in base..whole.len() {
|
||||||
|
let mut bad = whole.clone();
|
||||||
|
bad[at] ^= 0x5a;
|
||||||
|
files.push(bad);
|
||||||
|
}
|
||||||
|
let mut ok = 0;
|
||||||
|
for f in &files {
|
||||||
|
let st = CountingStorage::new(f.clone());
|
||||||
|
let want_h = BTreeV2Header::parse(f, base, 8, 8);
|
||||||
|
let got_h = BTreeV2Header::parse_in(&st, base as u64, 8, 8);
|
||||||
|
assert_eq!(format!("{got_h:?}"), format!("{want_h:?}"));
|
||||||
|
// The nodes of the intact header, over each damaged file.
|
||||||
|
let want = collect_btree_v2_records(f, &hdr, 8, 8);
|
||||||
|
st.reset();
|
||||||
|
let got = collect_btree_v2_records_in(&st, &hdr, 8, 8);
|
||||||
|
assert_eq!(format!("{got:?}"), format!("{want:?}"));
|
||||||
|
if want.is_ok() {
|
||||||
|
ok += 1;
|
||||||
|
assert!(st.reads() <= 1 + n as u64 / 3, "{} reads", st.reads());
|
||||||
|
}
|
||||||
|
for k in [0u64, 7, 60, 119, 500] {
|
||||||
|
let want = find_btree_v2_records(f, &hdr, 8, &mut |r: &[u8]| key(r).cmp(&k));
|
||||||
|
let got = find_btree_v2_records_in(&st, &hdr, 8, &mut |r: &[u8]| key(r).cmp(&k));
|
||||||
|
assert_eq!(format!("{got:?}"), format!("{want:?}"));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
assert!(ok > 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn a_node_too_small_or_too_big_is_an_error() {
|
||||||
|
assert!(build_btree_v2(params(16, 11), &records(1, 11), 0, 8, 8).is_err());
|
||||||
|
// A leaf with room for more than 65 535 records.
|
||||||
|
assert!(build_btree_v2(params(1 << 20, 11), &records(1, 11), 0, 8, 8).is_err());
|
||||||
|
// Records that are not whole.
|
||||||
|
assert!(build_btree_v2(params(512, 11), &[0u8; 12], 0, 8, 8).is_err());
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,79 @@
|
|||||||
|
//! Large output buffers backed by transparent huge pages where the OS offers
|
||||||
|
//! them.
|
||||||
|
//!
|
||||||
|
//! A fresh multi-megabyte `Vec` is mapped lazily by the kernel: the first
|
||||||
|
//! write to each 4 KiB page takes a page fault, and the kernel zeroes the page
|
||||||
|
//! before handing it over. For a 64 MiB read that is 16384 faults, and they
|
||||||
|
//! cost far more than the copy that fills the buffer — single-threaded
|
||||||
|
//! contiguous reads ran at about a quarter of h5py's speed because of them.
|
||||||
|
//! numpy (so h5py) avoids this by asking for transparent huge pages
|
||||||
|
//! (`madvise(MADV_HUGEPAGE)`) on every allocation of 4 MiB or more, which
|
||||||
|
//! turns 512 faults into one; this module does the same.
|
||||||
|
//!
|
||||||
|
//! The advice only changes how the pages are backed, never their contents, so
|
||||||
|
//! it is harmless when it cannot be honoured (THP disabled, not Linux, a
|
||||||
|
//! region that is part of the heap): the buffer is then exactly what it would
|
||||||
|
//! have been without it.
|
||||||
|
|
||||||
|
#[cfg(not(feature = "std"))]
|
||||||
|
use alloc::vec::Vec;
|
||||||
|
|
||||||
|
/// Buffers smaller than this are left alone (numpy uses the same threshold).
|
||||||
|
#[cfg(any(target_os = "linux", test))]
|
||||||
|
pub(crate) const HUGE_PAGE_THRESHOLD: usize = 4 << 20;
|
||||||
|
|
||||||
|
/// Advise the kernel to back `[ptr, ptr + len)` with transparent huge pages,
|
||||||
|
/// when `len` is large enough to benefit. Call it before the first write so
|
||||||
|
/// the faults happen at huge-page granularity.
|
||||||
|
#[inline]
|
||||||
|
pub(crate) fn advise_huge_pages(ptr: *const u8, len: usize) {
|
||||||
|
#[cfg(target_os = "linux")]
|
||||||
|
if len >= HUGE_PAGE_THRESHOLD {
|
||||||
|
const PAGE: usize = 4096;
|
||||||
|
let start = (ptr as usize).next_multiple_of(PAGE);
|
||||||
|
let end = (ptr as usize + len) & !(PAGE - 1);
|
||||||
|
if end > start {
|
||||||
|
// SAFETY: `[start, end)` lies inside an allocation of `len` bytes
|
||||||
|
// at `ptr` that the caller owns, and is page aligned as madvise
|
||||||
|
// requires. MADV_HUGEPAGE does not change the memory's contents or
|
||||||
|
// validity; on failure (EINVAL when THP is compiled out, etc.) the
|
||||||
|
// region is simply left as it was, so the result is ignored.
|
||||||
|
unsafe {
|
||||||
|
libc::madvise(start as *mut libc::c_void, end - start, libc::MADV_HUGEPAGE);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
#[cfg(not(target_os = "linux"))]
|
||||||
|
let _ = (ptr, len);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `Vec::with_capacity(count)` for a buffer about to be filled in bulk, with
|
||||||
|
/// huge-page advice when it is large (see the module docs).
|
||||||
|
#[inline]
|
||||||
|
pub(crate) fn vec_for_bulk<T>(count: usize) -> Vec<T> {
|
||||||
|
let v: Vec<T> = Vec::with_capacity(count);
|
||||||
|
advise_huge_pages(
|
||||||
|
v.as_ptr().cast::<u8>(),
|
||||||
|
v.capacity().saturating_mul(core::mem::size_of::<T>()),
|
||||||
|
);
|
||||||
|
v
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn bulk_vec_is_an_ordinary_vec() {
|
||||||
|
for count in [0usize, 1, 1000, HUGE_PAGE_THRESHOLD / 4 + 3] {
|
||||||
|
let mut v: Vec<u32> = vec_for_bulk(count);
|
||||||
|
assert!(v.capacity() >= count);
|
||||||
|
v.extend((0..count as u32).map(|i| i.wrapping_mul(2654435761)));
|
||||||
|
assert!(
|
||||||
|
v.iter()
|
||||||
|
.enumerate()
|
||||||
|
.all(|(i, &x)| x == (i as u32).wrapping_mul(2654435761))
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -14,6 +14,45 @@ pub fn jenkins_lookup3(data: &[u8]) -> u32 {
|
|||||||
hashlittle(data, 0)
|
hashlittle(data, 0)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// HDF5's Fletcher-32 checksum, as the Fletcher-32 I/O filter (filter id 3)
|
||||||
|
/// stores it after each chunk.
|
||||||
|
///
|
||||||
|
/// A line-for-line port of `H5_checksum_fletcher32` (H5checksum.c, libhdf5
|
||||||
|
/// 1.8 through 1.14): big-endian 16-bit words summed in blocks of 360, each
|
||||||
|
/// sum reduced after a block by the ones'-complement fold
|
||||||
|
/// `(s & 0xffff) + (s >> 16)` rather than `% 65535`, an odd trailing byte
|
||||||
|
/// taken as the high byte of a last word, and a final fold of both sums.
|
||||||
|
/// The fold and `% 65535` differ whenever a sum is a non-zero multiple of
|
||||||
|
/// 65535: the fold leaves 0xffff where the modulo gives 0, so the two
|
||||||
|
/// disagree on about one chunk in 32768 and libhdf5 rejects the other's
|
||||||
|
/// checksum. This must stay the only implementation.
|
||||||
|
pub fn fletcher32(data: &[u8]) -> u32 {
|
||||||
|
let mut sum1: u32 = 0;
|
||||||
|
let mut sum2: u32 = 0;
|
||||||
|
// 360 words keep both sums inside 32 bits between folds (the bound
|
||||||
|
// libhdf5 uses: after a fold sum1 < 0x10200, so sum2 stays below
|
||||||
|
// 360 * 361 / 2 * 0xffff + 360 * 0x10200 + 0x1fffe < 2^32). The adds wrap
|
||||||
|
// like the C unsigned arithmetic all the same.
|
||||||
|
let (words, odd) = data.as_chunks::<2>();
|
||||||
|
for block in words.chunks(360) {
|
||||||
|
for w in block {
|
||||||
|
sum1 = sum1.wrapping_add((u32::from(w[0]) << 8) | u32::from(w[1]));
|
||||||
|
sum2 = sum2.wrapping_add(sum1);
|
||||||
|
}
|
||||||
|
sum1 = (sum1 & 0xffff) + (sum1 >> 16);
|
||||||
|
sum2 = (sum2 & 0xffff) + (sum2 >> 16);
|
||||||
|
}
|
||||||
|
if let [last] = odd {
|
||||||
|
sum1 = sum1.wrapping_add(u32::from(*last) << 8);
|
||||||
|
sum2 = sum2.wrapping_add(sum1);
|
||||||
|
sum1 = (sum1 & 0xffff) + (sum1 >> 16);
|
||||||
|
sum2 = (sum2 & 0xffff) + (sum2 >> 16);
|
||||||
|
}
|
||||||
|
sum1 = (sum1 & 0xffff) + (sum1 >> 16);
|
||||||
|
sum2 = (sum2 & 0xffff) + (sum2 >> 16);
|
||||||
|
(sum2 << 16) | sum1
|
||||||
|
}
|
||||||
|
|
||||||
/// Compute CRC32 (IEEE / ISO 3309) over data.
|
/// Compute CRC32 (IEEE / ISO 3309) over data.
|
||||||
///
|
///
|
||||||
/// When the `fast-checksum` feature is enabled, this uses hardware CRC32
|
/// When the `fast-checksum` feature is enabled, this uses hardware CRC32
|
||||||
@@ -207,6 +246,19 @@ fn hashlittle(data: &[u8], initval: u32) -> u32 {
|
|||||||
mod tests {
|
mod tests {
|
||||||
use super::*;
|
use super::*;
|
||||||
|
|
||||||
|
/// Values of libhdf5's `H5_checksum_fletcher32` (h5py 3.x's bundled
|
||||||
|
/// libhdf5, called through ctypes). The first three are sums that are
|
||||||
|
/// multiples of 65535, where `% 65535` gave 0 instead of 0xffff.
|
||||||
|
#[test]
|
||||||
|
fn fletcher32_matches_libhdf5() {
|
||||||
|
assert_eq!(fletcher32(&[0x00, 0x01, 0xff, 0xfe]), 0x0001_ffff);
|
||||||
|
assert_eq!(fletcher32(&[0xff; 720]), 0xffff_ffff);
|
||||||
|
assert_eq!(fletcher32(&[0xff; 721]), 0xff00_ff00);
|
||||||
|
assert_eq!(fletcher32(&[0xff; 1441]), 0xff00_ff00);
|
||||||
|
assert_eq!(fletcher32(&[]), 0);
|
||||||
|
assert_eq!(fletcher32(&[7]), 0x0700_0700);
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn empty_input() {
|
fn empty_input() {
|
||||||
// Empty input should return the initial state after no mixing
|
// Empty input should return the initial state after no mixing
|
||||||
|
|||||||
@@ -223,13 +223,32 @@ pub const DEFAULT_CACHE_BYTES: usize = 16 * 1024 * 1024; // 16 MiB
|
|||||||
/// coordinate map and reduces collision chains compared to power-of-two sizes.
|
/// coordinate map and reduces collision chains compared to power-of-two sizes.
|
||||||
pub const DEFAULT_MAX_SLOTS: usize = 521;
|
pub const DEFAULT_MAX_SLOTS: usize = 521;
|
||||||
|
|
||||||
|
/// Most datasets whose chunk index a [`ChunkCache`] keeps at once.
|
||||||
|
pub const MAX_INDEXED_DATASETS: usize = 64;
|
||||||
|
|
||||||
|
/// Most chunk-index entries, summed over all datasets, a [`ChunkCache`] keeps.
|
||||||
|
/// Least-recently-used datasets' indexes are dropped past this (the dataset
|
||||||
|
/// being read is always kept), so a file with many or huge chunked datasets
|
||||||
|
/// cannot grow the cache without bound.
|
||||||
|
pub const MAX_INDEXED_CHUNKS: usize = 1 << 20;
|
||||||
|
|
||||||
|
/// The dataset key the address-less (legacy) methods use when
|
||||||
|
/// [`ChunkCache::ensure_dataset`] has not been called.
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
const UNBOUND_DATASET: u64 = u64::MAX;
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// LRU entry
|
// LRU entry
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// Decompressed chunks are keyed by dataset *and* coordinate: every chunked
|
||||||
|
/// dataset has a chunk at (0, 0, ...), so the coordinate alone is ambiguous.
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
type SlotKey = (u64, ChunkCoord);
|
||||||
|
|
||||||
#[cfg(feature = "std")]
|
#[cfg(feature = "std")]
|
||||||
struct CachedChunk {
|
struct CachedChunk {
|
||||||
coord: ChunkCoord,
|
key: SlotKey,
|
||||||
/// Shared so a cache hit is a refcount bump, not a copy of the whole
|
/// Shared so a cache hit is a refcount bump, not a copy of the whole
|
||||||
/// (potentially large) decompressed chunk.
|
/// (potentially large) decompressed chunk.
|
||||||
data: Arc<CacheAlignedBuffer>,
|
data: Arc<CacheAlignedBuffer>,
|
||||||
@@ -237,21 +256,48 @@ struct CachedChunk {
|
|||||||
last_access: u64,
|
last_access: u64,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Per-dataset index state.
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
#[derive(Default)]
|
||||||
|
struct DatasetEntry {
|
||||||
|
/// Chunk coordinate -> ChunkInfo (offset + size in file).
|
||||||
|
index: Option<Arc<HashMap<ChunkCoord, ChunkInfo>>>,
|
||||||
|
/// Pre-built chunk index for O(1) coordinate lookups.
|
||||||
|
chunk_index: Option<Arc<ChunkIndex>>,
|
||||||
|
/// Pre-computed chunk layout for fast assembly.
|
||||||
|
chunk_layout: Option<Arc<ChunkLayout>>,
|
||||||
|
/// Tick of the last use, for dropping the least recently used dataset.
|
||||||
|
last_used: u64,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
impl DatasetEntry {
|
||||||
|
fn weight(&self) -> usize {
|
||||||
|
self.index.as_ref().map_or(0, |m| m.len())
|
||||||
|
+ self.chunk_index.as_ref().map_or(0, |c| c.num_chunks())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// ChunkCache
|
// ChunkCache
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
/// A per-dataset chunk cache with hash-based index and LRU eviction.
|
/// A per-file chunk cache: chunk indexes per dataset, plus an LRU of
|
||||||
|
/// decompressed chunks, all keyed by dataset.
|
||||||
///
|
///
|
||||||
/// # Usage
|
/// A dataset is identified by the address of its chunk index (B-tree, fixed
|
||||||
|
/// or extensible array, ...), which is unique within a file. Every method
|
||||||
|
/// that takes an `addr` works on that dataset only, so threads reading
|
||||||
|
/// different datasets through one shared cache never see each other's
|
||||||
|
/// chunks. The address-less methods (`has_index`, `populate_index`,
|
||||||
|
/// `get_decompressed`, ...) act on the dataset last bound with
|
||||||
|
/// [`Self::ensure_dataset`]; that binding is shared state, so concurrent
|
||||||
|
/// readers must use the `*_in` / `*_for` methods instead (the chunked
|
||||||
|
/// readers in [`crate::chunked_read`] do).
|
||||||
///
|
///
|
||||||
/// ```ignore
|
/// Memory is bounded: decompressed data by `max_bytes`/`max_slots` across
|
||||||
/// let cache = ChunkCache::new();
|
/// all datasets, indexes by [`MAX_INDEXED_DATASETS`] and
|
||||||
/// // Pass &cache to read_chunked_data — it will populate the index lazily.
|
/// [`MAX_INDEXED_CHUNKS`].
|
||||||
/// ```
|
|
||||||
///
|
|
||||||
/// The cache is wrapped in `Mutex` internally so it can be mutated through
|
|
||||||
/// shared references (thread-safe).
|
|
||||||
///
|
///
|
||||||
/// Only available with the `std` feature because it requires `std::sync::Mutex`.
|
/// Only available with the `std` feature because it requires `std::sync::Mutex`.
|
||||||
#[cfg(feature = "std")]
|
#[cfg(feature = "std")]
|
||||||
@@ -261,26 +307,20 @@ pub struct ChunkCache {
|
|||||||
|
|
||||||
#[cfg(feature = "std")]
|
#[cfg(feature = "std")]
|
||||||
struct CacheInner {
|
struct CacheInner {
|
||||||
/// Hash index: chunk coordinate -> ChunkInfo (offset + size in file).
|
/// Per-dataset chunk indexes, keyed by chunk-index address.
|
||||||
/// Populated once per dataset on first access.
|
datasets: HashMap<u64, DatasetEntry>,
|
||||||
index: Option<HashMap<ChunkCoord, ChunkInfo>>,
|
|
||||||
|
|
||||||
/// Address of the dataset (its chunk-index base address) that the cached
|
/// Dataset the address-less methods act on (see `ensure_dataset`).
|
||||||
/// index, chunk index, layout, and decompressed slots currently belong to.
|
current: Option<u64>,
|
||||||
/// The cache is shared per file across datasets, so every cached-read entry
|
|
||||||
/// checks this and resets the per-dataset state when the dataset changes —
|
|
||||||
/// otherwise one dataset's chunk index (with its own rank) would be reused
|
|
||||||
/// for another, corrupting reads.
|
|
||||||
index_addr: Option<u64>,
|
|
||||||
|
|
||||||
/// LRU cache of decompressed chunk data.
|
/// LRU cache of decompressed chunk data.
|
||||||
slots: Vec<CachedChunk>,
|
slots: Vec<CachedChunk>,
|
||||||
|
|
||||||
/// Coordinate -> index into `slots`, for O(1) lookup instead of a linear
|
/// Key -> index into `slots`, for O(1) lookup instead of a linear
|
||||||
/// scan. Kept in sync with `slots` on every insert/evict/clear — in
|
/// scan. Kept in sync with `slots` on every insert/evict/clear — in
|
||||||
/// particular, `slots.swap_remove(i)` moves the last element into slot
|
/// particular, `slots.swap_remove(i)` moves the last element into slot
|
||||||
/// `i`, so the moved element's index entry must be updated too.
|
/// `i`, so the moved element's index entry must be updated too.
|
||||||
slot_index: HashMap<ChunkCoord, usize>,
|
slot_index: HashMap<SlotKey, usize>,
|
||||||
|
|
||||||
/// Current total bytes of cached decompressed data.
|
/// Current total bytes of cached decompressed data.
|
||||||
current_bytes: usize,
|
current_bytes: usize,
|
||||||
@@ -294,17 +334,145 @@ struct CacheInner {
|
|||||||
/// Monotonic counter for LRU ordering.
|
/// Monotonic counter for LRU ordering.
|
||||||
tick: u64,
|
tick: u64,
|
||||||
|
|
||||||
/// Last accessed chunk coordinate (for sequential detection).
|
/// Last accessed chunk (for sequential detection).
|
||||||
last_coord: Option<ChunkCoord>,
|
last_coord: Option<SlotKey>,
|
||||||
|
|
||||||
/// Access pattern statistics.
|
/// Access pattern statistics.
|
||||||
stats: AccessStats,
|
stats: AccessStats,
|
||||||
|
}
|
||||||
|
|
||||||
/// Pre-built chunk index for O(1) coordinate lookups.
|
#[cfg(feature = "std")]
|
||||||
chunk_index: Option<ChunkIndex>,
|
impl CacheInner {
|
||||||
|
fn current(&self) -> u64 {
|
||||||
|
self.current.unwrap_or(UNBOUND_DATASET)
|
||||||
|
}
|
||||||
|
|
||||||
/// Pre-computed chunk layout for fast assembly.
|
fn touch(&mut self, addr: u64) -> &mut DatasetEntry {
|
||||||
chunk_layout: Option<ChunkLayout>,
|
self.tick += 1;
|
||||||
|
let tick = self.tick;
|
||||||
|
let entry = self.datasets.entry(addr).or_default();
|
||||||
|
entry.last_used = tick;
|
||||||
|
entry
|
||||||
|
}
|
||||||
|
|
||||||
|
fn entry(&self, addr: u64) -> Option<&DatasetEntry> {
|
||||||
|
self.datasets.get(&addr)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Drop least-recently-used datasets' indexes (never `keep`'s) until the
|
||||||
|
/// dataset and chunk-entry budgets hold.
|
||||||
|
fn trim_datasets(&mut self, keep: u64) {
|
||||||
|
loop {
|
||||||
|
let total: usize = self.datasets.values().map(DatasetEntry::weight).sum();
|
||||||
|
if self.datasets.len() <= MAX_INDEXED_DATASETS && total <= MAX_INDEXED_CHUNKS {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
let victim = self
|
||||||
|
.datasets
|
||||||
|
.iter()
|
||||||
|
.filter(|(a, _)| **a != keep)
|
||||||
|
.min_by_key(|(_, e)| e.last_used)
|
||||||
|
.map(|(a, _)| *a);
|
||||||
|
match victim {
|
||||||
|
Some(a) => {
|
||||||
|
self.datasets.remove(&a);
|
||||||
|
}
|
||||||
|
None => return,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn get_decompressed(&mut self, addr: u64, coord: &[u64]) -> Option<Arc<CacheAlignedBuffer>> {
|
||||||
|
self.tick += 1;
|
||||||
|
let tick = self.tick;
|
||||||
|
|
||||||
|
// Track sequential vs random access
|
||||||
|
let is_sequential = self.last_coord.as_ref().is_some_and(|(prev_addr, prev)| {
|
||||||
|
// Sequential if exactly one dimension changed
|
||||||
|
let changes: usize = prev
|
||||||
|
.iter()
|
||||||
|
.zip(coord.iter())
|
||||||
|
.filter(|(a, b)| a != b)
|
||||||
|
.count();
|
||||||
|
*prev_addr == addr && changes <= 1
|
||||||
|
});
|
||||||
|
if is_sequential {
|
||||||
|
self.stats.sequential_count += 1;
|
||||||
|
} else if self.last_coord.is_some() {
|
||||||
|
self.stats.random_count += 1;
|
||||||
|
}
|
||||||
|
let key: SlotKey = (addr, coord.to_vec());
|
||||||
|
let found = if let Some(&idx) = self.slot_index.get(&key) {
|
||||||
|
self.slots[idx].last_access = tick;
|
||||||
|
Some(Arc::clone(&self.slots[idx].data))
|
||||||
|
} else {
|
||||||
|
None
|
||||||
|
};
|
||||||
|
self.last_coord = Some(key);
|
||||||
|
if let Some(ref data) = found {
|
||||||
|
self.stats.hits += 1;
|
||||||
|
self.stats.bytes_read += data.len() as u64;
|
||||||
|
} else {
|
||||||
|
self.stats.misses += 1;
|
||||||
|
}
|
||||||
|
found
|
||||||
|
}
|
||||||
|
|
||||||
|
fn put_decompressed(
|
||||||
|
&mut self,
|
||||||
|
key: SlotKey,
|
||||||
|
data: Arc<CacheAlignedBuffer>,
|
||||||
|
) -> Arc<CacheAlignedBuffer> {
|
||||||
|
let data_len = data.len();
|
||||||
|
|
||||||
|
// Don't cache if single chunk exceeds budget — still return the data
|
||||||
|
// to the caller, just don't retain it.
|
||||||
|
if data_len > self.max_bytes {
|
||||||
|
return data;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Check if already present
|
||||||
|
self.tick += 1;
|
||||||
|
let tick = self.tick;
|
||||||
|
if let Some(&idx) = self.slot_index.get(&key) {
|
||||||
|
self.slots[idx].last_access = tick;
|
||||||
|
return Arc::clone(&self.slots[idx].data); // already cached
|
||||||
|
}
|
||||||
|
|
||||||
|
// Evict until we have room
|
||||||
|
while self.slots.len() >= self.max_slots
|
||||||
|
|| (self.current_bytes + data_len > self.max_bytes && !self.slots.is_empty())
|
||||||
|
{
|
||||||
|
// Find LRU slot
|
||||||
|
let lru_idx = self
|
||||||
|
.slots
|
||||||
|
.iter()
|
||||||
|
.enumerate()
|
||||||
|
.min_by_key(|(_, s)| s.last_access)
|
||||||
|
.map(|(i, _)| i)
|
||||||
|
.unwrap();
|
||||||
|
let removed = self.slots.swap_remove(lru_idx);
|
||||||
|
self.slot_index.remove(&removed.key);
|
||||||
|
// swap_remove moved the former last element into `lru_idx` (unless
|
||||||
|
// it *was* the last element) — fix up that element's index entry.
|
||||||
|
if lru_idx < self.slots.len() {
|
||||||
|
let moved_key = self.slots[lru_idx].key.clone();
|
||||||
|
self.slot_index.insert(moved_key, lru_idx);
|
||||||
|
}
|
||||||
|
self.current_bytes -= removed.data.len();
|
||||||
|
self.stats.evictions += 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
self.current_bytes += data_len;
|
||||||
|
let new_idx = self.slots.len();
|
||||||
|
self.slot_index.insert(key.clone(), new_idx);
|
||||||
|
self.slots.push(CachedChunk {
|
||||||
|
key,
|
||||||
|
data: Arc::clone(&data),
|
||||||
|
last_access: tick,
|
||||||
|
});
|
||||||
|
data
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Access pattern statistics tracked by the chunk cache.
|
/// Access pattern statistics tracked by the chunk cache.
|
||||||
@@ -356,8 +524,8 @@ impl ChunkCache {
|
|||||||
pub fn with_capacity(max_bytes: usize, max_slots: usize) -> Self {
|
pub fn with_capacity(max_bytes: usize, max_slots: usize) -> Self {
|
||||||
Self {
|
Self {
|
||||||
inner: std::sync::Mutex::new(CacheInner {
|
inner: std::sync::Mutex::new(CacheInner {
|
||||||
index: None,
|
datasets: HashMap::new(),
|
||||||
index_addr: None,
|
current: None,
|
||||||
slots: Vec::with_capacity(max_slots.min(64)),
|
slots: Vec::with_capacity(max_slots.min(64)),
|
||||||
slot_index: HashMap::with_capacity(max_slots.min(64)),
|
slot_index: HashMap::with_capacity(max_slots.min(64)),
|
||||||
current_bytes: 0,
|
current_bytes: 0,
|
||||||
@@ -366,340 +534,331 @@ impl ChunkCache {
|
|||||||
tick: 0,
|
tick: 0,
|
||||||
last_coord: None,
|
last_coord: None,
|
||||||
stats: AccessStats::default(),
|
stats: AccessStats::default(),
|
||||||
chunk_index: None,
|
|
||||||
chunk_layout: None,
|
|
||||||
}),
|
}),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// ----- Index operations -----
|
fn lock(&self) -> std::sync::MutexGuard<'_, CacheInner> {
|
||||||
|
self.inner.lock().unwrap_or_else(|e| e.into_inner())
|
||||||
|
}
|
||||||
|
|
||||||
/// The most decompressed bytes this cache will hold.
|
/// The most decompressed bytes this cache will hold.
|
||||||
pub fn max_bytes(&self) -> usize {
|
pub fn max_bytes(&self) -> usize {
|
||||||
self.inner.lock().map(|g| g.max_bytes).unwrap_or(0)
|
self.lock().max_bytes
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Bind the cache to the dataset at chunk-index address `addr`.
|
// ----- Dataset-keyed operations (safe to use concurrently) -----
|
||||||
|
|
||||||
|
/// The chunk list of the dataset whose chunk index is at `addr`.
|
||||||
///
|
///
|
||||||
/// The cache is shared per file across all of its datasets. If the cache
|
/// On the first call for a dataset, `build` scans its chunk index; the
|
||||||
/// currently holds state for a different dataset, all per-dataset state
|
/// result is kept (offsets truncated to `rank` for the lookup key), so
|
||||||
/// (chunk index, chunk-index map, layout, and decompressed slots) is
|
/// later calls skip the scan. `build` runs without the cache lock held;
|
||||||
/// dropped so the next access rebuilds it for this dataset. Reading the
|
/// if two threads race to build the same dataset's index, the first
|
||||||
/// same dataset again is a no-op, preserving the cache's benefit for
|
/// stored one wins and both return equivalent lists.
|
||||||
/// repeated/sequential access. Returns `true` if a reset occurred.
|
pub fn chunks_for<E>(
|
||||||
pub fn ensure_dataset(&self, addr: u64) -> bool {
|
&self,
|
||||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
addr: u64,
|
||||||
if inner.index_addr == Some(addr) {
|
rank: usize,
|
||||||
return false;
|
build: impl FnOnce() -> Result<Vec<ChunkInfo>, E>,
|
||||||
|
) -> Result<Vec<ChunkInfo>, E> {
|
||||||
|
Ok(self
|
||||||
|
.index_for(addr, rank, build)?
|
||||||
|
.values()
|
||||||
|
.cloned()
|
||||||
|
.collect())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn index_for<E>(
|
||||||
|
&self,
|
||||||
|
addr: u64,
|
||||||
|
rank: usize,
|
||||||
|
build: impl FnOnce() -> Result<Vec<ChunkInfo>, E>,
|
||||||
|
) -> Result<Arc<HashMap<ChunkCoord, ChunkInfo>>, E> {
|
||||||
|
if let Some(index) = self.lock().touch(addr).index.clone() {
|
||||||
|
return Ok(index);
|
||||||
}
|
}
|
||||||
inner.index = None;
|
let chunks = build()?;
|
||||||
inner.chunk_index = None;
|
let map: HashMap<ChunkCoord, ChunkInfo> = chunks
|
||||||
inner.chunk_layout = None;
|
.into_iter()
|
||||||
inner.slots.clear();
|
.map(|ci| (ci.offsets.iter().take(rank).copied().collect(), ci))
|
||||||
inner.slot_index.clear();
|
.collect();
|
||||||
inner.current_bytes = 0;
|
let mut inner = self.lock();
|
||||||
inner.last_coord = None;
|
let entry = inner.touch(addr);
|
||||||
inner.index_addr = Some(addr);
|
let index = Arc::clone(entry.index.get_or_insert_with(|| Arc::new(map)));
|
||||||
true
|
inner.trim_datasets(addr);
|
||||||
|
Ok(index)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Returns `true` if the chunk index has been built.
|
/// The pre-computed assembly layout of the dataset at `addr`, building
|
||||||
|
/// its chunk index (via `build`, as in [`Self::chunks_for`]) and layout on
|
||||||
|
/// first use.
|
||||||
|
pub fn chunk_layout_for<E>(
|
||||||
|
&self,
|
||||||
|
addr: u64,
|
||||||
|
rank: usize,
|
||||||
|
build: impl FnOnce() -> Result<Vec<ChunkInfo>, E>,
|
||||||
|
ds_dims: &[usize],
|
||||||
|
chunk_dims: &[usize],
|
||||||
|
elem_size: usize,
|
||||||
|
) -> Result<Arc<ChunkLayout>, E> {
|
||||||
|
let (layout, chunk_index) = {
|
||||||
|
let mut inner = self.lock();
|
||||||
|
let entry = inner.touch(addr);
|
||||||
|
(entry.chunk_layout.clone(), entry.chunk_index.clone())
|
||||||
|
};
|
||||||
|
if let Some(layout) = layout {
|
||||||
|
return Ok(layout);
|
||||||
|
}
|
||||||
|
let chunk_index = match chunk_index {
|
||||||
|
Some(ci) => ci,
|
||||||
|
None => {
|
||||||
|
let index = self.index_for(addr, rank, build)?;
|
||||||
|
let chunks: Vec<ChunkInfo> = index.values().cloned().collect();
|
||||||
|
Arc::new(ChunkIndex::build(&chunks, rank))
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let layout = ChunkLayout::build(&chunk_index, ds_dims, chunk_dims, elem_size);
|
||||||
|
let mut inner = self.lock();
|
||||||
|
let entry = inner.touch(addr);
|
||||||
|
entry.chunk_index.get_or_insert(chunk_index);
|
||||||
|
let layout = Arc::clone(entry.chunk_layout.get_or_insert_with(|| Arc::new(layout)));
|
||||||
|
inner.trim_datasets(addr);
|
||||||
|
Ok(layout)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Cached decompressed chunk at `coord` of the dataset at `addr`.
|
||||||
|
///
|
||||||
|
/// O(1) lookup; the clone is an `Arc` refcount bump, not a copy of the
|
||||||
|
/// underlying decompressed data.
|
||||||
|
pub fn get_decompressed_in(&self, addr: u64, coord: &[u64]) -> Option<Arc<CacheAlignedBuffer>> {
|
||||||
|
self.lock().get_decompressed(addr, coord)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Cache decompressed chunk data for `coord` of the dataset at `addr`.
|
||||||
|
/// Returns the `Arc`-shared buffer now cached (or already cached).
|
||||||
|
pub fn put_decompressed_in(
|
||||||
|
&self,
|
||||||
|
addr: u64,
|
||||||
|
coord: ChunkCoord,
|
||||||
|
data: Vec<u8>,
|
||||||
|
) -> Arc<CacheAlignedBuffer> {
|
||||||
|
self.put_decompressed_aligned_in(addr, coord, CacheAlignedBuffer::from_vec(data))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`Self::put_decompressed_in`] for an already-aligned buffer.
|
||||||
|
pub fn put_decompressed_aligned_in(
|
||||||
|
&self,
|
||||||
|
addr: u64,
|
||||||
|
coord: ChunkCoord,
|
||||||
|
data: CacheAlignedBuffer,
|
||||||
|
) -> Arc<CacheAlignedBuffer> {
|
||||||
|
let data = Arc::new(data);
|
||||||
|
self.lock().put_decompressed((addr, coord), data)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Record that the given chunk coordinates of the dataset at `addr` are
|
||||||
|
/// predicted to be accessed soon (bookkeeping only).
|
||||||
|
///
|
||||||
|
/// This does **not** prefetch or pre-decompress anything — it only
|
||||||
|
/// checks whether each coordinate is already in the chunk index and
|
||||||
|
/// updates access-pattern stats accordingly.
|
||||||
|
pub fn prefetch_hint_in(&self, addr: u64, next_coords: &[ChunkCoord]) {
|
||||||
|
let mut inner = self.lock();
|
||||||
|
let Some(index) = inner.entry(addr).and_then(|e| e.index.clone()) else {
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
let known = next_coords
|
||||||
|
.iter()
|
||||||
|
.filter(|c| index.contains_key(*c))
|
||||||
|
.count();
|
||||||
|
inner.stats.sequential_count += known as u64;
|
||||||
|
}
|
||||||
|
|
||||||
|
// ----- Address-less operations on the bound dataset -----
|
||||||
|
|
||||||
|
/// Bind the address-less methods to the dataset at chunk-index address
|
||||||
|
/// `addr`. Returns `true` if this changed the bound dataset.
|
||||||
|
///
|
||||||
|
/// Each dataset's state is kept separately, so switching loses nothing
|
||||||
|
/// and never exposes one dataset's index or chunks to another. The
|
||||||
|
/// binding itself is shared, though: concurrent readers should use the
|
||||||
|
/// `addr`-taking methods rather than bind and then call these.
|
||||||
|
pub fn ensure_dataset(&self, addr: u64) -> bool {
|
||||||
|
let mut inner = self.lock();
|
||||||
|
let changed = inner.current != Some(addr);
|
||||||
|
inner.current = Some(addr);
|
||||||
|
changed
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Returns `true` if the bound dataset's chunk index has been built.
|
||||||
pub fn has_index(&self) -> bool {
|
pub fn has_index(&self) -> bool {
|
||||||
self.inner
|
let inner = self.lock();
|
||||||
.lock()
|
inner
|
||||||
.unwrap_or_else(|e| e.into_inner())
|
.entry(inner.current())
|
||||||
.index
|
.is_some_and(|e| e.index.is_some())
|
||||||
.is_some()
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Build the chunk index from a pre-collected list of `ChunkInfo`.
|
/// Build the bound dataset's chunk index from a pre-collected list of
|
||||||
|
/// `ChunkInfo`.
|
||||||
///
|
///
|
||||||
/// The `rank` parameter is used to truncate offsets to spatial dims only
|
/// The `rank` parameter is used to truncate offsets to spatial dims only
|
||||||
/// (B-tree v1 stores rank+1 offsets).
|
/// (B-tree v1 stores rank+1 offsets).
|
||||||
pub fn populate_index(&self, chunks: &[ChunkInfo], rank: usize) {
|
pub fn populate_index(&self, chunks: &[ChunkInfo], rank: usize) {
|
||||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let addr = self.lock().current();
|
||||||
if inner.index.is_some() {
|
let _ = self.index_for::<core::convert::Infallible>(addr, rank, || Ok(chunks.to_vec()));
|
||||||
return; // already populated
|
|
||||||
}
|
|
||||||
let mut map = HashMap::with_capacity(chunks.len());
|
|
||||||
|
|
||||||
for ci in chunks {
|
|
||||||
let coord: ChunkCoord = ci.offsets.iter().take(rank).copied().collect();
|
|
||||||
map.insert(coord, ci.clone());
|
|
||||||
}
|
|
||||||
inner.index = Some(map);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Look up a chunk by its spatial coordinate in the index.
|
/// Look up a chunk by its spatial coordinate in the bound dataset's index.
|
||||||
pub fn lookup_index(&self, coord: &[u64]) -> Option<ChunkInfo> {
|
pub fn lookup_index(&self, coord: &[u64]) -> Option<ChunkInfo> {
|
||||||
let inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let inner = self.lock();
|
||||||
inner.index.as_ref()?.get(coord).cloned()
|
inner
|
||||||
|
.entry(inner.current())?
|
||||||
|
.index
|
||||||
|
.as_ref()?
|
||||||
|
.get(coord)
|
||||||
|
.cloned()
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Return all indexed chunks as a `Vec<ChunkInfo>` (order unspecified).
|
/// Return all of the bound dataset's indexed chunks (order unspecified).
|
||||||
pub fn all_indexed_chunks(&self) -> Option<Vec<ChunkInfo>> {
|
pub fn all_indexed_chunks(&self) -> Option<Vec<ChunkInfo>> {
|
||||||
let inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let inner = self.lock();
|
||||||
inner.index.as_ref().map(|m| m.values().cloned().collect())
|
let index = inner.entry(inner.current())?.index.as_ref()?;
|
||||||
|
Some(index.values().cloned().collect())
|
||||||
}
|
}
|
||||||
|
|
||||||
// ----- Chunk index (pre-built coordinate → ChunkInfo map) -----
|
/// Returns `true` if the bound dataset's `ChunkIndex` has been built.
|
||||||
|
|
||||||
/// Returns `true` if the chunk B-tree index has been built.
|
|
||||||
pub fn has_chunk_index(&self) -> bool {
|
pub fn has_chunk_index(&self) -> bool {
|
||||||
self.inner
|
let inner = self.lock();
|
||||||
.lock()
|
inner
|
||||||
.unwrap_or_else(|e| e.into_inner())
|
.entry(inner.current())
|
||||||
.chunk_index
|
.is_some_and(|e| e.chunk_index.is_some())
|
||||||
.is_some()
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Build and store the chunk B-tree index from a pre-collected list of `ChunkInfo`.
|
/// Build and store the bound dataset's `ChunkIndex`.
|
||||||
pub fn populate_chunk_index(&self, chunks: &[ChunkInfo], rank: usize) {
|
pub fn populate_chunk_index(&self, chunks: &[ChunkInfo], rank: usize) {
|
||||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let built = Arc::new(ChunkIndex::build(chunks, rank));
|
||||||
if inner.chunk_index.is_some() {
|
let mut inner = self.lock();
|
||||||
return;
|
let addr = inner.current();
|
||||||
}
|
inner.touch(addr).chunk_index.get_or_insert(built);
|
||||||
inner.chunk_index = Some(ChunkIndex::build(chunks, rank));
|
inner.trim_datasets(addr);
|
||||||
}
|
}
|
||||||
|
|
||||||
// ----- Chunk layout (pre-computed assembly plan) -----
|
/// Returns `true` if the bound dataset's chunk layout has been computed.
|
||||||
|
|
||||||
/// Returns `true` if the chunk layout has been computed.
|
|
||||||
pub fn has_chunk_layout(&self) -> bool {
|
pub fn has_chunk_layout(&self) -> bool {
|
||||||
self.inner
|
let inner = self.lock();
|
||||||
.lock()
|
inner
|
||||||
.unwrap_or_else(|e| e.into_inner())
|
.entry(inner.current())
|
||||||
.chunk_layout
|
.is_some_and(|e| e.chunk_layout.is_some())
|
||||||
.is_some()
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Build and store the pre-computed chunk layout for fast assembly.
|
/// Build and store the bound dataset's chunk layout (needs its
|
||||||
|
/// `ChunkIndex`; does nothing without one).
|
||||||
pub fn populate_chunk_layout(&self, ds_dims: &[usize], chunk_dims: &[usize], elem_size: usize) {
|
pub fn populate_chunk_layout(&self, ds_dims: &[usize], chunk_dims: &[usize], elem_size: usize) {
|
||||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let mut inner = self.lock();
|
||||||
if inner.chunk_layout.is_some() {
|
let addr = inner.current();
|
||||||
|
let entry = inner.touch(addr);
|
||||||
|
if entry.chunk_layout.is_some() {
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
if let Some(ref idx) = inner.chunk_index {
|
if let Some(idx) = entry.chunk_index.clone() {
|
||||||
inner.chunk_layout = Some(ChunkLayout::build(idx, ds_dims, chunk_dims, elem_size));
|
entry.chunk_layout = Some(Arc::new(ChunkLayout::build(
|
||||||
|
&idx, ds_dims, chunk_dims, elem_size,
|
||||||
|
)));
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Execute a function with a reference to the chunk layout.
|
/// Execute a function with a reference to the bound dataset's chunk
|
||||||
///
|
/// layout. Returns `None` if the layout hasn't been computed yet.
|
||||||
/// Returns `None` if the layout hasn't been computed yet.
|
|
||||||
pub fn with_chunk_layout<F, R>(&self, f: F) -> Option<R>
|
pub fn with_chunk_layout<F, R>(&self, f: F) -> Option<R>
|
||||||
where
|
where
|
||||||
F: FnOnce(&ChunkLayout) -> R,
|
F: FnOnce(&ChunkLayout) -> R,
|
||||||
{
|
{
|
||||||
let inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let layout = {
|
||||||
inner.chunk_layout.as_ref().map(f)
|
let inner = self.lock();
|
||||||
|
inner.entry(inner.current())?.chunk_layout.clone()?
|
||||||
|
};
|
||||||
|
Some(f(&layout))
|
||||||
}
|
}
|
||||||
|
|
||||||
// ----- Decompressed data cache (LRU) -----
|
/// Try to get cached decompressed data for a chunk of the bound dataset.
|
||||||
|
|
||||||
/// Try to get cached decompressed data for a chunk coordinate.
|
|
||||||
///
|
///
|
||||||
/// O(1) lookup. Returns an owned copy for API compatibility with callers
|
/// Returns an owned copy; prefer [`Self::get_decompressed_aligned`] when
|
||||||
/// that need a `Vec<u8>`; prefer [`Self::get_decompressed_aligned`] when
|
/// an `Arc`-shared buffer works for the caller.
|
||||||
/// an `Arc`-shared buffer works for the caller, since that avoids the
|
|
||||||
/// copy entirely.
|
|
||||||
pub fn get_decompressed(&self, coord: &[u64]) -> Option<Vec<u8>> {
|
pub fn get_decompressed(&self, coord: &[u64]) -> Option<Vec<u8>> {
|
||||||
self.get_decompressed_aligned(coord)
|
self.get_decompressed_aligned(coord)
|
||||||
.map(|arc| arc.as_slice().to_vec())
|
.map(|arc| arc.as_slice().to_vec())
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Try to get a reference-counted clone of the aligned buffer for a chunk.
|
/// Reference-counted cached buffer for a chunk of the bound dataset.
|
||||||
///
|
|
||||||
/// O(1) index lookup; the clone is an `Arc` refcount bump, not a copy of
|
|
||||||
/// the underlying decompressed data.
|
|
||||||
pub fn get_decompressed_aligned(&self, coord: &[u64]) -> Option<Arc<CacheAlignedBuffer>> {
|
pub fn get_decompressed_aligned(&self, coord: &[u64]) -> Option<Arc<CacheAlignedBuffer>> {
|
||||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let mut inner = self.lock();
|
||||||
inner.tick += 1;
|
let addr = inner.current();
|
||||||
let tick = inner.tick;
|
inner.get_decompressed(addr, coord)
|
||||||
|
|
||||||
// Track sequential vs random access
|
|
||||||
let is_sequential = inner.last_coord.as_ref().is_some_and(|prev| {
|
|
||||||
// Sequential if exactly one dimension changed
|
|
||||||
let changes: usize = prev
|
|
||||||
.iter()
|
|
||||||
.zip(coord.iter())
|
|
||||||
.filter(|(a, b)| a != b)
|
|
||||||
.count();
|
|
||||||
changes <= 1
|
|
||||||
});
|
|
||||||
if is_sequential {
|
|
||||||
inner.stats.sequential_count += 1;
|
|
||||||
} else if inner.last_coord.is_some() {
|
|
||||||
inner.stats.random_count += 1;
|
|
||||||
}
|
|
||||||
inner.last_coord = Some(coord.to_vec());
|
|
||||||
|
|
||||||
let found = if let Some(&idx) = inner.slot_index.get(coord) {
|
|
||||||
inner.slots[idx].last_access = tick;
|
|
||||||
Some(Arc::clone(&inner.slots[idx].data))
|
|
||||||
} else {
|
|
||||||
None
|
|
||||||
};
|
|
||||||
if let Some(ref data) = found {
|
|
||||||
inner.stats.hits += 1;
|
|
||||||
inner.stats.bytes_read += data.len() as u64;
|
|
||||||
} else {
|
|
||||||
inner.stats.misses += 1;
|
|
||||||
}
|
|
||||||
found
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Insert decompressed chunk data into the LRU cache.
|
/// Insert decompressed chunk data for the bound dataset into the LRU
|
||||||
///
|
/// cache, returning the `Arc`-shared buffer now cached.
|
||||||
/// The data is stored in a [`CacheAlignedBuffer`] so subsequent reads
|
|
||||||
/// return cache-line-aligned memory. Returns the `Arc`-shared buffer that
|
|
||||||
/// is now cached (or already was), so the caller can reuse it directly
|
|
||||||
/// instead of holding a separate copy of the same data.
|
|
||||||
pub fn put_decompressed(&self, coord: ChunkCoord, data: Vec<u8>) -> Arc<CacheAlignedBuffer> {
|
pub fn put_decompressed(&self, coord: ChunkCoord, data: Vec<u8>) -> Arc<CacheAlignedBuffer> {
|
||||||
let aligned = CacheAlignedBuffer::from_vec(data);
|
self.put_decompressed_aligned(coord, CacheAlignedBuffer::from_vec(data))
|
||||||
self.put_decompressed_aligned(coord, aligned)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Insert an already-aligned buffer into the LRU cache.
|
/// Insert an already-aligned buffer for the bound dataset.
|
||||||
///
|
|
||||||
/// Returns the `Arc`-shared buffer now held by the cache (the one just
|
|
||||||
/// inserted, or the existing cached copy if `coord` was already present).
|
|
||||||
pub fn put_decompressed_aligned(
|
pub fn put_decompressed_aligned(
|
||||||
&self,
|
&self,
|
||||||
coord: ChunkCoord,
|
coord: ChunkCoord,
|
||||||
data: CacheAlignedBuffer,
|
data: CacheAlignedBuffer,
|
||||||
) -> Arc<CacheAlignedBuffer> {
|
) -> Arc<CacheAlignedBuffer> {
|
||||||
let data = Arc::new(data);
|
let data = Arc::new(data);
|
||||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let mut inner = self.lock();
|
||||||
let data_len = data.len();
|
let addr = inner.current();
|
||||||
|
inner.put_decompressed((addr, coord), data)
|
||||||
// Don't cache if single chunk exceeds budget — still return the data
|
|
||||||
// to the caller, just don't retain it.
|
|
||||||
if data_len > inner.max_bytes {
|
|
||||||
return data;
|
|
||||||
}
|
|
||||||
|
|
||||||
// Check if already present
|
|
||||||
inner.tick += 1;
|
|
||||||
let tick = inner.tick;
|
|
||||||
if let Some(&idx) = inner.slot_index.get(&coord) {
|
|
||||||
inner.slots[idx].last_access = tick;
|
|
||||||
return Arc::clone(&inner.slots[idx].data); // already cached
|
|
||||||
}
|
|
||||||
|
|
||||||
// Evict until we have room
|
|
||||||
while inner.slots.len() >= inner.max_slots
|
|
||||||
|| (inner.current_bytes + data_len > inner.max_bytes && !inner.slots.is_empty())
|
|
||||||
{
|
|
||||||
// Find LRU slot
|
|
||||||
let lru_idx = inner
|
|
||||||
.slots
|
|
||||||
.iter()
|
|
||||||
.enumerate()
|
|
||||||
.min_by_key(|(_, s)| s.last_access)
|
|
||||||
.map(|(i, _)| i)
|
|
||||||
.unwrap();
|
|
||||||
let removed = inner.slots.swap_remove(lru_idx);
|
|
||||||
inner.slot_index.remove(&removed.coord);
|
|
||||||
// swap_remove moved the former last element into `lru_idx` (unless
|
|
||||||
// it *was* the last element) — fix up that element's index entry.
|
|
||||||
if lru_idx < inner.slots.len() {
|
|
||||||
let moved_coord = inner.slots[lru_idx].coord.clone();
|
|
||||||
inner.slot_index.insert(moved_coord, lru_idx);
|
|
||||||
}
|
|
||||||
inner.current_bytes -= removed.data.len();
|
|
||||||
inner.stats.evictions += 1;
|
|
||||||
}
|
|
||||||
|
|
||||||
inner.current_bytes += data_len;
|
|
||||||
let new_idx = inner.slots.len();
|
|
||||||
inner.slot_index.insert(coord.clone(), new_idx);
|
|
||||||
inner.slots.push(CachedChunk {
|
|
||||||
coord,
|
|
||||||
data: Arc::clone(&data),
|
|
||||||
last_access: tick,
|
|
||||||
});
|
|
||||||
data
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Clear the entire cache (index + decompressed data).
|
/// [`Self::prefetch_hint_in`] for the bound dataset.
|
||||||
|
pub fn prefetch_hint(&self, next_coords: &[ChunkCoord]) {
|
||||||
|
let addr = self.lock().current();
|
||||||
|
self.prefetch_hint_in(addr, next_coords);
|
||||||
|
}
|
||||||
|
|
||||||
|
// ----- Whole-cache operations -----
|
||||||
|
|
||||||
|
/// Clear the entire cache (indexes + decompressed data + stats).
|
||||||
pub fn clear(&self) {
|
pub fn clear(&self) {
|
||||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let mut inner = self.lock();
|
||||||
inner.index = None;
|
inner.datasets.clear();
|
||||||
inner.index_addr = None;
|
inner.current = None;
|
||||||
inner.slots.clear();
|
inner.slots.clear();
|
||||||
inner.slot_index.clear();
|
inner.slot_index.clear();
|
||||||
inner.current_bytes = 0;
|
inner.current_bytes = 0;
|
||||||
inner.tick = 0;
|
inner.tick = 0;
|
||||||
inner.last_coord = None;
|
inner.last_coord = None;
|
||||||
inner.stats = AccessStats::default();
|
inner.stats = AccessStats::default();
|
||||||
inner.chunk_index = None;
|
|
||||||
inner.chunk_layout = None;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Record that the given chunk coordinates are predicted to be accessed
|
|
||||||
/// soon (bookkeeping only).
|
|
||||||
///
|
|
||||||
/// This does **not** prefetch or pre-decompress anything — it only
|
|
||||||
/// checks whether each coordinate is already in the chunk index and
|
|
||||||
/// updates access-pattern stats accordingly. Real prefetching (e.g.
|
|
||||||
/// background pre-decompression) is not implemented.
|
|
||||||
pub fn prefetch_hint(&self, next_coords: &[ChunkCoord]) {
|
|
||||||
let inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
|
||||||
if inner.index.is_none() {
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
drop(inner);
|
|
||||||
// For each predicted coordinate, verify it exists in the index.
|
|
||||||
// The index is already populated, so this is a no-op for known chunks.
|
|
||||||
// The purpose is to signal intent — callers can pre-decompress if needed.
|
|
||||||
// We touch the stats to record that prefetch hints were issued.
|
|
||||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
|
||||||
for coord in next_coords {
|
|
||||||
let exists = inner
|
|
||||||
.index
|
|
||||||
.as_ref()
|
|
||||||
.map(|idx| idx.contains_key(coord))
|
|
||||||
.unwrap_or(false);
|
|
||||||
if exists {
|
|
||||||
inner.stats.sequential_count += 1;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Return the current access pattern statistics.
|
/// Return the current access pattern statistics.
|
||||||
pub fn access_stats(&self) -> AccessStats {
|
pub fn access_stats(&self) -> AccessStats {
|
||||||
self.inner
|
self.lock().stats.clone()
|
||||||
.lock()
|
|
||||||
.unwrap_or_else(|e| e.into_inner())
|
|
||||||
.stats
|
|
||||||
.clone()
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Update the sweep direction label in the access stats.
|
/// Update the sweep direction label in the access stats.
|
||||||
pub fn set_sweep_direction(&self, direction: &'static str) {
|
pub fn set_sweep_direction(&self, direction: &'static str) {
|
||||||
self.inner
|
self.lock().stats.sweep_direction = Some(direction);
|
||||||
.lock()
|
|
||||||
.unwrap_or_else(|e| e.into_inner())
|
|
||||||
.stats
|
|
||||||
.sweep_direction = Some(direction);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Number of decompressed chunks currently cached.
|
/// Number of decompressed chunks currently cached (all datasets).
|
||||||
pub fn cached_chunk_count(&self) -> usize {
|
pub fn cached_chunk_count(&self) -> usize {
|
||||||
self.inner
|
self.lock().slots.len()
|
||||||
.lock()
|
|
||||||
.unwrap_or_else(|e| e.into_inner())
|
|
||||||
.slots
|
|
||||||
.len()
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Total bytes of decompressed data currently cached.
|
/// Total bytes of decompressed data currently cached (all datasets).
|
||||||
pub fn cached_bytes(&self) -> usize {
|
pub fn cached_bytes(&self) -> usize {
|
||||||
self.inner
|
self.lock().current_bytes
|
||||||
.lock()
|
}
|
||||||
.unwrap_or_else(|e| e.into_inner())
|
|
||||||
.current_bytes
|
/// Number of datasets whose chunk index is currently kept.
|
||||||
|
pub fn indexed_dataset_count(&self) -> usize {
|
||||||
|
self.lock().datasets.len()
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -808,6 +967,92 @@ mod tests {
|
|||||||
assert_eq!(cache.cached_bytes(), 0);
|
assert_eq!(cache.cached_bytes(), 0);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn datasets_sharing_coordinates_stay_separate() {
|
||||||
|
let cache = ChunkCache::new();
|
||||||
|
let a = vec![make_chunk(vec![0, 0], 0x100, 8)];
|
||||||
|
let b = vec![make_chunk(vec![0, 0], 0x900, 8)];
|
||||||
|
let got_a = cache.chunks_for::<()>(1, 1, || Ok(a.clone())).unwrap();
|
||||||
|
let got_b = cache.chunks_for::<()>(2, 1, || Ok(b.clone())).unwrap();
|
||||||
|
assert_eq!(got_a[0].address, 0x100);
|
||||||
|
assert_eq!(got_b[0].address, 0x900);
|
||||||
|
// Built once per dataset: a second lookup doesn't call the builder.
|
||||||
|
let again = cache
|
||||||
|
.chunks_for::<()>(1, 1, || panic!("index rebuilt"))
|
||||||
|
.unwrap();
|
||||||
|
assert_eq!(again[0].address, 0x100);
|
||||||
|
|
||||||
|
cache.put_decompressed_in(1, vec![0], vec![1; 4]);
|
||||||
|
cache.put_decompressed_in(2, vec![0], vec![2; 4]);
|
||||||
|
assert_eq!(
|
||||||
|
cache.get_decompressed_in(1, &[0]).unwrap().as_slice(),
|
||||||
|
&[1; 4]
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
cache.get_decompressed_in(2, &[0]).unwrap().as_slice(),
|
||||||
|
&[2; 4]
|
||||||
|
);
|
||||||
|
assert!(cache.get_decompressed_in(3, &[0]).is_none());
|
||||||
|
assert_eq!(cache.cached_chunk_count(), 2);
|
||||||
|
|
||||||
|
// The bound-dataset methods see only the bound dataset.
|
||||||
|
cache.ensure_dataset(2);
|
||||||
|
assert_eq!(cache.lookup_index(&[0]).unwrap().address, 0x900);
|
||||||
|
assert_eq!(cache.get_decompressed(&[0]).unwrap(), vec![2; 4]);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn dataset_indexes_are_bounded() {
|
||||||
|
let cache = ChunkCache::new();
|
||||||
|
for addr in 0..(MAX_INDEXED_DATASETS as u64 + 10) {
|
||||||
|
cache
|
||||||
|
.chunks_for::<()>(addr, 1, || Ok(vec![make_chunk(vec![0], addr, 8)]))
|
||||||
|
.unwrap();
|
||||||
|
}
|
||||||
|
assert_eq!(cache.indexed_dataset_count(), MAX_INDEXED_DATASETS);
|
||||||
|
|
||||||
|
// One huge index evicts the others but is itself kept.
|
||||||
|
let huge: Vec<ChunkInfo> = (0..MAX_INDEXED_CHUNKS as u64)
|
||||||
|
.map(|i| make_chunk(vec![i], i, 8))
|
||||||
|
.collect();
|
||||||
|
let got = cache.chunks_for::<()>(9999, 1, || Ok(huge)).unwrap();
|
||||||
|
assert_eq!(got.len(), MAX_INDEXED_CHUNKS);
|
||||||
|
assert_eq!(cache.indexed_dataset_count(), 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn concurrent_readers_of_different_datasets_see_their_own_chunks() {
|
||||||
|
let cache = std::sync::Arc::new(ChunkCache::with_capacity(1 << 20, 64));
|
||||||
|
let handles: Vec<_> = (0..8u64)
|
||||||
|
.map(|t| {
|
||||||
|
let cache = std::sync::Arc::clone(&cache);
|
||||||
|
std::thread::spawn(move || {
|
||||||
|
for round in 0..500u64 {
|
||||||
|
let addr = (t + round) % 16;
|
||||||
|
let coord = vec![round % 4];
|
||||||
|
let chunks = cache
|
||||||
|
.chunks_for::<()>(addr, 1, || {
|
||||||
|
Ok((0..4).map(|c| make_chunk(vec![c], addr, 8)).collect())
|
||||||
|
})
|
||||||
|
.unwrap();
|
||||||
|
assert!(chunks.iter().all(|c| c.address == addr));
|
||||||
|
let want = vec![addr as u8; 8];
|
||||||
|
let got = match cache.get_decompressed_in(addr, &coord) {
|
||||||
|
Some(hit) => hit.to_vec(),
|
||||||
|
None => cache
|
||||||
|
.put_decompressed_in(addr, coord, want.clone())
|
||||||
|
.to_vec(),
|
||||||
|
};
|
||||||
|
assert_eq!(got, want);
|
||||||
|
}
|
||||||
|
})
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
for h in handles {
|
||||||
|
h.join().unwrap();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn duplicate_insert_is_noop() {
|
fn duplicate_insert_is_noop() {
|
||||||
let cache = ChunkCache::new();
|
let cache = ChunkCache::new();
|
||||||
|
|||||||
@@ -0,0 +1,200 @@
|
|||||||
|
//! Chunk-index linearisation shared by the Fixed Array and Extensible Array
|
||||||
|
//! chunk indexes (reader and writer).
|
||||||
|
//!
|
||||||
|
//! Both indexes store one element per chunk at a *linear* index, and the
|
||||||
|
//! library derives that index from the chunk's scaled coordinates
|
||||||
|
//! (`offset / chunk_dim`) using the dataset's **maximum** dimensions, not its
|
||||||
|
//! current ones (`H5D__farray_idx_get_addr` / `H5D__earray_idx_get_addr`,
|
||||||
|
//! via `layout->max_down_chunks`). A dataset whose current shape is smaller
|
||||||
|
//! than its maxshape therefore has gaps in the index, and laying it out by the
|
||||||
|
//! current shape puts every chunk after the first row in the wrong place.
|
||||||
|
//!
|
||||||
|
//! The Extensible Array adds one more step: its one unlimited dimension has no
|
||||||
|
//! finite chunk count, so the library *swizzles* the coordinates to make that
|
||||||
|
//! dimension the slowest-varying one (`H5VM_swizzle_coords`, which moves
|
||||||
|
//! `coords[unlim_dim]` to the front and shifts the dimensions before it right
|
||||||
|
//! by one) before linearising with `swizzled_max_down_chunks`. When the
|
||||||
|
//! unlimited dimension is already dimension 0 no swizzle happens.
|
||||||
|
|
||||||
|
#[cfg(not(feature = "std"))]
|
||||||
|
extern crate alloc;
|
||||||
|
|
||||||
|
#[cfg(not(feature = "std"))]
|
||||||
|
use alloc::{vec, vec::Vec};
|
||||||
|
|
||||||
|
use crate::error::FormatError;
|
||||||
|
|
||||||
|
/// How a chunk index maps linear element indexes to chunk coordinates.
|
||||||
|
#[derive(Debug, Clone)]
|
||||||
|
pub(crate) struct ChunkGrid {
|
||||||
|
/// Spatial chunk dimensions, in dataset order.
|
||||||
|
chunk_dims: Vec<u64>,
|
||||||
|
/// Chunks per dimension covering the *current* extent, in dataset order.
|
||||||
|
cur_chunks: Vec<u64>,
|
||||||
|
/// Dataset dimension stored at each linearisation position (slowest
|
||||||
|
/// first). The identity except for a swizzled Extensible Array.
|
||||||
|
order: Vec<usize>,
|
||||||
|
/// Linear stride of each linearisation position.
|
||||||
|
down: Vec<u64>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl ChunkGrid {
|
||||||
|
/// Grid for a Fixed Array index: row-major over the chunk counts of the
|
||||||
|
/// maximum dimensions (`max_dims`, falling back to the current dimensions
|
||||||
|
/// when the dataspace records none).
|
||||||
|
pub(crate) fn fixed_array(
|
||||||
|
cur_dims: &[u64],
|
||||||
|
max_dims: Option<&[u64]>,
|
||||||
|
chunk_dims: &[u64],
|
||||||
|
) -> Result<Self, FormatError> {
|
||||||
|
Self::build(cur_dims, max_dims, chunk_dims, None)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Grid for an Extensible Array index: like the Fixed Array, but the
|
||||||
|
/// unlimited dimension (the one whose maximum is `H5S_UNLIMITED`) is moved
|
||||||
|
/// to the slowest-varying position first.
|
||||||
|
pub(crate) fn extensible_array(
|
||||||
|
cur_dims: &[u64],
|
||||||
|
max_dims: Option<&[u64]>,
|
||||||
|
chunk_dims: &[u64],
|
||||||
|
) -> Result<Self, FormatError> {
|
||||||
|
let unlim = max_dims.and_then(|m| m.iter().position(|&d| d == u64::MAX));
|
||||||
|
Self::build(cur_dims, max_dims, chunk_dims, unlim)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn build(
|
||||||
|
cur_dims: &[u64],
|
||||||
|
max_dims: Option<&[u64]>,
|
||||||
|
chunk_dims: &[u64],
|
||||||
|
unlim: Option<usize>,
|
||||||
|
) -> Result<Self, FormatError> {
|
||||||
|
let rank = chunk_dims.len();
|
||||||
|
if cur_dims.len() != rank || max_dims.is_some_and(|m| m.len() != rank) {
|
||||||
|
return Err(FormatError::ChunkedReadError(
|
||||||
|
"chunk index rank does not match the dataspace".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
if chunk_dims.contains(&0) {
|
||||||
|
return Err(FormatError::ChunkedReadError(
|
||||||
|
"chunk dimension is zero".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
let cur_chunks: Vec<u64> = cur_dims
|
||||||
|
.iter()
|
||||||
|
.zip(chunk_dims)
|
||||||
|
.map(|(&d, &c)| d.div_ceil(c))
|
||||||
|
.collect();
|
||||||
|
// Chunk counts of the maximum extent. An unlimited dimension has no
|
||||||
|
// finite count; it only ever sits in the slowest position, where its
|
||||||
|
// count never enters a stride. A (corrupt) maximum smaller than the
|
||||||
|
// current extent is widened so no allocated chunk becomes unreachable.
|
||||||
|
let max_chunks: Vec<u64> = (0..rank)
|
||||||
|
.map(|d| {
|
||||||
|
let max = max_dims.map_or(cur_dims[d], |m| m[d]);
|
||||||
|
if max == u64::MAX {
|
||||||
|
u64::MAX
|
||||||
|
} else {
|
||||||
|
max.div_ceil(chunk_dims[d]).max(cur_chunks[d])
|
||||||
|
}
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
|
||||||
|
let mut order: Vec<usize> = (0..rank).collect();
|
||||||
|
if let Some(u) = unlim {
|
||||||
|
order.remove(u);
|
||||||
|
order.insert(0, u);
|
||||||
|
}
|
||||||
|
let mut down = vec![1u64; rank];
|
||||||
|
for p in (0..rank.saturating_sub(1)).rev() {
|
||||||
|
let next = max_chunks[order[p + 1]];
|
||||||
|
if next == u64::MAX {
|
||||||
|
// Only reachable with more than one unlimited dimension, which
|
||||||
|
// neither index type can describe.
|
||||||
|
return Err(FormatError::ChunkedReadError(
|
||||||
|
"array chunk index with more than one unlimited dimension".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
down[p] = down[p + 1].checked_mul(next).ok_or_else(|| {
|
||||||
|
FormatError::Overflow("chunk index linear stride overflows u64".into())
|
||||||
|
})?;
|
||||||
|
}
|
||||||
|
Ok(Self {
|
||||||
|
chunk_dims: chunk_dims.to_vec(),
|
||||||
|
cur_chunks,
|
||||||
|
order,
|
||||||
|
down,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Dataset-space offsets of the chunk stored at linear `index`, or `None`
|
||||||
|
/// when that chunk lies outside the current extent (the index still has a
|
||||||
|
/// slot for it; the library ignores such chunks on read).
|
||||||
|
pub(crate) fn offsets(&self, index: u64) -> Option<Vec<u64>> {
|
||||||
|
let rank = self.chunk_dims.len();
|
||||||
|
let mut offsets = vec![0u64; rank];
|
||||||
|
let mut rem = index;
|
||||||
|
for p in 0..rank {
|
||||||
|
let d = self.order[p];
|
||||||
|
let scaled = rem / self.down[p];
|
||||||
|
rem %= self.down[p];
|
||||||
|
if scaled >= self.cur_chunks[d] {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
offsets[d] = scaled * self.chunk_dims[d];
|
||||||
|
}
|
||||||
|
Some(offsets)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Linear index of the chunk with scaled coordinates `scaled`
|
||||||
|
/// (`offset / chunk_dim` per dimension, in dataset order).
|
||||||
|
pub(crate) fn linear_index(&self, scaled: &[u64]) -> u64 {
|
||||||
|
self.order
|
||||||
|
.iter()
|
||||||
|
.zip(&self.down)
|
||||||
|
.map(|(&d, &stride)| scaled[d] * stride)
|
||||||
|
.sum()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn fixed_array_uses_max_dims() {
|
||||||
|
// shape (4, 6), chunks (2, 3), maxshape (20, 10): 10 x 4 chunk grid.
|
||||||
|
let g = ChunkGrid::fixed_array(&[4, 6], Some(&[20, 10]), &[2, 3]).unwrap();
|
||||||
|
assert_eq!(g.offsets(0), Some(vec![0, 0]));
|
||||||
|
assert_eq!(g.offsets(1), Some(vec![0, 3]));
|
||||||
|
assert_eq!(g.offsets(2), None); // column chunk 2 is beyond the extent
|
||||||
|
assert_eq!(g.offsets(4), Some(vec![2, 0]));
|
||||||
|
assert_eq!(g.offsets(5), Some(vec![2, 3]));
|
||||||
|
assert_eq!(g.offsets(8), None); // row chunk 2 is beyond the extent
|
||||||
|
assert_eq!(g.linear_index(&[1, 1]), 5);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn extensible_array_swizzles_unlimited_dim() {
|
||||||
|
// maxshape (10, None): dim 1 is unlimited and becomes slowest.
|
||||||
|
let g = ChunkGrid::extensible_array(&[4, 6], Some(&[10, u64::MAX]), &[2, 3]).unwrap();
|
||||||
|
// max chunks of dim 0 = 5, so index = c1 * 5 + c0.
|
||||||
|
assert_eq!(g.linear_index(&[1, 0]), 1);
|
||||||
|
assert_eq!(g.linear_index(&[0, 1]), 5);
|
||||||
|
assert_eq!(g.offsets(5), Some(vec![0, 3]));
|
||||||
|
assert_eq!(g.offsets(6), Some(vec![2, 3]));
|
||||||
|
assert_eq!(g.offsets(2), None);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn extensible_array_unlimited_first_is_row_major() {
|
||||||
|
let g = ChunkGrid::extensible_array(&[4, 6], Some(&[u64::MAX, 30]), &[2, 3]).unwrap();
|
||||||
|
// max chunks of dim 1 = 10.
|
||||||
|
assert_eq!(g.linear_index(&[1, 1]), 11);
|
||||||
|
assert_eq!(g.offsets(11), Some(vec![2, 3]));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn rejects_two_unlimited_dims_after_the_first() {
|
||||||
|
assert!(ChunkGrid::fixed_array(&[4, 6], Some(&[u64::MAX, u64::MAX]), &[2, 3]).is_err());
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -18,6 +18,7 @@ use alloc::collections::BTreeMap;
|
|||||||
#[cfg(feature = "std")]
|
#[cfg(feature = "std")]
|
||||||
use std::collections::HashMap;
|
use std::collections::HashMap;
|
||||||
|
|
||||||
|
use crate::addr::to_usize;
|
||||||
use crate::chunk_cache::ChunkCoord;
|
use crate::chunk_cache::ChunkCoord;
|
||||||
use crate::chunked_read::ChunkInfo;
|
use crate::chunked_read::ChunkInfo;
|
||||||
|
|
||||||
@@ -167,7 +168,15 @@ impl ChunkLayout {
|
|||||||
|
|
||||||
for (_coord, ci) in index.iter() {
|
for (_coord, ci) in index.iter() {
|
||||||
let coord: ChunkCoord = ci.offsets.iter().take(rank).copied().collect();
|
let coord: ChunkCoord = ci.offsets.iter().take(rank).copied().collect();
|
||||||
let chunk_offsets: Vec<usize> = coord.iter().map(|&o| o as usize).collect();
|
// `ds_dims` are `usize`: a chunk at an offset past `usize::MAX`
|
||||||
|
// (only on a 32-bit target) lies outside the dataset.
|
||||||
|
let Ok(chunk_offsets) = coord
|
||||||
|
.iter()
|
||||||
|
.map(|&o| to_usize(o))
|
||||||
|
.collect::<Result<Vec<usize>, _>>()
|
||||||
|
else {
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
|
||||||
let copies = if rank == 0 {
|
let copies = if rank == 0 {
|
||||||
// Scalar dataset — single copy
|
// Scalar dataset — single copy
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -1,12 +1,14 @@
|
|||||||
//! HDF5 Data Layout message parsing (message type 0x0008).
|
//! HDF5 Data Layout message parsing (message type 0x0008).
|
||||||
|
|
||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
use alloc::{string::String, vec::Vec};
|
use alloc::{format, string::String, vec::Vec};
|
||||||
|
|
||||||
#[cfg(feature = "std")]
|
#[cfg(feature = "std")]
|
||||||
use std::string::String;
|
use std::string::String;
|
||||||
|
|
||||||
|
use crate::addr::to_usize;
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
|
use crate::storage::Storage;
|
||||||
|
|
||||||
/// A single VDS (Virtual Dataset) source mapping.
|
/// A single VDS (Virtual Dataset) source mapping.
|
||||||
///
|
///
|
||||||
@@ -24,6 +26,34 @@ pub struct VdsMapping {
|
|||||||
pub virtual_selection: Vec<u8>,
|
pub virtual_selection: Vec<u8>,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Most dimensions a layout message can list (libhdf5 `H5O_LAYOUT_NDIMS`):
|
||||||
|
/// 32 dataspace dimensions plus the element size.
|
||||||
|
const MAX_LAYOUT_NDIMS: usize = 33;
|
||||||
|
|
||||||
|
/// libhdf5's checks on a chunked layout message's dimensions
|
||||||
|
/// (`H5O__layout_decode`): at most [`MAX_LAYOUT_NDIMS`], no dimension 0, and
|
||||||
|
/// before version 4 at least one dataspace dimension plus the element size.
|
||||||
|
/// A zero chunk dimension used to read the dataset as all fill values.
|
||||||
|
fn check_chunk_dims(dims: Vec<u32>, layout_version: u8) -> Result<Vec<u32>, FormatError> {
|
||||||
|
if dims.len() > MAX_LAYOUT_NDIMS {
|
||||||
|
return Err(FormatError::InvalidChunkDimensions(
|
||||||
|
"dimensionality is too large".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
if layout_version < 4 && dims.len() < 2 {
|
||||||
|
return Err(FormatError::InvalidChunkDimensions(
|
||||||
|
"bad dimensions for chunked storage".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
if let Some(u) = dims.iter().position(|&d| d == 0) {
|
||||||
|
return Err(FormatError::InvalidChunkDimensions(format!(
|
||||||
|
"bad chunk dimension value when parsing layout message - chunk dimension must be \
|
||||||
|
positive: mesg->u.chunk.dim[{u}] = 0"
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
Ok(dims)
|
||||||
|
}
|
||||||
|
|
||||||
/// Parsed HDF5 data layout message.
|
/// Parsed HDF5 data layout message.
|
||||||
#[derive(Debug, Clone, PartialEq)]
|
#[derive(Debug, Clone, PartialEq)]
|
||||||
pub enum DataLayout {
|
pub enum DataLayout {
|
||||||
@@ -45,7 +75,9 @@ pub enum DataLayout {
|
|||||||
chunk_dimensions: Vec<u32>,
|
chunk_dimensions: Vec<u32>,
|
||||||
/// B-tree address, or `None` if undefined.
|
/// B-tree address, or `None` if undefined.
|
||||||
btree_address: Option<u64>,
|
btree_address: Option<u64>,
|
||||||
/// Layout version (3 or 4).
|
/// Layout version (3 or 4). Version 1/2 messages (HDF5 1.4/1.6-era)
|
||||||
|
/// use the same version-1 B-tree chunk index as version 3 and are
|
||||||
|
/// reported as 3.
|
||||||
version: u8,
|
version: u8,
|
||||||
/// Chunk index type (v4 only).
|
/// Chunk index type (v4 only).
|
||||||
chunk_index_type: Option<u8>,
|
chunk_index_type: Option<u8>,
|
||||||
@@ -53,6 +85,11 @@ pub enum DataLayout {
|
|||||||
single_chunk_filtered_size: Option<u64>,
|
single_chunk_filtered_size: Option<u64>,
|
||||||
/// Filter mask for v4 single chunk with filters.
|
/// Filter mask for v4 single chunk with filters.
|
||||||
single_chunk_filter_mask: Option<u32>,
|
single_chunk_filter_mask: Option<u32>,
|
||||||
|
/// Layout v4 flag bit 0 (`H5D_CHUNK_DONT_FILTER_PARTIAL_CHUNKS`):
|
||||||
|
/// partial edge chunks — those extending past the dataset's current
|
||||||
|
/// extent in some dimension — are stored without the filter pipeline,
|
||||||
|
/// even though their filter mask is 0. Always `false` for v3.
|
||||||
|
dont_filter_partial_edge_chunks: bool,
|
||||||
},
|
},
|
||||||
/// Virtual dataset layout (v4 only).
|
/// Virtual dataset layout (v4 only).
|
||||||
Virtual {
|
Virtual {
|
||||||
@@ -67,21 +104,33 @@ pub enum DataLayout {
|
|||||||
},
|
},
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Version-1 VDS mapping flag: the source file name is stored by an earlier
|
||||||
|
/// entry, whose index follows in place of the name.
|
||||||
|
const VDS_SOURCE_FILE_SHARED: u8 = 0x01;
|
||||||
|
/// Version-1 VDS mapping flag: likewise for the source dataset name.
|
||||||
|
const VDS_SOURCE_DSET_SHARED: u8 = 0x02;
|
||||||
|
/// Version-1 VDS mapping flag: the source is in the virtual file itself
|
||||||
|
/// (`"."`); no file name is stored.
|
||||||
|
const VDS_SOURCE_SAME_FILE: u8 = 0x04;
|
||||||
|
const VDS_ALL_FLAGS: u8 = VDS_SOURCE_FILE_SHARED | VDS_SOURCE_DSET_SHARED | VDS_SOURCE_SAME_FILE;
|
||||||
|
|
||||||
/// Parse VDS mappings from global-heap object data.
|
/// Parse VDS mappings from global-heap object data.
|
||||||
///
|
///
|
||||||
/// The global-heap block holding a VDS mapping list is laid out as
|
/// The global-heap block holding a VDS mapping list is laid out as
|
||||||
/// (reverse-engineered and validated against HDF5 2.0):
|
/// (`H5D__virtual_store_layout` / `H5D__virtual_load_layout` in libhdf5):
|
||||||
///
|
///
|
||||||
/// ```text
|
/// ```text
|
||||||
/// version(1) · nused(length_size, LE) · entry[nused] · checksum(4)
|
/// version(1) · nused(length_size, LE) · entry[nused] · checksum(4)
|
||||||
/// ```
|
/// ```
|
||||||
///
|
///
|
||||||
/// Each entry is:
|
/// Each entry is:
|
||||||
/// - source file name — a null-terminated string in **block version 0**; in
|
/// - **block version 1 only:** a flags byte. `0x04`: the source is in the
|
||||||
/// **block version 1** a same-file reference is encoded as a single `0x04`
|
/// virtual file itself and no file name is stored; `0x01`/`0x02`: the
|
||||||
/// marker byte (the source file is the virtual file itself) in place of the
|
/// source file/dataset name is that of an earlier entry, whose index
|
||||||
/// name;
|
/// (`length_size` bytes) is stored instead of the name. libhdf5 2.0 writes
|
||||||
/// - source dataset name (null-terminated string);
|
/// version 1 when the file's low version bound is 2.0 and it saves space;
|
||||||
|
/// - source file name (null-terminated string, unless flagged above);
|
||||||
|
/// - source dataset name (null-terminated string, unless flagged above);
|
||||||
/// - source selection (serialized `H5S` dataspace selection — self-describing
|
/// - source selection (serialized `H5S` dataspace selection — self-describing
|
||||||
/// in length);
|
/// in length);
|
||||||
/// - virtual selection (serialized `H5S` dataspace selection).
|
/// - virtual selection (serialized `H5S` dataspace selection).
|
||||||
@@ -107,7 +156,7 @@ pub fn parse_vds_mappings(
|
|||||||
// `nused` is untrusted; don't pre-allocate from it. Each entry consumes at
|
// `nused` is untrusted; don't pre-allocate from it. Each entry consumes at
|
||||||
// least a few bytes, so the loop is naturally bounded by the heap data and
|
// least a few bytes, so the loop is naturally bounded by the heap data and
|
||||||
// a bogus `nused` simply errors out on the first short read.
|
// a bogus `nused` simply errors out on the first short read.
|
||||||
let mut mappings = Vec::new();
|
let mut mappings: Vec<VdsMapping> = Vec::new();
|
||||||
// Reads one self-describing selection at `pos`, returning its raw bytes and
|
// Reads one self-describing selection at `pos`, returning its raw bytes and
|
||||||
// advancing past it — bounds-checked so a corrupt selection can't overrun.
|
// advancing past it — bounds-checked so a corrupt selection can't overrun.
|
||||||
let read_selection = |heap_data: &[u8], pos: &mut usize| -> Result<Vec<u8>, FormatError> {
|
let read_selection = |heap_data: &[u8], pos: &mut usize| -> Result<Vec<u8>, FormatError> {
|
||||||
@@ -127,17 +176,57 @@ pub fn parse_vds_mappings(
|
|||||||
Ok(bytes)
|
Ok(bytes)
|
||||||
};
|
};
|
||||||
|
|
||||||
for _ in 0..nused {
|
if version > 1 {
|
||||||
// Source file name (with the version-1 same-file marker handled).
|
return Err(FormatError::ChunkedReadError(
|
||||||
let source_file = if version >= 1 && heap_data.get(pos) == Some(&0x04) {
|
"unsupported VDS mapping block version".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
for i in 0..nused {
|
||||||
|
// Version 1 prefixes each entry with a flags byte; a name may then be
|
||||||
|
// omitted (same file) or replaced by the index of an earlier entry
|
||||||
|
// holding the same name (`H5D__virtual_load_layout`).
|
||||||
|
let flags = if version >= 1 {
|
||||||
|
let f = *heap_data.get(pos).ok_or(FormatError::UnexpectedEof {
|
||||||
|
expected: pos + 1,
|
||||||
|
available: heap_data.len(),
|
||||||
|
})?;
|
||||||
pos += 1;
|
pos += 1;
|
||||||
|
if f & !VDS_ALL_FLAGS != 0 {
|
||||||
|
return Err(FormatError::ChunkedReadError(
|
||||||
|
"unknown VDS mapping flags".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
f
|
||||||
|
} else {
|
||||||
|
0
|
||||||
|
};
|
||||||
|
// Index of an earlier entry, for a shared name.
|
||||||
|
let earlier = |pos: &mut usize| -> Result<usize, FormatError> {
|
||||||
|
let idx = read_length(heap_data, *pos, length_size)?;
|
||||||
|
*pos += ls;
|
||||||
|
if idx >= i {
|
||||||
|
return Err(FormatError::ChunkedReadError(
|
||||||
|
"VDS mapping shares a name with a later entry".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
to_usize(idx)
|
||||||
|
};
|
||||||
|
|
||||||
|
let source_file = if flags & VDS_SOURCE_SAME_FILE != 0 {
|
||||||
String::from(".")
|
String::from(".")
|
||||||
|
} else if flags & VDS_SOURCE_FILE_SHARED != 0 {
|
||||||
|
let idx = earlier(&mut pos)?;
|
||||||
|
mappings[idx].source_file.clone()
|
||||||
} else {
|
} else {
|
||||||
read_null_terminated_string(heap_data, &mut pos)?
|
read_null_terminated_string(heap_data, &mut pos)?
|
||||||
};
|
};
|
||||||
|
|
||||||
// Source dataset name.
|
let source_dataset = if flags & VDS_SOURCE_DSET_SHARED != 0 {
|
||||||
let source_dataset = read_null_terminated_string(heap_data, &mut pos)?;
|
let idx = earlier(&mut pos)?;
|
||||||
|
mappings[idx].source_dataset.clone()
|
||||||
|
} else {
|
||||||
|
read_null_terminated_string(heap_data, &mut pos)?
|
||||||
|
};
|
||||||
|
|
||||||
// Source selection, then virtual selection (both self-describing length).
|
// Source selection, then virtual selection (both self-describing length).
|
||||||
let source_selection = read_selection(heap_data, &mut pos)?;
|
let source_selection = read_selection(heap_data, &mut pos)?;
|
||||||
@@ -222,6 +311,16 @@ impl DataLayout {
|
|||||||
&mut self,
|
&mut self,
|
||||||
file_data: &[u8],
|
file_data: &[u8],
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
|
) -> Result<(), FormatError> {
|
||||||
|
self.resolve_vds_mappings_in(file_data, length_size)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`Self::resolve_vds_mappings`] over any [`Storage`]: one read of the
|
||||||
|
/// global heap collection holding the mappings.
|
||||||
|
pub fn resolve_vds_mappings_in<S: Storage + ?Sized>(
|
||||||
|
&mut self,
|
||||||
|
file_data: &S,
|
||||||
|
length_size: u8,
|
||||||
) -> Result<(), FormatError> {
|
) -> Result<(), FormatError> {
|
||||||
if let DataLayout::Virtual {
|
if let DataLayout::Virtual {
|
||||||
global_heap_address,
|
global_heap_address,
|
||||||
@@ -231,11 +330,8 @@ impl DataLayout {
|
|||||||
} = self
|
} = self
|
||||||
&& let Some(addr) = *global_heap_address
|
&& let Some(addr) = *global_heap_address
|
||||||
{
|
{
|
||||||
let coll = crate::global_heap::GlobalHeapCollection::parse(
|
let coll =
|
||||||
file_data,
|
crate::global_heap::GlobalHeapCollection::parse_in(file_data, addr, length_size)?;
|
||||||
addr as usize,
|
|
||||||
length_size,
|
|
||||||
)?;
|
|
||||||
let obj = coll.get_object(*global_heap_index as u16).ok_or(
|
let obj = coll.get_object(*global_heap_index as u16).ok_or(
|
||||||
FormatError::GlobalHeapObjectNotFound {
|
FormatError::GlobalHeapObjectNotFound {
|
||||||
collection_address: addr,
|
collection_address: addr,
|
||||||
@@ -256,6 +352,7 @@ impl DataLayout {
|
|||||||
let layout_class = data[1];
|
let layout_class = data[1];
|
||||||
|
|
||||||
match version {
|
match version {
|
||||||
|
1 | 2 => Self::parse_v1_v2(data, offset_size),
|
||||||
3 => Self::parse_v3(data, layout_class, offset_size, length_size),
|
3 => Self::parse_v3(data, layout_class, offset_size, length_size),
|
||||||
// v5 (emitted by HDF5 1.14+/2.0 with `libver=latest`) uses the same
|
// v5 (emitted by HDF5 1.14+/2.0 with `libver=latest`) uses the same
|
||||||
// message structure as v4 — only the version number was bumped.
|
// message structure as v4 — only the version number was bumped.
|
||||||
@@ -264,6 +361,87 @@ impl DataLayout {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Layout message versions 1 and 2 (HDF5 before 1.6.3):
|
||||||
|
///
|
||||||
|
/// ```text
|
||||||
|
/// version(1) · dimensionality(1) · layout class(1) · reserved(5)
|
||||||
|
/// · address(offset_size) — contiguous and chunked only
|
||||||
|
/// · dimension sizes(4 × dimensionality)
|
||||||
|
/// · compact data size(4) · compact raw data — compact only
|
||||||
|
/// ```
|
||||||
|
///
|
||||||
|
/// The dimension sizes are the dataset's (contiguous/compact) or the
|
||||||
|
/// chunk's (chunked) extent plus a trailing element-size dimension, as in
|
||||||
|
/// version 3's chunked form. libhdf5 ignores them for contiguous storage
|
||||||
|
/// and sizes the data from the dataspace; the product of the stored
|
||||||
|
/// dimensions is that same size, and a disagreement (a dimension that was
|
||||||
|
/// truncated to 32 bits) is caught by the reader's size check rather than
|
||||||
|
/// returning wrong data.
|
||||||
|
fn parse_v1_v2(data: &[u8], offset_size: u8) -> Result<DataLayout, FormatError> {
|
||||||
|
ensure_len(data, 0, 8)?;
|
||||||
|
let dimensionality = data[1] as usize;
|
||||||
|
let layout_class = data[2];
|
||||||
|
// H5O_LAYOUT_NDIMS: 32 dataspace dimensions + the element-size one.
|
||||||
|
if dimensionality > 33 {
|
||||||
|
return Err(FormatError::Overflow(format!(
|
||||||
|
"data layout dimensionality {dimensionality} exceeds 33"
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
let mut p = 8;
|
||||||
|
let os = offset_size as usize;
|
||||||
|
let address = match layout_class {
|
||||||
|
1 | 2 => {
|
||||||
|
ensure_len(data, p, os)?;
|
||||||
|
let a = if is_undefined(data, p, offset_size) {
|
||||||
|
None
|
||||||
|
} else {
|
||||||
|
Some(read_offset(data, p, offset_size)?)
|
||||||
|
};
|
||||||
|
p += os;
|
||||||
|
a
|
||||||
|
}
|
||||||
|
0 => None,
|
||||||
|
_ => return Err(FormatError::InvalidLayoutClass(layout_class)),
|
||||||
|
};
|
||||||
|
ensure_len(data, p, dimensionality * 4)?;
|
||||||
|
let dims: Vec<u32> = data[p..p + dimensionality * 4]
|
||||||
|
.as_chunks::<4>()
|
||||||
|
.0
|
||||||
|
.iter()
|
||||||
|
.map(|c| u32::from_le_bytes(*c))
|
||||||
|
.collect();
|
||||||
|
p += dimensionality * 4;
|
||||||
|
match layout_class {
|
||||||
|
0 => {
|
||||||
|
ensure_len(data, p, 4)?;
|
||||||
|
let size =
|
||||||
|
u32::from_le_bytes([data[p], data[p + 1], data[p + 2], data[p + 3]]) as usize;
|
||||||
|
ensure_len(data, p + 4, size)?;
|
||||||
|
Ok(DataLayout::Compact {
|
||||||
|
data: data[p + 4..p + 4 + size].to_vec(),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
1 => {
|
||||||
|
let size = dims
|
||||||
|
.iter()
|
||||||
|
.try_fold(1u64, |acc, &d| acc.checked_mul(d as u64))
|
||||||
|
.ok_or_else(|| {
|
||||||
|
FormatError::Overflow(format!("contiguous layout size {dims:?}"))
|
||||||
|
})?;
|
||||||
|
Ok(DataLayout::Contiguous { address, size })
|
||||||
|
}
|
||||||
|
_ => Ok(DataLayout::Chunked {
|
||||||
|
chunk_dimensions: check_chunk_dims(dims, 2)?,
|
||||||
|
btree_address: address,
|
||||||
|
version: 3,
|
||||||
|
chunk_index_type: None,
|
||||||
|
single_chunk_filtered_size: None,
|
||||||
|
single_chunk_filter_mask: None,
|
||||||
|
dont_filter_partial_edge_chunks: false,
|
||||||
|
}),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
fn parse_v3(
|
fn parse_v3(
|
||||||
data: &[u8],
|
data: &[u8],
|
||||||
layout_class: u8,
|
layout_class: u8,
|
||||||
@@ -316,12 +494,13 @@ impl DataLayout {
|
|||||||
p += 4;
|
p += 4;
|
||||||
}
|
}
|
||||||
Ok(DataLayout::Chunked {
|
Ok(DataLayout::Chunked {
|
||||||
chunk_dimensions,
|
chunk_dimensions: check_chunk_dims(chunk_dimensions, 3)?,
|
||||||
btree_address,
|
btree_address,
|
||||||
version: 3,
|
version: 3,
|
||||||
chunk_index_type: None,
|
chunk_index_type: None,
|
||||||
single_chunk_filtered_size: None,
|
single_chunk_filtered_size: None,
|
||||||
single_chunk_filter_mask: None,
|
single_chunk_filter_mask: None,
|
||||||
|
dont_filter_partial_edge_chunks: false,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
_ => Err(FormatError::InvalidLayoutClass(layout_class)),
|
_ => Err(FormatError::InvalidLayoutClass(layout_class)),
|
||||||
@@ -364,47 +543,40 @@ impl DataLayout {
|
|||||||
let dimensionality = data[pos + 1] as usize;
|
let dimensionality = data[pos + 1] as usize;
|
||||||
let dim_size_encoded_length = data[pos + 2] as usize;
|
let dim_size_encoded_length = data[pos + 2] as usize;
|
||||||
let mut p = pos + 3;
|
let mut p = pos + 3;
|
||||||
|
if dimensionality > MAX_LAYOUT_NDIMS {
|
||||||
|
return Err(FormatError::InvalidChunkDimensions(
|
||||||
|
"dimensionality is too large".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
// dimension sizes
|
// Each dimension takes 1 to 8 bytes (libhdf5 writes the
|
||||||
|
// fewest that hold the largest one, so 3, 5, 6 and 7 occur:
|
||||||
|
// a chunk dimension of 70 000 takes 3). libhdf5 refuses 0
|
||||||
|
// and more than 8.
|
||||||
|
if dim_size_encoded_length == 0 || dim_size_encoded_length > 8 {
|
||||||
|
return Err(FormatError::InvalidChunkDimensions(
|
||||||
|
"encoded chunk dimension size is too large".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
ensure_len(data, p, dimensionality * dim_size_encoded_length)?;
|
ensure_len(data, p, dimensionality * dim_size_encoded_length)?;
|
||||||
let mut chunk_dimensions = Vec::with_capacity(dimensionality);
|
let mut chunk_dimensions = Vec::with_capacity(dimensionality);
|
||||||
for _ in 0..dimensionality {
|
for _ in 0..dimensionality {
|
||||||
let val = match dim_size_encoded_length {
|
let val = data[p..p + dim_size_encoded_length]
|
||||||
1 => data[p] as u32,
|
.iter()
|
||||||
2 => u16::from_le_bytes([data[p], data[p + 1]]) as u32,
|
.rev()
|
||||||
4 => u32::from_le_bytes([data[p], data[p + 1], data[p + 2], data[p + 3]]),
|
.fold(0u64, |acc, &b| (acc << 8) | u64::from(b));
|
||||||
8 => {
|
// Chunk dimensions are held as u32; HDF5 2.0 can write
|
||||||
// V4 chunked encodes dimension sizes as 8 bytes, but
|
// larger ones (layout version 5), which are refused
|
||||||
// our ChunkedStorageV4 stores them as u32. We read only
|
// rather than truncated.
|
||||||
// the low 4 bytes (little-endian). This silently
|
let val = u32::try_from(val).map_err(|_| {
|
||||||
// truncates dimensions > 4 GiB, which are not expected
|
FormatError::InvalidChunkDimensions(format!(
|
||||||
// in practice (HDF5 chunk dimensions are always small).
|
"chunk dimension {val} is larger than 2^32 - 1, which is not supported"
|
||||||
// If the high bytes are non-zero, the file is malformed
|
))
|
||||||
// or uses dimensions we cannot represent.
|
})?;
|
||||||
let high = u32::from_le_bytes([
|
|
||||||
data[p + 4],
|
|
||||||
data[p + 5],
|
|
||||||
data[p + 6],
|
|
||||||
data[p + 7],
|
|
||||||
]);
|
|
||||||
if high != 0 {
|
|
||||||
return Err(FormatError::UnexpectedEof {
|
|
||||||
expected: p + 8,
|
|
||||||
available: data.len(),
|
|
||||||
});
|
|
||||||
}
|
|
||||||
u32::from_le_bytes([data[p], data[p + 1], data[p + 2], data[p + 3]])
|
|
||||||
}
|
|
||||||
_ => {
|
|
||||||
return Err(FormatError::UnexpectedEof {
|
|
||||||
expected: p + dim_size_encoded_length,
|
|
||||||
available: data.len(),
|
|
||||||
});
|
|
||||||
}
|
|
||||||
};
|
|
||||||
chunk_dimensions.push(val);
|
chunk_dimensions.push(val);
|
||||||
p += dim_size_encoded_length;
|
p += dim_size_encoded_length;
|
||||||
}
|
}
|
||||||
|
let chunk_dimensions = check_chunk_dims(chunk_dimensions, 4)?;
|
||||||
|
|
||||||
// chunk index type
|
// chunk index type
|
||||||
ensure_len(data, p, 1)?;
|
ensure_len(data, p, 1)?;
|
||||||
@@ -505,6 +677,7 @@ impl DataLayout {
|
|||||||
chunk_index_type: Some(chunk_index_type),
|
chunk_index_type: Some(chunk_index_type),
|
||||||
single_chunk_filtered_size,
|
single_chunk_filtered_size,
|
||||||
single_chunk_filter_mask,
|
single_chunk_filter_mask,
|
||||||
|
dont_filter_partial_edge_chunks: flags & 0x01 != 0,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
3 => {
|
3 => {
|
||||||
@@ -539,6 +712,202 @@ impl DataLayout {
|
|||||||
mod tests {
|
mod tests {
|
||||||
use super::*;
|
use super::*;
|
||||||
|
|
||||||
|
/// Version 1/2 header: version, dimensionality, class, reserved(5).
|
||||||
|
fn v1v2_header(version: u8, ndims: u8, class: u8) -> Vec<u8> {
|
||||||
|
vec![version, ndims, class, 0, 0, 0, 0, 0]
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn v2_compact() {
|
||||||
|
let mut buf = v1v2_header(2, 2, 0);
|
||||||
|
// dims (3 elements of 2 bytes) — no address for compact
|
||||||
|
buf.extend_from_slice(&3u32.to_le_bytes());
|
||||||
|
buf.extend_from_slice(&2u32.to_le_bytes());
|
||||||
|
buf.extend_from_slice(&6u32.to_le_bytes()); // compact size (u32 in v1/v2)
|
||||||
|
buf.extend_from_slice(&[1, 0, 2, 0, 3, 0]);
|
||||||
|
assert_eq!(
|
||||||
|
DataLayout::parse(&buf, 8, 8).unwrap(),
|
||||||
|
DataLayout::Compact {
|
||||||
|
data: vec![1, 0, 2, 0, 3, 0]
|
||||||
|
}
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn v1_contiguous_size_from_dimensions() {
|
||||||
|
let mut buf = v1v2_header(1, 3, 1);
|
||||||
|
buf.extend_from_slice(&0x800u32.to_le_bytes()); // 4-byte address
|
||||||
|
for d in [10u32, 20, 4] {
|
||||||
|
buf.extend_from_slice(&d.to_le_bytes());
|
||||||
|
}
|
||||||
|
assert_eq!(
|
||||||
|
DataLayout::parse(&buf, 4, 4).unwrap(),
|
||||||
|
DataLayout::Contiguous {
|
||||||
|
address: Some(0x800),
|
||||||
|
size: 800,
|
||||||
|
}
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn v1_contiguous_undefined_address() {
|
||||||
|
let mut buf = v1v2_header(1, 2, 1);
|
||||||
|
buf.extend_from_slice(&[0xFF; 8]);
|
||||||
|
buf.extend_from_slice(&5u32.to_le_bytes());
|
||||||
|
buf.extend_from_slice(&8u32.to_le_bytes());
|
||||||
|
assert_eq!(
|
||||||
|
DataLayout::parse(&buf, 8, 8).unwrap(),
|
||||||
|
DataLayout::Contiguous {
|
||||||
|
address: None,
|
||||||
|
size: 40,
|
||||||
|
}
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn v1_chunked_maps_to_btree_v1_index() {
|
||||||
|
let mut buf = v1v2_header(1, 3, 2);
|
||||||
|
buf.extend_from_slice(&0x1234u64.to_le_bytes());
|
||||||
|
for d in [50u32, 50, 4] {
|
||||||
|
buf.extend_from_slice(&d.to_le_bytes());
|
||||||
|
}
|
||||||
|
assert_eq!(
|
||||||
|
DataLayout::parse(&buf, 8, 8).unwrap(),
|
||||||
|
DataLayout::Chunked {
|
||||||
|
chunk_dimensions: vec![50, 50, 4],
|
||||||
|
btree_address: Some(0x1234),
|
||||||
|
version: 3,
|
||||||
|
chunk_index_type: None,
|
||||||
|
single_chunk_filtered_size: None,
|
||||||
|
single_chunk_filter_mask: None,
|
||||||
|
dont_filter_partial_edge_chunks: false,
|
||||||
|
}
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A v3 chunked layout message with these dims (element size last).
|
||||||
|
fn v3_chunked_msg(dims: &[u32]) -> Vec<u8> {
|
||||||
|
let mut buf = vec![3u8, 2, dims.len() as u8];
|
||||||
|
buf.extend_from_slice(&0x1000u64.to_le_bytes());
|
||||||
|
for d in dims {
|
||||||
|
buf.extend_from_slice(&d.to_le_bytes());
|
||||||
|
}
|
||||||
|
buf
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn chunk_dimensions_are_checked_when_the_layout_is_parsed() {
|
||||||
|
assert!(DataLayout::parse(&v3_chunked_msg(&[4, 4, 8]), 8, 8).is_ok());
|
||||||
|
// A zero chunk dimension used to read as all fill values.
|
||||||
|
let err = DataLayout::parse(&v3_chunked_msg(&[4, 0, 8]), 8, 8).unwrap_err();
|
||||||
|
assert!(
|
||||||
|
matches!(&err, FormatError::InvalidChunkDimensions(m) if m.contains("dim[1] = 0")),
|
||||||
|
"{err:?}"
|
||||||
|
);
|
||||||
|
// Only the element-size dimension: libhdf5 "bad dimensions".
|
||||||
|
assert_eq!(
|
||||||
|
DataLayout::parse(&v3_chunked_msg(&[8]), 8, 8).unwrap_err(),
|
||||||
|
FormatError::InvalidChunkDimensions("bad dimensions for chunked storage".into())
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
DataLayout::parse(&v3_chunked_msg(&[1; 34]), 8, 8).unwrap_err(),
|
||||||
|
FormatError::InvalidChunkDimensions("dimensionality is too large".into())
|
||||||
|
);
|
||||||
|
// v1/v2 and v4 messages get the zero check too.
|
||||||
|
let mut v1 = v1v2_header(1, 2, 2);
|
||||||
|
v1.extend_from_slice(&0x1000u64.to_le_bytes());
|
||||||
|
v1.extend_from_slice(&0u32.to_le_bytes());
|
||||||
|
v1.extend_from_slice(&8u32.to_le_bytes());
|
||||||
|
assert!(matches!(
|
||||||
|
DataLayout::parse(&v1, 8, 8),
|
||||||
|
Err(FormatError::InvalidChunkDimensions(_))
|
||||||
|
));
|
||||||
|
let mut v4 = vec![4u8, 2, 0, 2, 4];
|
||||||
|
v4.extend_from_slice(&0u32.to_le_bytes());
|
||||||
|
v4.extend_from_slice(&8u32.to_le_bytes());
|
||||||
|
v4.push(3); // fixed array index
|
||||||
|
v4.push(0); // page bits
|
||||||
|
v4.extend_from_slice(&0x1000u64.to_le_bytes());
|
||||||
|
assert!(matches!(
|
||||||
|
DataLayout::parse(&v4, 8, 8),
|
||||||
|
Err(FormatError::InvalidChunkDimensions(_))
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A v4 chunked layout (fixed array index) whose `dims` are each
|
||||||
|
/// encoded in `width` bytes.
|
||||||
|
fn v4_chunked_msg(width: u8, dims: &[u64]) -> Vec<u8> {
|
||||||
|
let mut m = vec![4u8, 2, 0, dims.len() as u8, width];
|
||||||
|
for &d in dims {
|
||||||
|
m.extend_from_slice(&d.to_le_bytes()[..width.min(8) as usize]);
|
||||||
|
}
|
||||||
|
m.push(3); // fixed array index
|
||||||
|
m.push(0); // page bits
|
||||||
|
m.extend_from_slice(&0x1000u64.to_le_bytes());
|
||||||
|
m
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn v4_chunk_dimensions_take_1_to_8_bytes() {
|
||||||
|
// libhdf5 encodes each dimension in the fewest bytes that hold the
|
||||||
|
// largest: a chunk dimension of 70 000 takes 3, and 3, 5, 6 and 7
|
||||||
|
// were refused ("UnexpectedEof").
|
||||||
|
for width in 1..=8u8 {
|
||||||
|
let dims = [if width >= 3 { 70_000 } else { 200 }, 8];
|
||||||
|
let layout = DataLayout::parse(&v4_chunked_msg(width, &dims), 8, 8)
|
||||||
|
.unwrap_or_else(|e| panic!("width {width}: {e:?}"));
|
||||||
|
assert!(
|
||||||
|
matches!(&layout, DataLayout::Chunked { chunk_dimensions, .. }
|
||||||
|
if chunk_dimensions.iter().map(|&d| u64::from(d)).eq(dims)),
|
||||||
|
"width {width}: {layout:?}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
// libhdf5 refuses 0 and more than 8 bytes.
|
||||||
|
for width in [0u8, 9] {
|
||||||
|
assert_eq!(
|
||||||
|
DataLayout::parse(&v4_chunked_msg(width, &[4, 8]), 8, 8).unwrap_err(),
|
||||||
|
FormatError::InvalidChunkDimensions(
|
||||||
|
"encoded chunk dimension size is too large".into()
|
||||||
|
)
|
||||||
|
);
|
||||||
|
}
|
||||||
|
// A dimension past u32 cannot be represented and is refused, not
|
||||||
|
// truncated.
|
||||||
|
assert!(matches!(
|
||||||
|
DataLayout::parse(&v4_chunked_msg(5, &[1 << 32, 8]), 8, 8),
|
||||||
|
Err(FormatError::InvalidChunkDimensions(m)) if m.contains("2^32")
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn v1v2_rejects_bad_class_dimensionality_and_truncation() {
|
||||||
|
assert_eq!(
|
||||||
|
DataLayout::parse(&v1v2_header(1, 1, 3), 8, 8).unwrap_err(),
|
||||||
|
FormatError::InvalidLayoutClass(3)
|
||||||
|
);
|
||||||
|
assert!(matches!(
|
||||||
|
DataLayout::parse(&v1v2_header(2, 34, 1), 8, 8).unwrap_err(),
|
||||||
|
FormatError::Overflow(_)
|
||||||
|
));
|
||||||
|
// Chunked, dims cut short.
|
||||||
|
let mut buf = v1v2_header(1, 2, 2);
|
||||||
|
buf.extend_from_slice(&0x10u64.to_le_bytes());
|
||||||
|
buf.extend_from_slice(&7u32.to_le_bytes());
|
||||||
|
assert!(matches!(
|
||||||
|
DataLayout::parse(&buf, 8, 8).unwrap_err(),
|
||||||
|
FormatError::UnexpectedEof { .. }
|
||||||
|
));
|
||||||
|
// Compact, raw data shorter than its declared size.
|
||||||
|
let mut buf = v1v2_header(2, 1, 0);
|
||||||
|
buf.extend_from_slice(&4u32.to_le_bytes());
|
||||||
|
buf.extend_from_slice(&100u32.to_le_bytes());
|
||||||
|
buf.extend_from_slice(&[0; 4]);
|
||||||
|
assert!(matches!(
|
||||||
|
DataLayout::parse(&buf, 8, 8).unwrap_err(),
|
||||||
|
FormatError::UnexpectedEof { .. }
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn v3_compact() {
|
fn v3_compact() {
|
||||||
let mut buf = vec![3u8, 0]; // version=3, class=0 (compact)
|
let mut buf = vec![3u8, 0]; // version=3, class=0 (compact)
|
||||||
@@ -602,6 +971,7 @@ mod tests {
|
|||||||
chunk_index_type: None,
|
chunk_index_type: None,
|
||||||
single_chunk_filtered_size: None,
|
single_chunk_filtered_size: None,
|
||||||
single_chunk_filter_mask: None,
|
single_chunk_filter_mask: None,
|
||||||
|
dont_filter_partial_edge_chunks: false,
|
||||||
}
|
}
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
@@ -679,10 +1049,35 @@ mod tests {
|
|||||||
chunk_index_type: Some(1),
|
chunk_index_type: Some(1),
|
||||||
single_chunk_filtered_size: None,
|
single_chunk_filtered_size: None,
|
||||||
single_chunk_filter_mask: None,
|
single_chunk_filter_mask: None,
|
||||||
|
dont_filter_partial_edge_chunks: false,
|
||||||
}
|
}
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn v4_chunked_dont_filter_partial_edge_chunks_flag() {
|
||||||
|
let mut buf = vec![4u8, 2]; // version=4, class=2
|
||||||
|
buf.push(0x01); // flags bit 0 = don't filter partial edge chunks
|
||||||
|
buf.push(2); // dimensionality=2
|
||||||
|
buf.push(4); // dim_size_encoded_length=4
|
||||||
|
buf.extend_from_slice(&5u32.to_le_bytes());
|
||||||
|
buf.extend_from_slice(&4u32.to_le_bytes());
|
||||||
|
buf.push(3); // Fixed Array
|
||||||
|
buf.push(10); // max_dblk_page_nelmts_bits
|
||||||
|
buf.extend_from_slice(&0x3000u64.to_le_bytes());
|
||||||
|
match DataLayout::parse(&buf, 8, 8).unwrap() {
|
||||||
|
DataLayout::Chunked {
|
||||||
|
dont_filter_partial_edge_chunks,
|
||||||
|
btree_address,
|
||||||
|
..
|
||||||
|
} => {
|
||||||
|
assert!(dont_filter_partial_edge_chunks);
|
||||||
|
assert_eq!(btree_address, Some(0x3000));
|
||||||
|
}
|
||||||
|
other => panic!("expected Chunked, got {other:?}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn v4_chunked_single_chunk_with_filters() {
|
fn v4_chunked_single_chunk_with_filters() {
|
||||||
let mut buf = vec![4u8, 2]; // version=4, class=2
|
let mut buf = vec![4u8, 2]; // version=4, class=2
|
||||||
@@ -705,6 +1100,7 @@ mod tests {
|
|||||||
chunk_index_type: Some(1),
|
chunk_index_type: Some(1),
|
||||||
single_chunk_filtered_size: Some(1024),
|
single_chunk_filtered_size: Some(1024),
|
||||||
single_chunk_filter_mask: Some(0),
|
single_chunk_filter_mask: Some(0),
|
||||||
|
dont_filter_partial_edge_chunks: false,
|
||||||
}
|
}
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
@@ -815,6 +1211,62 @@ mod tests {
|
|||||||
assert_eq!(v1.iter_linear_1d(8).unwrap(), vec![4, 5, 6, 7]);
|
assert_eq!(v1.iter_linear_1d(8).unwrap(), vec![4, 5, 6, 7]);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn parse_vds_mappings_v1_shared_names() {
|
||||||
|
// Written by HDF5 2.0 (h5py, libver=("v200", "v200")) for three
|
||||||
|
// mappings from `a_rather_long_source_file.h5:a_rather_long_dataset_name`
|
||||||
|
// and one from the same file: the entries carry flags 0x00, 0x03, 0x03
|
||||||
|
// and 0x06, so names after the first are stored as entry indices.
|
||||||
|
let blob: &[u8] = &[
|
||||||
|
0x01, 0x04, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x61, 0x5f, 0x72, 0x61,
|
||||||
|
0x74, 0x68, 0x65, 0x72, 0x5f, 0x6c, 0x6f, 0x6e, 0x67, 0x5f, 0x73, 0x6f, 0x75, 0x72,
|
||||||
|
0x63, 0x65, 0x5f, 0x66, 0x69, 0x6c, 0x65, 0x2e, 0x68, 0x35, 0x00, 0x61, 0x5f, 0x72,
|
||||||
|
0x61, 0x74, 0x68, 0x65, 0x72, 0x5f, 0x6c, 0x6f, 0x6e, 0x67, 0x5f, 0x64, 0x61, 0x74,
|
||||||
|
0x61, 0x73, 0x65, 0x74, 0x5f, 0x6e, 0x61, 0x6d, 0x65, 0x00, 0x02, 0x00, 0x00, 0x00,
|
||||||
|
0x03, 0x00, 0x00, 0x00, 0x01, 0x02, 0x01, 0x00, 0x00, 0x00, 0x00, 0x00, 0x01, 0x00,
|
||||||
|
0x01, 0x00, 0x04, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00, 0x00, 0x00, 0x01, 0x02,
|
||||||
|
0x02, 0x00, 0x00, 0x00, 0x00, 0x00, 0x01, 0x00, 0x01, 0x00, 0x01, 0x00, 0x00, 0x00,
|
||||||
|
0x01, 0x00, 0x01, 0x00, 0x04, 0x00, 0x03, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
|
||||||
|
0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03,
|
||||||
|
0x00, 0x00, 0x00, 0x01, 0x02, 0x01, 0x00, 0x00, 0x00, 0x04, 0x00, 0x01, 0x00, 0x01,
|
||||||
|
0x00, 0x04, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00, 0x00, 0x00, 0x01, 0x02, 0x02,
|
||||||
|
0x00, 0x00, 0x00, 0x01, 0x00, 0x01, 0x00, 0x01, 0x00, 0x01, 0x00, 0x00, 0x00, 0x01,
|
||||||
|
0x00, 0x01, 0x00, 0x04, 0x00, 0x03, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
|
||||||
|
0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00,
|
||||||
|
0x00, 0x00, 0x01, 0x02, 0x01, 0x00, 0x00, 0x00, 0x08, 0x00, 0x01, 0x00, 0x01, 0x00,
|
||||||
|
0x04, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00, 0x00, 0x00, 0x01, 0x02, 0x02, 0x00,
|
||||||
|
0x00, 0x00, 0x02, 0x00, 0x01, 0x00, 0x01, 0x00, 0x01, 0x00, 0x00, 0x00, 0x01, 0x00,
|
||||||
|
0x01, 0x00, 0x04, 0x00, 0x06, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x02,
|
||||||
|
0x00, 0x00, 0x00, 0x03, 0x00, 0x00, 0x00, 0x01, 0x02, 0x01, 0x00, 0x00, 0x00, 0x00,
|
||||||
|
0x00, 0x01, 0x00, 0x01, 0x00, 0x04, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00, 0x00,
|
||||||
|
0x00, 0x01, 0x02, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00, 0x01, 0x00, 0x01, 0x00, 0x01,
|
||||||
|
0x00, 0x00, 0x00, 0x01, 0x00, 0x01, 0x00, 0x04, 0x00, 0x8e, 0xa7, 0xea, 0x7a,
|
||||||
|
];
|
||||||
|
let mappings = parse_vds_mappings(blob, 8).unwrap();
|
||||||
|
let names: Vec<(&str, &str)> = mappings
|
||||||
|
.iter()
|
||||||
|
.map(|m| (m.source_file.as_str(), m.source_dataset.as_str()))
|
||||||
|
.collect();
|
||||||
|
let (file, dset) = ("a_rather_long_source_file.h5", "a_rather_long_dataset_name");
|
||||||
|
assert_eq!(
|
||||||
|
names,
|
||||||
|
vec![(file, dset), (file, dset), (file, dset), (".", dset)]
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn parse_vds_mappings_v1_forward_reference_is_error() {
|
||||||
|
// Entry 0 claiming to share entry 0's file name must not index past
|
||||||
|
// the entries decoded so far.
|
||||||
|
let mut blob = vec![0x01u8, 1, 0, 0, 0, 0, 0, 0, 0, 0x01];
|
||||||
|
blob.extend_from_slice(&[0u8; 8]);
|
||||||
|
blob.extend_from_slice(b"d\0");
|
||||||
|
assert!(parse_vds_mappings(&blob, 8).is_err());
|
||||||
|
// Unknown flag bits are refused.
|
||||||
|
let blob = [0x01u8, 1, 0, 0, 0, 0, 0, 0, 0, 0x08, b'd', 0];
|
||||||
|
assert!(parse_vds_mappings(&blob, 8).is_err());
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn parse_vds_mappings_external_v0() {
|
fn parse_vds_mappings_external_v0() {
|
||||||
// Block version 0 with an explicit (external) source file name.
|
// Block version 0 with an explicit (external) source file name.
|
||||||
@@ -862,4 +1314,44 @@ mod tests {
|
|||||||
let blob = [0x01u8, 0, 0, 0, 0, 0, 0, 0, 0];
|
let blob = [0x01u8, 0, 0, 0, 0, 0, 0, 0, 0];
|
||||||
assert!(parse_vds_mappings(&blob, 8).unwrap().is_empty());
|
assert!(parse_vds_mappings(&blob, 8).unwrap().is_empty());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// A virtual dataset's mappings resolve identically through a
|
||||||
|
/// read_at-only CountingStorage, in two reads of the global heap.
|
||||||
|
#[test]
|
||||||
|
fn vds_mappings_through_storage_match_slice() {
|
||||||
|
use crate::message_type::MessageType;
|
||||||
|
use crate::object_header::ObjectHeader;
|
||||||
|
use crate::storage::CountingStorage;
|
||||||
|
let file: &[u8] = include_bytes!("../tests/fixtures/vds_same_file.h5");
|
||||||
|
let sb = crate::superblock::Superblock::parse(file, 0).unwrap();
|
||||||
|
let (os, ls) = (sb.offset_size, sb.length_size);
|
||||||
|
let storage = CountingStorage::new(file.to_vec());
|
||||||
|
let mut virtuals = 0;
|
||||||
|
for child in
|
||||||
|
crate::group_v2::resolve_group_children(file, &sb, sb.root_group_address).unwrap()
|
||||||
|
{
|
||||||
|
let h =
|
||||||
|
ObjectHeader::parse(file, child.object_header_address as usize, os, ls).unwrap();
|
||||||
|
let Some(msg) = h
|
||||||
|
.messages
|
||||||
|
.iter()
|
||||||
|
.find(|m| m.msg_type == MessageType::DataLayout)
|
||||||
|
else {
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
let mut want = DataLayout::parse(&msg.data, os, ls).unwrap();
|
||||||
|
if !matches!(want, DataLayout::Virtual { .. }) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let mut got = want.clone();
|
||||||
|
want.resolve_vds_mappings(file, ls).unwrap();
|
||||||
|
storage.reset();
|
||||||
|
got.resolve_vds_mappings_in(&storage, ls).unwrap();
|
||||||
|
assert_eq!(format!("{got:?}"), format!("{want:?}"));
|
||||||
|
assert!(matches!(&got, DataLayout::Virtual { mappings, .. } if !mappings.is_empty()));
|
||||||
|
assert_eq!(storage.reads(), 2);
|
||||||
|
virtuals += 1;
|
||||||
|
}
|
||||||
|
assert!(virtuals >= 1);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
+1093
-509
File diff suppressed because it is too large
Load Diff
@@ -7,6 +7,9 @@ use alloc::vec::Vec;
|
|||||||
|
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
|
|
||||||
|
/// Most dimensions a dataspace can have (`H5S_MAX_RANK`).
|
||||||
|
pub const MAX_RANK: u8 = 32;
|
||||||
|
|
||||||
/// Type of dataspace.
|
/// Type of dataspace.
|
||||||
#[derive(Debug, Clone, PartialEq)]
|
#[derive(Debug, Clone, PartialEq)]
|
||||||
pub enum DataspaceType {
|
pub enum DataspaceType {
|
||||||
@@ -67,6 +70,12 @@ impl Dataspace {
|
|||||||
let version = data[0];
|
let version = data[0];
|
||||||
let rank = data[1];
|
let rank = data[1];
|
||||||
let flags = data[2];
|
let flags = data[2];
|
||||||
|
// H5O__sdspace_decode's checks.
|
||||||
|
if rank > MAX_RANK {
|
||||||
|
return Err(FormatError::InvalidDataspace(
|
||||||
|
"simple dataspace dimensionality is too large",
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
let (space_type, header_size) = match version {
|
let (space_type, header_size) = match version {
|
||||||
1 => {
|
1 => {
|
||||||
@@ -88,6 +97,11 @@ impl Dataspace {
|
|||||||
2 => DataspaceType::Null,
|
2 => DataspaceType::Null,
|
||||||
_ => return Err(FormatError::InvalidDataspaceType(type_byte)),
|
_ => return Err(FormatError::InvalidDataspaceType(type_byte)),
|
||||||
};
|
};
|
||||||
|
if st != DataspaceType::Simple && rank > 0 {
|
||||||
|
return Err(FormatError::InvalidDataspace(
|
||||||
|
"invalid rank for scalar or NULL dataspace",
|
||||||
|
));
|
||||||
|
}
|
||||||
(st, 4usize)
|
(st, 4usize)
|
||||||
}
|
}
|
||||||
_ => return Err(FormatError::InvalidDataspaceVersion(version)),
|
_ => return Err(FormatError::InvalidDataspaceVersion(version)),
|
||||||
@@ -107,8 +121,13 @@ impl Dataspace {
|
|||||||
// Read max dimensions if flags bit 0 is set
|
// Read max dimensions if flags bit 0 is set
|
||||||
let max_dimensions = if flags & 0x01 != 0 {
|
let max_dimensions = if flags & 0x01 != 0 {
|
||||||
let mut max_dims = Vec::with_capacity(rank as usize);
|
let mut max_dims = Vec::with_capacity(rank as usize);
|
||||||
for _ in 0..rank {
|
for &dim in &dimensions {
|
||||||
let val = read_length(data, pos, length_size)?;
|
let val = read_length(data, pos, length_size)?;
|
||||||
|
if dim > val {
|
||||||
|
return Err(FormatError::InvalidDataspace(
|
||||||
|
"dataspace dimension size is greater than its maximum size",
|
||||||
|
));
|
||||||
|
}
|
||||||
max_dims.push(val);
|
max_dims.push(val);
|
||||||
pos += ls;
|
pos += ls;
|
||||||
}
|
}
|
||||||
@@ -176,7 +195,6 @@ impl Dataspace {
|
|||||||
match self.space_type {
|
match self.space_type {
|
||||||
DataspaceType::Null => Ok(0),
|
DataspaceType::Null => Ok(0),
|
||||||
DataspaceType::Scalar => Ok(1),
|
DataspaceType::Scalar => Ok(1),
|
||||||
DataspaceType::Simple if self.dimensions.is_empty() => Ok(0),
|
|
||||||
DataspaceType::Simple => self
|
DataspaceType::Simple => self
|
||||||
.dimensions
|
.dimensions
|
||||||
.iter()
|
.iter()
|
||||||
@@ -195,18 +213,14 @@ impl Dataspace {
|
|||||||
match self.space_type {
|
match self.space_type {
|
||||||
DataspaceType::Null => 0,
|
DataspaceType::Null => 0,
|
||||||
DataspaceType::Scalar => 1,
|
DataspaceType::Scalar => 1,
|
||||||
DataspaceType::Simple => {
|
// A simple dataspace of rank 0 holds one element, as in libhdf5
|
||||||
if self.dimensions.is_empty() {
|
// (the product of no dimensions). Saturate rather than wrap: a
|
||||||
0
|
// wrapped product could under-size a buffer. Size-critical
|
||||||
} else {
|
// callers use `checked_num_elements`.
|
||||||
// Saturate rather than wrap: a wrapped product could
|
DataspaceType::Simple => self
|
||||||
// under-size a buffer. Size-critical callers use
|
.dimensions
|
||||||
// `checked_num_elements`.
|
.iter()
|
||||||
self.dimensions
|
.fold(1u64, |acc, &d| acc.saturating_mul(d)),
|
||||||
.iter()
|
|
||||||
.fold(1u64, |acc, &d| acc.saturating_mul(d))
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -352,4 +366,44 @@ mod tests {
|
|||||||
let ds = Dataspace::parse(&data, 8).unwrap();
|
let ds = Dataspace::parse(&data, 8).unwrap();
|
||||||
assert_eq!(ds.max_dimensions, Some(vec![10]));
|
assert_eq!(ds.max_dimensions, Some(vec![10]));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// A simple dataspace of rank 0 (cve-2020-18494's `/dset1`) holds one
|
||||||
|
/// element in libhdf5, which h5py reads as shape `()`. It was 0.
|
||||||
|
#[test]
|
||||||
|
fn simple_rank_zero_holds_one_element() {
|
||||||
|
let data = build_v2_dataspace(0, 0, 1, &[], None);
|
||||||
|
let ds = Dataspace::parse(&data, 8).unwrap();
|
||||||
|
assert_eq!(ds.space_type, DataspaceType::Simple);
|
||||||
|
assert_eq!(ds.num_elements(), 1);
|
||||||
|
assert_eq!(ds.checked_num_elements().unwrap(), 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `H5O__sdspace_decode`'s checks.
|
||||||
|
#[test]
|
||||||
|
fn refuses_what_libhdf5_refuses() {
|
||||||
|
let too_many = build_v2_dataspace(33, 0, 1, &[1; 33], None);
|
||||||
|
assert!(matches!(
|
||||||
|
Dataspace::parse(&too_many, 8),
|
||||||
|
Err(FormatError::InvalidDataspace(_))
|
||||||
|
));
|
||||||
|
let scalar_with_rank = build_v2_dataspace(1, 0, 0, &[4], None);
|
||||||
|
assert!(matches!(
|
||||||
|
Dataspace::parse(&scalar_with_rank, 8),
|
||||||
|
Err(FormatError::InvalidDataspace(_))
|
||||||
|
));
|
||||||
|
let null_with_rank = build_v2_dataspace(1, 0, 2, &[4], None);
|
||||||
|
assert!(matches!(
|
||||||
|
Dataspace::parse(&null_with_rank, 8),
|
||||||
|
Err(FormatError::InvalidDataspace(_))
|
||||||
|
));
|
||||||
|
let over_max = build_v1_dataspace(2, 0x01, &[5, 20], Some(&[10, 10]));
|
||||||
|
assert!(matches!(
|
||||||
|
Dataspace::parse(&over_max, 8),
|
||||||
|
Err(FormatError::InvalidDataspace(_))
|
||||||
|
));
|
||||||
|
// 32 dimensions, and a size equal to the maximum or unlimited, are fine.
|
||||||
|
assert!(Dataspace::parse(&build_v2_dataspace(32, 0, 1, &[1; 32], None), 8).is_ok());
|
||||||
|
let at_max = build_v1_dataspace(2, 0x01, &[10, 20], Some(&[10, u64::MAX]));
|
||||||
|
assert!(Dataspace::parse(&at_max, 8).is_ok());
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -3,11 +3,14 @@
|
|||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
extern crate alloc;
|
extern crate alloc;
|
||||||
|
|
||||||
|
use crate::addr::saturating_usize;
|
||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
use alloc::{vec, vec::Vec};
|
use alloc::{vec, vec::Vec};
|
||||||
|
|
||||||
use crate::checksum::jenkins_lookup3;
|
use crate::checksum::jenkins_lookup3;
|
||||||
use crate::chunked_write::WrittenChunk;
|
use crate::chunked_write::{
|
||||||
|
WrittenChunk, filtered_chunk_size_len, push_addr, push_index_element, push_v4_chunk_dims,
|
||||||
|
};
|
||||||
|
|
||||||
/// Serialize a v4 Extensible Array layout message.
|
/// Serialize a v4 Extensible Array layout message.
|
||||||
pub(crate) fn serialize_v4_extensible_array(
|
pub(crate) fn serialize_v4_extensible_array(
|
||||||
@@ -24,45 +27,17 @@ pub(crate) fn serialize_v4_extensible_array(
|
|||||||
let ndims = chunk_dims.len() as u8 + 1;
|
let ndims = chunk_dims.len() as u8 + 1;
|
||||||
buf.push(ndims);
|
buf.push(ndims);
|
||||||
|
|
||||||
let max_dim = chunk_dims
|
push_v4_chunk_dims(&mut buf, chunk_dims, element_size);
|
||||||
.iter()
|
|
||||||
.map(|&d| d as u64)
|
|
||||||
.chain(core::iter::once(element_size as u64))
|
|
||||||
.max()
|
|
||||||
.unwrap_or(1);
|
|
||||||
let dim_encoded_len: u8 = if max_dim <= 0xFF {
|
|
||||||
1
|
|
||||||
} else if max_dim <= 0xFFFF {
|
|
||||||
2
|
|
||||||
} else {
|
|
||||||
4
|
|
||||||
};
|
|
||||||
buf.push(dim_encoded_len);
|
|
||||||
|
|
||||||
for &d in chunk_dims {
|
|
||||||
match dim_encoded_len {
|
|
||||||
1 => buf.push(d as u8),
|
|
||||||
2 => buf.extend_from_slice(&(d as u16).to_le_bytes()),
|
|
||||||
4 => buf.extend_from_slice(&d.to_le_bytes()),
|
|
||||||
_ => unreachable!("unexpected dim_encoded_len: {dim_encoded_len}"),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
match dim_encoded_len {
|
|
||||||
1 => buf.push(element_size as u8),
|
|
||||||
2 => buf.extend_from_slice(&(element_size as u16).to_le_bytes()),
|
|
||||||
4 => buf.extend_from_slice(&element_size.to_le_bytes()),
|
|
||||||
_ => unreachable!("unexpected dim_encoded_len: {dim_encoded_len}"),
|
|
||||||
}
|
|
||||||
|
|
||||||
// chunk index type = 4 (Extensible Array)
|
// chunk index type = 4 (Extensible Array)
|
||||||
buf.push(4);
|
buf.push(4);
|
||||||
|
|
||||||
// EA creation parameters (must match AEHD and HDF5 C library defaults)
|
// EA creation parameters (must match AEHD and HDF5 C library defaults)
|
||||||
buf.push(32); // max_nelmts_bits
|
buf.push(MAX_NELMTS_BITS);
|
||||||
buf.push(4); // idx_blk_elmts
|
buf.push(IDX_BLK_ELMTS);
|
||||||
buf.push(4); // super_blk_min_data_ptrs
|
buf.push(SUP_BLK_MIN_DATA_PTRS);
|
||||||
buf.push(16); // data_blk_min_elmts
|
buf.push(DATA_BLK_MIN_ELMTS);
|
||||||
buf.push(10); // max_dblk_page_nelmts_bits
|
buf.push(MAX_DBLK_PAGE_NELMTS_BITS);
|
||||||
|
|
||||||
// EA header address
|
// EA header address
|
||||||
match offset_size {
|
match offset_size {
|
||||||
@@ -74,304 +49,281 @@ pub(crate) fn serialize_v4_extensible_array(
|
|||||||
buf
|
buf
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// EA creation parameters — the HDF5 library's defaults for chunk indexes
|
||||||
|
// (`H5D_EARRAY_*`); the layout message above and the header must agree.
|
||||||
|
const MAX_NELMTS_BITS: u8 = 32;
|
||||||
|
const IDX_BLK_ELMTS: u8 = 4;
|
||||||
|
const SUP_BLK_MIN_DATA_PTRS: u8 = 4;
|
||||||
|
const DATA_BLK_MIN_ELMTS: u8 = 16;
|
||||||
|
const MAX_DBLK_PAGE_NELMTS_BITS: u8 = 10;
|
||||||
|
|
||||||
|
/// One data block of the array: its first element (relative to the end of
|
||||||
|
/// the index block's own elements), element count, and address when it is
|
||||||
|
/// allocated.
|
||||||
|
struct DataBlock {
|
||||||
|
start: usize,
|
||||||
|
nelmts: usize,
|
||||||
|
addr: Option<u64>,
|
||||||
|
}
|
||||||
|
|
||||||
/// Build a complete Extensible Array at a known absolute address.
|
/// Build a complete Extensible Array at a known absolute address.
|
||||||
///
|
///
|
||||||
/// For simplicity, we put all elements inline in the index block when the
|
/// `slots[i]` is the element at linear index `i` (see `chunk_grid`); `None`
|
||||||
/// number of chunks is small (up to idx_blk_elmts), otherwise use inline +
|
/// marks an unallocated chunk. The first `IDX_BLK_ELMTS` elements live in
|
||||||
/// direct data blocks.
|
/// the index block, the rest in data blocks grouped by super block level
|
||||||
|
/// exactly as `H5EA__hdr_init` sizes them: level `u` has `2^(u/2)` data
|
||||||
|
/// blocks of `DATA_BLK_MIN_ELMTS * 2^ceil(u/2)` elements. The data blocks of
|
||||||
|
/// the first levels are addressed straight from the index block; later
|
||||||
|
/// levels go through a super block (EASB). Data blocks larger than a page
|
||||||
|
/// (`2^MAX_DBLK_PAGE_NELMTS_BITS` elements) are paged, with their page-init
|
||||||
|
/// bits kept in the owning super block. Only blocks holding a defined element
|
||||||
|
/// are allocated; the rest keep the undefined address, as in a file the
|
||||||
|
/// library wrote.
|
||||||
pub fn build_extensible_array_at(
|
pub fn build_extensible_array_at(
|
||||||
chunks: &[WrittenChunk],
|
slots: &[Option<WrittenChunk>],
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
has_filters: bool,
|
has_filters: bool,
|
||||||
ea_base_address: u64,
|
ea_base_address: u64,
|
||||||
) -> Vec<u8> {
|
) -> Vec<u8> {
|
||||||
let os = offset_size as usize;
|
let os = offset_size as usize;
|
||||||
let num_elements = chunks.len();
|
let chunk_size_bytes = has_filters.then(|| filtered_chunk_size_len(slots));
|
||||||
|
let elem_size = os + chunk_size_bytes.map_or(0, |n| n + 4);
|
||||||
// Compute element encoding size (same logic as Fixed Array)
|
|
||||||
let chunk_size_bytes: usize = if has_filters {
|
|
||||||
let max_raw = chunks.iter().map(|c| c.raw_size).max().unwrap_or(1);
|
|
||||||
let log2_val = if max_raw <= 1 {
|
|
||||||
0
|
|
||||||
} else {
|
|
||||||
63 - max_raw.leading_zeros()
|
|
||||||
};
|
|
||||||
let len = 1 + ((log2_val + 8) / 8) as usize;
|
|
||||||
len.min(8)
|
|
||||||
} else {
|
|
||||||
0
|
|
||||||
};
|
|
||||||
|
|
||||||
let elem_size = if has_filters {
|
|
||||||
os + chunk_size_bytes + 4
|
|
||||||
} else {
|
|
||||||
os
|
|
||||||
};
|
|
||||||
|
|
||||||
let client_id: u8 = if has_filters { 1 } else { 0 };
|
let client_id: u8 = if has_filters { 1 } else { 0 };
|
||||||
|
let arr_off_size = (MAX_NELMTS_BITS as usize).div_ceil(8);
|
||||||
|
let page_nelmts = 1usize << MAX_DBLK_PAGE_NELMTS_BITS;
|
||||||
|
let idx_blk = IDX_BLK_ELMTS as usize;
|
||||||
|
|
||||||
// EA creation parameters — must match HDF5 C library defaults exactly
|
// Elements past the last defined one are never realised
|
||||||
let max_nelmts_bits: u8 = 32;
|
// (`max_idx_set` is one past the highest index ever set).
|
||||||
let idx_blk_elmts: u8 = 4;
|
let max_idx_set = slots.iter().rposition(Option::is_some).map_or(0, |i| i + 1);
|
||||||
let min_dblk_nelmts: u8 = 16;
|
let slots = &slots[..max_idx_set];
|
||||||
let super_blk_min_nelmts: u8 = 4;
|
let defined_in = |start: usize, n: usize| -> bool {
|
||||||
let max_dblk_nelmts_bits: u8 = 10;
|
let lo = idx_blk.saturating_add(start).min(slots.len());
|
||||||
|
let hi = idx_blk
|
||||||
|
.saturating_add(start)
|
||||||
|
.saturating_add(n)
|
||||||
|
.min(slots.len());
|
||||||
|
slots[lo..hi].iter().any(Option::is_some)
|
||||||
|
};
|
||||||
|
|
||||||
// EAHD size: fixed(12) + 6 stats(6*length_size) + addr(offset_size) + checksum(4)
|
// Super block levels: (ndblks, dblk_nelmts, first element).
|
||||||
|
let log2_dmin = (DATA_BLK_MIN_ELMTS as u32).trailing_zeros() as usize;
|
||||||
|
let nsblks = 1 + MAX_NELMTS_BITS as usize - log2_dmin;
|
||||||
|
let ndblk_addrs = 2 * (SUP_BLK_MIN_DATA_PTRS as usize - 1);
|
||||||
|
let mut levels: Vec<(usize, usize, usize)> = Vec::with_capacity(nsblks);
|
||||||
|
let mut start = 0usize;
|
||||||
|
for u in 0..nsblks {
|
||||||
|
let ndblks = 1usize << (u / 2);
|
||||||
|
let nelmts = (DATA_BLK_MIN_ELMTS as usize) << u.div_ceil(2);
|
||||||
|
levels.push((ndblks, nelmts, start));
|
||||||
|
// Saturate: on 32-bit targets the last levels only need to compare
|
||||||
|
// as "beyond the end".
|
||||||
|
start = start.saturating_add(ndblks.saturating_mul(nelmts));
|
||||||
|
}
|
||||||
|
// Levels whose data blocks the index block addresses directly.
|
||||||
|
let mut direct_levels = 0;
|
||||||
|
let mut n = 0;
|
||||||
|
while n < ndblk_addrs {
|
||||||
|
n += levels[direct_levels].0;
|
||||||
|
direct_levels += 1;
|
||||||
|
}
|
||||||
|
let nsblk_addrs = nsblks - direct_levels;
|
||||||
|
|
||||||
|
let dblk_size = |nelmts: usize| -> usize {
|
||||||
|
let prefix = 4 + 1 + 1 + os + arr_off_size + 4;
|
||||||
|
if nelmts > page_nelmts {
|
||||||
|
prefix + (nelmts / page_nelmts) * (page_nelmts * elem_size + 4)
|
||||||
|
} else {
|
||||||
|
prefix + nelmts * elem_size
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let sblk_bitmap_len = |ndblks: usize, nelmts: usize| -> usize {
|
||||||
|
if nelmts > page_nelmts {
|
||||||
|
ndblks * (nelmts / page_nelmts).div_ceil(8)
|
||||||
|
} else {
|
||||||
|
0
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
// Plan addresses: header, index block, the direct data blocks, then each
|
||||||
|
// allocated super block followed by its allocated data blocks.
|
||||||
let aehd_size = 4 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 6 * length_size as usize + os + 4;
|
let aehd_size = 4 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 6 * length_size as usize + os + 4;
|
||||||
let aeib_address = ea_base_address + aehd_size as u64;
|
let aeib_address = ea_base_address + aehd_size as u64;
|
||||||
|
let aeib_size = 4 + 1 + 1 + os + idx_blk * elem_size + ndblk_addrs * os + nsblk_addrs * os + 4;
|
||||||
|
let mut cursor = aeib_address + aeib_size as u64;
|
||||||
|
|
||||||
// Determine how many elements go inline vs data blocks
|
let mut ndata_blks = 0u64;
|
||||||
let n_inline = (idx_blk_elmts as usize).min(num_elements);
|
let mut data_blk_size = 0u64;
|
||||||
let remaining_after_inline = num_elements.saturating_sub(n_inline);
|
let mut nsuper_blks = 0u64;
|
||||||
|
let mut super_blk_size = 0u64;
|
||||||
|
let mut realized = idx_blk as u64;
|
||||||
|
|
||||||
// Compute super block layout per HDF5 spec
|
let mut plan_dblk = |cursor: &mut u64, start: usize, nelmts: usize| -> DataBlock {
|
||||||
let sblk_min = super_blk_min_nelmts as usize;
|
let addr = defined_in(start, nelmts).then(|| {
|
||||||
let log2_dblk_min = if min_dblk_nelmts <= 1 {
|
let a = *cursor;
|
||||||
0
|
let size = dblk_size(nelmts) as u64;
|
||||||
} else {
|
*cursor += size;
|
||||||
(min_dblk_nelmts as u32).trailing_zeros() as usize
|
ndata_blks += 1;
|
||||||
|
data_blk_size += size;
|
||||||
|
realized += nelmts as u64;
|
||||||
|
a
|
||||||
|
});
|
||||||
|
DataBlock {
|
||||||
|
start,
|
||||||
|
nelmts,
|
||||||
|
addr,
|
||||||
|
}
|
||||||
};
|
};
|
||||||
let nsblks = (max_nelmts_bits as usize).saturating_sub(log2_dblk_min) + 1;
|
|
||||||
|
|
||||||
// Direct data block addresses (from super blocks 0..sblk_min-1)
|
let mut direct: Vec<DataBlock> = Vec::with_capacity(ndblk_addrs);
|
||||||
let mut dblk_sizes: Vec<usize> = Vec::new();
|
for &(ndblks, nelmts, first) in &levels[..direct_levels] {
|
||||||
for sblk_idx in 0..sblk_min.min(nsblks) {
|
for k in 0..ndblks {
|
||||||
let ndblks = 1usize << (sblk_idx / 2);
|
direct.push(plan_dblk(&mut cursor, first + k * nelmts, nelmts));
|
||||||
let dblk_nelmts = (min_dblk_nelmts as usize) * (1 << sblk_idx.div_ceil(2));
|
|
||||||
for _ in 0..ndblks {
|
|
||||||
dblk_sizes.push(dblk_nelmts);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
let n_direct_dblks = dblk_sizes.len();
|
// (super block address, level, its data blocks)
|
||||||
|
let mut supers: Vec<(Option<u64>, usize, Vec<DataBlock>)> = Vec::with_capacity(nsblk_addrs);
|
||||||
// Super block addresses (for super blocks sblk_min..nsblks-1)
|
for (u, &(ndblks, nelmts, first)) in levels.iter().enumerate().skip(direct_levels) {
|
||||||
let n_sblk_addrs = nsblks.saturating_sub(sblk_min);
|
if !defined_in(first, ndblks.saturating_mul(nelmts)) {
|
||||||
|
supers.push((None, u, Vec::new()));
|
||||||
// EAIB size
|
continue;
|
||||||
let aeib_size = 4
|
|
||||||
+ 1
|
|
||||||
+ 1
|
|
||||||
+ os
|
|
||||||
+ idx_blk_elmts as usize * elem_size
|
|
||||||
+ n_direct_dblks * os
|
|
||||||
+ n_sblk_addrs * os
|
|
||||||
+ 4;
|
|
||||||
|
|
||||||
// Build AEHD
|
|
||||||
let mut aehd = Vec::with_capacity(aehd_size);
|
|
||||||
aehd.extend_from_slice(b"EAHD");
|
|
||||||
aehd.push(0); // version
|
|
||||||
aehd.push(client_id);
|
|
||||||
aehd.push(elem_size as u8);
|
|
||||||
aehd.push(max_nelmts_bits);
|
|
||||||
aehd.push(idx_blk_elmts);
|
|
||||||
aehd.push(min_dblk_nelmts);
|
|
||||||
aehd.push(super_blk_min_nelmts);
|
|
||||||
aehd.push(max_dblk_nelmts_bits);
|
|
||||||
|
|
||||||
// Count data blocks that will have chunks
|
|
||||||
let n_active_dblks: u64 = if remaining_after_inline > 0 {
|
|
||||||
let mut count = 0u64;
|
|
||||||
let mut ci = n_inline;
|
|
||||||
for &sz in &dblk_sizes {
|
|
||||||
if ci < num_elements {
|
|
||||||
count += 1;
|
|
||||||
ci += sz;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
count
|
let sb_size =
|
||||||
} else {
|
4 + 1 + 1 + os + arr_off_size + sblk_bitmap_len(ndblks, nelmts) + ndblks * os + 4;
|
||||||
0
|
let sb_addr = cursor;
|
||||||
};
|
cursor += sb_size as u64;
|
||||||
let blk_off_size = (max_nelmts_bits as usize).div_ceil(8);
|
nsuper_blks += 1;
|
||||||
let aedb_header_overhead = 4 + 1 + 1 + os + blk_off_size + 4;
|
super_blk_size += sb_size as u64;
|
||||||
let data_blk_total_size: u64 = if remaining_after_inline > 0 {
|
let dblks = (0..ndblks)
|
||||||
let mut total = 0u64;
|
.map(|k| plan_dblk(&mut cursor, first + k * nelmts, nelmts))
|
||||||
let mut ci = n_inline;
|
.collect();
|
||||||
for &sz in &dblk_sizes {
|
supers.push((Some(sb_addr), u, dblks));
|
||||||
if ci < num_elements {
|
}
|
||||||
total += (aedb_header_overhead + sz * elem_size) as u64;
|
|
||||||
ci += sz;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
total
|
|
||||||
} else {
|
|
||||||
0
|
|
||||||
};
|
|
||||||
let max_idx_set: u64 = if remaining_after_inline > 0 {
|
|
||||||
let mut max_set = idx_blk_elmts as u64;
|
|
||||||
let mut ci = n_inline;
|
|
||||||
for &sz in &dblk_sizes {
|
|
||||||
if ci < num_elements {
|
|
||||||
max_set += sz as u64;
|
|
||||||
ci += sz;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
max_set
|
|
||||||
} else {
|
|
||||||
idx_blk_elmts as u64
|
|
||||||
};
|
|
||||||
|
|
||||||
|
let slot = |i: usize| slots.get(i).and_then(Option::as_ref);
|
||||||
let write_length = |buf: &mut Vec<u8>, val: u64| match length_size {
|
let write_length = |buf: &mut Vec<u8>, val: u64| match length_size {
|
||||||
4 => buf.extend_from_slice(&(val as u32).to_le_bytes()),
|
4 => buf.extend_from_slice(&(val as u32).to_le_bytes()),
|
||||||
_ => buf.extend_from_slice(&val.to_le_bytes()),
|
_ => buf.extend_from_slice(&val.to_le_bytes()),
|
||||||
};
|
};
|
||||||
let write_addr = |buf: &mut Vec<u8>, val: u64| match offset_size {
|
let write_addr_opt = |buf: &mut Vec<u8>, addr: Option<u64>| match addr {
|
||||||
4 => buf.extend_from_slice(&(val as u32).to_le_bytes()),
|
Some(a) => push_addr(buf, a, offset_size),
|
||||||
_ => buf.extend_from_slice(&val.to_le_bytes()),
|
None => buf.extend(core::iter::repeat_n(0xFF, os)),
|
||||||
|
};
|
||||||
|
let block_prefix = |buf: &mut Vec<u8>, sig: &[u8; 4], block_off: usize| {
|
||||||
|
buf.extend_from_slice(sig);
|
||||||
|
buf.push(0); // version
|
||||||
|
buf.push(client_id);
|
||||||
|
push_addr(buf, ea_base_address, offset_size);
|
||||||
|
buf.extend_from_slice(&(block_off as u64).to_le_bytes()[..arr_off_size]);
|
||||||
|
};
|
||||||
|
// Serialise one data block (paged or not) onto `out`.
|
||||||
|
let write_dblk = |out: &mut Vec<u8>, db: &DataBlock| {
|
||||||
|
let at = out.len();
|
||||||
|
block_prefix(out, b"EADB", db.start);
|
||||||
|
let first = idx_blk + db.start;
|
||||||
|
if db.nelmts > page_nelmts {
|
||||||
|
// Paged: the prefix carries only its own checksum; each page
|
||||||
|
// follows with one of its own.
|
||||||
|
let sum = jenkins_lookup3(&out[at..]);
|
||||||
|
out.extend_from_slice(&sum.to_le_bytes());
|
||||||
|
for p in 0..db.nelmts / page_nelmts {
|
||||||
|
let page_at = out.len();
|
||||||
|
for e in 0..page_nelmts {
|
||||||
|
let i = first + p * page_nelmts + e;
|
||||||
|
push_index_element(out, slot(i), offset_size, chunk_size_bytes);
|
||||||
|
}
|
||||||
|
let sum = jenkins_lookup3(&out[page_at..]);
|
||||||
|
out.extend_from_slice(&sum.to_le_bytes());
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
for i in first..first + db.nelmts {
|
||||||
|
push_index_element(out, slot(i), offset_size, chunk_size_bytes);
|
||||||
|
}
|
||||||
|
let sum = jenkins_lookup3(&out[at..]);
|
||||||
|
out.extend_from_slice(&sum.to_le_bytes());
|
||||||
|
}
|
||||||
|
debug_assert_eq!(out.len() - at, dblk_size(db.nelmts));
|
||||||
};
|
};
|
||||||
|
|
||||||
write_length(&mut aehd, 0);
|
// Header (EAHD). The six statistics are, in order: super blocks, their
|
||||||
write_length(&mut aehd, 0);
|
// bytes, data blocks, their bytes, max index set, elements realised.
|
||||||
write_length(&mut aehd, n_active_dblks);
|
let mut out = Vec::with_capacity(saturating_usize(cursor - ea_base_address));
|
||||||
write_length(&mut aehd, data_blk_total_size);
|
out.extend_from_slice(b"EAHD");
|
||||||
write_length(&mut aehd, num_elements as u64);
|
out.push(0); // version
|
||||||
write_length(&mut aehd, max_idx_set);
|
out.push(client_id);
|
||||||
|
out.push(elem_size as u8);
|
||||||
|
out.push(MAX_NELMTS_BITS);
|
||||||
|
out.push(IDX_BLK_ELMTS);
|
||||||
|
out.push(DATA_BLK_MIN_ELMTS);
|
||||||
|
out.push(SUP_BLK_MIN_DATA_PTRS);
|
||||||
|
out.push(MAX_DBLK_PAGE_NELMTS_BITS);
|
||||||
|
write_length(&mut out, nsuper_blks);
|
||||||
|
write_length(&mut out, super_blk_size);
|
||||||
|
write_length(&mut out, ndata_blks);
|
||||||
|
write_length(&mut out, data_blk_size);
|
||||||
|
write_length(&mut out, max_idx_set as u64);
|
||||||
|
write_length(&mut out, realized);
|
||||||
|
push_addr(&mut out, aeib_address, offset_size);
|
||||||
|
let sum = jenkins_lookup3(&out);
|
||||||
|
out.extend_from_slice(&sum.to_le_bytes());
|
||||||
|
debug_assert_eq!(out.len(), aehd_size);
|
||||||
|
|
||||||
write_addr(&mut aehd, aeib_address);
|
// Index block (EAIB): inline elements, data block and super block
|
||||||
|
// addresses.
|
||||||
let aehd_checksum = jenkins_lookup3(&aehd);
|
let ib_start = out.len();
|
||||||
aehd.extend_from_slice(&aehd_checksum.to_le_bytes());
|
out.extend_from_slice(b"EAIB");
|
||||||
debug_assert_eq!(aehd.len(), aehd_size);
|
out.push(0);
|
||||||
|
out.push(client_id);
|
||||||
// Build AEIB
|
push_addr(&mut out, ea_base_address, offset_size);
|
||||||
let mut aeib = Vec::with_capacity(aeib_size);
|
for i in 0..idx_blk {
|
||||||
aeib.extend_from_slice(b"EAIB");
|
push_index_element(&mut out, slot(i), offset_size, chunk_size_bytes);
|
||||||
aeib.push(0);
|
|
||||||
aeib.push(client_id);
|
|
||||||
|
|
||||||
match offset_size {
|
|
||||||
4 => aeib.extend_from_slice(&(ea_base_address as u32).to_le_bytes()),
|
|
||||||
8 => aeib.extend_from_slice(&ea_base_address.to_le_bytes()),
|
|
||||||
_ => aeib.extend_from_slice(&ea_base_address.to_le_bytes()),
|
|
||||||
}
|
}
|
||||||
|
for db in &direct {
|
||||||
// Inline elements
|
write_addr_opt(&mut out, db.addr);
|
||||||
#[allow(clippy::needless_range_loop)]
|
|
||||||
for i in 0..idx_blk_elmts as usize {
|
|
||||||
if i < n_inline {
|
|
||||||
write_chunk_element(
|
|
||||||
&mut aeib,
|
|
||||||
&chunks[i],
|
|
||||||
offset_size,
|
|
||||||
has_filters,
|
|
||||||
chunk_size_bytes,
|
|
||||||
);
|
|
||||||
} else {
|
|
||||||
write_undefined_element(&mut aeib, offset_size, has_filters, chunk_size_bytes);
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
for (sb_addr, _, _) in &supers {
|
||||||
|
write_addr_opt(&mut out, *sb_addr);
|
||||||
|
}
|
||||||
|
let sum = jenkins_lookup3(&out[ib_start..]);
|
||||||
|
out.extend_from_slice(&sum.to_le_bytes());
|
||||||
|
debug_assert_eq!(out.len() - ib_start, aeib_size);
|
||||||
|
|
||||||
// Data block addresses + build data blocks
|
for db in direct.iter().filter(|d| d.addr.is_some()) {
|
||||||
let mut data_blocks_buf = Vec::new();
|
write_dblk(&mut out, db);
|
||||||
let dblks_base = aeib_address + aeib_size as u64;
|
}
|
||||||
let mut dblk_cursor = dblks_base;
|
for (sb_addr, u, dblks) in &supers {
|
||||||
let mut chunk_idx = n_inline;
|
if sb_addr.is_none() {
|
||||||
|
|
||||||
for &nelmts in &dblk_sizes {
|
|
||||||
if chunk_idx >= num_elements {
|
|
||||||
match offset_size {
|
|
||||||
4 => aeib.extend_from_slice(&u32::MAX.to_le_bytes()),
|
|
||||||
8 => aeib.extend_from_slice(&u64::MAX.to_le_bytes()),
|
|
||||||
_ => aeib.extend_from_slice(&u64::MAX.to_le_bytes()),
|
|
||||||
}
|
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
let (ndblks, nelmts, first) = levels[*u];
|
||||||
match offset_size {
|
let sb_start = out.len();
|
||||||
4 => aeib.extend_from_slice(&(dblk_cursor as u32).to_le_bytes()),
|
block_prefix(&mut out, b"EASB", first);
|
||||||
8 => aeib.extend_from_slice(&dblk_cursor.to_le_bytes()),
|
if nelmts > page_nelmts {
|
||||||
_ => aeib.extend_from_slice(&dblk_cursor.to_le_bytes()),
|
// Page-init bits, `npages` per data block, packed MSB-first
|
||||||
}
|
// (`H5VM_bit_set`): every page of an allocated data block is
|
||||||
|
// written.
|
||||||
// Build EADB
|
let npages = nelmts / page_nelmts;
|
||||||
let mut aedb = Vec::new();
|
let mut bitmap = vec![0u8; sblk_bitmap_len(ndblks, nelmts)];
|
||||||
aedb.extend_from_slice(b"EADB");
|
for (k, db) in dblks.iter().enumerate() {
|
||||||
aedb.push(0);
|
if db.addr.is_some() {
|
||||||
aedb.push(client_id);
|
for p in 0..npages {
|
||||||
match offset_size {
|
let bit = k * npages + p;
|
||||||
4 => aedb.extend_from_slice(&(ea_base_address as u32).to_le_bytes()),
|
bitmap[bit / 8] |= 0x80 >> (bit % 8);
|
||||||
8 => aedb.extend_from_slice(&ea_base_address.to_le_bytes()),
|
}
|
||||||
_ => aedb.extend_from_slice(&ea_base_address.to_le_bytes()),
|
}
|
||||||
}
|
|
||||||
|
|
||||||
let blk_off_size = (max_nelmts_bits as usize).div_ceil(8);
|
|
||||||
let blk_off_val = (chunk_idx - n_inline) as u64;
|
|
||||||
aedb.extend_from_slice(&blk_off_val.to_le_bytes()[..blk_off_size]);
|
|
||||||
|
|
||||||
for slot in 0..nelmts {
|
|
||||||
if chunk_idx + slot < num_elements {
|
|
||||||
write_chunk_element(
|
|
||||||
&mut aedb,
|
|
||||||
&chunks[chunk_idx + slot],
|
|
||||||
offset_size,
|
|
||||||
has_filters,
|
|
||||||
chunk_size_bytes,
|
|
||||||
);
|
|
||||||
} else {
|
|
||||||
write_undefined_element(&mut aedb, offset_size, has_filters, chunk_size_bytes);
|
|
||||||
}
|
}
|
||||||
|
out.extend_from_slice(&bitmap);
|
||||||
}
|
}
|
||||||
|
for db in dblks {
|
||||||
let aedb_checksum = jenkins_lookup3(&aedb);
|
write_addr_opt(&mut out, db.addr);
|
||||||
aedb.extend_from_slice(&aedb_checksum.to_le_bytes());
|
}
|
||||||
|
let sum = jenkins_lookup3(&out[sb_start..]);
|
||||||
dblk_cursor += aedb.len() as u64;
|
out.extend_from_slice(&sum.to_le_bytes());
|
||||||
data_blocks_buf.extend_from_slice(&aedb);
|
for db in dblks.iter().filter(|d| d.addr.is_some()) {
|
||||||
chunk_idx += nelmts;
|
write_dblk(&mut out, db);
|
||||||
}
|
|
||||||
|
|
||||||
// Super block addresses (all undefined)
|
|
||||||
for _ in 0..n_sblk_addrs {
|
|
||||||
match offset_size {
|
|
||||||
4 => aeib.extend_from_slice(&u32::MAX.to_le_bytes()),
|
|
||||||
8 => aeib.extend_from_slice(&u64::MAX.to_le_bytes()),
|
|
||||||
_ => aeib.extend_from_slice(&u64::MAX.to_le_bytes()),
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
debug_assert_eq!(out.len() as u64, cursor - ea_base_address);
|
||||||
let aeib_checksum = jenkins_lookup3(&aeib);
|
out
|
||||||
aeib.extend_from_slice(&aeib_checksum.to_le_bytes());
|
|
||||||
debug_assert_eq!(aeib.len(), aeib_size);
|
|
||||||
|
|
||||||
let mut combined = aehd;
|
|
||||||
combined.extend_from_slice(&aeib);
|
|
||||||
combined.extend_from_slice(&data_blocks_buf);
|
|
||||||
combined
|
|
||||||
}
|
|
||||||
|
|
||||||
fn write_chunk_element(
|
|
||||||
buf: &mut Vec<u8>,
|
|
||||||
chunk: &WrittenChunk,
|
|
||||||
offset_size: u8,
|
|
||||||
has_filters: bool,
|
|
||||||
chunk_size_bytes: usize,
|
|
||||||
) {
|
|
||||||
match offset_size {
|
|
||||||
4 => buf.extend_from_slice(&(chunk.address as u32).to_le_bytes()),
|
|
||||||
8 => buf.extend_from_slice(&chunk.address.to_le_bytes()),
|
|
||||||
_ => buf.extend_from_slice(&chunk.address.to_le_bytes()),
|
|
||||||
}
|
|
||||||
if has_filters {
|
|
||||||
let cs_bytes = chunk.compressed_size.to_le_bytes();
|
|
||||||
buf.extend_from_slice(&cs_bytes[..chunk_size_bytes]);
|
|
||||||
buf.extend_from_slice(&chunk.filter_mask.to_le_bytes());
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn write_undefined_element(
|
|
||||||
buf: &mut Vec<u8>,
|
|
||||||
offset_size: u8,
|
|
||||||
has_filters: bool,
|
|
||||||
chunk_size_bytes: usize,
|
|
||||||
) {
|
|
||||||
let os = offset_size as usize;
|
|
||||||
// Use extend with repeat to avoid heap-allocating a temporary Vec on each call.
|
|
||||||
buf.extend(core::iter::repeat_n(0xFF, os));
|
|
||||||
if has_filters {
|
|
||||||
buf.extend(core::iter::repeat_n(0x00, chunk_size_bytes));
|
|
||||||
buf.extend_from_slice(&0u32.to_le_bytes());
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -12,7 +12,11 @@ use std::string::String;
|
|||||||
use core::fmt;
|
use core::fmt;
|
||||||
|
|
||||||
/// Errors that can occur when parsing HDF5 binary format structures.
|
/// Errors that can occur when parsing HDF5 binary format structures.
|
||||||
|
///
|
||||||
|
/// Non-exhaustive: new failure modes (new storage backends, new file
|
||||||
|
/// features) add variants, so a `match` needs a wildcard arm.
|
||||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||||
|
#[non_exhaustive]
|
||||||
pub enum FormatError {
|
pub enum FormatError {
|
||||||
/// The HDF5 magic signature was not found at any valid offset.
|
/// The HDF5 magic signature was not found at any valid offset.
|
||||||
SignatureNotFound,
|
SignatureNotFound,
|
||||||
@@ -80,6 +84,9 @@ pub enum FormatError {
|
|||||||
InvalidLocalHeapSignature,
|
InvalidLocalHeapSignature,
|
||||||
/// Invalid local heap version.
|
/// Invalid local heap version.
|
||||||
InvalidLocalHeapVersion(u8),
|
InvalidLocalHeapVersion(u8),
|
||||||
|
/// A local heap's free list points outside its data segment (libhdf5:
|
||||||
|
/// "bad heap free list").
|
||||||
|
InvalidLocalHeapFreeList,
|
||||||
/// Invalid B-tree v1 signature.
|
/// Invalid B-tree v1 signature.
|
||||||
InvalidBTreeSignature,
|
InvalidBTreeSignature,
|
||||||
/// Invalid B-tree node type.
|
/// Invalid B-tree node type.
|
||||||
@@ -117,6 +124,14 @@ pub enum FormatError {
|
|||||||
/// A message is marked shared but was parsed without access to the file,
|
/// A message is marked shared but was parsed without access to the file,
|
||||||
/// so the reference to the real message could not be followed.
|
/// so the reference to the real message could not be followed.
|
||||||
UnresolvedSharedMessage,
|
UnresolvedSharedMessage,
|
||||||
|
/// A shared-message reference points at an object header that holds no
|
||||||
|
/// (unshared) message of the referenced type (raw message type id).
|
||||||
|
SharedMessageTargetMissing(u16),
|
||||||
|
/// A superblock was parsed at a non-zero offset of the buffer (the file
|
||||||
|
/// has a user block of this many bytes). HDF5 addresses are relative to
|
||||||
|
/// the superblock, so the buffer must start there: see
|
||||||
|
/// `signature::split_user_block`.
|
||||||
|
UserBlockNotStripped(u64),
|
||||||
/// A selection does not fit the dataset it was applied to (wrong rank, or
|
/// A selection does not fit the dataset it was applied to (wrong rank, or
|
||||||
/// it reaches past a dimension's extent).
|
/// it reaches past a dimension's extent).
|
||||||
SelectionOutOfBounds(String),
|
SelectionOutOfBounds(String),
|
||||||
@@ -190,6 +205,55 @@ pub enum FormatError {
|
|||||||
DuplicateDatasetName(String),
|
DuplicateDatasetName(String),
|
||||||
/// Integer overflow in size computation (malformed data protection).
|
/// Integer overflow in size computation (malformed data protection).
|
||||||
Overflow(String),
|
Overflow(String),
|
||||||
|
/// An object header that libhdf5 refuses to load (the reason is
|
||||||
|
/// libhdf5's own error text): a misaligned or overrunning message, a
|
||||||
|
/// wrong message count, contradictory message flags, a message of a
|
||||||
|
/// class that cannot be shared flagged shareable, …
|
||||||
|
InvalidObjectHeader(&'static str),
|
||||||
|
/// A datatype message libhdf5 refuses to decode (the reason is
|
||||||
|
/// libhdf5's own error text): size 0, bit fields outside the type,
|
||||||
|
/// an empty enum name, a compound member outside its compound, …
|
||||||
|
InvalidDatatype(String),
|
||||||
|
/// A chunked layout whose chunk dimensions libhdf5 refuses: a zero
|
||||||
|
/// dimension, a rank that does not match the dataspace, an element size
|
||||||
|
/// that is not the datatype's, or a chunk of 4 GiB or more indexed by a
|
||||||
|
/// version-1 B-tree.
|
||||||
|
InvalidChunkDimensions(String),
|
||||||
|
/// The superblock's end-of-file address lies past the end of the file:
|
||||||
|
/// the file was truncated (libhdf5 refuses to open it).
|
||||||
|
TruncatedFile {
|
||||||
|
/// End of file recorded in the superblock (relative to byte 0).
|
||||||
|
stored_eof: u64,
|
||||||
|
/// The file's actual length in bytes.
|
||||||
|
actual_len: u64,
|
||||||
|
},
|
||||||
|
/// A link libhdf5 refuses to list: a symbol-table entry with an empty
|
||||||
|
/// name ("invalid link name"). Listing the group fails, as in libhdf5.
|
||||||
|
InvalidLinkName,
|
||||||
|
/// A dataspace message libhdf5 refuses to decode (the reason is
|
||||||
|
/// libhdf5's own error text): more than 32 dimensions, a rank on a
|
||||||
|
/// scalar or null dataspace, a dimension larger than its maximum.
|
||||||
|
InvalidDataspace(&'static str),
|
||||||
|
/// A dataset whose storage libhdf5 refuses when it opens the dataset
|
||||||
|
/// (the reason is libhdf5's own error text): an element count times
|
||||||
|
/// element size that overflows, contiguous storage past the end of the
|
||||||
|
/// file, compact data of the wrong size.
|
||||||
|
InvalidDatasetStorage(&'static str),
|
||||||
|
/// A superblock extension message libhdf5 refuses to decode when it
|
||||||
|
/// opens the file (the reason is libhdf5's own error text): a File Space
|
||||||
|
/// Info message that runs off its end or has a bad page size, a metadata
|
||||||
|
/// cache image outside the file, …
|
||||||
|
InvalidSuperblockExtension(&'static str),
|
||||||
|
/// A metadata cache image block libhdf5 refuses to load (the reason is
|
||||||
|
/// libhdf5's own error text).
|
||||||
|
InvalidCacheImage(&'static str),
|
||||||
|
/// The [`Storage`](crate::storage::Storage) backend failed to serve a
|
||||||
|
/// read (an I/O or network error, or a short read inside the file).
|
||||||
|
Storage(String),
|
||||||
|
/// The operation still needs the whole file as one slice and the
|
||||||
|
/// [`Storage`](crate::storage::Storage) backend has no contiguous view
|
||||||
|
/// (`as_contiguous()` is `None`); the text names the operation.
|
||||||
|
ContiguousStorageRequired(&'static str),
|
||||||
}
|
}
|
||||||
|
|
||||||
impl fmt::Display for FormatError {
|
impl fmt::Display for FormatError {
|
||||||
@@ -270,6 +334,9 @@ impl fmt::Display for FormatError {
|
|||||||
FormatError::InvalidLocalHeapSignature => {
|
FormatError::InvalidLocalHeapSignature => {
|
||||||
write!(f, "invalid local heap signature")
|
write!(f, "invalid local heap signature")
|
||||||
}
|
}
|
||||||
|
FormatError::InvalidLocalHeapFreeList => {
|
||||||
|
write!(f, "bad local heap free list")
|
||||||
|
}
|
||||||
FormatError::InvalidLocalHeapVersion(v) => {
|
FormatError::InvalidLocalHeapVersion(v) => {
|
||||||
write!(f, "invalid local heap version: {v}")
|
write!(f, "invalid local heap version: {v}")
|
||||||
}
|
}
|
||||||
@@ -339,6 +406,16 @@ impl fmt::Display for FormatError {
|
|||||||
FormatError::SelectionOutOfBounds(msg) => {
|
FormatError::SelectionOutOfBounds(msg) => {
|
||||||
write!(f, "selection out of bounds: {msg}")
|
write!(f, "selection out of bounds: {msg}")
|
||||||
}
|
}
|
||||||
|
FormatError::UserBlockNotStripped(n) => write!(
|
||||||
|
f,
|
||||||
|
"file has a {n}-byte user block: parse the bytes from the superblock on \
|
||||||
|
(signature::split_user_block)"
|
||||||
|
),
|
||||||
|
FormatError::SharedMessageTargetMissing(t) => write!(
|
||||||
|
f,
|
||||||
|
"shared message reference points at an object header with no message of type \
|
||||||
|
{t:#06x}"
|
||||||
|
),
|
||||||
FormatError::UnresolvedSharedMessage => write!(
|
FormatError::UnresolvedSharedMessage => write!(
|
||||||
f,
|
f,
|
||||||
"message is shared but no file data was available to resolve it"
|
"message is shared but no file data was available to resolve it"
|
||||||
@@ -382,9 +459,17 @@ impl fmt::Display for FormatError {
|
|||||||
FormatError::InvalidFilterPipelineVersion(v) => {
|
FormatError::InvalidFilterPipelineVersion(v) => {
|
||||||
write!(f, "invalid filter pipeline version: {v}")
|
write!(f, "invalid filter pipeline version: {v}")
|
||||||
}
|
}
|
||||||
FormatError::UnsupportedFilter(id) => {
|
FormatError::UnsupportedFilter(id) => match crate::filter_registry::known_filter(*id) {
|
||||||
write!(f, "unsupported filter: {id}")
|
Some((name, Some(feature))) => write!(
|
||||||
}
|
f,
|
||||||
|
"unsupported filter: {id} ({name}; this build lacks the `{feature}` feature)"
|
||||||
|
),
|
||||||
|
Some((name, None)) => write!(
|
||||||
|
f,
|
||||||
|
"unsupported filter: {id} ({name}, not implemented by clawhdf5)"
|
||||||
|
),
|
||||||
|
None => write!(f, "unsupported filter: {id}"),
|
||||||
|
},
|
||||||
FormatError::FilterError(msg) => {
|
FormatError::FilterError(msg) => {
|
||||||
write!(f, "filter error: {msg}")
|
write!(f, "filter error: {msg}")
|
||||||
}
|
}
|
||||||
@@ -421,6 +506,50 @@ impl fmt::Display for FormatError {
|
|||||||
FormatError::Overflow(msg) => {
|
FormatError::Overflow(msg) => {
|
||||||
write!(f, "integer overflow: {msg}")
|
write!(f, "integer overflow: {msg}")
|
||||||
}
|
}
|
||||||
|
FormatError::InvalidObjectHeader(why) => {
|
||||||
|
write!(f, "corrupt object header: {why}")
|
||||||
|
}
|
||||||
|
FormatError::InvalidDatatype(why) => {
|
||||||
|
write!(f, "invalid datatype: {why}")
|
||||||
|
}
|
||||||
|
FormatError::InvalidChunkDimensions(why) => {
|
||||||
|
write!(f, "invalid chunk dimensions: {why}")
|
||||||
|
}
|
||||||
|
FormatError::TruncatedFile {
|
||||||
|
stored_eof,
|
||||||
|
actual_len,
|
||||||
|
} => {
|
||||||
|
write!(
|
||||||
|
f,
|
||||||
|
"truncated file: the superblock records end of file {stored_eof}, \
|
||||||
|
but the file is {actual_len} bytes"
|
||||||
|
)
|
||||||
|
}
|
||||||
|
FormatError::InvalidLinkName => {
|
||||||
|
write!(f, "invalid link name: a group entry has an empty name")
|
||||||
|
}
|
||||||
|
FormatError::InvalidDataspace(why) => {
|
||||||
|
write!(f, "invalid dataspace: {why}")
|
||||||
|
}
|
||||||
|
FormatError::InvalidDatasetStorage(why) => {
|
||||||
|
write!(f, "invalid dataset storage: {why}")
|
||||||
|
}
|
||||||
|
FormatError::InvalidSuperblockExtension(why) => {
|
||||||
|
write!(f, "invalid superblock extension: {why}")
|
||||||
|
}
|
||||||
|
FormatError::InvalidCacheImage(why) => {
|
||||||
|
write!(f, "invalid metadata cache image: {why}")
|
||||||
|
}
|
||||||
|
FormatError::Storage(why) => {
|
||||||
|
write!(f, "storage read failed: {why}")
|
||||||
|
}
|
||||||
|
FormatError::ContiguousStorageRequired(what) => {
|
||||||
|
write!(
|
||||||
|
f,
|
||||||
|
"{what} needs the whole file in memory, which this storage backend does \
|
||||||
|
not provide"
|
||||||
|
)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user