Compare commits
429
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
00b6f76ee0 | ||
|
|
f9edf4d6ad | ||
|
|
56684ff147 | ||
|
|
b6bbe6b604 | ||
|
|
d279ee06a2 | ||
|
|
c033660d0d | ||
|
|
5e9160f82c | ||
|
|
bce07e9cb9 | ||
|
|
d0d5347cd9 | ||
|
|
31efac1ae1 | ||
|
|
3eca5d8334 | ||
|
|
5eae9ee60b | ||
|
|
d3d8d7ded3 | ||
|
|
c27a478e44 | ||
|
|
a48cb9f1a4 | ||
|
|
3100f0143b | ||
|
|
2bffd6b622 | ||
|
|
c53c43b14f | ||
|
|
14790487a7 | ||
|
|
3c557c9f0a | ||
|
|
419cb52287 | ||
|
|
cab952bb00 | ||
|
|
999cb86071 | ||
|
|
b55b24b7ba | ||
|
|
b660952421 | ||
|
|
b0b4018919 | ||
|
|
a90373ca84 | ||
|
|
4d1a43fd7a | ||
|
|
38107b90ed | ||
|
|
24cc9d14d8 | ||
|
|
ac0020594b | ||
|
|
06d8e2ee45 | ||
|
|
798331ddbc | ||
|
|
9b5803f587 | ||
|
|
694ee0a090 | ||
|
|
bf5a163dcf | ||
|
|
4f90ce02a7 | ||
|
|
5b45c60c9c | ||
|
|
2e5b059530 | ||
|
|
761bdbf24f | ||
|
|
6f5d14fd62 | ||
|
|
ca81c3ebfa | ||
|
|
05e1136027 | ||
|
|
ff0b2f8a4e | ||
|
|
96086add99 | ||
|
|
5c44630ea2 | ||
|
|
425585ee71 | ||
|
|
7179006aee | ||
|
|
b39a705e77 | ||
|
|
d239292655 | ||
|
|
7a8fae0357 | ||
|
|
175d3a1f50 | ||
|
|
4ad80736f4 | ||
|
|
fcf45b6845 | ||
|
|
e80b12521e | ||
|
|
5d42b4241c | ||
|
|
527a1a7ac6 | ||
|
|
18e9be0af7 | ||
|
|
7a4acbce12 | ||
|
|
2379e6163e | ||
|
|
581c6ddef8 | ||
|
|
d09e55e229 | ||
|
|
e553153e48 | ||
|
|
b0930708b6 | ||
|
|
173de3e0d2 | ||
|
|
39f25e5d4e | ||
|
|
bdf584abb2 | ||
|
|
1f7651644b | ||
|
|
1d065adf8b | ||
|
|
2e906ffe3a | ||
|
|
dbafa952ac | ||
|
|
8ef7e80473 | ||
|
|
4467d9dd32 | ||
|
|
ae6b960453 | ||
|
|
92fb0830e0 | ||
|
|
3f45755d58 | ||
|
|
21f104bb77 | ||
|
|
f825a89e23 | ||
|
|
b8c85f7627 | ||
|
|
5107583b97 | ||
|
|
910d81904c | ||
|
|
076feb089a | ||
|
|
e397376820 | ||
|
|
13b36e2bd7 | ||
|
|
a4c2aced55 | ||
|
|
ef428d756c | ||
|
|
4313917b4d | ||
|
|
011e0dbb96 | ||
|
|
f37e7ae326 | ||
|
|
7447dce121 | ||
|
|
93e2d5f365 | ||
|
|
2893b6c974 | ||
|
|
ea0508aaa5 | ||
|
|
75444950f3 | ||
|
|
8236b0e30a | ||
|
|
c2ae7846c9 | ||
|
|
a69c5be8b2 | ||
|
|
930921e8cb | ||
|
|
159e588550 | ||
|
|
6185874f9c | ||
|
|
89e7977943 | ||
|
|
67e72b30d7 | ||
|
|
0e98ffc498 | ||
|
|
efb88f94e3 | ||
|
|
ef480746da | ||
|
|
c5b2afbc35 | ||
|
|
b086dc3c2b | ||
|
|
30a1ed6b9c | ||
|
|
61e34927dc | ||
|
|
680c90b3a8 | ||
|
|
7d629f49e3 | ||
|
|
c04e34620e | ||
|
|
8df5b209a7 | ||
|
|
e8aaf050be | ||
|
|
5062b907bd | ||
|
|
4f5697fdd9 | ||
|
|
955dd1c691 | ||
|
|
ebe51f8e97 | ||
|
|
4e8109770d | ||
|
|
c513f7e6d7 | ||
|
|
a4f586e657 | ||
|
|
c54c64cc9b | ||
|
|
4ff3e40fea | ||
|
|
0aca0eb724 | ||
|
|
db2554dd81 | ||
|
|
955fdb660d | ||
|
|
304aed5813 | ||
|
|
1ffd013de9 | ||
|
|
dc9cfba6bb | ||
|
|
773f427f16 | ||
|
|
7e5e920c72 | ||
|
|
f191dc09d5 | ||
|
|
1c3ef98828 | ||
|
|
e9c71e5d2e | ||
|
|
17201e279d | ||
|
|
3fa5ed1dda | ||
|
|
42894bf93b | ||
|
|
8f59b2e1c2 | ||
|
|
0645dcf173 | ||
|
|
cadd27df5b | ||
|
|
b49ec39aff | ||
|
|
8fadb9f424 | ||
|
|
c233fbca6e | ||
|
|
437e81cfff | ||
|
|
234dd3e36c | ||
|
|
0e8522cfad | ||
|
|
895c79a2fe | ||
|
|
76c97f6c94 | ||
|
|
e5359354b7 | ||
|
|
052098bf36 | ||
|
|
fe377266e1 | ||
|
|
b668878129 | ||
|
|
485bea0f4f | ||
|
|
1ea9132e10 | ||
|
|
04a7f6f6c7 | ||
|
|
5b3d32b37d | ||
|
|
f7e2ab12f2 | ||
|
|
b6cbd2319f | ||
|
|
92c8285549 | ||
|
|
d2b25f154f | ||
|
|
85efde0b4a | ||
|
|
677dc5ec7c | ||
|
|
3c89a31df0 | ||
|
|
1b4a93f65a | ||
|
|
b41583113a | ||
|
|
02e89c1d2d | ||
|
|
5d17712adb | ||
|
|
476960f4b8 | ||
|
|
24f0c71939 | ||
|
|
5705866d40 | ||
|
|
2b7065998a | ||
|
|
6a535e2651 | ||
|
|
a65a2b7f18 | ||
|
|
2c292404d2 | ||
|
|
ba3f476be6 | ||
|
|
ff6d644391 | ||
|
|
cf2b408a63 | ||
|
|
d97b3d703a | ||
|
|
23a4784e72 | ||
|
|
a0160730f2 | ||
|
|
24cbf12f16 | ||
|
|
bcf3ae4856 | ||
|
|
06625b7470 | ||
|
|
aab7ea9e8f | ||
|
|
6a9bb02f37 | ||
|
|
0d908facd3 | ||
|
|
cd828725c7 | ||
|
|
512a6a753f | ||
|
|
6a4707d791 | ||
|
|
6248b411f0 | ||
|
|
479d8b47e0 | ||
|
|
f2ff2c424f | ||
|
|
c5334b1c97 | ||
|
|
d0e3beb3aa | ||
|
|
9a73299594 | ||
|
|
4b02e7d068 | ||
|
|
d7f07fa5c1 | ||
|
|
bdb2c0e36b | ||
|
|
55e0e7e9cf | ||
|
|
00f94d57ed | ||
|
|
4c01267b76 | ||
|
|
d493d4792e | ||
|
|
6559a91495 | ||
|
|
60502593b7 | ||
|
|
a6ed3a5c7d | ||
|
|
3da118d2ee | ||
|
|
7515e5dcbd | ||
|
|
742ed4dfb8 | ||
|
|
2ba4bc97d8 | ||
|
|
d9e4dfb6e6 | ||
|
|
22dc87b07c | ||
|
|
e05530a805 | ||
|
|
989335b67b | ||
|
|
bf4aefcd00 | ||
|
|
378afa1584 | ||
|
|
67958b08d9 | ||
|
|
f512bf3d09 | ||
|
|
8295d01614 | ||
|
|
94b6df986c | ||
|
|
6b3d003950 | ||
|
|
d110b1d945 | ||
|
|
b3058ca46e | ||
|
|
d3d73676c0 | ||
|
|
56abaec75e | ||
|
|
7334e21c93 | ||
|
|
b22b15f00a | ||
|
|
9e608b975c | ||
|
|
9e9b849dd7 | ||
|
|
193a5f8a82 | ||
|
|
3aab433edb | ||
|
|
1c1af460b6 | ||
|
|
c5cd14c2b2 | ||
|
|
16b7359485 | ||
|
|
1207df5189 | ||
|
|
f0db817678 | ||
|
|
9e59499c56 | ||
|
|
d63c76e7ab | ||
|
|
cc1c872a93 | ||
|
|
7acfb79584 | ||
|
|
de2a53f613 | ||
|
|
dda28d6c72 | ||
|
|
408f69ec1d | ||
|
|
73a01f1256 | ||
|
|
956e55c76a | ||
|
|
846c35455d | ||
|
|
ca779b2864 | ||
|
|
20bd381c87 | ||
|
|
5a202f3791 | ||
|
|
8dcce084ca | ||
|
|
45d617c39e | ||
|
|
d345ffbf80 | ||
|
|
17edfe2cf0 | ||
|
|
546fdb84fa | ||
|
|
05b0192a60 | ||
|
|
8bcae3c78e | ||
|
|
f0ecae38b6 | ||
|
|
400e3a9fec | ||
|
|
bd1d8f1a59 | ||
|
|
751edeb7e6 | ||
|
|
b43bd2e67f | ||
|
|
81a0e8685d | ||
|
|
41b7837d0a | ||
|
|
8c51b05b9c | ||
|
|
24412a0e59 | ||
|
|
3bcd443e63 | ||
|
|
37770f594a | ||
|
|
a5e41c1a53 | ||
|
|
b0a1e4f9a6 | ||
|
|
bd36fe883b | ||
|
|
d102c06306 | ||
|
|
6e8421a81e | ||
|
|
8ce6eca34d | ||
|
|
c3850a0b66 | ||
|
|
2bc4cb46a6 | ||
|
|
f7d88bb4fb | ||
|
|
f99587c27d | ||
|
|
2d4b211523 | ||
|
|
10da8f0d09 | ||
|
|
83cda847dd | ||
|
|
78c769f179 | ||
|
|
006bf3b131 | ||
|
|
8cbbef3fae | ||
|
|
55309dd242 | ||
|
|
63648c7000 | ||
|
|
91644d8aaf | ||
|
|
72306c6013 | ||
|
|
e60bde3579 | ||
|
|
c85a8222cc | ||
|
|
2b68791f6a | ||
|
|
f7c362cef5 | ||
|
|
13c095a3da | ||
|
|
591aa71d12 | ||
|
|
b9a2ce3077 | ||
|
|
743c32b512 | ||
|
|
993214723e | ||
|
|
3938f7f8a2 | ||
|
|
dd40bea467 | ||
|
|
afae86f3ea | ||
|
|
f713847e65 | ||
|
|
17fc8b1964 | ||
|
|
9238605661 | ||
|
|
738b9491b2 | ||
|
|
e10df68ed8 | ||
|
|
c4d96c1390 | ||
|
|
b8492bd28d | ||
|
|
a5bd70216c | ||
|
|
17f09375ad | ||
|
|
386bd1d41e | ||
|
|
b5e43bacd7 | ||
|
|
a14ccc36bf | ||
|
|
7f52a6f3ba | ||
|
|
f325d111f3 | ||
|
|
699ee9c447 | ||
|
|
9416c58723 | ||
|
|
6a8ee3ec7f | ||
|
|
a59d83d47d | ||
|
|
bb39be7f24 | ||
|
|
845a9d0125 | ||
|
|
7d7a7e75d4 | ||
|
|
0685037593 | ||
|
|
e92faa23a6 | ||
|
|
40968b3578 | ||
|
|
310448bfcb | ||
|
|
e73ac2af09 | ||
|
|
e01160299a | ||
|
|
5461a13984 | ||
|
|
056092b082 | ||
|
|
a5ca970015 | ||
|
|
e7a7951f1e | ||
|
|
1f71f3bcbc | ||
|
|
3cf8cd86f2 | ||
|
|
34987ec194 | ||
|
|
b58d61cfb7 | ||
|
|
1abd93e0f8 | ||
|
|
6dfd239011 | ||
|
|
07094e34a9 | ||
|
|
f4dee1cd08 | ||
|
|
e38f9123db | ||
|
|
a42b646689 | ||
|
|
74f9f50086 | ||
|
|
d16544b928 | ||
|
|
3b24e6753b | ||
|
|
e815eb922f | ||
|
|
c9c5337a62 | ||
|
|
bb78d70b99 | ||
|
|
a7de15534c | ||
|
|
10d1029ead | ||
|
|
883980f2bd | ||
|
|
d6e426e6d5 | ||
|
|
256e7b89e4 | ||
|
|
f2e704abf3 | ||
|
|
61f36516d7 | ||
|
|
45720fe5a6 | ||
|
|
adf961c883 | ||
|
|
b4a44a2e66 | ||
|
|
17fa783dce | ||
|
|
90e050944f | ||
|
|
a6e90f3ee3 | ||
|
|
0555794850 | ||
|
|
efc2dc53c9 | ||
|
|
5c2f656fe7 | ||
|
|
945b13a1f1 | ||
|
|
9179aa356e | ||
|
|
e94a52a88b | ||
|
|
d54a0f4737 | ||
|
|
aadfd18d4c | ||
|
|
2c6c6c176e | ||
|
|
190918a478 | ||
|
|
38d0d4de02 | ||
|
|
1c85986079 | ||
|
|
c7092722aa | ||
|
|
36356ba8a1 | ||
|
|
85eb7f5ce2 | ||
|
|
8196fab72a | ||
|
|
8ebd488d9e | ||
|
|
42b81d9f1c | ||
|
|
72b9cfb1e1 | ||
|
|
650f355219 | ||
|
|
7f5cfee281 | ||
|
|
c5302e587e | ||
|
|
e1115bc92a | ||
|
|
36d7a6f234 | ||
|
|
4b23ad697c | ||
|
|
e7f2d8575d | ||
|
|
7c1968a34a | ||
|
|
6db13c60b8 | ||
|
|
d99426be94 | ||
|
|
f5505fb03d | ||
|
|
95dcb04454 | ||
|
|
1dba7b465a | ||
|
|
57e938c438 | ||
|
|
3000b40cf3 | ||
|
|
bc820fbd8c | ||
|
|
9066d34eaa | ||
|
|
540fa08907 | ||
|
|
14876b8ae5 | ||
|
|
5935e13866 | ||
|
|
8c3ef996ea | ||
|
|
74fdf0582b | ||
|
|
44f5f8b5c5 | ||
|
|
2f252df084 | ||
|
|
c8c2930fc0 | ||
|
|
417c9516ca | ||
|
|
53dbddb07b | ||
|
|
081341b433 | ||
|
|
d074385944 | ||
|
|
4a1876faf2 | ||
|
|
be88e3fec7 | ||
|
|
e162c013fd | ||
|
|
06dda26d85 | ||
|
|
585e14d5e2 | ||
|
|
183d96ee26 | ||
|
|
bba1560416 | ||
|
|
b36998ef01 | ||
|
|
aef8e766ae | ||
|
|
9ea44d473d | ||
|
|
46203ea761 | ||
|
|
75bdb53342 | ||
|
|
dd5b3f6633 | ||
|
|
79dfa78e8f | ||
|
|
87d64588e5 | ||
|
|
bdadf3447c | ||
|
|
0c65a27b00 | ||
|
|
db9af7972c | ||
|
|
7706697feb | ||
|
|
c0f704c381 | ||
|
|
a7920bd4b3 | ||
|
|
7e43b5366c | ||
|
|
73bb068264 |
+14
-3
@@ -22,6 +22,9 @@ jobs:
|
|||||||
run: rustup component add rustfmt clippy
|
run: rustup component add rustfmt clippy
|
||||||
- name: Install thumbv7em-none-eabihf target
|
- name: Install thumbv7em-none-eabihf target
|
||||||
run: rustup target add thumbv7em-none-eabihf
|
run: rustup target add thumbv7em-none-eabihf
|
||||||
|
- name: Install wasm32-unknown-unknown target
|
||||||
|
# ci-test.sh builds the reader and clawhdf5-wasm for the browser.
|
||||||
|
run: rustup target add wasm32-unknown-unknown
|
||||||
- name: Install Python interop dependencies
|
- name: Install Python interop dependencies
|
||||||
# The interop suites used to skip silently when python3/h5py were
|
# The interop suites used to skip silently when python3/h5py were
|
||||||
# missing, so they never ran in CI. Install them and make a missing
|
# missing, so they never ran in CI. Install them and make a missing
|
||||||
@@ -31,12 +34,20 @@ jobs:
|
|||||||
# cmake builds libz-ng-sys for the opt-in `fast-deflate` (zlib-ng)
|
# cmake builds libz-ng-sys for the opt-in `fast-deflate` (zlib-ng)
|
||||||
# steps in ci-test.sh; rust:latest does not ship it. The default
|
# steps in ci-test.sh; rust:latest does not ship it. The default
|
||||||
# build (pure-Rust zlib-rs) does not need it.
|
# build (pure-Rust zlib-rs) does not need it.
|
||||||
apt-get install -y --no-install-recommends python3 python3-venv cmake
|
# hdf5-tools: h5ls/h5stat/h5dump/h5diff, which the h5rs
|
||||||
|
# (clawhdf5-tools) interop tests compare against.
|
||||||
|
apt-get install -y --no-install-recommends python3 python3-venv cmake hdf5-tools
|
||||||
python3 -m venv /opt/interop
|
python3 -m venv /opt/interop
|
||||||
/opt/interop/bin/pip install --no-cache-dir h5py numpy netCDF4 xarray
|
# maturin + pytest: ci-test.sh builds the Python package
|
||||||
|
# (crates/clawhdf5-py) and runs its tests against h5py.
|
||||||
|
/opt/interop/bin/pip install --no-cache-dir h5py numpy netCDF4 xarray h5netcdf hdf5plugin maturin pytest
|
||||||
echo "/opt/interop/bin" >> "$GITHUB_PATH"
|
echo "/opt/interop/bin" >> "$GITHUB_PATH"
|
||||||
- name: Show interop library versions
|
- name: Show interop library versions
|
||||||
run: /opt/interop/bin/python -c "import h5py, netCDF4; print('h5py', h5py.__version__, 'HDF5', h5py.version.hdf5_version, 'netCDF4', netCDF4.__version__)"
|
# h5dump's version too: the h5rs dump test requires its exact output
|
||||||
|
# (checked against Debian's 1.14.5 in rust:latest and 1.14.6).
|
||||||
|
run: |
|
||||||
|
/opt/interop/bin/python -c "import h5py, netCDF4, hdf5plugin; print('h5py', h5py.__version__, 'HDF5', h5py.version.hdf5_version, 'netCDF4', netCDF4.__version__, 'hdf5plugin', hdf5plugin.version)"
|
||||||
|
h5dump --version
|
||||||
- name: Run CI script
|
- name: Run CI script
|
||||||
env:
|
env:
|
||||||
# Name the interpreter outright rather than relying on $GITHUB_PATH
|
# Name the interpreter outright rather than relying on $GITHUB_PATH
|
||||||
|
|||||||
@@ -0,0 +1,58 @@
|
|||||||
|
name: Conformance
|
||||||
|
# Nightly: read every file of the pinned public HDF5 corpora with clawhdf5 and
|
||||||
|
# with h5py/libhdf5 and compare (conformance/run.sh; CONFORMANCE.md explains
|
||||||
|
# the method). Fails on any panic, hang, crash or out-of-memory in clawhdf5,
|
||||||
|
# and when the ok count drops below conformance/baseline.json or a file the
|
||||||
|
# baseline lists as ok stops being ok. The report is printed into the job log;
|
||||||
|
# nothing is uploaded (artifact actions are JavaScript, which rust:latest
|
||||||
|
# cannot run — see CLAUDE.md).
|
||||||
|
on:
|
||||||
|
schedule:
|
||||||
|
- cron: "17 3 * * *"
|
||||||
|
workflow_dispatch:
|
||||||
|
jobs:
|
||||||
|
conformance:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
container: rust:latest
|
||||||
|
timeout-minutes: 60
|
||||||
|
env:
|
||||||
|
CARGO_NET_RETRY: "10"
|
||||||
|
steps:
|
||||||
|
# Plain git, not actions/checkout (a JavaScript action; see ci.yml).
|
||||||
|
- name: Check out
|
||||||
|
run: |
|
||||||
|
git init -q .
|
||||||
|
git remote add origin "${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}.git"
|
||||||
|
for i in 1 2 3; do git fetch -q --depth 1 origin "${GITHUB_SHA}" && break; sleep 5; done
|
||||||
|
git checkout -q FETCH_HEAD
|
||||||
|
- name: Install h5py, h5dump and the probe's codec libraries
|
||||||
|
# hdf5-tools: h5dump for the CVE-corpus comparison. libaec-dev and
|
||||||
|
# pkg-config: the probe builds clawhdf5-format with `szip` (the core
|
||||||
|
# crates' default build needs neither).
|
||||||
|
run: |
|
||||||
|
apt-get update
|
||||||
|
apt-get install -y --no-install-recommends python3 python3-venv hdf5-tools libaec-dev pkg-config
|
||||||
|
python3 -m venv /opt/conformance
|
||||||
|
/opt/conformance/bin/pip install --no-cache-dir -r conformance/requirements.txt
|
||||||
|
/opt/conformance/bin/python -c "import h5py, hdf5plugin; print('h5py', h5py.__version__, 'HDF5', h5py.version.hdf5_version, 'hdf5plugin', hdf5plugin.version)"
|
||||||
|
h5dump --version
|
||||||
|
- name: Probe unit tests
|
||||||
|
run: cargo test --release --manifest-path conformance/probe/Cargo.toml
|
||||||
|
env:
|
||||||
|
CARGO_TARGET_DIR: conformance/.cache/target
|
||||||
|
- name: Reference-side tests
|
||||||
|
run: /opt/conformance/bin/python conformance/test_ref.py
|
||||||
|
- name: Sweep
|
||||||
|
# The corpora come from GitHub (pinned commits, conformance/corpus.txt),
|
||||||
|
# so this job needs a runner that reaches github.com.
|
||||||
|
env:
|
||||||
|
CLAWHDF5_PYTHON: /opt/conformance/bin/python
|
||||||
|
run: bash conformance/run.sh
|
||||||
|
- name: Report
|
||||||
|
if: always()
|
||||||
|
run: |
|
||||||
|
if [ -f CONFORMANCE.md ]; then cat CONFORMANCE.md; else echo "no report was generated"; fi
|
||||||
|
if [ -f conformance/.cache/results/summary.md ]; then
|
||||||
|
echo; echo "---- per-file detail (conformance/.cache/results/summary.md) ----"
|
||||||
|
cat conformance/.cache/results/summary.md
|
||||||
|
fi
|
||||||
@@ -5,3 +5,5 @@ benchmarks/longmemeval/*.json
|
|||||||
# Local model weights (MiniLM etc.) — large, not committed
|
# Local model weights (MiniLM etc.) — large, not committed
|
||||||
weights/
|
weights/
|
||||||
.venv
|
.venv
|
||||||
|
__pycache__/
|
||||||
|
.pytest_cache/
|
||||||
|
|||||||
+560
-31
@@ -30,12 +30,15 @@ target: Criterion stretched it where 5 s could not hold the samples it needed
|
|||||||
> and in memory), Consolidation Efficiency, Ephemeral Tier, Multi-modal Search,
|
> and in memory), Consolidation Efficiency, Ephemeral Tier, Multi-modal Search,
|
||||||
> the Search and Read harnesses, and the "h5bench-Equivalent I/O Benchmarks"
|
> the Search and Read harnesses, and the "h5bench-Equivalent I/O Benchmarks"
|
||||||
> and "Independent Validation: tank" sections. What does not yet meet that bar:
|
> and "Independent Validation: tank" sections. What does not yet meet that bar:
|
||||||
> the LongMemEval rows that need real embeddings (not re-run here, except the
|
> the Consolidation Efficiency 100K cycle row and
|
||||||
> dated float16 comparison), the Consolidation Efficiency 100K cycle row and
|
|
||||||
> memory-reduction part (the 2026-09-24 run was stopped before it produced
|
> memory-reduction part (the 2026-09-24 run was stopped before it produced
|
||||||
> them), the int8 side of "Quantising the index copy" (not re-run), and the
|
> them), the int8 side of "Quantising the index copy" (not re-run), and the
|
||||||
> i7-12650H and macOS M3 Max rows under Cross-Platform Notes. That is a
|
> i7-12650H and macOS M3 Max rows under Cross-Platform Notes. That is a
|
||||||
> known, tracked documentation gap, not a claim that those numbers are wrong.
|
> known, tracked documentation gap, not a claim that those numbers are wrong.
|
||||||
|
> The LongMemEval rows that need real embeddings (vector-only, hybrid, RRF,
|
||||||
|
> stemmed hybrid, re-ranking, the weight sweep, the oracle variant) were
|
||||||
|
> re-run on 2026-09-27 on tank; see "Re-run with real embeddings" under
|
||||||
|
> LongMemEval Results.
|
||||||
>
|
>
|
||||||
> **Correctness note (2026-08-06).** Being dated and reproducible is necessary but
|
> **Correctness note (2026-08-06).** Being dated and reproducible is necessary but
|
||||||
> not sufficient — a number can be perfectly reproducible and still measure the
|
> not sufficient — a number can be perfectly reproducible and still measure the
|
||||||
@@ -48,6 +51,31 @@ target: Criterion stretched it where 5 s could not hold the samples it needed
|
|||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
|
## Current headline numbers
|
||||||
|
|
||||||
|
The newest dated measurement of each headline figure, as of 2026-09-28.
|
||||||
|
Everything below this section is the dated record behind them; sections whose
|
||||||
|
figures a later run replaced are marked *Superseded*. Machine "tank" is an AMD
|
||||||
|
Ryzen 7 7800X3D (8C/16T); rows marked idle were run with the 1-minute load
|
||||||
|
average below 2.
|
||||||
|
|
||||||
|
| Figure | Value | Measured | Command | Details |
|
||||||
|
|---|---|---|---|---|
|
||||||
|
| Agent memory search, `HDF5Memory::hybrid_search` p50 | 0.49 ms at 10K, 4.69 ms at 100K records | 2026-09-24, tank, `5c8323c` | `cargo run --release -p clawhdf5-bench --bin search_harness -- --full` | [Current: search harness](#current-search-harness-2026-09-24) |
|
||||||
|
| LongMemEval `longmemeval_s` (full haystack), default hybrid 0.4/0.6, turn-level retrieval Hit@5 (not QA accuracy) | 81.4% | 2026-09-27, tank, search code of `7a8fae0` | `longmemeval_bench … --embeddings weights/all-minilm-l6-v2` | [Re-run with real embeddings](#re-run-with-real-embeddings-2026-09-27-tank), [Fusion method](#fusion-method--weighted-vs-rrf-full-haystack-n500) |
|
||||||
|
| Loaded store memory, 100K × 384 | 399 MiB (2.72x raw) with the `f32` index; 256 MiB (1.74x) with the int8 index (int8 side not re-run since it was first measured) | `f32`: 2026-09-24, tank, `5c8323c`; int8: 2026-09-19 (`c0a9206`), machine not recorded | `search_harness -- --footprint --full [--int8]` | [Memory footprint](#memory-footprint), [Quantising the index copy](#quantising-the-index-copy-quantized_index) |
|
||||||
|
| int8 index vs `f32` index, QPS at equal recall | 1.63x (x86-64 AVX2), 1.18x (Raspberry Pi 5, `SDOT`) | x86: 2026-09-20 (`dea02f5`), machine not recorded; Pi 5: 2026-09-21 (`114a2df`); not re-checked against the 2026-09-24 `f32` figure | `search_harness -- --full` | [Quantising the index copy](#quantising-the-index-copy-quantized_index), [On ARM](#on-arm-raspberry-pi-5-cortex-a76) |
|
||||||
|
| `float16` store file size, 100K × 384 | 80.8 MiB vs 154.0 MiB `f32` (48% smaller) | 2026-09-23, tank | `search_harness -- --float16-study --full` | [float16 embedding storage](#float16-embedding-storage-memoryconfigfloat16) |
|
||||||
|
| Full reads of chunked deflate data, 16 threads on one `File` | 4944 MB/s, 1.58x 16 h5py processes (noisy run: compare ratios, not MB/s) | 2026-09-26, tank, `c5334b1` | `concurrent_read` + `concurrent_read_h5py.py` | [Results after in-place chunk decoding](#results-after-in-place-chunk-decoding-2026-09-26-tank-c5334b1) |
|
||||||
|
| Same, clawhdf5 only, against the build before range-read M2/M3 | 8525 MB/s vs 6258 (+36%); contiguous and metadata reads at parity | 2026-09-27, tank (idle), `7a8fae0` vs `8f59b2e` | `concurrent_read --decode-threads 1 --reps 3` | [Local metadata and data reads after range-read M2/M3](#local-metadata-and-data-reads-after-range-read-m2m3-2026-09-27-tank) |
|
||||||
|
| `ObjectHeader::parse` (401 headers) | 23.5–23.6 µs, 1.0–2.6% below `8f59b2e` | 2026-09-27, tank (idle), `96086ad` | `cargo bench -p clawhdf5 --bench local_metadata_bench` | [`ObjectHeader::parse` back at 8f59b2e's speed](#objectheaderparse-back-at-8f59b2es-speed-2026-09-27-tank) |
|
||||||
|
| Selection reads, 64 MB chunked + deflate `f64` | full 63.2 ms; one 64 × 64 window 0.18 ms | 2026-09-24, tank, `5c8323c` | `cargo run --release -p clawhdf5-bench --bin read_harness` | [Current: read harness](#current-read-harness-2026-09-24) |
|
||||||
|
| Deflate backend, zlib-rs (default) vs zlib-ng | within 6% on every HDF5 read/write path | 2026-09-23, tank | `cargo bench -p clawhdf5-filters --bench deflate_bench` (and the two commands with it) | [Deflate backend](#deflate-backend-zlib-rs-vs-zlib-ng) |
|
||||||
|
| vs libhdf5 1.14.6: chunked deflate-6 write 512×512 / 128 attributes / 64 groups | 35x (1.46 vs 51.4 ms, pure-Rust deflate) / 10.3x / 10.6x | write 2026-09-23, tank; attributes and groups 2026-08-03, tank | `cargo bench -p clawhdf5-bench --bench h5bench_write --features libhdf5-compare -- '^write_2d_chunked/'`; `cargo bench -p clawhdf5-bench --features libhdf5-compare` | [Deflate backend](#deflate-backend-zlib-rs-vs-zlib-ng), [Independent Validation: tank](#independent-validation-tank-ryzen-7-7800x3d-2026-08-03) |
|
||||||
|
| Signed checkpoints | about 20% of a checkpoint (598 vs 495 ms at 100K) | 2026-09-25, tank | `search_harness -- --signing-study --full` | [Signed checkpoints](#signed-checkpoints) |
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
## Memory footprint
|
## Memory footprint
|
||||||
|
|
||||||
`cargo run --release -p clawhdf5-bench --bin search_harness -- --footprint --full`,
|
`cargo run --release -p clawhdf5-bench --bin search_harness -- --footprint --full`,
|
||||||
@@ -61,6 +89,10 @@ change at all. Measured that way a store holding the corpus twice and one
|
|||||||
holding it once came out *identical* (1.00x both), which is how the first
|
holding it once came out *identical* (1.00x both), which is how the first
|
||||||
attempt at this measurement went.
|
attempt at this measurement went.
|
||||||
|
|
||||||
|
> *Superseded* by the current figures below (2026-09-24): this table is the
|
||||||
|
> record of the double-copy fix (commit 2e7e045, undated); the store measured
|
||||||
|
> 2.72x, not 2.43x, by the time the int8 index landed.
|
||||||
|
|
||||||
| N | vectors (raw) | reopened, before | reopened, after |
|
| N | vectors (raw) | reopened, before | reopened, after |
|
||||||
|---:|---:|---:|---:|
|
|---:|---:|---:|---:|
|
||||||
| 1 000 | 1 MiB | 5 MiB (3.41x) | 4 MiB (2.39x) |
|
| 1 000 | 1 MiB | 5 MiB (3.41x) | 4 MiB (2.39x) |
|
||||||
@@ -243,6 +275,30 @@ index asked for ~16 000 candidates, where scanning the few hundred or thousand
|
|||||||
allowed records is exact and cheap. Re-ranking a 3k candidate pool and
|
allowed records is exact and cheap. Re-ranking a 3k candidate pool and
|
||||||
confidence rejection add about 3%.
|
confidence rejection add about 3%.
|
||||||
|
|
||||||
|
### Signed checkpoints
|
||||||
|
|
||||||
|
Measured 2026-09-25 on tank (AMD Ryzen 7 7800X3D). A default store (float16,
|
||||||
|
int8 index), 384-dim; each checkpoint rewrites the whole file, as every
|
||||||
|
checkpoint does. Medians of five checkpoints and three verifies; three runs
|
||||||
|
agreed to within the ranges shown.
|
||||||
|
|
||||||
|
```bash
|
||||||
|
cargo run --release -p clawhdf5-bench --bin search_harness -- --signing-study --full
|
||||||
|
```
|
||||||
|
|
||||||
|
| N | checkpoint, unsigned | checkpoint, signed | signing adds | `verify` | file size added |
|
||||||
|
|---:|---:|---:|---:|---:|---:|
|
||||||
|
| 1 000 | 5.4 ms | 6.4 ms | 0.7–1.0 ms | 2.1 ms | 0.03 MiB |
|
||||||
|
| 10 000 | 46 ms | 55 ms | 8.1–9.4 ms | 18.6 ms | 0.31 MiB |
|
||||||
|
| 100 000 | 495 ms | 598 ms | 89–112 ms | 247 ms | 3.05 MiB |
|
||||||
|
|
||||||
|
Signing costs about 20% of a checkpoint: every record is rehashed (SHA-256)
|
||||||
|
and the Merkle root recomputed each time; the Ed25519 signature itself is
|
||||||
|
microseconds. Caching per-record hashes between checkpoints would cut this to
|
||||||
|
the records that changed. The per-record hashes stored for locating edits are
|
||||||
|
32 bytes each (4% of a 100K float16 store). `verify` reads and rehashes the
|
||||||
|
whole checkpoint.
|
||||||
|
|
||||||
### float16 embedding storage (`MemoryConfig::float16`)
|
### float16 embedding storage (`MemoryConfig::float16`)
|
||||||
|
|
||||||
Measured 2026-09-23 on tank (AMD Ryzen 7 7800X3D). The same clustered
|
Measured 2026-09-23 on tank (AMD Ryzen 7 7800X3D). The same clustered
|
||||||
@@ -301,6 +357,19 @@ which of two gold sessions ranks first, out of ~320. Those flips show the
|
|||||||
half-precision path was in effect; they do not change a single hit. The f32
|
half-precision path was in effect; they do not change a single hit. The f32
|
||||||
run reproduces the published hybrid numbers exactly.
|
run reproduces the published hybrid numbers exactly.
|
||||||
|
|
||||||
|
**Re-checked 2026-09-27** (tank, commit 7a8fae0, the same pair of runs on
|
||||||
|
`longmemeval_s_cleaned.json`, which is the same file; the machine was not
|
||||||
|
idle, which does not affect recall): the result is the same. The table above
|
||||||
|
reproduced exactly. Every Hit@k and MRR of the eight modes matched between f32
|
||||||
|
and float16 at both levels, with two exceptions: RRF's session MRR (0.9253 vs
|
||||||
|
0.9254) and two per-type session MRRs in the fourth decimal. Three modes
|
||||||
|
differed by one question in the recency count. That re-run also corrects the
|
||||||
|
sentence above: two f32 runs on the same day differed by one question in
|
||||||
|
recency as well, so those flips are run-to-run variation and do not show that
|
||||||
|
the half-precision path was in effect. `--float16` is what shows that: the
|
||||||
|
harness prints "Stores use MemoryConfig::float16" and `MemoryConfig::float16`
|
||||||
|
is set on every store.
|
||||||
|
|
||||||
### Opening a store (`read_from_disk`)
|
### Opening a store (`read_from_disk`)
|
||||||
|
|
||||||
`HDF5Memory::open` memory-mapped the file, copied the whole mapping into a
|
`HDF5Memory::open` memory-mapped the file, copied the whole mapping into a
|
||||||
@@ -339,6 +408,9 @@ point: does a selection cost what the *selection* costs?
|
|||||||
|
|
||||||
### Baseline (v2.4.0): every selection decodes the whole dataset
|
### Baseline (v2.4.0): every selection decodes the whole dataset
|
||||||
|
|
||||||
|
> *Superseded* by [Current: read harness](#current-read-harness-2026-09-24)
|
||||||
|
> (2026-09-24). Kept as the before picture.
|
||||||
|
|
||||||
4096 x 2048 f64 (64 MB per dataset), chunks 256 x 256, file 129 MB
|
4096 x 2048 f64 (64 MB per dataset), chunks 256 x 256, file 129 MB
|
||||||
|
|
||||||
| layout | read | selected | time ms | MB/s of selection | vs full read |
|
| layout | read | selected | time ms | MB/s of selection | vs full read |
|
||||||
@@ -364,6 +436,9 @@ point: does a selection cost what the *selection* costs?
|
|||||||
|
|
||||||
### After: partial reads
|
### After: partial reads
|
||||||
|
|
||||||
|
> *Superseded* by [Current: read harness](#current-read-harness-2026-09-24)
|
||||||
|
> (2026-09-24).
|
||||||
|
|
||||||
Only the rows of a contiguous dataset, or the chunks, that overlap the
|
Only the rows of a contiguous dataset, or the chunks, that overlap the
|
||||||
selection's bounding box are read/decoded. A 64 x 64 window of the compressed
|
selection's bounding box are read/decoded. A 64 x 64 window of the compressed
|
||||||
dataset: **105 -> 0.39 ms**; one row: **106 -> 2.7 ms**; one column:
|
dataset: **105 -> 0.39 ms**; one row: **106 -> 2.7 ms**; one column:
|
||||||
@@ -395,6 +470,9 @@ because the machine's speed drifted; compare the *vs full read* column.)
|
|||||||
|
|
||||||
### After: parallel cached decode, fewer copies (full reads)
|
### After: parallel cached decode, fewer copies (full reads)
|
||||||
|
|
||||||
|
> *Superseded* by [Current: read harness](#current-read-harness-2026-09-24)
|
||||||
|
> (2026-09-24).
|
||||||
|
|
||||||
Full-read times, old and new binaries run alternately at the same moment (this
|
Full-read times, old and new binaries run alternately at the same moment (this
|
||||||
machine's absolute speed drifts over a long session, so only same-moment
|
machine's absolute speed drifts over a long session, so only same-moment
|
||||||
comparisons mean anything):
|
comparisons mean anything):
|
||||||
@@ -458,8 +536,338 @@ The rows and columns of the uncompressed layouts are within 20% (chunked
|
|||||||
column 0.45 -> 0.49 ms, contiguous column 2.55 -> 2.61 ms). This run does not
|
column 0.45 -> 0.49 ms, contiguous column 2.55 -> 2.61 ms). This run does not
|
||||||
explain the slower windows.
|
explain the slower windows.
|
||||||
|
|
||||||
|
## Local file speed after range reads
|
||||||
|
|
||||||
|
### `ObjectHeader::parse` back at 8f59b2e's speed (2026-09-27, tank)
|
||||||
|
|
||||||
|
The remaining 4% (below) was the call to the version-1 message loop, which
|
||||||
|
`4313917` kept out of line with `#[inline(never)]`. Found with A/B builds
|
||||||
|
changing one piece at a time (perf is not available: `perf_event_paranoid`
|
||||||
|
4): `#[inline]` on `parse_v1_messages` alone brought
|
||||||
|
`object_header_parse_x401` from about 24.5–24.9 µs to 23.6–24.0 µs against
|
||||||
|
8f59b2e's 23.6–24.1 µs (short 4-second rounds); no attribute measured like
|
||||||
|
`#[inline(never)]`;
|
||||||
|
creating the chunk list only when a continuation is found measured no
|
||||||
|
faster on top and was not kept.
|
||||||
|
|
||||||
|
Same method as below: `8f59b2e` built in its own worktree and target
|
||||||
|
directory, separate binaries alternating, `taskset -c 5
|
||||||
|
local_metadata_bench --bench --warm-up-time 3 --measurement-time 10`, every
|
||||||
|
binary started with the 1-minute load average below 2 (0.19–1.86) and no
|
||||||
|
`rustc` running. Candidate: `96086ad` (this change). Median (range) of 3
|
||||||
|
rounds; run 2 also alternated `main` `425585e`.
|
||||||
|
|
||||||
|
| function | 8f59b2e | 425585e (main) | 96086ad | vs 8f59b2e |
|
||||||
|
|---|---:|---:|---:|---:|
|
||||||
|
| run 1: `object_header_parse_x401` | 23.81 µs (23.76–24.00) | | 23.57 µs (23.23–23.89) | **−1.0%** |
|
||||||
|
| run 1: `snod_parse_all` | 1.840 µs (1.837–1.854) | | 1.839 µs (1.837–1.873) | 0.0% |
|
||||||
|
| run 1: `btree_v1_walk` | 343 ns (338–349) | | 352 ns (344–363) | +2.7% |
|
||||||
|
| run 1: `facade_list_400_groups` | 8.03 ms (8.01–8.10) | | 8.05 ms (8.01–8.15) | +0.2% |
|
||||||
|
| run 2: `object_header_parse_x401` | 24.15 µs (23.74–24.41) | 24.93 µs (24.72–25.05) | 23.52 µs (23.34–23.68) | **−2.6%** |
|
||||||
|
| run 2: `snod_parse_all` | 1.843 µs (1.830–1.847) | 1.856 µs (1.850–1.865) | 1.861 µs (1.858–1.869) | +1.0% |
|
||||||
|
| run 2: `btree_v1_walk` | 346 ns (339–366) | 356 ns (347–356) | 357 ns (350–359) | +3.2% |
|
||||||
|
| run 2: `facade_list_400_groups` | 8.09 ms (8.02–8.10) | 8.12 ms (7.97–8.19) | 8.08 ms (7.98–8.12) | −0.1% |
|
||||||
|
|
||||||
|
- `ObjectHeader::parse` is at or below 8f59b2e (−1.0%, −2.6%) and 5.6%
|
||||||
|
faster than `main` in the same run.
|
||||||
|
- `btree_v1_walk` (one walk of a 350 ns B-tree) is 3% above 8f59b2e in
|
||||||
|
both runs, with overlapping ranges, and is the same on `main` (+0.2%
|
||||||
|
between `main` and this change): not from this change. The walk's code
|
||||||
|
changed in `e553153` (after a failed child the siblings are only read,
|
||||||
|
so the error returns after them; the fixture never takes that path, but
|
||||||
|
the loop carries the extra state); left as is.
|
||||||
|
- `snod_parse_all` and the facade listing are within noise.
|
||||||
|
|
||||||
|
### Local metadata and data reads after range-read M2/M3 (2026-09-27, tank)
|
||||||
|
|
||||||
|
> The `object_header_parse_x401` row (+4.2%) is *superseded* by
|
||||||
|
> [`ObjectHeader::parse` back at 8f59b2e's speed](#objectheaderparse-back-at-8f59b2es-speed-2026-09-27-tank)
|
||||||
|
> (2026-09-27, `96086ad`); the other rows are current.
|
||||||
|
|
||||||
|
`main` just before range-read M2/M3 (`8f59b2e`, PR #17) against `main`
|
||||||
|
`7a8fae0` (PRs #18 and #19), each built in its own worktree and run as
|
||||||
|
separate binaries, alternating base and candidate. Machine: tank (AMD Ryzen
|
||||||
|
7 7800X3D, 16 threads). **Idle:** every round started with the 1-minute load
|
||||||
|
average below 2 (1.05–1.98; `target/ab-results2/load.log`). Criterion:
|
||||||
|
`taskset -c 5 local_metadata_bench --bench --warm-up-time 3
|
||||||
|
--measurement-time 10`, 3 rounds each. Reads: `concurrent_read --dir
|
||||||
|
~/.cache/concurrent-read --decode-threads 1 --reps 3`, 3 rounds each.
|
||||||
|
Median (range) over the rounds.
|
||||||
|
|
||||||
|
`local_metadata_bench` (the 400-group v1 fixture):
|
||||||
|
|
||||||
|
| function | 8f59b2e | 7a8fae0 | change |
|
||||||
|
|---|---:|---:|---:|
|
||||||
|
| `object_header_parse_x401` | 23.86 µs (23.79–23.96) | 24.86 µs (24.69–24.98) | **+4.2%** |
|
||||||
|
| `snod_parse_all` | 1.842 µs (1.835–1.847) | 1.833 µs (1.833–1.854) | −0.5% |
|
||||||
|
| `btree_v1_walk` | 344 ns (338–346) | 348 ns (337–356) | +1.1% |
|
||||||
|
| `facade_list_400_groups` | 8.24 ms (8.04–8.28) | 8.10 ms (8.04–8.13) | −1.7% |
|
||||||
|
|
||||||
|
`concurrent_read`, MB/s (64 datasets of 64 MiB `f32`; deflate chunks
|
||||||
|
256 x 256, level 4):
|
||||||
|
|
||||||
|
| layout | mode | threads | 8f59b2e | 7a8fae0 | change |
|
||||||
|
|---|---|---:|---:|---:|---:|
|
||||||
|
| deflate | distinct | 1 | 891 (880–892) | 907 (906–908) | +1.7% |
|
||||||
|
| deflate | distinct | 2 | 1692 (1680–1699) | 1777 (1767–1779) | +5.0% |
|
||||||
|
| deflate | distinct | 4 | 3102 (3099–3102) | 3348 (3347–3350) | +8.0% |
|
||||||
|
| deflate | distinct | 8 | 5306 (5168–5345) | 6240 (6226–6248) | +17.6% |
|
||||||
|
| deflate | distinct | 16 | 6258 (6109–6442) | 8525 (8513–8561) | **+36.2%** |
|
||||||
|
| deflate | same | 1 | 234 (234–235) | 233 (233–234) | −0.5% |
|
||||||
|
| deflate | same | 16 | 2443 (2038–2444) | 2468 (2424–2473) | +1.0% |
|
||||||
|
| contiguous | distinct | 1 | 13477 (12949–13874) | 13302 (13287–13578) | −1.3% |
|
||||||
|
| contiguous | distinct | 16 | 12501 (12484–12530) | 12566 (12449–12567) | +0.5% |
|
||||||
|
| contiguous | same | 1 | 29866 (28832–30207) | 29364 (28838–29780) | −1.7% |
|
||||||
|
| contiguous | same | 16 | 163415 (161097–229146) | 233280 (159377–238440) | (noise) |
|
||||||
|
|
||||||
|
What this shows:
|
||||||
|
- **Local metadata reads are at parity or faster.** Listing the 400-group
|
||||||
|
file through the facade is 1.7% faster than before M2/M3; the +7–10%
|
||||||
|
listing regression found while merging #18 is gone.
|
||||||
|
- **`ObjectHeader::parse` alone is 4.2% slower** (about 2.5 ns per header;
|
||||||
|
the base and candidate ranges do not overlap). It is the cost of reading
|
||||||
|
continuation chunks from a bounded queue (the fix for unbounded reads on
|
||||||
|
crafted headers) and does not show in the listing. (Fixed later the
|
||||||
|
same day; see the section above and `docs/known-issues.md`.)
|
||||||
|
- **Full reads of deflate data got faster** after #18 (in-place chunk
|
||||||
|
decoding into the typed output and per-thread scratch buffers): +1.7% on
|
||||||
|
one thread, +36% at 16.
|
||||||
|
- Single-thread contiguous hyperslabs are within noise (−1.7%, overlapping
|
||||||
|
ranges). The multi-thread `contiguous same` rows read one 64 MiB dataset
|
||||||
|
out of the CPU caches and swing widely between rounds of the same build.
|
||||||
|
|
||||||
|
An earlier run the same day at load 2.3–3.3 (two orphaned h5py processes,
|
||||||
|
since stopped, each using a core) reported that single-thread contiguous
|
||||||
|
hyperslab row as −5.6%; the idle rerun above does not reproduce it.
|
||||||
|
|
||||||
|
### Results after in-place chunk decoding (2026-09-26, tank, `c5334b1`)
|
||||||
|
|
||||||
|
Same machine, files and commands, re-run after chunked reads started
|
||||||
|
decoding into reusable per-thread buffers straight into the (typed) output,
|
||||||
|
with the calling thread decoding alongside the pool. Load average 1.78 at
|
||||||
|
the start; it rose to 6-9 during the runs (the clawhdf5 runs' own threads,
|
||||||
|
and it stayed around 5-6 through the h5py runs, so something else was
|
||||||
|
active). **This run was noisier than the previous one: h5py's own contiguous
|
||||||
|
figures are about 40% lower than in the run below, and ours dropped
|
||||||
|
similarly, so compare ratios within a run rather than MB/s across runs.**
|
||||||
|
h5py was re-run in the same session.
|
||||||
|
|
||||||
|
Each read decoding on its calling thread (`--decode-threads 1`, like h5py):
|
||||||
|
|
||||||
|
| layout | mode | threads | clawhdf5 MB/s (eff) | h5py threads MB/s (eff) | h5py processes MB/s (eff) | vs h5py processes |
|
||||||
|
|---|---|---:|---:|---:|---:|---:|
|
||||||
|
| deflate | distinct | 1 | 670 (1.00) | 410 (1.00) | 397 (1.00) | 1.69x |
|
||||||
|
| deflate | distinct | 4 | 2434 (0.91) | 406 (0.25) | 1470 (0.93) | 1.66x |
|
||||||
|
| deflate | distinct | 8 | 3749 (0.70) | 406 (0.12) | 2398 (0.76) | 1.56x |
|
||||||
|
| deflate | distinct | 16 | 4944 (0.46) | 390 (0.06) | 3135 (0.49) | 1.58x |
|
||||||
|
| deflate | same | 1 | 211 (1.00) | 125 (1.00) | 124 (1.00) | 1.70x |
|
||||||
|
| deflate | same | 16 | 1835 (0.54) | 122 (0.06) | 961 (0.48) | 1.91x |
|
||||||
|
| contiguous | distinct | 1 | 6718 (1.00) | 5545 (1.00) | 5200 (1.00) | 1.29x |
|
||||||
|
| contiguous | distinct | 16 | 11035 (0.10) | 4950 (0.06) | 10558 (0.13) | 1.05x |
|
||||||
|
| contiguous | same | 1 | 14483 (1.00) | 2593 (1.00) | 2737 (1.00) | 5.29x |
|
||||||
|
| contiguous | same | 16 | 132175 (0.57) | 2224 (0.05) | 14809 (0.34) | 8.93x |
|
||||||
|
|
||||||
|
With the default rayon pool, deflate `distinct` reads 6143 MB/s from a single
|
||||||
|
thread (15x h5py's 410 on one call) and 4556 MB/s at 16 threads (1.45x h5py
|
||||||
|
processes); the other rows are within the noise of the table above.
|
||||||
|
|
||||||
|
What changed: full reads of chunked datasets were 0.69x-0.76x of h5py
|
||||||
|
processes at 16 threads in the run below, and are 1.58x here; with one
|
||||||
|
thread they were 1.44x and are 1.69x. Minor page faults for the 16-thread
|
||||||
|
run fell from about 4.6M to 0.2M (`/usr/bin/time -v`, provisional, loaded
|
||||||
|
machine). clawhdf5 now reads faster than 16 h5py processes in every row of
|
||||||
|
this benchmark except contiguous full reads at 16 threads, where both
|
||||||
|
saturate memory bandwidth (1.05x).
|
||||||
|
|
||||||
|
### Results after the read fixes (2026-09-26, tank, `408f69e`)
|
||||||
|
|
||||||
|
> *Superseded* by [Results after in-place chunk decoding](#results-after-in-place-chunk-decoding-2026-09-26-tank-c5334b1)
|
||||||
|
> (2026-09-26, `c5334b1`), which closed the 16-thread gap listed at the end
|
||||||
|
> of this section.
|
||||||
|
|
||||||
|
Same machine, files and commands as the first run below, re-run on an idle
|
||||||
|
tank (load average 1.60 at the start; the 1-minute figure rose to about 5
|
||||||
|
during the clawhdf5 runs, mostly their own threads) after two fixes:
|
||||||
|
contiguous reads back their output with transparent huge pages and copy
|
||||||
|
hyperslabs run by run, and full chunked reads no longer queue behind a
|
||||||
|
one-thread rayon pool. h5py was re-run in the same session.
|
||||||
|
|
||||||
|
Each read decoding on its calling thread (`--decode-threads 1`, like h5py):
|
||||||
|
|
||||||
|
| layout | mode | threads | clawhdf5 MB/s (eff) | h5py threads MB/s (eff) | h5py processes MB/s (eff) |
|
||||||
|
|---|---|---:|---:|---:|---:|
|
||||||
|
| deflate | distinct | 1 | 606 (1.00) | 432 (1.00) | 421 (1.00) |
|
||||||
|
| deflate | distinct | 4 | 1816 (0.75) | 428 (0.25) | 1654 (0.98) |
|
||||||
|
| deflate | distinct | 8 | 2943 (0.61) | 428 (0.12) | 3042 (0.90) |
|
||||||
|
| deflate | distinct | 16 | 2142 (0.22) | 375 (0.05) | 3083 (0.46) |
|
||||||
|
| deflate | same | 1 | 154 (1.00) | 130 (1.00) | 129 (1.00) |
|
||||||
|
| deflate | same | 4 | 599 (0.98) | 129 (0.25) | 499 (0.97) |
|
||||||
|
| deflate | same | 16 | 1592 (0.65) | 128 (0.06) | 1399 (0.68) |
|
||||||
|
| contiguous | distinct | 1 | 13665 (1.00) | 9490 (1.00) | 8781 (1.00) |
|
||||||
|
| contiguous | distinct | 16 | 12674 (0.06) | 2285 (0.02) | 6942 (0.05) |
|
||||||
|
| contiguous | same | 1 | 31991 (1.00) | 5087 (1.00) | 5078 (1.00) |
|
||||||
|
| contiguous | same | 16 | 237151 (0.46) | 4304 (0.05) | 35772 (0.44) |
|
||||||
|
|
||||||
|
With the default rayon pool: deflate `distinct` 2117 MB/s at 1 thread (4.9x
|
||||||
|
h5py), 3163 at 4, 2341 at 16 (0.76x h5py processes); deflate `same` 1439 MB/s
|
||||||
|
at 16; contiguous as above within a few percent.
|
||||||
|
|
||||||
|
Before -> after for clawhdf5 (`--decode-threads 1` unless noted):
|
||||||
|
contiguous full read at 1 thread 2495 -> 13665 MB/s (0.25x -> 1.44x h5py);
|
||||||
|
contiguous 256 x 256 hyperslabs at 1 thread 624 -> 31991 MB/s (0.12x ->
|
||||||
|
6.3x); deflate full reads at 8 threads 887 -> 2943 MB/s; deflate
|
||||||
|
hyperslabs at 16 threads 1244 -> 1592 MB/s.
|
||||||
|
|
||||||
|
Read with care:
|
||||||
|
- `contiguous same` reads 1024 slabs of one 64 MiB dataset over and over, so
|
||||||
|
it mostly measures copies out of the CPU's caches (the 7800X3D has 96 MiB
|
||||||
|
of L3); the per-call overhead is what differs (h5py's is about 50 us).
|
||||||
|
- At 16 threads every tool dropped in this run (h5py threads on contiguous
|
||||||
|
data from 8002 to 2285 MB/s, processes from 12846 to 6942), so the
|
||||||
|
16-thread rows are noisier than the others.
|
||||||
|
- Still behind at this commit: full reads of chunked data at 16 threads
|
||||||
|
(0.69x-0.76x h5py processes); fixed by `c5334b1` (above), recorded as
|
||||||
|
fixed in `docs/known-issues.md`.
|
||||||
|
|
||||||
|
### First run, before the read fixes (2026-09-26, tank, `91644d8`)
|
||||||
|
|
||||||
|
> *Superseded* results: the tables and "What this shows" are the before
|
||||||
|
> picture for [Results after in-place chunk decoding](#results-after-in-place-chunk-decoding-2026-09-26-tank-c5334b1)
|
||||||
|
> (2026-09-26). The workload description and the **Run** box below are
|
||||||
|
> still how every `concurrent_read` figure in this file is produced.
|
||||||
|
|
||||||
|
Measured on tank (AMD Ryzen 7 7800X3D, 8 cores / 16 threads, 61 GiB, Linux
|
||||||
|
7.0) at commit `91644d8`, load average 1.84 when the run started (the
|
||||||
|
1-minute figure rose to 3.7 during the runs; that is mostly the benchmark's
|
||||||
|
own threads). Warm page cache. clawhdf5 2.7.0 (workspace), h5py 3.16.0 on
|
||||||
|
HDF5 2.0.0. Commands exactly as in the **Run** box below; files at their
|
||||||
|
defaults (64 datasets of 16384 x 1024 `f32`, 64 MiB each; deflate chunks
|
||||||
|
256 x 256, level 4). MB/s is decoded data, the median of the repetitions;
|
||||||
|
eff is scaling efficiency against the same tool's 1-thread row.
|
||||||
|
|
||||||
|
Each read decoding on its calling thread (`--decode-threads 1`, like h5py):
|
||||||
|
|
||||||
|
| layout | mode | threads | clawhdf5 MB/s (eff) | h5py threads MB/s (eff) | h5py processes MB/s (eff) |
|
||||||
|
|---|---|---:|---:|---:|---:|
|
||||||
|
| deflate | distinct | 1 | 421 (1.00) | 433 (1.00) | 421 (1.00) |
|
||||||
|
| deflate | distinct | 4 | 890 (0.53) | 428 (0.25) | 1651 (0.98) |
|
||||||
|
| deflate | distinct | 16 | 880 (0.13) | 427 (0.06) | 4424 (0.66) |
|
||||||
|
| deflate | same | 1 | 151 (1.00) | 130 (1.00) | 129 (1.00) |
|
||||||
|
| deflate | same | 4 | 490 (0.81) | 129 (0.25) | 497 (0.96) |
|
||||||
|
| deflate | same | 16 | 1244 (0.52) | 128 (0.06) | 1402 (0.68) |
|
||||||
|
| contiguous | distinct | 1 | 2495 (1.00) | 9789 (1.00) | 9169 (1.00) |
|
||||||
|
| contiguous | distinct | 16 | 8083 (0.20) | 8096 (0.05) | 12272 (0.08) |
|
||||||
|
| contiguous | same | 1 | 624 (1.00) | 5022 (1.00) | 5172 (1.00) |
|
||||||
|
| contiguous | same | 16 | 4778 (0.48) | 4411 (0.05) | 37138 (0.45) |
|
||||||
|
|
||||||
|
With the default rayon pool decoding inside each read, deflate `distinct`
|
||||||
|
is 912 MB/s at 1 thread (2.1x h5py) and 2824 MB/s at 16 (6.6x h5py threads,
|
||||||
|
0.64x h5py processes); the other rows are within a few percent of the table
|
||||||
|
above. Full tables (2, 4, 8 threads, both decode modes) come from
|
||||||
|
`compare_concurrent_read.py` on the JSON files.
|
||||||
|
|
||||||
|
What this shows:
|
||||||
|
- **h5py threads do not scale** (flat at about 430 MB/s on deflate, every
|
||||||
|
thread count): libhdf5's global lock.
|
||||||
|
- **clawhdf5 threads on one `File` do, for hyperslab reads of compressed
|
||||||
|
data:** 1244 MB/s at 16 threads, 9.7x h5py threads and 0.89x h5py
|
||||||
|
processes, without a process pool.
|
||||||
|
- **Where clawhdf5 was behind** at `91644d8` (both since fixed; see
|
||||||
|
`docs/known-issues.md`, "Concurrent and contiguous read performance"):
|
||||||
|
- *Full reads of chunked datasets stop scaling at about 4 threads*
|
||||||
|
(about 880 MB/s) while h5py processes reach 4424 MB/s. Hyperslab
|
||||||
|
reads, which bypass the `File`'s chunk cache, keep scaling, so the
|
||||||
|
cache (one mutex and one 16 MiB budget per `File`, thrashed by 64 MiB
|
||||||
|
datasets) is the suspect. The cause of the `--decode-threads 1`
|
||||||
|
ceiling was not the cache: every full read queued its chunks for the
|
||||||
|
pool's single rayon worker. That case was fixed after these
|
||||||
|
measurements (2026-09-26, not yet re-measured here). With the default
|
||||||
|
pool the gap to h5py processes remains (see `docs/known-issues.md`).
|
||||||
|
- *Contiguous reads are slow*: 2.5 GB/s for a single-threaded full read
|
||||||
|
against h5py's 9.8 GB/s (0.25x), and 0.12x for 256 x 256 hyperslabs.
|
||||||
|
Threads close the gap (about 1.0x h5py at 16), but single-thread
|
||||||
|
contiguous I/O is a real deficit.
|
||||||
|
|
||||||
|
The question: libhdf5's threadsafe build serialises every API call under one
|
||||||
|
global mutex, and h5py holds a global lock around every call too, so threads
|
||||||
|
reading through h5py cannot decode in parallel; h5py users scale with
|
||||||
|
processes. A clawhdf5 `File` is `Send + Sync`, and nothing on the read paths
|
||||||
|
this harness uses (`read_f32`, `read_f32_selection`) takes a library-wide
|
||||||
|
lock: the one mutex is the `File`'s chunk cache (keyed per dataset), taken by
|
||||||
|
full reads of chunked datasets for each chunk's O(1) lookup and insert, never
|
||||||
|
across a decode; hyperslab reads do not use the cache. How does
|
||||||
|
decoded throughput scale with threads on one open file, against h5py threads
|
||||||
|
and h5py processes on the same files?
|
||||||
|
|
||||||
|
Workload (`crates/clawhdf5-bench/src/bin/concurrent_read.rs`; the h5py script
|
||||||
|
mirrors it): `<dir>/deflate.h5` and `<dir>/contiguous.h5`, each with 64 `f32`
|
||||||
|
datasets of 64 MiB decoded (`[16384, 1024]`; the deflate file chunked
|
||||||
|
`256 x 256`, level 4), written by clawhdf5 on first use and reused while
|
||||||
|
`manifest.json` matches. The data is a slowly varying ramp plus 8 bits of
|
||||||
|
noise per element, every value exact in `f32`, so both harnesses check what
|
||||||
|
they read; it deflates about 3.1x (128 MiB -> 40.7 MiB for two 64 MiB
|
||||||
|
datasets). For each layout and thread count
|
||||||
|
(1, 2, 4, 8, 16; fixed total work per repetition, split among the threads):
|
||||||
|
|
||||||
|
- `distinct`: every dataset read in full once, thread `t` taking datasets
|
||||||
|
`t, t + T, ...`;
|
||||||
|
- `same`: 1024 random `256 x 256` hyperslabs of `d00` in total, from a seeded
|
||||||
|
splitmix64 stream that both harnesses generate identically.
|
||||||
|
|
||||||
|
Reported per row: MB/s of decoded (selected) data from the median of the
|
||||||
|
repetitions, and scaling efficiency `MB/s(T) / (T x MB/s(1))`. Each worker
|
||||||
|
times itself from a start barrier; a repetition spans the earliest start to
|
||||||
|
the latest finish. Page cache: warm by default (each file is read once before
|
||||||
|
timing); `--cold` evicts the files with `posix_fadvise(POSIX_FADV_DONTNEED)`
|
||||||
|
before every repetition (no root needed; best effort). clawhdf5 opens one
|
||||||
|
`File` per repetition, shared by all threads; h5py threads share one
|
||||||
|
`h5py.File`; h5py processes (spawned before timing) each open the file inside
|
||||||
|
the timed region.
|
||||||
|
|
||||||
|
Decode inside a single clawhdf5 read is itself parallel in this binary
|
||||||
|
(clawhdf5-format's `parallel` feature, enabled here through clawhdf5-agent;
|
||||||
|
it is off in the facade's default features), so a 1-thread clawhdf5 full read
|
||||||
|
of the deflate file already uses the whole rayon pool. Run both
|
||||||
|
`--decode-threads 1` (each read decodes on its calling thread, like h5py —
|
||||||
|
this isolates the API's own scaling) and the default pool.
|
||||||
|
|
||||||
|
> **Run** (from the repository root). The default files take about 5.4 GiB
|
||||||
|
> of disk (4 GiB contiguous + about 1.3 GiB deflate). Generating them is
|
||||||
|
> memory-hungry because `FileBuilder` holds a whole file in memory: peak RSS
|
||||||
|
> was 676 MB for `--datasets 2 --mib 64` (2026-09-25, tank,
|
||||||
|
> `/usr/bin/time -f %M`), about 5x one file's decoded size, so expect about
|
||||||
|
> 21 GB at the defaults (once; later runs reuse the files). Put `--dir` on a
|
||||||
|
> real disk, not tmpfs, if `--cold` is to mean anything.
|
||||||
|
>
|
||||||
|
> ```bash
|
||||||
|
> DIR=/path/on/disk/concurrent-read
|
||||||
|
> BENCH=crates/clawhdf5-bench/scripts
|
||||||
|
> PY=.venv/bin/python # h5py 3.16 / HDF5 2.0 in this repo
|
||||||
|
> cargo build --release -p clawhdf5-bench --bin concurrent_read
|
||||||
|
> B=target/release/concurrent_read
|
||||||
|
> $B --dir $DIR --json claw-pool.json # generates on first run
|
||||||
|
> $B --dir $DIR --decode-threads 1 --json claw-1.json
|
||||||
|
> $PY $BENCH/concurrent_read_h5py.py --dir $DIR --executor threads --json h5py-threads.json
|
||||||
|
> $PY $BENCH/concurrent_read_h5py.py --dir $DIR --executor processes --json h5py-procs.json
|
||||||
|
> $PY $BENCH/compare_concurrent_read.py claw-1.json h5py-threads.json h5py-procs.json
|
||||||
|
> $PY $BENCH/compare_concurrent_read.py claw-pool.json h5py-threads.json h5py-procs.json
|
||||||
|
> ```
|
||||||
|
>
|
||||||
|
> Cold page cache: add `--cold` to every harness command. Smoke test (seconds):
|
||||||
|
> `$B --dir /tmp/cr --datasets 4 --mib 1 --threads 1,2,4 --slabs 16 --reps 1`
|
||||||
|
> and the same `--threads/--slabs/--reps` to the h5py script.
|
||||||
|
|
||||||
|
Other flags (both harnesses): `--threads`, `--reps`, `--slab`, `--slabs`,
|
||||||
|
`--seed`, `--modes distinct,same`, `--layouts deflate,contiguous`; sizes
|
||||||
|
(`--datasets`, `--mib`) only on the Rust harness, which writes the files.
|
||||||
|
|
||||||
## Search harness baseline (v2.3.0)
|
## Search harness baseline (v2.3.0)
|
||||||
|
|
||||||
|
> *Historical.* This baseline and the "After: …" subsections that follow
|
||||||
|
> record each step of the search work; they are *superseded* by
|
||||||
|
> [Current: search harness](#current-search-harness-2026-09-24) (2026-09-24),
|
||||||
|
> the last subsection of this part.
|
||||||
|
|
||||||
Produced by `cargo run --release -p clawhdf5-bench --bin search_harness -- --full`
|
Produced by `cargo run --release -p clawhdf5-bench --bin search_harness -- --full`
|
||||||
on deterministic **clustered** synthetic data (384-dim, unit-normalised; points =
|
on deterministic **clustered** synthetic data (384-dim, unit-normalised; points =
|
||||||
cluster centre + noise — uniform random vectors are nearly equidistant in high
|
cluster centre + noise — uniform random vectors are nearly equidistant in high
|
||||||
@@ -522,10 +930,11 @@ build: 9752.6 ms (10254 vectors/s) · exact scan: 40 QPS, p50 24648 µs
|
|||||||
| 1000 | 11 | 3.9 | 0.9 | 68.1 | 5.48 | 5.57 | 182.5 |
|
| 1000 | 11 | 3.9 | 0.9 | 68.1 | 5.48 | 5.57 | 182.5 |
|
||||||
| 10000 | 114 | 32.2 | 10.9 | 845.0 | 48.56 | 78.65 | 19.8 |
|
| 10000 | 114 | 32.2 | 10.9 | 845.0 | 48.56 | 78.65 | 19.8 |
|
||||||
| 100000 | 1486 | 713.0 | 354.5 | 10486.5 | 883.51 | 975.23 | 1.1 |
|
| 100000 | 1486 | 713.0 | 354.5 | 10486.5 | 883.51 | 975.23 | 1.1 |
|
||||||
wrote /tmp/claude-1000/-home-osobh-projects-clawhdf5/422f755e-dd25-4c35-8613-5439087e3aaa/scratchpad/baseline_full.json
|
|
||||||
|
|
||||||
### After: HNSW neighbour-selection heuristic
|
### After: HNSW neighbour-selection heuristic
|
||||||
|
|
||||||
|
> *Superseded* by [Current: search harness](#current-search-harness-2026-09-24) (2026-09-24).
|
||||||
|
|
||||||
Same harness, same data, after replacing closest-M neighbour selection with the
|
Same harness, same data, after replacing closest-M neighbour selection with the
|
||||||
HNSW paper's diversity heuristic (Algorithm 4, keeping pruned connections) for
|
HNSW paper's diversity heuristic (Algorithm 4, keeping pruned connections) for
|
||||||
both new links and back-link pruning. Recall@10 at `ef = 64`: **0.87 → 1.00**
|
both new links and back-link pruning. Recall@10 at `ef = 64`: **0.87 → 1.00**
|
||||||
@@ -571,6 +980,8 @@ build: 36472.8 ms (2742 vectors/s) · exact scan: 40 QPS, p50 24644 µs
|
|||||||
|
|
||||||
### After: persistent keyword index, no store rewrite per query
|
### After: persistent keyword index, no store rewrite per query
|
||||||
|
|
||||||
|
> *Superseded* by [Current: search harness](#current-search-harness-2026-09-24) (2026-09-24).
|
||||||
|
|
||||||
`hybrid_search` used to rebuild the BM25 index from scratch (re-tokenising every
|
`hybrid_search` used to rebuild the BM25 index from scratch (re-tokenising every
|
||||||
record) and rewrite the whole `.h5` file on **every query**. The index is now
|
record) and rewrite the whole `.h5` file on **every query**. The index is now
|
||||||
kept for the life of the store and updated incrementally, and activation boosts
|
kept for the life of the store and updated incrementally, and activation boosts
|
||||||
@@ -591,6 +1002,8 @@ index removes that.
|
|||||||
|
|
||||||
### After: vector index persisted with the checkpoint
|
### After: vector index persisted with the checkpoint
|
||||||
|
|
||||||
|
> *Superseded* by [Current: search harness](#current-search-harness-2026-09-24) (2026-09-24).
|
||||||
|
|
||||||
The HNSW graph (not the vectors, which the store already holds) is saved to
|
The HNSW graph (not the vectors, which the store already holds) is saved to
|
||||||
`<store>.h5.ann` at each checkpoint and reloaded by `open()`, tied to that
|
`<store>.h5.ann` at each checkpoint and reloaded by `open()`, tied to that
|
||||||
checkpoint by a generation id. The index is now built once per store (the *cold
|
checkpoint by a generation id. The index is now built once per store (the *cold
|
||||||
@@ -608,6 +1021,8 @@ index incrementally.
|
|||||||
|
|
||||||
### After: unit-vector dot product, reusable visited set
|
### After: unit-vector dot product, reusable visited set
|
||||||
|
|
||||||
|
> *Superseded* by [Current: search harness](#current-search-harness-2026-09-24) (2026-09-24).
|
||||||
|
|
||||||
Cosine distance recomputed both vector norms on every evaluation; the index now
|
Cosine distance recomputed both vector norms on every evaluation; the index now
|
||||||
stores unit vectors and uses a plain dot product. The per-call `HashSet` of
|
stores unit vectors and uses a plain dot product. The per-call `HashSet` of
|
||||||
visited nodes became a reusable epoch-stamped array. Recall is unchanged.
|
visited nodes became a reusable epoch-stamped array. Recall is unchanged.
|
||||||
@@ -653,6 +1068,8 @@ build: 21084.6 ms (4743 vectors/s) · exact scan: 39 QPS, p50 24739 µs
|
|||||||
|
|
||||||
### After: unranked keyword scores, top-k merge (rankings unchanged)
|
### After: unranked keyword scores, top-k merge (rankings unchanged)
|
||||||
|
|
||||||
|
> *Superseded* by [Current: search harness](#current-search-harness-2026-09-24) (2026-09-24).
|
||||||
|
|
||||||
A fusion study (`search_harness --fusion-study`) showed that capping the
|
A fusion study (`search_harness --fusion-study`) showed that capping the
|
||||||
keyword candidate pool is **not** a safe optimisation: against the current
|
keyword candidate pool is **not** a safe optimisation: against the current
|
||||||
full-corpus normalisation the final top-10 overlap is only 0.83-0.92 and the
|
full-corpus normalisation the final top-10 overlap is only 0.83-0.92 and the
|
||||||
@@ -675,6 +1092,8 @@ results.
|
|||||||
|
|
||||||
### After: batched bulk build (optionally parallel); deletions handled in search
|
### After: batched bulk build (optionally parallel); deletions handled in search
|
||||||
|
|
||||||
|
> *Superseded* by [Current: search harness](#current-search-harness-2026-09-24) (2026-09-24).
|
||||||
|
|
||||||
Profiling showed **90% of a build's distance evaluations are in back-link
|
Profiling showed **90% of a build's distance evaluations are in back-link
|
||||||
pruning**. The bulk build now inserts in batches: plan each node's neighbours
|
pruning**. The bulk build now inserts in batches: plan each node's neighbours
|
||||||
against the graph as it stood at the start of the batch, link, then prune every
|
against the graph as it stood at the start of the batch, link, then prune every
|
||||||
@@ -1173,10 +1592,75 @@ turn-level row plus its session Hit@1 in the tokenizer table. The run also
|
|||||||
produced figures this document does not publish (stemmed session Hit@5,
|
produced figures this document does not publish (stemmed session Hit@5,
|
||||||
Hit@10 and MRR, and per-type Hit@5/Hit@10/MRR for both modes), so there was
|
Hit@10 and MRR, and per-type Hit@5/Hit@10/MRR for both modes), so there was
|
||||||
nothing to compare them with. Rows that need real embeddings (vector-only,
|
nothing to compare them with. Rows that need real embeddings (vector-only,
|
||||||
hybrid, RRF, re-ranking, the weight sweep) were not re-run.
|
hybrid, RRF, re-ranking, the weight sweep) were not re-run then; they were on
|
||||||
|
2026-09-27 (next section).
|
||||||
|
|
||||||
|
### Re-run with real embeddings (2026-09-27, tank)
|
||||||
|
|
||||||
|
Every recall row in this section that needs real embeddings was measured
|
||||||
|
again on 2026-09-27 on tank (AMD Ryzen 7 7800X3D, 16 threads; MiniLM
|
||||||
|
embeddings on an RTX 5060 Ti, retrieval on the CPU), with the search code of
|
||||||
|
commit 7a8fae0. Six runs:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
cargo build --release -p clawhdf5-bench --bin longmemeval_bench --features embeddings-cuda
|
||||||
|
B=target/release/longmemeval_bench W=weights/all-minilm-l6-v2
|
||||||
|
$B benchmarks/longmemeval/longmemeval_oracle.json --embeddings $W
|
||||||
|
$B benchmarks/longmemeval/longmemeval_s_cleaned.json --embeddings $W
|
||||||
|
$B benchmarks/longmemeval/longmemeval_s_cleaned.json --embeddings $W --float16
|
||||||
|
$B benchmarks/longmemeval/longmemeval_oracle.json --embeddings $W --sweep
|
||||||
|
$B benchmarks/longmemeval/longmemeval_s_cleaned.json --embeddings $W --sweep
|
||||||
|
$B benchmarks/longmemeval/longmemeval_s_cleaned.json --embeddings $W --rerank-sweep
|
||||||
|
```
|
||||||
|
|
||||||
|
(`longmemeval_s.json`, used by the commands elsewhere in this section, is the
|
||||||
|
same file as `longmemeval_s_cleaned.json`.)
|
||||||
|
|
||||||
|
**The machine was not idle.** The 1-minute load never fell below 2 in a
|
||||||
|
2-hour wait, because two stray test processes were each holding a core; it
|
||||||
|
was 2.1–2.7 when each run started and up to 8.4 while the runs were going.
|
||||||
|
Recall does not depend on load. Latency does, so no latency figure in this
|
||||||
|
file was updated from these runs.
|
||||||
|
|
||||||
|
**What reproduced exactly:** every published full-haystack Hit@1/5/10 and MRR
|
||||||
|
at turn and session level for BM25, vector-only, hybrid 0.4/0.6, both stemmed
|
||||||
|
modes, and every re-ranking row; RRF except as below; the float16 table; 4 of
|
||||||
|
the 11 weight-sweep rows (0.0, 0.1, 0.3 and 0.9); and the oracle vector-only
|
||||||
|
figure. The headline,
|
||||||
|
hybrid 0.4/0.6 turn Hit@5 **81.4%**, is unchanged.
|
||||||
|
|
||||||
|
**What changed.** Old values are kept here; the tables below show the new ones.
|
||||||
|
|
||||||
|
| Figure | Published | 2026-09-27 | Why |
|
||||||
|
|---|---:|---:|---|
|
||||||
|
| Oracle, hybrid turn Hit@5 | 85.2% (2026-08-07, c913cd1) | **86.8%** | Weights. 85.2% was measured at 0.7/0.3, the default then. The default has been 0.4/0.6 since 29baabb. Today's oracle sweep gives 85.2% at 0.7/0.3 and 86.8% at 0.4/0.6. |
|
||||||
|
| Oracle, BM25-only turn Hit@5 with embeddings | 84.2% (2026-08-07) | 84.4% | Now equal to the zero-embedding figure, as it already was on the full haystack. At c913cd1, tied candidates were ordered by `HashMap` iteration. Since 3ed0489 (2026-09-19) they are broken by index. 3ed0489 is the likely cause; it was not bisected. |
|
||||||
|
| Ablation / sweep, hybrid 0.7/0.3, turn | 44.4% / 79.2% / 86.0% / 0.5868 | 44.2% / 79.2% / 85.8% / 0.5856 | The same tie-breaking change. 29baabb's re-run on 2026-09-19, made after 3ed0489 that same day, already had 44.2% / 85.8% / 0.5856. |
|
||||||
|
| Ablation, hybrid 0.7/0.3, session | 88.2% / 95.8% / 97.8% / 0.9158 | 88.0% / 95.8% / 97.6% / 0.9146 | Same. |
|
||||||
|
| Ablation / sweep, vector-only turn MRR | 0.5027 | 0.5031 | Same. The fusion and float16 tables already had 0.5031. |
|
||||||
|
| Sweep 0.2/0.8, turn | 53.6% / 78.2% / 0.6440 | 53.4% / 78.0% / 0.6429 | Same (the sweep was measured 2026-08-07, 1537a94). |
|
||||||
|
| Sweep 0.4/0.6, 0.5/0.5, 0.8/0.2 turn MRR | 0.6429 / 0.6234 / 0.5571 | 0.6430 / 0.6232 / 0.5574 | Same. |
|
||||||
|
| Sweep 0.6/0.4 | Hit@5 79.8%, MRR 0.6069, session Hit@5 96.6% | 79.6%, 0.6067, 96.4% | Same. |
|
||||||
|
| RRF turn MRR | 0.5967 (2026-09-19, aa92fef) | 0.5969 | Not explained. It may be a later search-path change, such as the int8 index becoming the default in 8b85d93. It may also be the run-to-run variation described next. It was not bisected. |
|
||||||
|
|
||||||
|
**Not exactly deterministic.** Between two f32 runs today, the recency count
|
||||||
|
(the share of `knowledge-update` questions where the newest gold session
|
||||||
|
ranks first) differed by one question in two modes: hybrid 0.4/0.6 was
|
||||||
|
144/320 in one run and 145/320 in the other, and re-rank with a 1-day
|
||||||
|
half-life was 165/319 and 166/319. Each question gets a fresh store, so this
|
||||||
|
is not state carried between modes. The cause was not found; a candidate is
|
||||||
|
that the GPU embeddings are not bit-for-bit identical from run to run. Hit@k
|
||||||
|
and MRR agreed in every run that repeated a mode (hybrid 0.4/0.6 was measured
|
||||||
|
four times: three f32 runs and one float16 run), but a last-digit change in an
|
||||||
|
MRR, or a one-question change in recency, is within this variation.
|
||||||
|
|
||||||
### Full haystack — `longmemeval_s`, n=500 (the number to cite)
|
### Full haystack — `longmemeval_s`, n=500 (the number to cite)
|
||||||
|
|
||||||
|
This table is **BM25-only** (zero embeddings). With real embeddings and the
|
||||||
|
default hybrid 0.4/0.6 the same corpus gives turn Hit@5 **81.4%** (2026-09-27;
|
||||||
|
see [Fusion method](#fusion-method--weighted-vs-rrf-full-haystack-n500)),
|
||||||
|
which is the headline figure.
|
||||||
|
|
||||||
47.7 sessions and 493.5 turns per question; 4.0% of haystack sessions are evidence
|
47.7 sessions and 493.5 turns per question; 4.0% of haystack sessions are evidence
|
||||||
sessions, so retrieval has to actually discriminate.
|
sessions, so retrieval has to actually discriminate.
|
||||||
|
|
||||||
@@ -1203,13 +1687,16 @@ the question.
|
|||||||
|
|
||||||
Real 384-d `all-MiniLM-L6-v2` embeddings, 190,015 unique texts encoded once on an
|
Real 384-d `all-MiniLM-L6-v2` embeddings, 190,015 unique texts encoded once on an
|
||||||
RTX 5060 Ti (~13 min; the same work on the 8-core CPU was still unfinished after
|
RTX 5060 Ti (~13 min; the same work on the 8-core CPU was still unfinished after
|
||||||
30 minutes, so the GPU path is not a convenience here). Turn-level:
|
30 minutes, so the GPU path is not a convenience here). First measured
|
||||||
|
2026-08-07 (c913cd1); the values below are the 2026-09-27 re-run, which moved
|
||||||
|
the vector-only MRR and the `0.7/0.3` row (see the re-run section above).
|
||||||
|
Turn-level:
|
||||||
|
|
||||||
| Mode | Hit@1 | Hit@5 | Hit@10 | MRR |
|
| Mode | Hit@1 | Hit@5 | Hit@10 | MRR |
|
||||||
|------|-------|-------|--------|-----|
|
|------|-------|-------|--------|-----|
|
||||||
| BM25 only (`0.0`/`1.0`) | **53.8%** | 75.0% | 81.6% | **0.6320** |
|
| BM25 only (`0.0`/`1.0`) | **53.8%** | 75.0% | 81.6% | **0.6320** |
|
||||||
| Vector only (`1.0`/`0.0`) | 36.0% | 71.8% | 81.6% | 0.5027 |
|
| Vector only (`1.0`/`0.0`) | 36.0% | 71.8% | 81.6% | 0.5031 |
|
||||||
| Hybrid (`0.7`/`0.3`) | 44.4% | **79.2%** | **86.0%** | 0.5868 |
|
| Hybrid (`0.7`/`0.3`) | 44.2% | **79.2%** | **85.8%** | 0.5856 |
|
||||||
|
|
||||||
Session-level:
|
Session-level:
|
||||||
|
|
||||||
@@ -1217,7 +1704,7 @@ Session-level:
|
|||||||
|------|-------|-------|--------|-----|
|
|------|-------|-------|--------|-----|
|
||||||
| BM25 only | 86.2% | 93.6% | 96.6% | 0.8948 |
|
| BM25 only | 86.2% | 93.6% | 96.6% | 0.8948 |
|
||||||
| Vector only | 85.4% | 94.2% | 96.6% | 0.8901 |
|
| Vector only | 85.4% | 94.2% | 96.6% | 0.8901 |
|
||||||
| Hybrid | **88.2%** | **95.8%** | **97.8%** | **0.9158** |
|
| Hybrid | **88.0%** | **95.8%** | **97.6%** | **0.9146** |
|
||||||
|
|
||||||
### Fusion method — weighted vs. RRF, full haystack, n=500
|
### Fusion method — weighted vs. RRF, full haystack, n=500
|
||||||
|
|
||||||
@@ -1231,7 +1718,7 @@ takes a `Fusion`, and both run over the same HNSW + BM25 candidates:
|
|||||||
| BM25 only | **53.8%** | 75.0% | 81.6% | 0.6320 | 86.2% | 0.8948 |
|
| BM25 only | **53.8%** | 75.0% | 81.6% | 0.6320 | 86.2% | 0.8948 |
|
||||||
| Vector only | 36.0% | 71.8% | 81.6% | 0.5031 | 85.4% | 0.8901 |
|
| Vector only | 36.0% | 71.8% | 81.6% | 0.5031 | 85.4% | 0.8901 |
|
||||||
| **Weighted 0.4 / 0.6** | 51.6% | **81.4%** | **87.8%** | **0.6430** | **91.0%** | **0.9347** |
|
| **Weighted 0.4 / 0.6** | 51.6% | **81.4%** | **87.8%** | **0.6430** | **91.0%** | **0.9347** |
|
||||||
| RRF (k=60) | 45.0% | 78.8% | 87.6% | 0.5967 | 89.6% | 0.9253 |
|
| RRF (k=60) | 45.0% | 78.8% | 87.6% | 0.5969 | 89.6% | 0.9253 |
|
||||||
|
|
||||||
**RRF loses to the tuned weighted sum** — 6.6pp of turn Hit@1 and 0.046 of MRR
|
**RRF loses to the tuned weighted sum** — 6.6pp of turn Hit@1 and 0.046 of MRR
|
||||||
— and lands almost exactly where the old `0.7/0.3` weighting did (44.2% /
|
— and lands almost exactly where the old `0.7/0.3` weighting did (44.2% /
|
||||||
@@ -1281,7 +1768,9 @@ over rank-1 precision.
|
|||||||
activation. Until now its combined score contained **no relevance term at
|
activation. Until now its combined score contained **no relevance term at
|
||||||
all** — `RerankInput` did not carry the retrieval score — so a caller that
|
all** — `RerankInput` did not carry the retrieval score — so a caller that
|
||||||
re-ranked its candidates threw the retriever's ordering away and returned them
|
re-ranked its candidates threw the retriever's ordering away and returned them
|
||||||
ordered by age. The OpenClaw backend did exactly that on every search.
|
ordered by age. `ClawhdfBackend` (the `openclaw` module) did exactly that on
|
||||||
|
every search. (OpenClaw itself never integrated clawhdf5; see
|
||||||
|
`docs/openclaw.md`.)
|
||||||
|
|
||||||
Measuring that is unambiguous. "Recency" below is the share of
|
Measuring that is unambiguous. "Recency" below is the share of
|
||||||
`knowledge-update` questions where the newest gold session outranked the stale
|
`knowledge-update` questions where the newest gold session outranked the stale
|
||||||
@@ -1296,6 +1785,12 @@ one (see `newest_gold_first`); ~45% is chance.
|
|||||||
| + re-rank, relevance-led, half-life 30 days | 51.8% | 81.0% | 87.8% | 0.6427 | 51.4% |
|
| + re-rank, relevance-led, half-life 30 days | 51.8% | 81.0% | 87.8% | 0.6427 | 51.4% |
|
||||||
| + re-rank, relevance-led, half-life 90 days | **52.0%** | 80.4% | 87.8% | 0.6425 | 50.8% |
|
| + re-rank, relevance-led, half-life 90 days | **52.0%** | 80.4% | 87.8% | 0.6425 | 50.8% |
|
||||||
|
|
||||||
|
Re-run on 2026-09-27 (`--rerank-sweep`, tank, commit 7a8fae0): every Hit@k and
|
||||||
|
MRR above reproduced exactly. The recency column came out 45.3%, 87.5%,
|
||||||
|
52.0%, 52.2%, 51.7% and 50.5%. Each of those is within one question of the
|
||||||
|
value in the table, which is the run-to-run variation described under
|
||||||
|
"Re-run with real embeddings" above, so the table was left as it was.
|
||||||
|
|
||||||
**The pre-fix row is the finding.** Ordering candidates by recency alone costs
|
**The pre-fix row is the finding.** Ordering candidates by recency alone costs
|
||||||
40.6pp of Hit@1 and two thirds of MRR: the results are the newest memories in
|
40.6pp of Hit@1 and two thirds of MRR: the results are the newest memories in
|
||||||
the pool rather than the ones that answer the question. It does ace the recency
|
the pool rather than the ones that answer the question. It does ace the recency
|
||||||
@@ -1309,9 +1804,9 @@ cannot reach the 87.5% the degenerate ordering gets. Those two rows are the
|
|||||||
ends of a trade-off, and the default sits deliberately near the relevance end.
|
ends of a trade-off, and the default sits deliberately near the relevance end.
|
||||||
|
|
||||||
**Half-life is not a sensitive knob.** Across 1, 7, 30 and 90 days recency
|
**Half-life is not a sensitive knob.** Across 1, 7, 30 and 90 days recency
|
||||||
moves 1.4pp and MRR 0.003 — inside the noise of a 500-question run — because
|
moves 1.4pp (1.7pp in the 2026-09-27 re-run) and MRR 0.003 — inside the
|
||||||
the temporal term is capped by its weight (0.3) while relevance differences
|
noise of a 500-question run — because the temporal term is capped by its
|
||||||
between candidates are larger. The 24-hour default is kept; there is no
|
weight (0.3) while relevance differences between candidates are larger. The 24-hour default is kept; there is no
|
||||||
measured reason to change it, and a corpus-matched value is not the lever it
|
measured reason to change it, and a corpus-matched value is not the lever it
|
||||||
looks like.
|
looks like.
|
||||||
|
|
||||||
@@ -1319,24 +1814,28 @@ looks like.
|
|||||||
|
|
||||||
`0.7/0.3` was a documented default, never a searched one. Sweeping
|
`0.7/0.3` was a documented default, never a searched one. Sweeping
|
||||||
`vector_weight` from 0.0 to 1.0 (`--sweep`, reusing the one-time embedding
|
`vector_weight` from 0.0 to 1.0 (`--sweep`, reusing the one-time embedding
|
||||||
table) shows it is not merely suboptimal but **strictly dominated**:
|
table) shows it is not merely suboptimal but **strictly dominated**. First
|
||||||
|
measured 2026-08-07 (1537a94); the values below are the 2026-09-27 re-run,
|
||||||
|
which changed the 0.2, 0.4, 0.5, 0.6, 0.7, 0.8 and 1.0 rows in the last digit
|
||||||
|
or by one or two questions (see the re-run section above):
|
||||||
|
|
||||||
| vector / keyword | Hit@1 | Hit@5 | Hit@10 | MRR | session Hit@5 |
|
| vector / keyword | Hit@1 | Hit@5 | Hit@10 | MRR | session Hit@5 |
|
||||||
|---|---|---|---|---|---|
|
|---|---|---|---|---|---|
|
||||||
| 0.0 / 1.0 (BM25) | **53.8%** | 75.0% | 81.6% | 0.6320 | 93.6% |
|
| 0.0 / 1.0 (BM25) | **53.8%** | 75.0% | 81.6% | 0.6320 | 93.6% |
|
||||||
| 0.1 / 0.9 | 53.2% | 77.4% | 83.8% | 0.6374 | 95.0% |
|
| 0.1 / 0.9 | 53.2% | 77.4% | 83.8% | 0.6374 | 95.0% |
|
||||||
| 0.2 / 0.8 | 53.6% | 78.2% | 85.6% | 0.6440 | 95.4% |
|
| 0.2 / 0.8 | 53.4% | 78.0% | 85.6% | 0.6429 | 95.4% |
|
||||||
| 0.3 / 0.7 | 53.2% | 78.8% | 87.2% | **0.6463** | 96.0% |
|
| 0.3 / 0.7 | 53.2% | 78.8% | 87.2% | **0.6463** | 96.0% |
|
||||||
| **0.4 / 0.6** | 51.6% | **81.4%** | 87.8% | 0.6429 | 96.8% |
|
| **0.4 / 0.6** | 51.6% | **81.4%** | 87.8% | 0.6430 | 96.8% |
|
||||||
| 0.5 / 0.5 | 48.2% | **81.4%** | **88.2%** | 0.6234 | **97.4%** |
|
| 0.5 / 0.5 | 48.2% | **81.4%** | **88.2%** | 0.6232 | **97.4%** |
|
||||||
| 0.6 / 0.4 | 46.6% | 79.8% | 87.4% | 0.6069 | 96.6% |
|
| 0.6 / 0.4 | 46.6% | 79.6% | 87.4% | 0.6067 | 96.4% |
|
||||||
| 0.7 / 0.3 *(old default)* | 44.4% | 79.2% | 86.0% | 0.5868 | 95.8% |
|
| 0.7 / 0.3 *(old default)* | 44.2% | 79.2% | 85.8% | 0.5856 | 95.8% |
|
||||||
| 0.8 / 0.2 | 40.6% | 76.2% | 85.4% | 0.5571 | 95.2% |
|
| 0.8 / 0.2 | 40.6% | 76.2% | 85.4% | 0.5574 | 95.2% |
|
||||||
| 0.9 / 0.1 | 37.8% | 73.4% | 84.6% | 0.5289 | 94.2% |
|
| 0.9 / 0.1 | 37.8% | 73.4% | 84.6% | 0.5289 | 94.2% |
|
||||||
| 1.0 / 0.0 (vector) | 36.0% | 71.8% | 81.6% | 0.5027 | 94.2% |
|
| 1.0 / 0.0 (vector) | 36.0% | 71.8% | 81.6% | 0.5031 | 94.2% |
|
||||||
|
|
||||||
**`0.4/0.6` beats `0.7/0.3` on every metric at both granularities** — Hit@1
|
**`0.4/0.6` beats `0.7/0.3` on every metric at both granularities** — Hit@1
|
||||||
+7.2pp, Hit@5 +2.2, Hit@10 +1.8, MRR +0.056. There is no trade being made; the
|
+7.4pp, Hit@5 +2.2, Hit@10 +2.0, MRR +0.057 (2026-09-27 figures; +7.2pp,
|
||||||
|
+2.2, +1.8 and +0.056 as measured on 2026-08-07). There is no trade being made; the
|
||||||
old default was simply on the wrong side of the peak. **`0.4/0.6` is the
|
old default was simply on the wrong side of the peak. **`0.4/0.6` is the
|
||||||
recommended setting**, with `0.3/0.7` preferable if rank-1 precision matters
|
recommended setting**, with `0.3/0.7` preferable if rank-1 precision matters
|
||||||
most (it takes the best MRR in the sweep and gives up only 0.6pp of Hit@1
|
most (it takes the best MRR in the sweep and gives up only 0.6pp of Hit@1
|
||||||
@@ -1369,7 +1868,7 @@ worth stating plainly rather than hiding: LongMemEval questions share substantia
|
|||||||
vocabulary with their evidence turns, which is close to the best case for lexical
|
vocabulary with their evidence turns, which is close to the best case for lexical
|
||||||
matching, and MiniLM at 384 dimensions is a small embedding model.
|
matching, and MiniLM at 384 dimensions is a small embedding model.
|
||||||
|
|
||||||
> **Run:** `cargo run --release --bin longmemeval_bench --features embeddings -- \
|
> **Run:** `cargo run --release -p clawhdf5-bench --bin longmemeval_bench --features embeddings -- \
|
||||||
> benchmarks/longmemeval/longmemeval_s_cleaned.json --embeddings weights/all-minilm-l6-v2`
|
> benchmarks/longmemeval/longmemeval_s_cleaned.json --embeddings weights/all-minilm-l6-v2`
|
||||||
> For the GPU path use `--features embeddings-cuda`. That requires `nvcc` on
|
> For the GPU path use `--features embeddings-cuda`. That requires `nvcc` on
|
||||||
> `PATH` at *build* time — cudarc's build script shells out to it. The toolkit
|
> `PATH` at *build* time — cudarc's build script shells out to it. The toolkit
|
||||||
@@ -1397,11 +1896,27 @@ price of the harder corpus, and is the reason oracle-only numbers should not be
|
|||||||
presented as LongMemEval results. Session-level figures on this variant are
|
presented as LongMemEval results. Session-level figures on this variant are
|
||||||
degenerate — see below.
|
degenerate — see below.
|
||||||
|
|
||||||
With real embeddings the same oracle corpus gives BM25-only 84.2% / vector-only
|
With real embeddings, the same oracle corpus gives these turn-level figures
|
||||||
80.4% / hybrid **85.2%** Hit@5 turn-level — hybrid ahead at Hit@5 and Hit@10 and
|
(2026-09-27, tank, commit 7a8fae0; command and load in "Re-run with real
|
||||||
behind at Hit@1, matching the full-haystack pattern above. (BM25-only reads 84.2%
|
embeddings" above):
|
||||||
here against 84.4% with zero embedding vectors: one question of 500 changes rank,
|
|
||||||
with MRR identical at 0.6597. On the full haystack the two agree exactly.)
|
| Mode | Hit@1 | Hit@5 | Hit@10 | MRR |
|
||||||
|
|---|---:|---:|---:|---:|
|
||||||
|
| BM25 only | **52.6%** | 84.4% | 90.4% | 0.6597 |
|
||||||
|
| Vector only | 39.6% | 80.4% | 91.0% | 0.5605 |
|
||||||
|
| Hybrid 0.4 / 0.6 (default) | 52.4% | **86.8%** | **92.4%** | **0.6678** |
|
||||||
|
| Hybrid 0.7 / 0.3 (old default) | 48.4% | 85.2% | 92.2% | 0.6382 |
|
||||||
|
|
||||||
|
Hybrid 0.4/0.6 leads at Hit@5, Hit@10 and MRR and is 0.2pp (one question)
|
||||||
|
behind BM25 at Hit@1, as on the full haystack.
|
||||||
|
|
||||||
|
Until 2026-09-27 this paragraph gave BM25-only 84.2%, vector-only 80.4% and
|
||||||
|
hybrid **85.2%** Hit@5. Those were measured on 2026-08-07 (c913cd1), when the
|
||||||
|
hybrid default was 0.7/0.3; today's 0.7/0.3 row reproduces the 85.2%. The
|
||||||
|
0.4/0.6 default (29baabb) is what moves hybrid to 86.8%. The earlier BM25-only
|
||||||
|
84.2% with embeddings, one question below the zero-embedding 84.4%, predates
|
||||||
|
the index tie-break of 3ed0489; the two now agree, as they always did on the
|
||||||
|
full haystack.
|
||||||
|
|
||||||
### Retracted: session-level recall and the MemX comparison
|
### Retracted: session-level recall and the MemX comparison
|
||||||
|
|
||||||
@@ -1772,7 +2287,8 @@ The tank row was measured 2026-09-24 on tank (AMD Ryzen 7 7800X3D), commit
|
|||||||
### Reproducibility
|
### Reproducibility
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
rustup override set nightly
|
# Any stable toolchain at or above the MSRV (1.92) works; the original
|
||||||
|
# 2026-07-01 run used a nightly, later runs stable.
|
||||||
|
|
||||||
# Latency benchmarks (Criterion)
|
# Latency benchmarks (Criterion)
|
||||||
cargo bench -p clawhdf5-agent
|
cargo bench -p clawhdf5-agent
|
||||||
@@ -1875,7 +2391,9 @@ libhdf5 reads from a temp file including `open` + `read` + `close` overhead.
|
|||||||
| clawhdf5 hyperslab (f64, 10% slice) | — | 4.09 µs / **1.8 GiB/s** | 50.1 µs / **1.5 GiB/s** |
|
| clawhdf5 hyperslab (f64, 10% slice) | — | 4.09 µs / **1.8 GiB/s** | 50.1 µs / **1.5 GiB/s** |
|
||||||
|
|
||||||
libhdf5 f64 comparison excluded — clawhdf5's datatype encoding differs from libhdf5's (known
|
libhdf5 f64 comparison excluded — clawhdf5's datatype encoding differs from libhdf5's (known
|
||||||
gap), making cross-format reads unreliable for comparison.
|
gap), making cross-format reads unreliable for comparison. (That gap was the float sign-bit
|
||||||
|
bug, fixed 2026-09-23: `docs/known-issues.md`, "Every `f32` dataset we wrote was unreadable by
|
||||||
|
h5py / libhdf5". The comparison has not been re-run since.)
|
||||||
|
|
||||||
### Chunked Read Throughput
|
### Chunked Read Throughput
|
||||||
|
|
||||||
@@ -1983,6 +2501,13 @@ global file mutex and flushes to disk on every attribute write or group creation
|
|||||||
|
|
||||||
## vs libhdf5 Summary
|
## vs libhdf5 Summary
|
||||||
|
|
||||||
|
> Measured on the original i7-12650H (clawhdf5 2026-07-01, libhdf5
|
||||||
|
> 2026-06-30). The newest run of this table is
|
||||||
|
> [Independent Validation: tank](#independent-validation-tank-ryzen-7-7800x3d-2026-08-03)
|
||||||
|
> (2026-08-03), which reproduces every row within ~15% except chunked
|
||||||
|
> write (45.3x on tank; 35x on 2026-09-23 with the pure-Rust deflate, see
|
||||||
|
> [Deflate backend](#deflate-backend-zlib-rs-vs-zlib-ng)).
|
||||||
|
|
||||||
| Workload | clawhdf5 | libhdf5 | Speedup |
|
| Workload | clawhdf5 | libhdf5 | Speedup |
|
||||||
|----------|----------|---------|---------|
|
|----------|----------|---------|---------|
|
||||||
| Sequential read, 1K f32 | 634 ns | 45.2 µs | **71×** |
|
| Sequential read, 1K f32 | 634 ns | 45.2 µs | **71×** |
|
||||||
@@ -2014,7 +2539,7 @@ to the page cache. There is no algorithmic headroom above ~1.7 GiB/s on this har
|
|||||||
|
|
||||||
### Caveats
|
### Caveats
|
||||||
|
|
||||||
- libhdf5 f64 read comparison excluded — clawhdf5's f32 datatype encoding differs from libhdf5's (known compatibility gap). f64 results are clawhdf5-only.
|
- libhdf5 f64 read comparison excluded — clawhdf5's f32 datatype encoding differs from libhdf5's (known compatibility gap at the time; fixed 2026-09-23, see [Sequential Read Throughput](#sequential-read-throughput)). f64 results are clawhdf5-only.
|
||||||
- Serial benchmarks. clawhdf5 uses Rayon for chunk compression when > 2 chunks; that parallelism is already reflected in the chunked write numbers.
|
- Serial benchmarks. clawhdf5 uses Rayon for chunk compression when > 2 chunks; that parallelism is already reflected in the chunked write numbers.
|
||||||
- clawhdf5 reads from `Vec<u8>` (zero-copy from mmap in production); libhdf5 reads from a temp file. This gives clawhdf5 a structural read advantage that reflects realistic API usage.
|
- clawhdf5 reads from `Vec<u8>` (zero-copy from mmap in production); libhdf5 reads from a temp file. This gives clawhdf5 a structural read advantage that reflects realistic API usage.
|
||||||
|
|
||||||
@@ -2211,6 +2736,10 @@ Same not-like-for-like caveat as the "Comparison to MemX" section at the top of
|
|||||||
file applies — MemX's figure is end-to-end, these are a single component. Ratios are
|
file applies — MemX's figure is end-to-end, these are a single component. Ratios are
|
||||||
an order-of-magnitude indication, not a benchmark result.
|
an order-of-magnitude indication, not a benchmark result.
|
||||||
|
|
||||||
|
> The Ratio column below was retracted afterwards: see
|
||||||
|
> [Comparison to MemX](#comparison-to-memx-arxiv260316171). Kept as recorded
|
||||||
|
> on 2026-08-05; do not cite it.
|
||||||
|
|
||||||
| Metric | MemX (claimed, end-to-end) | ClawhDF5 (tank, component only) | Ratio |
|
| Metric | MemX (claimed, end-to-end) | ClawhDF5 (tank, component only) | Ratio |
|
||||||
|--------|----------------------------|----------------------------------|-------|
|
|--------|----------------------------|----------------------------------|-------|
|
||||||
| 100K flat search | <90 ms | 6.60 ms | ~14x |
|
| 100K flat search | <90 ms | 6.60 ms | ~14x |
|
||||||
|
|||||||
+2296
File diff suppressed because it is too large
Load Diff
@@ -1,176 +1,275 @@
|
|||||||
# clawhdf5
|
# clawhdf5
|
||||||
|
|
||||||
## Purpose
|
## Purpose
|
||||||
Pure-Rust HDF5 format implementation with HNSW vector search, WAL-backed persistence, agent memory storage, and GPU-accelerated I/O. Used by ZeroClaw as its persistent memory and knowledge graph backend.
|
Pure-Rust HDF5 implementation (read, write, in-place edit, remote and browser
|
||||||
|
reads) plus agent memory on top of it: HNSW vector search, a WAL-backed store,
|
||||||
|
and GPU vector distances. A standalone library. Its one verified consumer is
|
||||||
|
ClawBrainHub (`.brain` files); no agent framework integrates it (see
|
||||||
|
*Standing rules*).
|
||||||
|
|
||||||
## Architecture
|
## Architecture
|
||||||
|
|
||||||
Cargo workspace with 16 crates under `crates/` (plus `libaec-sys`, an internal FFI bindings crate for the optional `szip` feature):
|
Cargo workspace, 19 crates under `crates/` (plus `libaec-sys`, the FFI crate
|
||||||
|
behind the optional `szip` feature). MSRV 1.92 (`rust-version`, checked in CI).
|
||||||
|
|
||||||
| Crate | Role |
|
| Crate | Role |
|
||||||
|-------|------|
|
|-------|------|
|
||||||
| `clawhdf5-format` | HDF5 binary spec parser (superblock, B-tree, heap) — also holds shared type definitions and physical constants |
|
| `clawhdf5-format` | The HDF5 format: parsers and writer (superblock, headers, B-trees, heaps, chunk indexes), the `Storage` trait, the filter pipeline and registry (`filter_registry`), every codec except deflate (LZ4, Zstd, SZIP, N-Bit, scale-offset, pcodec; pure-Rust LZF, bitshuffle, bzip2, Blosc 1; Blosc2 and ZFP read-only), `float16`, `checksum` |
|
||||||
| `clawhdf5-io` | Read/write implementation |
|
| `clawhdf5-filters` | Deflate backends (zlib-rs default, zlib-ng, Apple Compression) |
|
||||||
| `clawhdf5-filters` | Compression filters (gzip, LZ4, Zstd, Blosc) |
|
| `clawhdf5-io` | I/O adapters (buffers, mmap, prefetch) |
|
||||||
| `clawhdf5-derive` | Proc-macro derive for HDF5-serializable structs |
|
| `clawhdf5-derive` | `#[derive(H5Type)]` for compound types |
|
||||||
| `clawhdf5` | Main facade crate |
|
| `clawhdf5` | Facade: `File`, `FileBuilder`, `Dataset`, `FileEditor` (`src/edit/`), SWMR reading (`src/swmr.rs`) |
|
||||||
| `clawhdf5-netcdf4` | NetCDF-4 compatibility layer |
|
| `clawhdf5-netcdf4` | NetCDF-4 read support |
|
||||||
| `clawhdf5-ann` | HNSW approximate nearest-neighbor vector index |
|
| `clawhdf5-remote` | `open_url`: HTTP(S) range requests and object stores (S3, GCS, Azure) through `BlockCache` |
|
||||||
| `clawhdf5-agent` | Agent memory, session history, knowledge graph storage |
|
| `clawhdf5-tools` | `h5rs`: `ls`, `dump` (DDL / hdf5-json), `stat`, `diff`, `check` |
|
||||||
| `clawhdf5-gpu` | GPU-accelerated I/O via wgpu (hand-written WGSL compute shaders) |
|
| `clawhdf5-py` | PyO3 bindings (h5py-like API, remote files, `'r+'` editing) |
|
||||||
| `clawhdf5-accel` | CPU SIMD acceleration path |
|
| `clawhdf5-wasm` | wasm-bindgen browser reader (`open(bytes)`, `openUrl(url)`); demo in `examples/wasm-viewer/` |
|
||||||
| `clawhdf5-migrate` | SQLite → HDF5 agent-memory migration |
|
| `clawhdf5-ann` | HNSW index |
|
||||||
|
| `clawhdf5-agent` | Agent memory store (`HDF5Memory`), sessions, knowledge graph, BM25 |
|
||||||
|
| `clawhdf5-accel` | CPU SIMD kernels (AVX2, NEON) |
|
||||||
|
| `clawhdf5-gpu` | wgpu vector distances (WGSL) — not dataset I/O; HDF5 I/O is CPU-only |
|
||||||
|
| `clawhdf5-migrate` | SQLite → agent store migration |
|
||||||
|
| `clawhdf5-cli` | Agent-memory CLI |
|
||||||
|
| `clawhdf5-napi` | Node.js addon (the `packages/clawhdf5-node` wrapper is broken; `docs/known-issues.md`) |
|
||||||
| `clawhdf5-android` | Android JNI bindings |
|
| `clawhdf5-android` | Android JNI bindings |
|
||||||
| `clawhdf5-cli` | Command-line interface |
|
| `clawhdf5-bench` | Benchmarks and harnesses (`search_harness`, `read_harness`, `concurrent_read`, `longmemeval_bench`, …) |
|
||||||
| `clawhdf5-napi` | Node.js native addon bindings |
|
|
||||||
| `clawhdf5-py` | PyO3 Python bindings |
|
|
||||||
| `clawhdf5-bench` | Benchmark suite |
|
|
||||||
|
|
||||||
## Key Features
|
Reference docs: `docs/known-issues.md` (open issues table first — check it
|
||||||
- Zero-C-dependency HDF5 read/write: no libhdf5, and deflate defaults to
|
before calling something a bug or a feature), `BENCHMARKS.md` (headline
|
||||||
pure-Rust zlib-rs (`fast-deflate` opts into zlib-ng, which needs cmake).
|
numbers first), `CONFORMANCE.md` (generated), `docs/design/range-reads.md`
|
||||||
`ci-test.sh` fails if a C-building crate enters the core crates' default
|
and `docs/design/swmr.md`, `CHANGELOG.md` (full detail of every fix).
|
||||||
tree. flate2 must keep `runtime_detection` with zlib-rs — without it zlib-rs
|
|
||||||
loses SIMD and inflates 3.5x slower. MSRV is 1.92 (`rust-version`, checked
|
## Standing rules
|
||||||
in CI).
|
|
||||||
- HNSW vector index for semantic similarity search over agent memories — the
|
- **No C in the default build.** No libhdf5; deflate defaults to pure-Rust
|
||||||
`clawhdf5-agent` `hnsw` feature is **on by default**, so `hybrid_search` uses
|
zlib-rs (`fast-deflate` opts into zlib-ng, which needs cmake). `ci-test.sh`
|
||||||
the approximate `clawhdf5-ann` index for the vector stage (the index mirrors
|
fails if a C-building crate enters the core crates' default tree. Zstd,
|
||||||
the cache and self-heals on drift). Build the agent with
|
SZIP, `https` (ring) and `s3`/`gcs`/`azure` (aws-lc-rs) are opt-in. flate2
|
||||||
`--no-default-features --features float16` to force the exact linear cosine scan.
|
must keep `runtime_detection` with zlib-rs — without it zlib-rs loses SIMD
|
||||||
The agent's `parallel` feature (also default) builds the index on a thread
|
and inflates 3.5x slower.
|
||||||
pool; the graph is identical with or without it.
|
- **Every file we write must open in h5py/libhdf5.** Interop tests compare
|
||||||
The index uses the HNSW paper's diversity heuristic for neighbour selection
|
against h5py and h5dump; `f32` and empty datasets did not open until
|
||||||
(plain closest-M capped recall on clustered data: 0.31 recall@10 at 100K). Its
|
2026-09-23.
|
||||||
graph is saved to `<store>.h5.ann` at each checkpoint and reloaded by `open()`
|
- **float16 has one implementation:** `clawhdf5_format::float16`.
|
||||||
(tied to the checkpoint by a generation id; stale/damaged sidecars are
|
- **Claims need evidence.** Performance and integration claims in docs must
|
||||||
ignored and the index rebuilt). `MemoryConfig::quantized_index` (**on by
|
be measured, dated (with machine and command), or withdrawn. Benchmark
|
||||||
default** for new stores, persisted; stores predating the setting load as
|
numbers are dated records: never edit a measured value, add a new dated
|
||||||
`false` and keep their f32 index — guarded by
|
section and mark the old one superseded.
|
||||||
`tests/fixtures/store_v2_5_0.h5`; CLI opt-out is `create --f32-index`)
|
- **OpenClaw is not supported** (decided 2026-09-25): clawhdf5 is not and
|
||||||
stores the index's own copy of the embeddings as `i8`,
|
never was an OpenClaw memory plugin; the old `memory.backend = "clawhdf5"`
|
||||||
which roughly halves a loaded store's memory (2.72x -> 1.74x the raw vectors
|
config was never valid. `docs/openclaw.md` records what a real plugin would
|
||||||
at 100K); because quantised distances are approximate and `ef` cannot
|
need. The `openclaw` module's `ClawhdfBackend` is just `search` with
|
||||||
compensate, the query path then re-scores the candidate pool against the
|
re-rank + confidence on.
|
||||||
exact embeddings, which holds recall at the f32 index's level. It is also
|
- **ZeroClaw does not use clawhdf5** (checked 2026-09-25 against upstream
|
||||||
faster at equal recall: 1.63x the QPS on x86-64 (AVX2) and 1.18x on a
|
v0.8.5 and the `osobh/zeroclaw` fork and their history): its memory
|
||||||
Raspberry Pi 5 (`clawhdf5_accel::dot_i8`, NEON `SDOT` via inline asm since
|
backends are its own; `clawhdf5-migrate`'s default SQLite layout is not
|
||||||
the intrinsic is unstable; plain NEON on pre-dotprod cores). The aarch64
|
ZeroClaw's schema. Don't reintroduce integration claims without an
|
||||||
code is `cfg`'d out on x86, so x86 CI never compiles or lints it — test it
|
integration and a test against the real consumer.
|
||||||
on real ARM (`rpivision02`, 10.0.2.3, is a Pi 5). `hybrid_search` keeps one incremental BM25
|
- **known-issues.md:** one entry per bug; when fixed, record it in
|
||||||
index for the life of the store and never writes the store: Hebbian
|
`CHANGELOG.md` and move the entry to *Fixed (history)* with date, PR,
|
||||||
activation boosts are persisted by the next checkpoint (or on drop), not per
|
affected releases and what users must do — never delete it.
|
||||||
query. Measure any search-path change with
|
|
||||||
`cargo run --release -p clawhdf5-bench --bin search_harness` (baselines in
|
## HDF5 library: invariants and gotchas
|
||||||
`BENCHMARKS.md`).
|
|
||||||
- WAL (write-ahead log) for crash-safe persistence, with a chained CRC32
|
- **Remote/range reads** (`docs/design/range-reads.md`, M0-M5 merged in PRs
|
||||||
trailer per entry (each entry's CRC folds in the previous entry's CRC) so a
|
#17-#19, M4 listing costs cut in #21): every format-crate read path goes through `Storage`
|
||||||
corrupted, reordered, duplicated, or spliced entry stops replay cleanly
|
(`read_at`/`read_ranges`/`hint`). `File::open_storage` takes any
|
||||||
instead of loading bad or tampered data. The pre-chaining per-entry-CRC
|
`Storage`; `clawhdf5_remote::open_url` wraps HTTP (`HttpStorage`, ureq) or
|
||||||
format (v2) is still fully readable; the oldest no-CRC format (v1) is only
|
`ObjectStoreStorage` in `BlockCache` (1 MiB blocks, LRU budget, in-flight
|
||||||
reachable through the one-time migration path in `HDF5Memory::open`, not
|
dedup, coalesced runs). Remote files are pinned by ETag/Last-Modified and
|
||||||
through the public `WalFile::read_entries`.
|
length (`RemoteError::FileChanged`). Zero-copy APIs and `File::as_bytes`
|
||||||
**What the WAL guarantees:** integrity, ordering, and recovery from a
|
need an in-memory file. Parse through `File::storage()` and the `*_in`
|
||||||
*process* crash at any point — including between a checkpoint and the WAL
|
functions, not `as_bytes`, in new code (the Python bindings do).
|
||||||
truncate (each checkpoint records a `WalMark` in `/meta`, and `open()` skips
|
`ObjectStoreStorage` runs reads on its own small tokio runtime, so it
|
||||||
the WAL prefix the `.h5` already contains, so entries are never applied
|
works from any thread.
|
||||||
twice). Checkpoints and snapshots are made durable as a unit (temp file
|
- **SWMR** (`docs/design/swmr.md`): `File::open_swmr` reads a file a libhdf5
|
||||||
synced, renamed, directory synced). **What it does not guarantee:**
|
SWMR writer is appending to — positioned reads, no chunk cache, bounded
|
||||||
individual WAL appends are *not* fsynced (a deliberate latency trade-off), so
|
retries (100), `Dataset::refresh()`. clawhdf5 has no SWMR writer; remote
|
||||||
saves made since the last checkpoint can be lost on power failure or kernel
|
SWMR is out of scope.
|
||||||
panic. Current header version is 4 (adds the `Update` record used by
|
- **Browser** (`clawhdf5-wasm`, read-only, no Zstd/SZIP): `openUrl` reads
|
||||||
`save_or_update`); v3 files are read and upgraded in place.
|
through the restartable "NeedBytes" cache (`src/lazy.rs`: a call is re-run
|
||||||
- A store has a **single writer**: `HDF5Memory::create`/`open` hold an exclusive
|
after each wave of misses; no block is evicted while a call runs); the HTTP
|
||||||
advisory lock on `<store>.h5.lock` and a second opener gets
|
is JavaScript (`js/remote.js`).
|
||||||
`MemoryError::Locked`. Use `HDF5Memory::open_read_only` for a lock-free,
|
- **In-place editing** (`clawhdf5::FileEditor`): overwrites values, grows and
|
||||||
never-writing point-in-time view (the CLI's `recall`/`stats`/`agents-md`/
|
shrinks chunked datasets (every chunk index) and sets attributes (compact
|
||||||
`export` do). An unreadable WAL (torn header, bad magic) is quarantined to
|
and dense) without rewriting the file, changing indexes and heaps as
|
||||||
`<store>.h5.wal.corrupt-<ts>` rather than blocking `open()`; a WAL with an
|
libhdf5 does; freed space is reused within one editor. Anything it cannot
|
||||||
unknown *newer* version still fails and is left untouched.
|
do safely is `Error::Unsupported` before any write (limits in
|
||||||
- `MemoryConfig::float16` (**on by default** for new stores, persisted;
|
`docs/known-issues.md`). The algorithms follow libhdf5 `hdf5_1_14_6`
|
||||||
existing stores keep their recorded `false` — guarded by the v2.5.0
|
(github.com/HDFGroup/hdf5). Test changes with `cargo test -p
|
||||||
fixture in `tests/float16_store.rs`; CLI opt-out is `create --f32`) writes
|
clawhdf5-tools --test edit_interop --test edit_coverage_interop`.
|
||||||
`/memory/embeddings` as IEEE half precision (48% smaller file at 100K;
|
- **Provenance:** `Dataset::verify_provenance()` (facade `provenance`
|
||||||
LongMemEval with real MiniLM embeddings identical to f32).
|
feature, default) re-hashes a dataset against its `_provenance_sha256`
|
||||||
`MemoryCache::half_precision` rounds each embedding as it enters the cache (push, update, WAL replay, and on load of a store still
|
attribute (`DatasetBuilder::with_provenance`). Opt-in per call; unkeyed
|
||||||
`f32` on disk), so memory and file agree bit for bit; the conversions live
|
hash — tamper-evident, not tamper-proof.
|
||||||
in `clawhdf5_format::float16` and must stay the single implementation.
|
|
||||||
Values beyond ±65504 are `MemoryError::InvalidEntry`. Interop: every file
|
## Agent memory: invariants and gotchas
|
||||||
must open in h5py — `f32` datasets and empty datasets did not until
|
|
||||||
2026-09-23 (see `docs/known-issues.md`); the agent's `h5py_interop` test
|
- **Search.** `HDF5Memory::search(query_emb, text, &SearchOptions)` is the
|
||||||
guards a whole store.
|
full path: optional source-channel filter (before ranking; exact scan of
|
||||||
- `HDF5Memory::search(query_emb, text, &SearchOptions)` is the full search
|
the allowed records when cheaper than `pool × M` index distance
|
||||||
path: optional source-channel filter (applied before ranking; exact scan of
|
|
||||||
the allowed records whenever cheaper than `pool × M` index distance
|
|
||||||
evaluations, and as the fallback when the pool comes back short), fusion,
|
evaluations, and as the fallback when the pool comes back short), fusion,
|
||||||
activation scaling, optional re-ranking and confidence rejection.
|
activation scaling, optional re-ranking and confidence rejection.
|
||||||
`hybrid_search`/`hybrid_search_with` are thin wrappers; the OpenClaw
|
`hybrid_search`/`hybrid_search_with` are thin wrappers. It keeps one
|
||||||
backend is `search` with re-rank + confidence on. Measure changes with
|
incremental BM25 index for the life of the store and never writes the
|
||||||
`search_harness --options-study`.
|
store: Hebbian activation boosts are persisted by the next checkpoint (or
|
||||||
- `MemoryConfig::compression` is off by default; when on, embeddings are
|
on drop).
|
||||||
deflate-compressed, or Zstd with the agent's `zstd` feature (links libzstd).
|
- **HNSW** (`hnsw` feature, default): the approximate `clawhdf5-ann` index
|
||||||
- `Dataset::verify_provenance()` (clawhdf5 facade, `provenance` feature, on by
|
mirrors the cache and self-heals on drift; build the agent with
|
||||||
default) recomputes a dataset's SHA-256 and compares it against the
|
`--no-default-features --features float16` for the exact linear scan.
|
||||||
`_provenance_sha256` attribute written automatically on save when
|
`parallel` (default) builds it on a thread pool with an identical graph.
|
||||||
`DatasetBuilder::with_provenance` is used. It's opt-in per call, not run
|
Neighbour selection uses the HNSW paper's diversity heuristic (closest-M
|
||||||
automatically on open — it decodes and hashes the whole dataset. The hash
|
capped recall at 0.31 recall@10 at 100K on clustered data). The graph is
|
||||||
is unkeyed (tamper-*evident*, not tamper-*proof*): it detects accidental
|
saved to `<store>.h5.ann` at each checkpoint, tied to it by a generation
|
||||||
corruption, not a deliberate actor able to modify both the data and the
|
id; a stale or damaged sidecar is ignored and the index rebuilt.
|
||||||
stored hash.
|
- **`MemoryConfig::quantized_index`** (default on for new stores, persisted;
|
||||||
- `clawhdf5-agent`'s `HDF5Memory::save`/`save_batch`/`save_or_update` run every
|
older stores load as `false` — guarded by `tests/fixtures/store_v2_5_0.h5`;
|
||||||
write through an in-memory (session-scoped, not persisted to disk)
|
CLI `create --f32-index`): the index's copy of the embeddings is `i8`, and
|
||||||
provenance ledger and write-anomaly detector: a content hash per record
|
the query path re-scores candidates against the exact embeddings. The
|
||||||
(`provenance.rs`) for detecting accidental mid-session corruption, plus
|
aarch64 kernels (`clawhdf5_accel::dot_i8`, NEON `SDOT` via inline asm) are
|
||||||
rate-limit/injection-pattern/source-distribution checks (`anomaly.rs`).
|
`cfg`'d out on x86, so x86 CI never compiles them — test on real ARM
|
||||||
Alerts never block a save — drain them with `HDF5Memory::take_anomaly_alerts`.
|
(`rpivision02`, 10.0.2.3, a Pi 5) or rely on the `test-arm64` job.
|
||||||
`MemorySource` for this bookkeeping is inferred from the caller-supplied
|
- **`MemoryConfig::float16`** (default on for new stores, persisted; older
|
||||||
`source_channel` string (a heuristic, not an authenticated trust boundary).
|
stores keep `false` — guarded in `tests/float16_store.rs`; CLI `create
|
||||||
- GPU-accelerated batch I/O for large dataset processing
|
--f32`): `/memory/embeddings` is IEEE half. `MemoryCache::half_precision`
|
||||||
- Python and Node.js bindings for cross-language use
|
rounds each embedding as it enters the cache (push, update, WAL replay, and
|
||||||
- NetCDF-4 compatibility for scientific data interop
|
load of a store still `f32` on disk) so memory and file agree bit for bit.
|
||||||
|
Values beyond ±65504 are `MemoryError::InvalidEntry`. The agent's
|
||||||
|
`h5py_interop` test guards that a whole store opens in h5py.
|
||||||
|
- **WAL.** Chained CRC32 per entry (a corrupted, reordered, duplicated or
|
||||||
|
spliced entry stops replay cleanly). Header version 4 (`Update` record for
|
||||||
|
`save_or_update`); v3 is upgraded in place, v2 read, v1 only through the
|
||||||
|
one-time migration in `HDF5Memory::open`. Each checkpoint records a
|
||||||
|
`WalMark` in `/meta` so `open()` never applies an entry twice; checkpoints
|
||||||
|
and snapshots are durable as a unit (temp file synced, renamed, directory
|
||||||
|
synced). Individual WAL appends are **not** fsynced (deliberate): saves
|
||||||
|
since the last checkpoint can be lost on power failure or kernel panic.
|
||||||
|
- **Single writer.** `create`/`open` hold an exclusive lock on
|
||||||
|
`<store>.h5.lock` (`MemoryError::Locked` for a second opener);
|
||||||
|
`open_read_only` is a lock-free point-in-time view (CLI `recall`/`stats`/
|
||||||
|
`agents-md`/`export`). An unreadable WAL is quarantined to
|
||||||
|
`<store>.h5.wal.corrupt-<ts>`; a WAL of an unknown newer version fails and
|
||||||
|
is left untouched.
|
||||||
|
- **Signed checkpoints** (`signing` module): with `set_signing_key` each
|
||||||
|
checkpoint stores an Ed25519-signed manifest (per-record SHA-256 in a
|
||||||
|
Merkle tree plus settings/sessions/graph hashes; `/integrity/record_hashes`);
|
||||||
|
`HDF5Memory::verify(path, &pk)` locates edits. The hashes must cover
|
||||||
|
exactly what the file persists in the form the loader returns it (strings
|
||||||
|
lose trailing NULs; an empty WAL mark is not written) —
|
||||||
|
`tests/signed_store.rs` round-trips awkward strings. The key is never
|
||||||
|
persisted; a signed store refuses to checkpoint without it
|
||||||
|
(`MemoryError::SigningKeyRequired`; `MemoryError` is `#[non_exhaustive]`).
|
||||||
|
WAL entries after the checkpoint are not covered.
|
||||||
|
- **Write bookkeeping.** `save`/`save_batch`/`save_or_update` feed an
|
||||||
|
in-memory, session-scoped provenance ledger and anomaly detector
|
||||||
|
(`provenance.rs`, `anomaly.rs`); alerts never block a save
|
||||||
|
(`take_anomaly_alerts`). `MemorySource` is inferred from the caller's
|
||||||
|
`source_channel` string — a heuristic, not a trust boundary.
|
||||||
|
- `MemoryConfig::compression` is off by default (deflate, or Zstd with the
|
||||||
|
agent's `zstd` feature, which links libzstd).
|
||||||
|
|
||||||
## Workflows
|
## Workflows
|
||||||
|
|
||||||
### Build
|
Put `$HOME/.cargo/bin` on `PATH`. The h5py/netCDF4 interop tests find their
|
||||||
|
Python through `CLAWHDF5_PYTHON` (or `.venv/bin/python`); create it with
|
||||||
|
`python3 -m venv .venv && .venv/bin/pip install h5py numpy netCDF4 hdf5plugin`.
|
||||||
|
Set `CLAWHDF5_REQUIRE_INTEROP=1` to make a missing interpreter a failure.
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
cargo build --release
|
cargo build --release
|
||||||
```
|
|
||||||
|
|
||||||
### Test
|
|
||||||
```bash
|
|
||||||
cargo test --workspace
|
cargo test --workspace
|
||||||
|
bash scripts/ci-test.sh # everything CI runs (see below)
|
||||||
```
|
```
|
||||||
|
|
||||||
### CI
|
### CI (`.gitea/workflows/`)
|
||||||
`.gitea/workflows/ci.yml` has two jobs, both green as of 2026-09-22:
|
- **`ci.yml` `test`** (`ubuntu-latest`, `rust:latest` container; runners
|
||||||
- **`test`** (`ubuntu-latest`, in `rust:latest`) runs `scripts/ci-test.sh` with
|
`tank`, `architect`): installs h5py/netCDF4/xarray/hdf5plugin/maturin/pytest,
|
||||||
the h5py/netCDF4 interop suites required (`CLAWHDF5_REQUIRE_INTEROP=1`).
|
`hdf5-tools` and `cmake`, then runs `scripts/ci-test.sh` with
|
||||||
Served by the `tank` and `architect` runners.
|
`CLAWHDF5_REQUIRE_INTEROP=1`. The script runs: fmt; clippy (workspace, the
|
||||||
- **`test-arm64`** (`linux_arm64`) lints and tests the aarch64 code — the NEON
|
format feature matrix, each plugin filter alone, parallel, fast-deflate,
|
||||||
kernels are `cfg`'d out on x86, so this is the only place they are built.
|
remote with all backends, h5rs remote); "no C in the default build";
|
||||||
Served by `vision-01` (host mode) and `vision-02` (Docker), so steps must
|
wasm32 build and clippy; `check-32bit-casts.sh`; the wasm package under
|
||||||
work in both.
|
Node when `node` and `wasm-bindgen` exist (not in CI); the MSRV check;
|
||||||
|
`cargo test` (workspace plus feature variants: format matrix, parallel,
|
||||||
|
remote/object_store, h5rs URLs, ann parallel, fast-deflate); the h5py
|
||||||
|
interop suites (`writer_h5py_tests --include-ignored`, plugin filters,
|
||||||
|
ZFP); the Python package (clippy, `maturin build`, pytest vs h5py);
|
||||||
|
`cargo bench --no-run`; `check-nostd.sh`; an optional fuzz smoke run
|
||||||
|
(`CLAWHDF5_FUZZ_SECONDS`).
|
||||||
|
- **`ci.yml` `test-arm64`** (`linux_arm64`; `vision-01` host mode,
|
||||||
|
`vision-02` Docker — steps must work in both): clippy of
|
||||||
|
`clawhdf5-accel`, tests of `-accel`, `-ann`, `-format`; the only place the NEON kernels build.
|
||||||
|
- **`conformance.yml`** (nightly 03:17 UTC and manual): probe unit tests,
|
||||||
|
`conformance/test_ref.py`, then `conformance/run.sh` (gate:
|
||||||
|
`conformance/check.py` against `baseline.json`).
|
||||||
|
|
||||||
Keep workflows free of JavaScript actions (`actions/checkout`, `actions/cache`,
|
Keep workflows free of JavaScript actions (`actions/checkout`,
|
||||||
…): `rust:latest` has no `node`, and not every runner reaches GitHub, where
|
`actions/cache`, …): `rust:latest` has no `node` and not every runner reaches
|
||||||
they are fetched from. Check out with plain `git` instead. The `test` job
|
GitHub. Check out with plain `git`. Runners are `gitea-runner` 3.5.0 from
|
||||||
installs `cmake` for the opt-in `fast-deflate` (zlib-ng) steps; the default
|
`docker.gitea.com/act_runner` (`gitea/act_runner:latest` on Docker Hub is
|
||||||
build needs no C toolchain, so `test-arm64` does not.
|
frozen at 0.6.1).
|
||||||
All runners are on `gitea-runner` 3.5.0, from `docker.gitea.com/act_runner`
|
|
||||||
— `gitea/act_runner:latest` on Docker Hub is frozen at 0.6.1.
|
### Conformance
|
||||||
|
```bash
|
||||||
|
CLAWHDF5_PYTHON=.venv/bin/python bash conformance/run.sh --no-fetch # writes CONFORMANCE.md
|
||||||
|
```
|
||||||
|
Reads 697 files of eight pinned corpora with clawhdf5 and h5py and compares
|
||||||
|
them object by object (602 ok in the run of 2026-09-28). `CONFORMANCE.md` is
|
||||||
|
generated — never hand-edit it (its wording lives in `conformance/report.py`).
|
||||||
|
Use `--update-baseline` only after an intended change in results.
|
||||||
|
`CONFORMANCE_CACHE` points at an existing corpus cache (`conformance/.cache`,
|
||||||
|
about 450 MB). See `conformance/README.md`.
|
||||||
|
|
||||||
|
### HDF5 tools (`h5rs`)
|
||||||
|
```bash
|
||||||
|
cargo run -p clawhdf5-tools -- ls -r file.h5 # also dump [--json], stat, diff, check
|
||||||
|
bash scripts/h5rs-fuzz.sh # every subcommand over the CVE corpus: no panic/crash/hang
|
||||||
|
bash scripts/h5rs-check-ok-files.sh --data # check passes every fully-read conformance file
|
||||||
|
```
|
||||||
|
Interop tests compare against h5ls/h5stat/h5dump/h5diff (Debian `hdf5-tools`);
|
||||||
|
`dump` must stay byte-identical to h5dump on the test files.
|
||||||
|
|
||||||
|
### Remote and browser tests
|
||||||
|
- `clawhdf5-remote` tests run a std-only HTTP server
|
||||||
|
(`tests/common/server.rs`, also the `range_server` example);
|
||||||
|
`CLAWHDF5_REMOTE_CORPUS=conformance/.cache/corpus` compares every corpus
|
||||||
|
file over HTTP with `File::open`.
|
||||||
|
- wasm: `bash examples/wasm-viewer/test/run.sh` builds the package (needs the
|
||||||
|
`wasm-bindgen` CLI at the crate's exact version) and tests it under Node and
|
||||||
|
headless Chromium (Playwright's download in `~/.cache/ms-playwright` on
|
||||||
|
tank) against `test/serve.py` (range server with request counts). CI has
|
||||||
|
neither, so it runs the native `h5py_interop` and `lazy` tests
|
||||||
|
(`CLAWHDF5_WASM_CORPUS=conformance/.cache/corpus` for the corpus).
|
||||||
|
|
||||||
|
### Python bindings
|
||||||
|
```bash
|
||||||
|
cd crates/clawhdf5-py && maturin develop
|
||||||
|
python -m pytest crates/clawhdf5-py/tests # compares with h5py; editing tests want CLAWHDF5_H5RS=<path to h5rs>
|
||||||
|
```
|
||||||
|
|
||||||
|
### Benchmarks
|
||||||
|
- Search path: `cargo run --release -p clawhdf5-bench --bin search_harness`
|
||||||
|
(`--full`, `--options-study`, `--footprint`, …); reads: `read_harness`,
|
||||||
|
`concurrent_read`; criterion benches with `cargo bench -p <crate>`.
|
||||||
|
- Run on an idle machine (1-minute load average below 2; wait otherwise),
|
||||||
|
alternate base and candidate binaries for A/B comparisons, and record date,
|
||||||
|
machine, commit and command with every number in `BENCHMARKS.md`.
|
||||||
|
- `BENCHMARKS.md` is written by hand from dated runs; no script regenerates
|
||||||
|
it (the old `scripts/run-benchmarks.sh`, which benchmarked the pre-rename
|
||||||
|
`rustyhdf5-format` and overwrote the file, was removed on 2026-09-28).
|
||||||
|
|
||||||
### CLI
|
### CLI
|
||||||
```bash
|
```bash
|
||||||
cargo run -p clawhdf5-cli -- --help
|
cargo run -p clawhdf5-cli -- --help
|
||||||
# create, save, search, recall, stats, flush-wal, agents-md, export, snapshot subcommands
|
# create, save, search, recall, stats, flush-wal, agents-md, export, snapshot, keygen, verify
|
||||||
```
|
|
||||||
|
|
||||||
### Python bindings
|
|
||||||
```bash
|
|
||||||
cd crates/clawhdf5-py
|
|
||||||
maturin develop
|
|
||||||
python -c "import clawhdf5; print(clawhdf5.__version__)"
|
|
||||||
```
|
```
|
||||||
|
|
||||||
## Integration
|
## Integration
|
||||||
ZeroClaw imports this as a Cargo feature (`clawhdf5` feature flag) to persist agent memory with HNSW vector search for context retrieval.
|
- **ClawBrainHub** (`clawverse/clawbrainhub` on git.redclaw.dev) is the one
|
||||||
|
verified consumer: `cbh-core` reads and writes `.brain` files through the
|
||||||
|
facade (`File`, `FileBuilder`, `AttrValue`, `Selection`), `cbh-scanner`
|
||||||
|
uses the facade, and `cbh-cli` uses `clawhdf5_agent::bm25::BM25Index`. It
|
||||||
|
depends on this repo by path (`../clawhdf5`), so changes to those APIs
|
||||||
|
reach it directly. Verified 2026-09-25 against main: builds, and its 204
|
||||||
|
tests pass.
|
||||||
|
- OpenClaw and ZeroClaw integrate nothing (see *Standing rules*).
|
||||||
|
|||||||
+305
@@ -0,0 +1,305 @@
|
|||||||
|
# clawhdf5 conformance report
|
||||||
|
|
||||||
|
Every HDF5 file of eight public corpora (pinned by commit) is read twice — by
|
||||||
|
clawhdf5 (`conformance/probe`, the same `clawhdf5-format` calls the facade
|
||||||
|
makes) and by h5py/libhdf5 (`conformance/ref.py`) — and the two readings are
|
||||||
|
compared object by object: the set of hard-linked objects, each dataset's and
|
||||||
|
attribute's shape, and a SHA-256 of its values in a canonical encoding. The
|
||||||
|
CVE corpus is also run through `h5dump`. Each side runs under a timeout and an
|
||||||
|
address-space limit, so a hang, crash or runaway allocation is recorded, not
|
||||||
|
fatal. This file is generated by `conformance/run.sh`; do not edit it by hand.
|
||||||
|
|
||||||
|
## Run
|
||||||
|
|
||||||
|
| | |
|
||||||
|
|---|---|
|
||||||
|
| date | 2026-09-28 04:29 UTC |
|
||||||
|
| clawhdf5 commit | `bf5a163dcf7fe28d651ada6545d8136ffeffc825` |
|
||||||
|
| machine | `tank`: AMD Ryzen 7 7800X3D 8-Core Processor, 16 CPUs, 61 GiB, Linux 7.0.0-34-generic x86_64 |
|
||||||
|
| command | `conformance/run.sh --no-fetch --update-baseline` |
|
||||||
|
| rustc | rustc 1.98.1 (48a229cea 2026-09-01) |
|
||||||
|
| reference | h5py 3.16.0, HDF5 2.0.0, numpy 2.5.3, hdf5plugin 7.1.0, Python 3.14.4 |
|
||||||
|
| h5dump | Version 1.14.6 (CVE corpus only) |
|
||||||
|
| limits | 20 s timeout (SIGKILL), 4096 MiB address space, per process; 16 files in parallel |
|
||||||
|
| runtime | 20 s probing + comparing (5 s fetch/build before it) |
|
||||||
|
|
||||||
|
## Results
|
||||||
|
|
||||||
|
A file's class is the first that applies:
|
||||||
|
|
||||||
|
- **panic / hang / crash / oom** — clawhdf5 panicked (caught per object or not), hit the timeout, died on a signal, or failed an allocation. The CI gate fails on any of these.
|
||||||
|
- **h5py-cannot-read** — libhdf5 could not open the file (or itself crashed or hung). Nothing to compare against; most are the deliberately malformed CVE reproducers.
|
||||||
|
- **ref-bug** — every difference is an object clawhdf5 refuses that h5py reads only through a libhdf5 bug: the values h5py returns for it change with the reading process's heap, re-checked in every run (see *Reference bugs*).
|
||||||
|
- **our-error** — clawhdf5 returned an error for something h5py reads.
|
||||||
|
- **mismatch** — both read it, but the shapes, values, object set or attribute set differ.
|
||||||
|
- **ok** — every object h5py reads, clawhdf5 reads identically.
|
||||||
|
|
||||||
|
| corpus | files | ok | our-error | mismatch | h5py-cannot-read | ref-bug | panic | hang | crash | oom |
|
||||||
|
|---|---|---|---|---|---|---|---|---|---|---|
|
||||||
|
| NCAS-CMS_pyfive | 33 | 33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 |
|
||||||
|
| cve_hdf5 | 147 | 113 | 0 | 0 | 32 | 2 | 0 | 0 | 0 | 0 |
|
||||||
|
| h5py_data | 4 | 4 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 |
|
||||||
|
| hdf5 | 466 | 405 | 1 | 0 | 60 | 0 | 0 | 0 | 0 | 0 |
|
||||||
|
| netcdf-c | 20 | 20 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 |
|
||||||
|
| netcdf4-python | 18 | 18 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 |
|
||||||
|
| usnistgov_h5wasm | 5 | 5 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 |
|
||||||
|
| xarray-data | 4 | 4 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 |
|
||||||
|
| **all** | **697** | **602** | **1** | **0** | **92** | **2** | **0** | **0** | **0** | **0** |
|
||||||
|
|
||||||
|
**Our errors and mismatches: 1.** Files not ok: 1 our-error, 92 h5py-cannot-read, 2 ref-bug. 2 object(s) were compared against h5py's values corrected for a known h5py bug (2 identical to clawhdf5's; see *Reference bugs*).
|
||||||
|
|
||||||
|
Corpora (fetched by `conformance/fetch-corpus.sh` into the gitignored `conformance/.cache/`):
|
||||||
|
|
||||||
|
| corpus | source | commit |
|
||||||
|
|---|---|---|
|
||||||
|
| hdf5 | https://github.com/HDFGroup/hdf5 | `a3cf1ea82cc7` |
|
||||||
|
| cve_hdf5 | https://github.com/HDFGroup/cve_hdf5 | `3fd1f5ae3869` |
|
||||||
|
| netcdf-c | https://github.com/Unidata/netcdf-c | `beb7b9585273` |
|
||||||
|
| NCAS-CMS_pyfive | https://github.com/NCAS-CMS/pyfive | `8cf07b874913` |
|
||||||
|
| usnistgov_h5wasm | https://github.com/usnistgov/h5wasm | `02f6336527d2` |
|
||||||
|
| netcdf4-python | https://github.com/Unidata/netcdf4-python | `6e67576d39ae` |
|
||||||
|
| xarray-data | https://github.com/pydata/xarray-data | `a35297e9da2c` |
|
||||||
|
| h5py_data | https://github.com/h5py/h5py (`h5py/tests/data_files`) | `b2f0347c4200` |
|
||||||
|
|
||||||
|
## Panics, hangs, crashes, out-of-memory
|
||||||
|
|
||||||
|
None.
|
||||||
|
|
||||||
|
## Our-error root causes
|
||||||
|
|
||||||
|
Grouped by normalised error message. *files* counts files whose class this cause affects.
|
||||||
|
|
||||||
|
| files | objects | error | examples |
|
||||||
|
|---:|---:|---|---|
|
||||||
|
| 1 | 1 | `ChunkedReadError("…")` | `hdf5/test/testfiles/bad_nbit_parms_walk.h5` |
|
||||||
|
|
||||||
|
## Mismatch root causes
|
||||||
|
|
||||||
|
None.
|
||||||
|
|
||||||
|
## CVE corpus: clawhdf5 vs h5dump vs h5py
|
||||||
|
|
||||||
|
The 147 files of [HDFGroup/cve_hdf5](https://github.com/HDFGroup/cve_hdf5) — reproducers for
|
||||||
|
published libhdf5 CVEs and fuzzer finds. *read* = produced output (possibly with per-object
|
||||||
|
errors), *error* = refused cleanly. h5dump exits non-zero on any error anywhere in a file, so
|
||||||
|
its read/error split is not comparable with the other two rows; the panic, crash, hang and oom
|
||||||
|
columns are.
|
||||||
|
|
||||||
|
| tool | read | error | panic | crash | hang | oom |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
| clawhdf5 | 121 | 26 | 0 | 0 | 0 | 0 |
|
||||||
|
| h5dump 1.14.6 | 16 | 129 | 0 | 2 | 0 | 0 |
|
||||||
|
| h5py 3.16.0 / HDF5 2.0.0 | 115 | 31 | 0 | 1 | 0 | 0 |
|
||||||
|
|
||||||
|
<details><summary>Per-file outcomes</summary>
|
||||||
|
|
||||||
|
| file | h5dump | h5py | clawhdf5 | class |
|
||||||
|
|---|---|---|---|---|
|
||||||
|
| cvefiles/cve-2016-4330.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2016-4331.h5 | error exit | read 25 obj, 1 errors | read 25 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2016-4332-mtime-new.h5 | error exit | read 25 obj, 1 errors | read 25 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2016-4332-mtime.h5 | error exit | read 4 obj, 3 errors | read 4 obj, 3 errors | ok |
|
||||||
|
| cvefiles/cve-2016-4332-stab.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2016-4333.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2017-17505.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2017-17506.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2017-17507.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2017-17508.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2017-17509.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-11202.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-11203.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-11204.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-11205.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-11206-new.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-11206-old.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-11207.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-13866.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2018-13867.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-13868.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-13869.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-13870.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-13871.h5 | error exit | read 2 obj | read 2 obj | ok |
|
||||||
|
| cvefiles/cve-2018-13872.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-13873.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-13874.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2018-13875.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-13876.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2018-14031.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-14033.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-14034.h5 | error exit | read 1 obj, 2 errors | read 1 obj | ok |
|
||||||
|
| cvefiles/cve-2018-14035.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-14460.h5 | error exit | read 3 obj, 2 errors | read 3 obj, 2 errors | ok |
|
||||||
|
| cvefiles/cve-2018-15671.h5 | ok | read 1 obj | read 1 obj | ok |
|
||||||
|
| cvefiles/cve-2018-15672.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-16438.h5 | error exit | read 1 obj, 1 errors | read 1 obj | ok |
|
||||||
|
| cvefiles/cve-2018-17233.h5 | error exit | read 6 obj, 1 errors | read 6 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-17234.h5 | error exit | read 6 obj, 1 errors | read 6 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-17237.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-17432.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-17433 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2018-17434.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-17435.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-17436 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2018-17437.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-17438 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-17439 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2019-8396.h5 | error exit | read 3 obj, 2 errors | read 3 obj, 2 errors | ok |
|
||||||
|
| cvefiles/cve-2019-8397.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2019-8398.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2019-9151.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 2 errors | ok |
|
||||||
|
| cvefiles/cve-2019-9152.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2020-10809 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2020-10810.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2020-10811.h5 | error exit | read 25 obj, 1 errors | read 25 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2020-10812.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2020-18232.h5 | error exit | read 3 obj, 2 errors | read 3 obj, 2 errors | ok |
|
||||||
|
| cvefiles/cve-2020-18494.h5 | ok | read 2 obj | read 2 obj | ok |
|
||||||
|
| cvefiles/cve-2021-36977.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2021-37501.h5 | error exit | read 18 obj, 1 errors | read 18 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2021-45829.h5 | error exit | read 1 obj, 2 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2021-45830.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2021-45833.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2021-46242.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2021-46243.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2021-46244.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29157.h5 | error exit | read 4 obj, 7 errors | read 4 obj, 7 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29158.h5 | ok | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29159.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29160.h5 | error exit | read 4 obj, 1 errors | read 4 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29161.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29162.h5 | error exit | read 17 obj, 4 errors | read 17 obj, 4 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29163.h5 | error exit | read 7 obj, 1 errors | read 7 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29164.h5 | ok | read 3 obj | read 3 obj | ok |
|
||||||
|
| cvefiles/cve-2024-29165.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29166.h5 | error exit | read 17 obj, 2 errors | read 17 obj | ok |
|
||||||
|
| cvefiles/cve-2024-32605.h5 | ok | read 6 obj, 1 errors | read 6 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32606.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32607-1.h5 | ok | read 10 obj | read 10 obj | ok |
|
||||||
|
| cvefiles/cve-2024-32607-2.h5 | error exit | read 9 obj, 1 errors | read 9 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32608.h5 | error exit | read 6 obj, 1 errors | read 6 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32609.h5 | error exit | SIGSEGV | read 3 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2024-32610.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32611.h5 | ok | read 6 obj | read 6 obj | ok |
|
||||||
|
| cvefiles/cve-2024-32612.h5 | ok | read 3 obj | read 3 obj | ok |
|
||||||
|
| cvefiles/cve-2024-32613.h5 | error exit | read 7 obj, 1 errors | read 7 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32614.h5 | error exit | read 25 obj, 2 errors | read 25 obj, 2 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32615.h5 | error exit | read 4 obj, 1 errors | read 4 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32616.h5 | error exit | read 10 obj, 7 errors | read 10 obj, 6 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32617.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32618.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32619.h5 | error exit | read 3 obj, 2 errors | read 3 obj, 2 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32620.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32621.h5 | ok | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32622.h5 | ok | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32623.h5 | ok | read 6 obj | read 6 obj | ok |
|
||||||
|
| cvefiles/cve-2024-32624.h5 | error exit | read 6 obj, 1 errors | read 6 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-33873.h5 | error exit | read 4 obj, 1 errors | read 4 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-33874.h5 | ok | read 6 obj, 1 errors | read 6 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-33875.h5 | ok | read 2 obj | read 2 obj | ok |
|
||||||
|
| cvefiles/cve-2024-33876.h5 | ok | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-33877.h5 | error exit | read 8 obj, 1 errors | read 8 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-2153.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-2308.h5 | error exit | read 25 obj, 1 errors | read 25 obj, 2 errors | ref-bug |
|
||||||
|
| cvefiles/cve-2025-2309.h5 | ok | read 6 obj, 1 errors | read 6 obj | ok |
|
||||||
|
| cvefiles/cve-2025-2310.h5 | error exit | read 24 obj, 8 errors | read 24 obj, 8 errors | ok |
|
||||||
|
| cvefiles/cve-2025-2912.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-2913.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-2914.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-2915.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-2923.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-2924.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-2925.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-2926.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-44904.h5 | error exit | read 25 obj, 1 errors | read 25 obj, 2 errors | ref-bug |
|
||||||
|
| cvefiles/cve-2025-44905.h5 | error exit | read 25 obj, 3 errors | read 25 obj, 3 errors | ok |
|
||||||
|
| cvefiles/cve-2025-6269-1.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-6269-2.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-6269-3.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-6269-4.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-6270-1.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-6270-2.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-6270-3.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-6516.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-6750.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-6816.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-6817.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-6818.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-6856.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-6857.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-6858.h5 | SIGSEGV | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-7067.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-7068.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-7069.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2026-26200.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2026-34734.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2026-92627.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/unknown-1.h5 | error exit | read 11 obj, 1 errors | read 11 obj, 1 errors | ok |
|
||||||
|
| fuzzerfiles/gh-4431-poc-03.h5 | error exit | read 1 obj | read 1 obj | ok |
|
||||||
|
| fuzzerfiles/gh-4432-poc-05.h5 | SIGSEGV | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| fuzzerfiles/gh-4433-poc-08.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| fuzzerfiles/gh-4434-poc-09.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| fuzzerfiles/gh-4435-poc-10.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| fuzzerfiles/gh-4585.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| fuzzerfiles/gh_2649_flawed.h5 | error exit | read 9 obj, 1 errors | read 9 obj, 1 errors | ok |
|
||||||
|
| fuzzerfiles/gh_2649_plain_model.h5 | ok | read 10 obj | read 10 obj | ok |
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
## Reference bugs
|
||||||
|
|
||||||
|
### Objects h5py reads only through a libhdf5 bug (*ref-bug*)
|
||||||
|
|
||||||
|
clawhdf5 refuses these objects; h5py 3.16 / HDF5 2.0 returns values for them. `conformance/ref_bugs.py`
|
||||||
|
re-reads each with h5py in six fresh processes whose heaps differ (h5py imported before numpy, three
|
||||||
|
times and twice more with `MALLOC_PERTURB_`, and numpy imported first). Values the file determines
|
||||||
|
come out the same every time; these do not, so they are memory libhdf5 over-reads, not the file's
|
||||||
|
data. A file is *ref-bug* only while every one of its differences is such an object confirmed in
|
||||||
|
the same run; an object that reads the same every time goes back to *our-error*. Reproducer:
|
||||||
|
`python conformance/ref_bugs.py conformance/.cache/corpus` (prints every read's outcome).
|
||||||
|
|
||||||
|
| file | object | distinct results in 6 reads | confirmed | what goes wrong |
|
||||||
|
|---|---|---:|---|---|
|
||||||
|
| `cve_hdf5/cvefiles/cve-2025-2308.h5` | `/Scale_offset_long_long_data_le` | 6 | yes | the first chunk records minbits 11: its 12 values need 17 bytes of codes, and the 26-byte chunk holds 5 after its 21-byte header; libhdf5's scale-offset decoder reads past its buffer, and develop refuses the chunk ("Buffer too short") |
|
||||||
|
| `cve_hdf5/cvefiles/cve-2025-44904.h5` | `/Scale_offset_float_data_le` | 6 | yes | unfiltered chunks stored as 38 and 37 bytes for 48-byte chunks: 1.14/2.0 read the stored bytes into a buffer of that size and use it as the whole chunk (H5D__chunk_lock), so the rest is heap memory; develop refuses them ("incorrect chunk size returned from index for unfiltered chunk") |
|
||||||
|
| `hdf5/test/testfiles/bad_nbit_parms_walk.h5` | `/Nbit_int_data_le` | 1 | **no** | the N-Bit parameter list holds 7 values (cd_values[0] = 7) where an integer needs 8: the decoder takes the bit offset from cd_values[7], past the list; libhdf5's own test (`test_filter_bad_params`, test/dsets.c on develop) requires the read to fail |
|
||||||
|
|
||||||
|
### Values corrected for a known h5py bug
|
||||||
|
|
||||||
|
- **h5py big-endian variable-length sequences.** h5py returns the elements of a VL sequence
|
||||||
|
whose base type is big-endian with the file's big-endian bytes but a native (little-endian)
|
||||||
|
numpy dtype: a `h5py.vlen_dtype(np.dtype('>f4'))` dataset holding `[1.0, 2.0]` reads back as
|
||||||
|
`[4.6e-41, 9.0e-44]`; `h5dump` prints the file's values. `ref.py` checks that the installed
|
||||||
|
h5py still does this (by writing and reading exactly that dataset in memory) and, if so,
|
||||||
|
relabels such elements with the file's byte order before hashing, so the values are still
|
||||||
|
compared. Corrected objects: `NCAS-CMS_pyfive/tests/data/attr_datatypes.hdf5` `/@vlen_uint64` (same as clawhdf5), `hdf5/tools/test/testfiles/tcomplex_be.h5` `/VariableLengthDatasetFloatComplex` (same as clawhdf5).
|
||||||
|
|
||||||
|
## Other comparison rules
|
||||||
|
|
||||||
|
- **Non-IEEE floats and partial-precision integers (N-Bit).** libhdf5 converts a float whose
|
||||||
|
bit layout is not IEEE (e.g. `H5Tset_precision` for the N-Bit filter) or an integer with a
|
||||||
|
bit offset / reduced precision into the plain numpy type of the same size. The probe
|
||||||
|
compares such values as converted numbers, not raw file bytes (before 2026-09-25 it compared
|
||||||
|
raw bytes, which reported every N-Bit float dataset as a mismatch).
|
||||||
|
- **Types h5py widens.** Where h5py reads a type into a numpy type of a different size
|
||||||
|
(FP8 -> float16, bfloat16 -> float32, x87 long double -> float128) the values are not
|
||||||
|
compared (shape and presence still are): dataset file type size 1 -> numpy float16 (2) (15x), attr file type size 1 -> numpy float16 (2) (15x), dataset file type size 2 -> numpy float32 (4) (2x), dataset file type size 8 -> numpy float128 (16) (1x), dataset file type size 12 -> numpy float128 (16) (1x), attr file type size 2 -> numpy float32 (4) (1x), dataset file type size 2 -> numpy >f4 (4) (1x), attr file type size 2 -> numpy >f4 (4) (1x).
|
||||||
|
- **References** are compared by presence only (`R`), not by target.
|
||||||
|
|
||||||
|
## Objects h5py fails on but clawhdf5 reads
|
||||||
|
|
||||||
|
- 19 x `OSError: Can't synchronously read data (no appropriate function for conversion path)`
|
||||||
|
- 1 x `TypeError: unhandled dtype kind M (dtype('…'))`
|
||||||
|
- 1 x `TypeError: No NumPy equivalent for TypeTimeID exists`
|
||||||
|
- 1 x `ValueError: Insufficient precision in available types to represent (N, N, N, N, N)`
|
||||||
|
|
||||||
|
## Reproduce
|
||||||
|
|
||||||
|
```sh
|
||||||
|
# needs: Rust, python3 with h5py numpy hdf5plugin (conformance/requirements.txt), h5dump (hdf5-tools), git
|
||||||
|
CLAWHDF5_PYTHON=/path/to/venv/bin/python conformance/run.sh
|
||||||
|
```
|
||||||
|
|
||||||
|
The corpus (about 450 MB of sparse checkouts) is cached in `conformance/.cache/`; results for
|
||||||
|
every file, both sides' raw JSON and stderr, are in `conformance/.cache/results/`.
|
||||||
|
`conformance/baseline.json` holds the ok files the nightly CI job (`.gitea/workflows/conformance.yml`)
|
||||||
|
must keep; `conformance/run.sh --update-baseline` rewrites it.
|
||||||
+12
@@ -16,6 +16,9 @@ members = [
|
|||||||
"crates/clawhdf5-cli",
|
"crates/clawhdf5-cli",
|
||||||
"crates/clawhdf5-napi",
|
"crates/clawhdf5-napi",
|
||||||
"crates/clawhdf5-bench",
|
"crates/clawhdf5-bench",
|
||||||
|
"crates/clawhdf5-tools",
|
||||||
|
"crates/clawhdf5-wasm",
|
||||||
|
"crates/clawhdf5-remote",
|
||||||
"crates/libaec-sys",
|
"crates/libaec-sys",
|
||||||
]
|
]
|
||||||
resolver = "2"
|
resolver = "2"
|
||||||
@@ -34,3 +37,12 @@ tempfile = "3"
|
|||||||
criterion = { version = "0.5", features = ["html_reports"] }
|
criterion = { version = "0.5", features = ["html_reports"] }
|
||||||
half = "2.7"
|
half = "2.7"
|
||||||
serde = { version = "1", features = ["derive"] }
|
serde = { version = "1", features = ["derive"] }
|
||||||
|
|
||||||
|
# The browser build of clawhdf5-wasm (examples/wasm-viewer/build.sh): size
|
||||||
|
# over speed, whole-program optimisation. Native profiles are unaffected.
|
||||||
|
[profile.wasm-release]
|
||||||
|
inherits = "release"
|
||||||
|
opt-level = "s"
|
||||||
|
lto = true
|
||||||
|
codegen-units = 1
|
||||||
|
panic = "abort"
|
||||||
|
|||||||
+144
-171
@@ -1,187 +1,160 @@
|
|||||||
# ClawhDF5 Roadmap — Agent Memory Evolution
|
# clawhdf5 roadmap
|
||||||
|
|
||||||
> Making clawhdf5 the defacto agentic memory solution.
|
What has shipped, and what is genuinely next. Everything here is checked
|
||||||
> Single file. Pure Rust. Zero dependencies. Trusted everywhere.
|
against `CHANGELOG.md`, `git log` and [`docs/known-issues.md`](docs/known-issues.md);
|
||||||
|
dates are merge dates on `main`. Nothing after v2.7.0 has been released:
|
||||||
|
the work since then is on `main` under `CHANGELOG.md` "Unreleased".
|
||||||
|
|
||||||
|
_Last updated: 2026-09-28 (at `9b5803f`, PR #21)._
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
## Track 1: Knowledge Graph in HDF5
|
## Done
|
||||||
**Status:** 🟢 Phase 1 Complete
|
|
||||||
**Priority:** Critical
|
|
||||||
**Crate:** `clawhdf5-agent`
|
|
||||||
|
|
||||||
- [x] **1.1** Entity storage — entities with properties, embeddings, timestamps (created_at/updated_at)
|
### Releases
|
||||||
- [x] **1.2** Relation storage — typed edges with RelationType enum (Temporal/Causal/Associative/Hierarchical/Custom), metadata, timestamps
|
|
||||||
- [x] **1.3** Entity extraction helpers — rule-based extraction (Person, Org, Location, Date, Technology, Project) with extract_and_store_entities() integration
|
|
||||||
- [x] **1.4** Entity resolution — fuzzy name matching (Levenshtein distance) via resolve_or_create()
|
|
||||||
- [x] **1.5** Graph traversal queries — BFS neighbors with depth, subgraph extraction from seeds
|
|
||||||
- [x] **1.6** Spreading activation — weighted activation propagation with configurable decay
|
|
||||||
- [x] **1.7** Graph-aware retrieval — get_entity_context() for formatted context injection
|
|
||||||
- [x] **1.8** Tests — comprehensive tests for all new features
|
|
||||||
|
|
||||||
**Research:** Graph-Native Cognitive Memory (2026), Graph-based Agent Memory survey (2026), SYNAPSE (2025)
|
| Version | Date | Headline |
|
||||||
|
|---|---|---|
|
||||||
|
| v2.0.0 | 2026-03-19 | rustyhdf5 (11 crates) and edgehdf5 (4 crates) unified into one workspace as `clawhdf5-*` |
|
||||||
|
| v2.1.0 | 2026-06-03 | HNSW backs the agent's vector search by default; live, mutable HNSW index |
|
||||||
|
| v2.2.0 – v2.7.0 | 2026-09-18 – 2026-09-20 | bounded decompression and read-path bounds checks, single-writer store locking, WAL v4, HNSW recall fix (0.31 -> 0.98 recall@10 at 100K), fusion weights tuned on LongMemEval, int8 index, Extensible Array read fix and chunk-index checksums |
|
||||||
|
|
||||||
|
Details per release: [`CHANGELOG.md`](CHANGELOG.md).
|
||||||
|
|
||||||
|
### Since v2.7.0 (unreleased, on `main`)
|
||||||
|
|
||||||
|
| PR | Merged | What |
|
||||||
|
|---|---|---|
|
||||||
|
| #3 | 2026-09-23 | pure-Rust deflate (zlib-rs) by default, no C in the core crates' default build (checked in CI), MSRV 1.92 |
|
||||||
|
| #4 | 2026-09-25 | files open in h5py again (every `f32` and every empty dataset clawhdf5 wrote was unreadable by libhdf5); float16 embedding storage |
|
||||||
|
| #5 | 2026-09-25 | `HDF5Memory::search` with `SearchOptions` (source filters, re-ranking, confidence); float16 on by default |
|
||||||
|
| #6 | 2026-09-25 | `clawhdf5-migrate` writes real agent stores; knowledge-graph fix; dated benchmark re-run |
|
||||||
|
| #7 | 2026-09-25 | consolidation benchmark completed (cheaper novelty scoring) |
|
||||||
|
| #8 | 2026-09-25 | Ed25519-signed checkpoints (`HDF5Memory::verify`) |
|
||||||
|
| #9, #10 | 2026-09-25 | OpenClaw and ZeroClaw integration claims withdrawn — neither ever integrated clawhdf5 |
|
||||||
|
| #11 | 2026-09-26 | silent wrong data and libhdf5 interop bugs found by the HDF5 audit fixed |
|
||||||
|
| #12 | 2026-09-26 | reproducible conformance sweep over eight public corpora, nightly CI job ([`CONFORMANCE.md`](CONFORMANCE.md)) |
|
||||||
|
| #13 | 2026-09-26 | reads HDF5 1.6-era layouts, user blocks, virtual datasets, dense attributes, very large groups |
|
||||||
|
| #14 | 2026-09-26 | `h5rs` tools (`ls`, `dump`, `stat`, `diff`, `check`), the browser reader (`clawhdf5-wasm`), libhdf5's header checks, plugin filters (LZF, bitshuffle, bzip2, Blosc), concurrency benchmark |
|
||||||
|
| #15 | 2026-09-26 | fast contiguous and concurrent reads, variable-length data, nested groups and links in the writer, Python bindings |
|
||||||
|
| #16 | 2026-09-26 | chunked full reads faster than an h5py process pool, writer B-trees of any size, Blosc2 (read), 599/697 conformance |
|
||||||
|
| #17 | 2026-09-26 | range reads M0/M1 (indexed name lookups, the `Storage` trait), ZFP (read), in-place editing (`FileEditor`) |
|
||||||
|
| #18 | 2026-09-27 | range reads M2/M3 (`File::open_storage`; `clawhdf5-remote`: HTTP(S), S3, GCS, Azure), in-place editing of every chunk index, shrinking, dense attributes |
|
||||||
|
| #19 | 2026-09-27 | remote files in the browser (`openUrl`, M4), SWMR reader (`File::open_swmr`, M5), Python remote reads and `'r+'` editing |
|
||||||
|
| #20 | 2026-09-28 | benchmarks re-measured: LongMemEval with real MiniLM embeddings, local reads on an idle machine |
|
||||||
|
| #21 | 2026-09-28 | remote files open in a few requests (group lookups down the B-tree, `Storage::hint`), `ObjectHeader::parse` back to its earlier speed, the last conformance mismatches resolved: 602/697 ok, 0 mismatch (the run of 2026-09-28 in [`CONFORMANCE.md`](CONFORMANCE.md) still counts 1 our-error, a corrupt N-Bit file libhdf5's own tests refuse) |
|
||||||
|
|
||||||
|
### Range reads (design: [`docs/design/range-reads.md`](docs/design/range-reads.md))
|
||||||
|
|
||||||
|
- [x] M0 — indexed name lookups (#17)
|
||||||
|
- [x] M1 — metadata parsed through the `Storage` trait (#17)
|
||||||
|
- [x] M2 — raw data through `Storage`, `File::open_storage` (#18)
|
||||||
|
- [x] M3 — `clawhdf5-remote`: HTTP(S) range requests and object stores through a block cache; `h5rs` URLs (#18); Python URLs (#19)
|
||||||
|
- [x] M4 — `openUrl` in the browser, restartable "NeedBytes" cache (#19; fewer round trips in #21)
|
||||||
|
- [x] M5 — reading files a SWMR writer is appending to ([`docs/design/swmr.md`](docs/design/swmr.md), #19)
|
||||||
|
|
||||||
|
### Agent memory (`clawhdf5-agent`)
|
||||||
|
|
||||||
|
Shipped before and during the v2 releases, and kept current since:
|
||||||
|
knowledge graph with entity extraction and resolution; three-tier
|
||||||
|
consolidation with decay; hybrid retrieval (HNSW + BM25, weighted or RRF
|
||||||
|
fusion, re-ranking, confidence rejection, query expansion); temporal index
|
||||||
|
and session DAG; per-save provenance ledger and write-anomaly detection;
|
||||||
|
multi-modal embeddings; WAL with chained CRC32; single-writer locking;
|
||||||
|
signed checkpoints. Retrieval is measured, not claimed: see
|
||||||
|
[`BENCHMARKS.md`](BENCHMARKS.md) ("LongMemEval Results" reports retrieval
|
||||||
|
recall, not QA accuracy; earlier headline numbers that compared different
|
||||||
|
granularities were retracted there).
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
## Track 2: Memory Consolidation Engine
|
## Next
|
||||||
**Status:** 🟢 Phase 1 Complete
|
|
||||||
**Priority:** Critical
|
|
||||||
**Crate:** `clawhdf5-agent`
|
|
||||||
|
|
||||||
- [x] **2.1** Importance scoring — surprise (novelty), correction boost, length scoring with configurable weights
|
Not scheduled; listed roughly by how much they unblock. None has a date.
|
||||||
- [x] **2.2** Three-tier memory model — Working → Episodic → Semantic with bounded capacities
|
|
||||||
- [x] **2.3** Time-decay with reactivation — exponential decay with configurable half-life, access resets timestamp
|
|
||||||
- [x] **2.4** Bounded memory with graceful degradation — evict lowest-decay entries when over capacity
|
|
||||||
- [x] **2.5** Consolidation cycles — promote/evict across tiers based on importance and access thresholds
|
|
||||||
- [x] **2.6** Memory statistics — ConsolidationStats with per-tier counts, eviction/promotion tracking
|
|
||||||
- [x] **2.7** Tests — comprehensive tests for all features
|
|
||||||
|
|
||||||
**Research:** CraniMem (2026), D-MEM (2026), AI Hippocampus survey (2026)
|
### Distribution
|
||||||
|
|
||||||
|
- [ ] **Publish the crates to crates.io.** Nothing is published; the READMEs
|
||||||
|
say to depend on git. Before publishing: no `publish` settings exist
|
||||||
|
(only `clawhdf5-wasm` has `publish = false`).
|
||||||
|
- [ ] **Publish Python wheels to PyPI.** `crates/clawhdf5-py` builds with
|
||||||
|
maturin and is tested in CI, but no wheel is published. The default wheel
|
||||||
|
reads plain `http://` only; `https`/`s3`/`gcs`/`azure` wheels compile C
|
||||||
|
(ring, aws-lc-rs).
|
||||||
|
- [ ] **The Node.js package** (`packages/clawhdf5-node` over
|
||||||
|
`clawhdf5-napi`) has never worked and is not in CI: fix it and add CI, or
|
||||||
|
remove it ([known issue](docs/known-issues.md)).
|
||||||
|
|
||||||
|
### HDF5 features
|
||||||
|
|
||||||
|
- [ ] **SWMR writing.** The reader is done (M5); writing a file while
|
||||||
|
libhdf5 readers follow it is not. Also not covered: remote SWMR (a remote
|
||||||
|
file is pinned at open), `MmapFile`/`LazyFile` SWMR reads, refreshing
|
||||||
|
groups or attributes.
|
||||||
|
- [ ] **MPI collective I/O.** `clawhdf5-io`'s `MpiVol` (`mpi-io`) is
|
||||||
|
root-read + broadcast and gather-to-root writes, not collective MPI-IO
|
||||||
|
(`MPI_File_read_at_all`/`write_at_all`).
|
||||||
|
- [ ] **Paged-metadata single-request reads.** Files written with paged
|
||||||
|
aggregation (`H5Pset_file_space_strategy(PAGE)`, `h5repack -S PAGE`)
|
||||||
|
keep their metadata in a few pages; range reads could fetch those in one
|
||||||
|
request and use the file's page size as the block size. Today the block
|
||||||
|
size is fixed (1 MiB) and only the first block is read ahead
|
||||||
|
(range-reads design, option (c) as a policy).
|
||||||
|
- [ ] **Blosc2 and ZFP encoders.** Both filters are read-only; the other
|
||||||
|
plugin filters (LZF, bitshuffle, bzip2, Blosc 1) read and write.
|
||||||
|
- [ ] **External links and external raw data** are explicit errors, not
|
||||||
|
followed.
|
||||||
|
- [ ] **Virtual datasets:** the "first missing" view and printf gaps other
|
||||||
|
than 0, source-to-virtual type conversion other than a byte swap, nested
|
||||||
|
virtual sources, source files outside the virtual file's directory.
|
||||||
|
- [ ] **Datatypes:** x87 long double and binary128 are refused.
|
||||||
|
- [ ] **Writer:** one attribute or link message over 65 515 bytes in dense
|
||||||
|
storage is an error (huge fractal-heap objects); no option to write
|
||||||
|
files HDF5 1.8 can read.
|
||||||
|
- [ ] **`FileEditor`:** new chunks in implicit indexes, variable-length and
|
||||||
|
reference data, filters it cannot encode (scale-offset, N-Bit, SZIP),
|
||||||
|
some dense-attribute heap layouts, creating or deleting objects and
|
||||||
|
attributes (also from Python `'r+'`), and no journal (a crash mid-edit
|
||||||
|
can leave the file inconsistent). Freed space is reused only within one
|
||||||
|
editor.
|
||||||
|
- [ ] **Selection reads** decode the whole dataset when the selection's
|
||||||
|
bounding box covers more than half of it (a strided `ds[::100]`), and
|
||||||
|
for compact/virtual datasets or a non-default fill value: correct, but
|
||||||
|
more work than needed.
|
||||||
|
- [ ] **Readers:** `LazyFile` and `MmapFile` still need the whole file;
|
||||||
|
the zero-copy methods need the file in memory.
|
||||||
|
|
||||||
|
### Remote and browser
|
||||||
|
|
||||||
|
- [ ] Run the `s3`/`gcs`/`azure` backends against real buckets (only built
|
||||||
|
and URL-parsing-tested so far).
|
||||||
|
- [ ] `h5rs` options for request headers and cache settings.
|
||||||
|
- [ ] Browser limits in [`docs/known-issues.md`](docs/known-issues.md)
|
||||||
|
("`clawhdf5-wasm` (browser) limits"): files of 4 GiB or more (wasm32),
|
||||||
|
compound/reference/opaque datasets, round trips per index level. The
|
||||||
|
package doubled in size with `openUrl`
|
||||||
|
([size table](examples/wasm-viewer/README.md#size)); dropping the
|
||||||
|
function-name section would take a third off the raw size (13% gzipped).
|
||||||
|
|
||||||
|
### Quality
|
||||||
|
|
||||||
|
- [ ] Scheduled fuzz campaigns: the cargo-fuzz targets
|
||||||
|
([`crates/clawhdf5-format/fuzz`](crates/clawhdf5-format/fuzz/README.md),
|
||||||
|
and the agent's WAL target) run only by hand or with
|
||||||
|
`CLAWHDF5_FUZZ_SECONDS`.
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
## Track 3: Hybrid Retrieval Pipeline
|
## Withdrawn
|
||||||
**Status:** 🟢 Phase 1 Complete
|
|
||||||
**Priority:** High
|
|
||||||
**Crate:** `clawhdf5-agent`
|
|
||||||
|
|
||||||
- [x] **3.1** Reciprocal Rank Fusion (RRF) — rrf_hybrid_search() with k=60 constant
|
- **OpenClaw integration** (withdrawn 2026-09-25, PR #9). clawhdf5 was
|
||||||
- [x] **3.2** Multi-factor re-ranking — temporal decay, source authority hierarchy, activation scores (reranker.rs)
|
never an OpenClaw memory plugin; the documented
|
||||||
- [x] **3.3** Low-confidence rejection — min_score threshold, gap filtering, max_results (confidence.rs)
|
`memory.backend = "clawhdf5"` was never valid. The Rust `ClawhdfBackend`
|
||||||
- [x] **3.4** Query expansion — synonyms, acronyms, temporal rewrites, morphological variants, knowledge graph aliases + expanded_search() with RRF merge
|
remains as a library API. [`docs/openclaw.md`](docs/openclaw.md) records
|
||||||
- [x] **3.5** Result explanation — ReRankResult with full score breakdown per factor
|
what a real plugin would need.
|
||||||
- [x] **3.6** Configurable pipeline — ReRankConfig + ConfidenceConfig with tunable weights/thresholds
|
- **ZeroClaw integration** (withdrawn 2026-09-25, PR #10). ZeroClaw has no
|
||||||
- [x] **3.7** Tests + MemX-comparable benchmarks — 5 integration tests (Hit@1≥90%, search<500ms@100K, BM25<200ms@100K, hybrid<50ms@10K, compact<200ms@10K)
|
clawhdf5 backend, and `clawhdf5-migrate`'s SQLite layout is not
|
||||||
|
ZeroClaw's schema.
|
||||||
|
|
||||||
**Research:** MemX (2026), SwiftMem (2026)
|
The old track-by-track tracker this file used to be (agent-memory
|
||||||
|
Tracks 1–8, mid-2026) is in git history (`git log -- ROADMAP.md`).
|
||||||
---
|
|
||||||
|
|
||||||
## Track 4: Temporal Reasoning
|
|
||||||
**Status:** 🟢 Phase 1 Complete
|
|
||||||
**Priority:** High
|
|
||||||
**Crate:** `clawhdf5-agent`
|
|
||||||
|
|
||||||
- [x] **4.1** Temporal index — sorted timestamp index with binary search, insert/remove
|
|
||||||
- [x] **4.2** Time-range queries — range_query, before, after, latest, earliest
|
|
||||||
- [x] **4.3** Session DAG — parent/child linking, chain walking, time-range overlap queries
|
|
||||||
- [x] **4.4** Temporal re-ranking — query hint enum (Latest/Earliest/Around/Between/None) with boost scoring
|
|
||||||
- [x] **4.5** Temporal entity tracking — EntityTimeline with state change history + point-in-time reconstruction
|
|
||||||
- [x] **4.6** Tests — comprehensive tests for all features
|
|
||||||
|
|
||||||
**Research:** MemX temporal gaps (≤43.6% Hit@5), MemoryArena multi-session tasks (2026)
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Track 5: Memory Security & Provenance
|
|
||||||
**Status:** 🟢 Phase 1 Complete
|
|
||||||
**Priority:** Medium-High
|
|
||||||
**Crate:** `clawhdf5-agent`
|
|
||||||
|
|
||||||
- [x] **5.1** Source attribution — MemoryProvenance with source, creator, session, FNV-1a content hash
|
|
||||||
- [x] **5.2** Write anomaly detection — rate limiting, 15 injection patterns, source distribution analysis
|
|
||||||
- [x] **5.3** Source isolation — per-MemorySource sub-stores preventing cross-contamination
|
|
||||||
- [x] **5.4** Memory integrity verification — content hash comparison via verify_integrity()
|
|
||||||
- [x] **5.5** Poisoning resistance — pattern detection for prompt injection attempts
|
|
||||||
- [x] **5.6** Tests — comprehensive tests including adversarial patterns
|
|
||||||
|
|
||||||
**Research:** MemoryGraft (2025), SSGM Framework (2026)
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Track 6: Multi-Modal Memory
|
|
||||||
**Status:** 🟢 Phase 1 Complete
|
|
||||||
**Priority:** Medium
|
|
||||||
**Crate:** `clawhdf5-agent`
|
|
||||||
|
|
||||||
- [x] **6.1** Image embedding storage — ModalEmbedding with model provenance (CLIP, SigLIP, etc.)
|
|
||||||
- [x] **6.2** Audio fingerprints — Audio modality with embedding storage
|
|
||||||
- [x] **6.3** Multi-modal search — search_by_modality (filtered) + search_cross_modal (all embeddings)
|
|
||||||
- [x] **6.4** Observation records — raw perception vs interpretation with confidence scoring
|
|
||||||
- [x] **6.5** Media reference storage — MediaRef with Path/Url/Inline, MIME types, FNV-1a checksums
|
|
||||||
- [x] **6.6** Tests — 35 comprehensive tests
|
|
||||||
|
|
||||||
**Research:** Neuro-Symbolic Memory (2026), RAGdb multi-modal RAG (2025)
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Track 7: OpenClaw Integration
|
|
||||||
**Status:** 🟢 Complete
|
|
||||||
**Priority:** Critical (for adoption)
|
|
||||||
**Crates:** `clawhdf5-agent`, `clawhdf5-napi`
|
|
||||||
|
|
||||||
- [x] **7.1** Memory backend trait — MemoryBackend with search/get/write/ingest/export/stats
|
|
||||||
- [x] **7.2** Hybrid retrieval pipeline — ClawhdfBackend wires RRF → reranker → confidence rejection
|
|
||||||
- [x] **7.3** Markdown import/export — MarkdownParser + MarkdownExporter with line tracking + metadata
|
|
||||||
- [x] **7.4** memory_search tool — backed by full hybrid retrieval pipeline
|
|
||||||
- [x] **7.5** memory_get tool — get() with path + line range support
|
|
||||||
- [x] **7.6** Compaction integration — run_compaction() (decay + compact + WAL flush), run_consolidation() (hippocampal engine), tick_session(), flush_wal()
|
|
||||||
- [x] **7.7** Config surface — `memory.backend = "clawhdf5"` schema documented in docs/openclaw-config.md
|
|
||||||
- [x] **7.8** Documentation + migration guide — docs/migration-guide.md, docs/openclaw-integration.md (architecture, full API reference, code patterns)
|
|
||||||
|
|
||||||
**Node.js bridge:** `clawhdf5-napi` (napi-rs) → `@redclaw/clawhdf5` npm package with full TypeScript types.
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Track 8: Benchmarking & Validation
|
|
||||||
**Status:** 🟢 Complete
|
|
||||||
**Priority:** High
|
|
||||||
**Crates:** `clawhdf5-agent`, `clawhdf5-bench`
|
|
||||||
|
|
||||||
- [x] **8.1** MemoryArena benchmark — 35 queries, 50 sessions, Hit@10=91.4%, MRR=0.547
|
|
||||||
- [x] **8.2** LongMemEval benchmark — 500 queries, session Hit@1=100%, turn Hit@5=84.4% (beats MemX 51.6%), MRR=0.660
|
|
||||||
- [x] **8.3** Latency benchmarks — vector search at 1K/10K/100K, hybrid/RRF, graph traversal, consolidation, temporal
|
|
||||||
- [x] **8.4** Memory footprint — 1.7 KB/record uncompressed, 282 B compressed (6.2x ratio), 100K+ rec/s ingestion
|
|
||||||
- [x] **8.5** Consolidation efficiency — 8.8x search speedup, 90% noise eviction, zero quality loss
|
|
||||||
- [x] **8.6** Cross-platform benchmarks — x86 measured, ARM estimated, cross_platform.sh script
|
|
||||||
- [x] **8.7** Published results in BENCHMARKS.md with ephemeral tier Redis comparison (70-140x faster)
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Implementation Order
|
|
||||||
|
|
||||||
**Phase 1:** ~~Tracks 1, 2, 3 — core memory intelligence~~ 🟢 Complete
|
|
||||||
**Phase 2:** ~~Track 4 (temporal) + Track 5 (security)~~ 🟢 Complete
|
|
||||||
**Phase 3:** ~~Track 6 (multi-modal) + Track 7 (OpenClaw integration)~~ 🟢 Complete
|
|
||||||
**Phase 4:** ~~Track 8 (benchmarking + validation)~~ 🟢 Complete
|
|
||||||
|
|
||||||
All 8 tracks delivered. 1,650+ tests passing, zero clippy warnings.
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## What's Next
|
|
||||||
|
|
||||||
Verified against current repo state on 2026-08-05 (see also `docs/superpowers/plans/` for the filter-codec/format-write/MPI-IO work, now shipped):
|
|
||||||
|
|
||||||
- [ ] TypeScript bridge not wired into CI — `packages/clawhdf5-node/` already has a complete, working napi-rs package (package.json, tsconfig, hand-written TS wrapper matching all 21 `#[napi]` items, Jest test suite, README); it isn't published to npm and has no committed lockfile
|
|
||||||
- [ ] Publish crates to crates.io — no `publish` config anywhere in the workspace yet
|
|
||||||
- [ ] Python wheel distribution via maturin — `crates/clawhdf5-py/pyproject.toml` exists (maturin-buildable locally) but wheels aren't published anywhere
|
|
||||||
- [ ] `chunked_read.rs`/`data_read.rs` full bounds-check audit + scheduled fuzz campaigns (the new `fuzz_dataset_read` target covers the two files' main entry points; a full manual audit of every indexing site is still open) — see Tier 4 below
|
|
||||||
- [ ] WAL per-entry checksum landed as CRC32 (see below); a stronger per-entry format (explicit length prefix, avoiding the read-then-verify restructuring) could still be revisited if profiling shows it matters
|
|
||||||
- [ ] HNSW build parallelism is still narrow (only `prune_connections`); the correctness-sensitive outer insert loop needs its own dedicated design pass before parallelizing
|
|
||||||
|
|
||||||
### Recently closed out (2026-08-05, Tier 3–4 hardening pass)
|
|
||||||
|
|
||||||
- [x] Academic benchmark cross-validation — LongMemEval reproduced against MemX on tank (Ryzen 7 7800X3D): turn-level Hit@5 84.4% vs MemX's 51.6%; recall numbers are deterministic and reproduce exactly across machines. SIMD/Parallelism and Vector Search sections also re-run and dated. See [BENCHMARKS.md § Independent Validation: tank — LongMemEval & Vector Search](BENCHMARKS.md#independent-validation-tank--longmemeval--vector-search-ryzen-7-7800x3d-2026-08-05)
|
|
||||||
- [x] Android JNI (`clawhdf5-android`): validate `embedding_len`/`query_embedding_len` against the handle's configured `embedding_dim` before constructing a slice from a raw pointer
|
|
||||||
- [x] `clawhdf5-py`: bumped pyo3/numpy 0.28 → 0.29, clearing two RUSTSEC advisories
|
|
||||||
- [x] WAL (`clawhdf5-agent`): length-prefix caps (`MAX_WAL_FIELD_LEN`) to reject a corrupted length claim before allocating, then a full per-entry CRC32 trailer (`WAL_VERSION` 2) so a bit-flip stops replay cleanly instead of loading corrupted data; old-format WAL files still read correctly and are migrated on next open
|
|
||||||
- [x] `chunked_read.rs`/`data_read.rs`/`local_heap.rs` bounds-check audit: added `ensure_len` overflow guards, a recursion-depth guard against cyclic B-trees, and a fix for an unguarded compound-datatype byte-offset overrun. Added a new `fuzz_dataset_read` cargo-fuzz target exercising the contiguous/chunked/compact read paths — it found and we fixed 3 real crash bugs (integer-overflow panics) within the first few runs
|
|
||||||
- [x] `clawhdf5-ann`: optional `parallel` feature (rayon) for HNSW's `prune_connections` neighbor-distance computation
|
|
||||||
- [x] `[workspace.dependencies]` added for `tempfile`/`criterion`/`half`/`serde`, fixing a real version skew on `half` (2 vs 2.7)
|
|
||||||
|
|
||||||
### Recently closed out (2026-08-05 hardening pass)
|
|
||||||
|
|
||||||
- [x] CI/CD pipeline — `.gitea/workflows/ci.yml` now runs `scripts/ci-test.sh` (fmt, clippy, tests, no_std check) on push/PR to `main`
|
|
||||||
- [x] Fixed no_std build breakage in `clawhdf5-format` (missing alloc imports, `AtomicU64` unsupported on thumbv7em, `f64::powi` requiring std/libm)
|
|
||||||
- [x] Fixed version skew: `clawhdf5-py` (pyproject.toml) and `packages/clawhdf5-node` (package.json) were both behind the actual crate version
|
|
||||||
|
|
||||||
### Recently closed out (2026-08-03 cleanup pass)
|
|
||||||
|
|
||||||
- [x] Removed `clawhdf5-types` — it was an empty 1-line stub crate; shared type definitions already live in `clawhdf5-format`, so CLAUDE.md and the workspace manifest were corrected instead of filling it in
|
|
||||||
- [x] Superblock v4 (page-buffer mode) read/write — the only unimplemented task from `docs/superpowers/plans/2026-06-29-format-write-extensions.md`; now done (`Superblock::parse_v4`/`serialize`, `FileWriter::with_page_size`)
|
|
||||||
- [x] Reconciled the three `docs/superpowers/plans/*.md` docs against actual shipped code — they were pre-work plans for `d6c4d4f` (2026-06-30), committed to git late; checkboxes now reflect reality
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
_Last updated: 2026-08-05_
|
|
||||||
|
|||||||
@@ -29,7 +29,8 @@
|
|||||||
# - Use wasm-pack with a custom bench harness
|
# - Use wasm-pack with a custom bench harness
|
||||||
# - Replace std::time::Instant with web_sys::Performance::now()
|
# - Replace std::time::Instant with web_sys::Performance::now()
|
||||||
# - Replace TempDir/HDF5 I/O with an in-memory backend (separate effort)
|
# - Replace TempDir/HDF5 I/O with an in-memory backend (separate effort)
|
||||||
# See ROADMAP.md §WASM for the full scope.
|
# Browser reads are tested (not benchmarked) by
|
||||||
|
# examples/wasm-viewer/test/run.sh; see examples/wasm-viewer/README.md.
|
||||||
|
|
||||||
set -euo pipefail
|
set -euo pipefail
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,3 @@
|
|||||||
|
/.cache/
|
||||||
|
# pin the probe's dependencies (the workspace lock is not committed)
|
||||||
|
!/probe/Cargo.lock
|
||||||
@@ -0,0 +1,70 @@
|
|||||||
|
# Conformance sweep
|
||||||
|
|
||||||
|
Reads every HDF5 file of eight public corpora with clawhdf5 and with
|
||||||
|
h5py/libhdf5, compares the two readings object by object, and writes
|
||||||
|
[`CONFORMANCE.md`](../CONFORMANCE.md).
|
||||||
|
|
||||||
|
```sh
|
||||||
|
CLAWHDF5_PYTHON=/path/to/venv/bin/python conformance/run.sh # ~30 s once the corpus is cached
|
||||||
|
conformance/run.sh --no-fetch # use the cached corpus as is
|
||||||
|
conformance/run.sh --update-baseline # after an intended change in results
|
||||||
|
```
|
||||||
|
|
||||||
|
Latest result (tank, 2026-09-28 04:29 UTC, `conformance/run.sh --no-fetch
|
||||||
|
--update-baseline`): 602 of 697 files ok, 1 our-error, 0 mismatch, 2
|
||||||
|
ref-bug, 92 h5py-cannot-read, and no panic, hang, crash or out-of-memory.
|
||||||
|
The our-error file is `bad_nbit_parms_walk.h5`, which flips between ref-bug
|
||||||
|
and our-error from run to run (see `docs/known-issues.md`). The report with every file is
|
||||||
|
[`CONFORMANCE.md`](../CONFORMANCE.md).
|
||||||
|
|
||||||
|
## Classes
|
||||||
|
|
||||||
|
`compare.py` puts each file in one class:
|
||||||
|
|
||||||
|
| class | meaning |
|
||||||
|
|---|---|
|
||||||
|
| **ok** | clawhdf5 and h5py read the same objects with the same values |
|
||||||
|
| **our-error** | h5py reads something clawhdf5 refuses |
|
||||||
|
| **mismatch** | both read it, with different values or structure |
|
||||||
|
| **h5py-cannot-read** | h5py (libhdf5) cannot read the file; not compared |
|
||||||
|
| **ref-bug** | h5py reads an object clawhdf5 refuses, but only through a libhdf5 over-read: `ref_bugs.py` re-reads it in six processes with different heaps (import order, `MALLOC_PERTURB_`) and its values change. The file is ref-bug only while that is confirmed in the same run; if the values become stable it counts as our-error again |
|
||||||
|
| **panic / hang / crash / oom** | a clawhdf5 failure under the timeout and address-space limit; the gate fails on any |
|
||||||
|
|
||||||
|
Where h5py itself returns wrong values through a known h5py bug (the
|
||||||
|
big-endian variable-length bug: elements returned with the file's bytes
|
||||||
|
under a little-endian dtype), `ref.py` checks that the installed h5py has
|
||||||
|
the bug, corrects the values before hashing and marks them `ref_fix`, so
|
||||||
|
those objects are still compared. The evidence for the three remaining
|
||||||
|
non-ok files (ref-bug or, for one, our-error) is under "Conformance: the last non-ok files" in
|
||||||
|
[`docs/known-issues.md`](../docs/known-issues.md).
|
||||||
|
|
||||||
|
Needs Rust, `git`, `h5dump` (Debian/Ubuntu `hdf5-tools`), `libaec` (for the
|
||||||
|
probe's `szip` feature; `libaec-dev`), and a Python with the packages in
|
||||||
|
`requirements.txt`. The first run downloads about 450 MB of sparse checkouts.
|
||||||
|
|
||||||
|
| file | role |
|
||||||
|
|---|---|
|
||||||
|
| `corpus.txt` | the corpora: git URL, pinned commit, swept root, sparse-checkout patterns |
|
||||||
|
| `fetch-corpus.sh` | shallow, sparse, blob-filtered checkout of each pinned commit into `.cache/src/` (gitignored); no-op when already there |
|
||||||
|
| `list_files.py` | which files are probed (HDF5/netCDF-4 extensions minus netCDF classic, plus the CVE reproducers) |
|
||||||
|
| `probe/` | the clawhdf5 side: a standalone crate (outside the workspace, so `cargo test --workspace` never builds it) that walks a file with `clawhdf5-format` and prints canonical JSON |
|
||||||
|
| `ref.py` | the h5py side: the same JSON from h5py (values corrected for a known h5py bug are marked `ref_fix`) |
|
||||||
|
| `ref_bugs.py` | re-reads the objects h5py reads only through a libhdf5 bug in six differently-set-up processes; an object whose values change is confirmed as a libhdf5 over-read |
|
||||||
|
| `test_ref.py` | tests of `ref.py`'s correction and `ref_bugs.py`'s confirmation (`python conformance/test_ref.py`) |
|
||||||
|
| `run_one.sh` | runs both sides on one file (and `h5dump` on the CVE corpus) under a timeout and an address-space limit |
|
||||||
|
| `compare.py` | classifies each file (ok / our-error / mismatch / h5py-cannot-read / ref-bug / panic / hang / crash / oom) and groups root causes |
|
||||||
|
| `report.py` | writes `CONFORMANCE.md` |
|
||||||
|
| `check.py` | the gate: fails on any panic/hang/crash/oom, on an ok count below `baseline.json`, or on a baseline-ok file that is no longer ok |
|
||||||
|
| `baseline.json` | the ok files the gate holds the line on |
|
||||||
|
| `requirements.txt` | pinned h5py / numpy / hdf5plugin / netCDF4 |
|
||||||
|
|
||||||
|
Results for every file (both sides' JSON and stderr, `results.csv`,
|
||||||
|
`results.json`, `summary.md`) are left in `.cache/results/`.
|
||||||
|
|
||||||
|
The nightly job is `.gitea/workflows/conformance.yml`; it prints the report
|
||||||
|
into the job log.
|
||||||
|
|
||||||
|
The canonical value encoding both sides hash is documented at the top of
|
||||||
|
`probe/src/main.rs`. Values are compared as libhdf5 presents them: a float
|
||||||
|
with a non-IEEE bit layout (N-Bit) or an integer with a bit offset is compared
|
||||||
|
as the converted number, not as raw file bytes.
|
||||||
@@ -0,0 +1,648 @@
|
|||||||
|
{
|
||||||
|
"comment": "conformance/run.sh fails if the ok count drops below `ok` or a file in `ok_files` stops being ok. Regenerate with `conformance/run.sh --update-baseline` after an intended change.",
|
||||||
|
"commit": "bf5a163dcf7fe28d651ada6545d8136ffeffc825",
|
||||||
|
"date": "2026-09-28 04:29 UTC",
|
||||||
|
"reference": "h5py 3.16.0 / HDF5 2.0.0",
|
||||||
|
"files": 697,
|
||||||
|
"ok": 602,
|
||||||
|
"counts": {
|
||||||
|
"h5py-cannot-read": 92,
|
||||||
|
"ok": 602,
|
||||||
|
"our-error": 1,
|
||||||
|
"ref-bug": 2
|
||||||
|
},
|
||||||
|
"per_corpus": {
|
||||||
|
"NCAS-CMS_pyfive": {
|
||||||
|
"ok": 33
|
||||||
|
},
|
||||||
|
"cve_hdf5": {
|
||||||
|
"h5py-cannot-read": 32,
|
||||||
|
"ok": 113,
|
||||||
|
"ref-bug": 2
|
||||||
|
},
|
||||||
|
"h5py_data": {
|
||||||
|
"ok": 4
|
||||||
|
},
|
||||||
|
"hdf5": {
|
||||||
|
"h5py-cannot-read": 60,
|
||||||
|
"ok": 405,
|
||||||
|
"our-error": 1
|
||||||
|
},
|
||||||
|
"netcdf-c": {
|
||||||
|
"ok": 20
|
||||||
|
},
|
||||||
|
"netcdf4-python": {
|
||||||
|
"ok": 18
|
||||||
|
},
|
||||||
|
"usnistgov_h5wasm": {
|
||||||
|
"ok": 5
|
||||||
|
},
|
||||||
|
"xarray-data": {
|
||||||
|
"ok": 4
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"ok_files": [
|
||||||
|
"NCAS-CMS_pyfive/tests/compact.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/attr_datatypes.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/btreev2.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/chunked.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/cmip_bad_eg.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/compressed.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/compressed_v1.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/dataset_datatypes.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/dataset_multidim.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/dim_scales.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/earliest.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/enum_h5variable.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/enum_variable.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/enum_variable.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/enums_from_netcdf.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/fillvalue_earliest.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/fillvalue_latest.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/filter_pipeline_v2.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/fletcher32.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/fractal_heap_no_mci_rlat.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/groups.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/h5netcdf_test.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/issue23_A.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/issue23_A_contiguous.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/issue23_B.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/latest.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/netcdf4_classic.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/new_style_groups.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/noy_AERmonZ_UKESM1-0-LL_piControl_r1i1p1f2_gnz_200001-200012.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/references.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/resizable.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/opaque_datetime.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/opaque_fixed.hdf5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2016-4330.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2016-4331.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2016-4332-mtime-new.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2016-4332-mtime.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2016-4333.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2017-17505.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2017-17506.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2017-17507.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2017-17508.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2017-17509.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-11202.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-11203.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-11204.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-11205.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-11206-new.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-11206-old.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-11207.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-13867.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-13868.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-13869.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-13870.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-13871.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-13872.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-13873.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-13875.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-14031.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-14033.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-14034.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-14035.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-14460.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-15671.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-15672.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-16438.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-17233.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-17234.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-17237.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-17432.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-17434.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-17435.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-17437.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-17438",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-17439",
|
||||||
|
"cve_hdf5/cvefiles/cve-2019-8396.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2019-8397.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2019-8398.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2019-9151.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2019-9152.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2020-10811.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2020-18232.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2020-18494.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2021-36977.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2021-37501.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2021-45829.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2021-45833.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2021-46243.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2021-46244.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29157.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29158.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29159.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29160.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29161.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29162.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29163.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29164.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29165.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29166.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32605.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32606.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32607-1.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32607-2.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32608.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32610.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32611.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32612.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32613.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32614.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32615.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32616.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32617.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32618.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32619.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32620.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32621.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32622.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32623.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32624.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-33873.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-33874.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-33875.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-33876.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-33877.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-2309.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-2310.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-2924.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-2925.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-44905.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-6269-1.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-6269-2.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-6269-3.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-6269-4.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-6516.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-6857.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-7067.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2026-26200.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2026-34734.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2026-92627.h5",
|
||||||
|
"cve_hdf5/cvefiles/unknown-1.h5",
|
||||||
|
"cve_hdf5/fuzzerfiles/gh-4431-poc-03.h5",
|
||||||
|
"cve_hdf5/fuzzerfiles/gh-4432-poc-05.h5",
|
||||||
|
"cve_hdf5/fuzzerfiles/gh-4433-poc-08.h5",
|
||||||
|
"cve_hdf5/fuzzerfiles/gh-4435-poc-10.h5",
|
||||||
|
"cve_hdf5/fuzzerfiles/gh_2649_flawed.h5",
|
||||||
|
"cve_hdf5/fuzzerfiles/gh_2649_plain_model.h5",
|
||||||
|
"h5py_data/compound-dtype-complex.h5",
|
||||||
|
"h5py_data/vlen_string_dset.h5",
|
||||||
|
"h5py_data/vlen_string_dset_utc.h5",
|
||||||
|
"h5py_data/vlen_string_s390x.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_bitgroom.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_blosc.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_blosc2.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_bshuf.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_bzip2.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_granularbr.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_jpeg.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_lz4.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_lzf.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_zfp.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_zstd.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5G/16/h5ex_g_iterate.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5G/16/h5ex_g_traverse.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5G/h5ex_g_iterate.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5G/h5ex_g_traverse.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5G/h5ex_g_visit.h5",
|
||||||
|
"hdf5/HDF5Examples/FORTRAN/H5G/h5ex_g_iterate.h5",
|
||||||
|
"hdf5/HDF5Examples/FORTRAN/H5G/h5ex_g_traverse.h5",
|
||||||
|
"hdf5/HDF5Examples/FORTRAN/H5G/h5ex_g_visit.h5",
|
||||||
|
"hdf5/HDF5Examples/JAVA/H5G/h5ex_g_iterate.h5",
|
||||||
|
"hdf5/HDF5Examples/JAVA/H5G/h5ex_g_visit.h5",
|
||||||
|
"hdf5/HDF5Examples/JAVA/compat/H5G/110/h5ex_g_iterate.h5",
|
||||||
|
"hdf5/HDF5Examples/JAVA/compat/H5G/110/h5ex_g_visit.h5",
|
||||||
|
"hdf5/HDF5Examples/JAVA/compat/H5G/h5ex_g_iterate.h5",
|
||||||
|
"hdf5/HDF5Examples/JAVA/compat/H5G/h5ex_g_visit.h5",
|
||||||
|
"hdf5/c++/test/th5s.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_ds_be.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_ds_be_new_ref-32bit.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_ds_be_new_ref.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_ds_le.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_ds_le_new_ref.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_ld.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_table_be.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_table_cray.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_table_le.h5",
|
||||||
|
"hdf5/test/testfiles/aggr.h5",
|
||||||
|
"hdf5/test/testfiles/bad_chunk_ndims.h5",
|
||||||
|
"hdf5/test/testfiles/bad_compound.h5",
|
||||||
|
"hdf5/test/testfiles/bad_offset.h5",
|
||||||
|
"hdf5/test/testfiles/be_data.h5",
|
||||||
|
"hdf5/test/testfiles/be_extlink1.h5",
|
||||||
|
"hdf5/test/testfiles/be_extlink2.h5",
|
||||||
|
"hdf5/test/testfiles/btree_idx_1_6.h5",
|
||||||
|
"hdf5/test/testfiles/btree_idx_1_8.h5",
|
||||||
|
"hdf5/test/testfiles/charsets.h5",
|
||||||
|
"hdf5/test/testfiles/corrupt_stab_msg.h5",
|
||||||
|
"hdf5/test/testfiles/deflate.h5",
|
||||||
|
"hdf5/test/testfiles/file_image_core_test.h5",
|
||||||
|
"hdf5/test/testfiles/filespace_1_6.h5",
|
||||||
|
"hdf5/test/testfiles/filespace_1_8.h5",
|
||||||
|
"hdf5/test/testfiles/fill18.h5",
|
||||||
|
"hdf5/test/testfiles/fill_old.h5",
|
||||||
|
"hdf5/test/testfiles/filter_error.h5",
|
||||||
|
"hdf5/test/testfiles/fsm_aggr_nopersist.h5",
|
||||||
|
"hdf5/test/testfiles/fsm_aggr_persist.h5",
|
||||||
|
"hdf5/test/testfiles/group_old.h5",
|
||||||
|
"hdf5/test/testfiles/h5fc_ext1_f.h5",
|
||||||
|
"hdf5/test/testfiles/h5fc_ext1_i.h5",
|
||||||
|
"hdf5/test/testfiles/h5fc_ext2_if.h5",
|
||||||
|
"hdf5/test/testfiles/h5fc_ext2_sf.h5",
|
||||||
|
"hdf5/test/testfiles/h5fc_ext3_isf.h5",
|
||||||
|
"hdf5/test/testfiles/h5fc_ext_none.h5",
|
||||||
|
"hdf5/test/testfiles/le_data.h5",
|
||||||
|
"hdf5/test/testfiles/le_extlink1.h5",
|
||||||
|
"hdf5/test/testfiles/le_extlink2.h5",
|
||||||
|
"hdf5/test/testfiles/memleak_H5O_dtype_decode_helper_H5Odtype.h5",
|
||||||
|
"hdf5/test/testfiles/mergemsg.h5",
|
||||||
|
"hdf5/test/testfiles/noencoder.h5",
|
||||||
|
"hdf5/test/testfiles/none.h5",
|
||||||
|
"hdf5/test/testfiles/paged_nopersist.h5",
|
||||||
|
"hdf5/test/testfiles/paged_persist.h5",
|
||||||
|
"hdf5/test/testfiles/specmetaread.h5",
|
||||||
|
"hdf5/test/testfiles/tarrold.h5",
|
||||||
|
"hdf5/test/testfiles/tbad_msg_count.h5",
|
||||||
|
"hdf5/test/testfiles/tbogus.h5",
|
||||||
|
"hdf5/test/testfiles/test_filters_be.h5",
|
||||||
|
"hdf5/test/testfiles/test_filters_le.h5",
|
||||||
|
"hdf5/test/testfiles/th5s.h5",
|
||||||
|
"hdf5/test/testfiles/tlayouto.h5",
|
||||||
|
"hdf5/test/testfiles/tmisc38a.h5",
|
||||||
|
"hdf5/test/testfiles/tmisc38b.h5",
|
||||||
|
"hdf5/test/testfiles/tmtimen.h5",
|
||||||
|
"hdf5/test/testfiles/tmtimeo.h5",
|
||||||
|
"hdf5/test/testfiles/tnullspace.h5",
|
||||||
|
"hdf5/test/testfiles/tsizeslheap.h5",
|
||||||
|
"hdf5/tools/test/testfiles/bigendian/tall.h5",
|
||||||
|
"hdf5/tools/test/testfiles/bigendian/tdset2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/binfp64.h5",
|
||||||
|
"hdf5/tools/test/testfiles/binin16.h5",
|
||||||
|
"hdf5/tools/test/testfiles/binin32.h5",
|
||||||
|
"hdf5/tools/test/testfiles/binin8.h5",
|
||||||
|
"hdf5/tools/test/testfiles/binin8w.h5",
|
||||||
|
"hdf5/tools/test/testfiles/binuin16.h5",
|
||||||
|
"hdf5/tools/test/testfiles/binuin32.h5",
|
||||||
|
"hdf5/tools/test/testfiles/bounds_latest_latest.h5",
|
||||||
|
"hdf5/tools/test/testfiles/charsets.h5",
|
||||||
|
"hdf5/tools/test/testfiles/compounds_array_vlen1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/compounds_array_vlen2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/err_attr_dspace.h5",
|
||||||
|
"hdf5/tools/test/testfiles/file_space.h5",
|
||||||
|
"hdf5/tools/test/testfiles/filter_fail.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5clear_fsm_persist_equal.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5clear_fsm_persist_less.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5clear_fsm_persist_noclose.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5clear_fsm_persist_user_equal.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5clear_fsm_persist_user_less.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5clear_mdc_image.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5clear_sec2_v0.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5clear_sec2_v2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5copy_extlinks_src.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5copy_extlinks_trg.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5copy_ref.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5copytst.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5copytst_new.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_attr1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_attr2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_attr3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_attr_v_level1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_attr_v_level2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_basic1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_basic2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_comp_vl_strs.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_danglelinks1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_danglelinks2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_dset1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_dset2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_dset3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_dset_zero_dim_size1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_dset_zero_dim_size2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_dtypes.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_empty.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_enum_invalid_values.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_eps1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_eps2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_exclude1-1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_exclude1-2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_exclude2-1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_exclude2-2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_exclude3-1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_exclude3-2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_ext2softlink_src.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_ext2softlink_trg.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_extlink_src.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_extlink_trg.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_grp_recurse1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_grp_recurse2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_grp_recurse_ext1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_grp_recurse_ext2-1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_grp_recurse_ext2-2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_grp_recurse_ext2-3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_hyper1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_hyper2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_linked_softlink.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_links.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_onion_dset_1d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_onion_dset_ext.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_onion_objs.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_softlinks.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_strings1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_strings2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_types.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_edge_v3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_err_level.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_ext1_f.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_ext1_i.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_ext1_s.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_ext2_if.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_ext2_is.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_ext2_sf.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_ext3_isf.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_ext_none.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_non_v3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_CVE-2018-14460.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_CVE-2018-17432.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_aggr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_attr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_attr_refs.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_deflate.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_early.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_ext.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_f32le.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_f32le_ex.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_fill.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_filters.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_fletcher.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_fsm_aggr_nopersist.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_fsm_aggr_persist.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_hlink.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_int32le_1d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_int32le_1d_ex.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_int32le_2d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_int32le_2d_ex.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_int32le_3d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_int32le_3d_ex.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_layout.UD.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_layout.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_layout2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_layout3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_layouto.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_named_dtypes.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_nbit.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_nested_8bit_enum.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_nested_8bit_enum_deflated.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_none.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_objs.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_paged_nopersist.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_paged_persist.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_refs.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_shuffle.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_soffset.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_szip.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_uint8be.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_uint8be_ex.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5stat_err_old_fill.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5stat_err_old_layout.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5stat_err_refcount.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5stat_filters.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5stat_idx.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5stat_newgrat.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5stat_threshold.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5stat_tsohm.h5",
|
||||||
|
"hdf5/tools/test/testfiles/mod_h5clear_mdc_image.h5",
|
||||||
|
"hdf5/tools/test/testfiles/non_comparables1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/non_comparables2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/old_h5fc_ext1_f.h5",
|
||||||
|
"hdf5/tools/test/testfiles/old_h5fc_ext1_i.h5",
|
||||||
|
"hdf5/tools/test/testfiles/old_h5fc_ext1_s.h5",
|
||||||
|
"hdf5/tools/test/testfiles/old_h5fc_ext2_if.h5",
|
||||||
|
"hdf5/tools/test/testfiles/old_h5fc_ext2_is.h5",
|
||||||
|
"hdf5/tools/test/testfiles/old_h5fc_ext2_sf.h5",
|
||||||
|
"hdf5/tools/test/testfiles/old_h5fc_ext3_isf.h5",
|
||||||
|
"hdf5/tools/test/testfiles/old_h5fc_ext_none.h5",
|
||||||
|
"hdf5/tools/test/testfiles/packedbits.h5",
|
||||||
|
"hdf5/tools/test/testfiles/t128bit_float.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tCVE-2021-37501_attr_decode.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tCVE_2018_11206_fill_new.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tCVE_2018_11206_fill_old.h5",
|
||||||
|
"hdf5/tools/test/testfiles/taindices.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tall.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tarray1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tarray1_big.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tarray2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tarray3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tarray4.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tarray5.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tarray8.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tattr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tattr2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tattr4_be.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tattrintsize.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tattrreg.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tbfloat16.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tbfloat16_be.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tbigdims.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tbinary.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tbitnopaque.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tchar.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tcmpdattrintsize.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tcmpdintarray.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tcmpdints.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tcmpdintsize.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tcomplex.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tcomplex_be.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tcompound.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tcompound_complex.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tcompound_complex2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tdatareg.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tdset.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tdset2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tdset_idx.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tempty.h5",
|
||||||
|
"hdf5/tools/test/testfiles/textlink.h5",
|
||||||
|
"hdf5/tools/test/testfiles/textlinkfar.h5",
|
||||||
|
"hdf5/tools/test/testfiles/textlinksrc.h5",
|
||||||
|
"hdf5/tools/test/testfiles/textlinktar.h5",
|
||||||
|
"hdf5/tools/test/testfiles/textpfe.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfcontents1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfcontents2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfilters.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfloat16.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfloat16_be.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfloat4.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfloat6.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfloat8.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfloatsattrs.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfpformat.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfvalues.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tgroup.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tgrp_comments.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tgrpnullspace.h5",
|
||||||
|
"hdf5/tools/test/testfiles/thlink.h5",
|
||||||
|
"hdf5/tools/test/testfiles/thyperslab.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tintascii.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tints4dims.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tintsattrs.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tintsnodata.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tlarge_objname.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tldouble.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tldouble_scalar.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tlonglinks.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tloop.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tnamed_dtype_attr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tnestedcmpddt.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tnestedcomp.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tno-subset.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tnullspace.h5",
|
||||||
|
"hdf5/tools/test/testfiles/torderattr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tordergr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_attr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_compat.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_ext1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_ext2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_grp.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_obj.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_obj_del.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_param.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_reg.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_reg_1d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tsaf.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tscalarattrintsize.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tscalarintattrsize.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tscalarintsize.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tscalarstring.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tslink.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tsoftlinks.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tst_onion_dset_1d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tst_onion_dset_ext.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tst_onion_objs.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tstr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tstr2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tstr3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tudfilter.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tudfilter2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tudlink.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tvldtypes1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tvldtypes2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tvldtypes3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tvldtypes4.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tvldtypes5.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tvlenstr_array.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tvlstr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tvms.h5",
|
||||||
|
"hdf5/tools/test/testfiles/twithub.h5",
|
||||||
|
"hdf5/tools/test/testfiles/twithub513.h5",
|
||||||
|
"hdf5/tools/test/testfiles/txtfp32.h5",
|
||||||
|
"hdf5/tools/test/testfiles/txtfp64.h5",
|
||||||
|
"hdf5/tools/test/testfiles/txtin16.h5",
|
||||||
|
"hdf5/tools/test/testfiles/txtin32.h5",
|
||||||
|
"hdf5/tools/test/testfiles/txtin8.h5",
|
||||||
|
"hdf5/tools/test/testfiles/txtstr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/txtuin16.h5",
|
||||||
|
"hdf5/tools/test/testfiles/txtuin32.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/1_a.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/1_b.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/1_c.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/1_d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/1_e.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/1_f.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/1_vds.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/2_a.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/2_b.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/2_c.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/2_d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/2_e.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/2_vds.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/3_1_vds.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/3_2_vds.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/4_0.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/4_1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/4_2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/4_vds.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/5_a.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/5_b.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/5_c.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/5_vds.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/a.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/b.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/c.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/f-0.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/f-3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/vds-eiger.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/vds-percival-unlim-maxmin.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tbitfields.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tcompound2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tdset2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tenum.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/test35.nc",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tloop2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tmany.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tname-amp.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tname-apos.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tname-gt.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tname-lt.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tname-quot.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tname-sp.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tnodata.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tobjref.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/topaque.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tref-escapes-at.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tref-escapes.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tref.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tstring-at.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tstring.h5",
|
||||||
|
"hdf5/tools/test/testfiles/zerodim.h5",
|
||||||
|
"netcdf-c/h5_test/ref_tst_h_compounds.h5",
|
||||||
|
"netcdf-c/h5_test/ref_tst_h_compounds2.h5",
|
||||||
|
"netcdf-c/nc_test4/ref_hdf5_compat1.nc",
|
||||||
|
"netcdf-c/nc_test4/ref_hdf5_compat2.nc",
|
||||||
|
"netcdf-c/nc_test4/ref_hdf5_compat3.nc",
|
||||||
|
"netcdf-c/nc_test4/ref_szip.h5",
|
||||||
|
"netcdf-c/nc_test4/ref_tst_compounds.nc",
|
||||||
|
"netcdf-c/nc_test4/ref_tst_dims.nc",
|
||||||
|
"netcdf-c/nc_test4/ref_tst_interops4.nc",
|
||||||
|
"netcdf-c/nc_test4/ref_tst_xplatform2_1.nc",
|
||||||
|
"netcdf-c/nc_test4/ref_tst_xplatform2_2.nc",
|
||||||
|
"netcdf-c/nc_test4/tdset.h5",
|
||||||
|
"netcdf-c/ncdump/ref_nc_test_netcdf4_4_0.nc",
|
||||||
|
"netcdf-c/ncdump/ref_no_ncproperty.nc",
|
||||||
|
"netcdf-c/ncdump/ref_provenance_v1.nc",
|
||||||
|
"netcdf-c/ncdump/ref_test_corrupt_magic.nc",
|
||||||
|
"netcdf-c/ncdump/ref_tst_compounds2.nc",
|
||||||
|
"netcdf-c/ncdump/ref_tst_compounds3.nc",
|
||||||
|
"netcdf-c/ncdump/ref_tst_compounds4.nc",
|
||||||
|
"netcdf-c/ncdump/ref_tst_irish_rover.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2000.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2001.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2002.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2003.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2004.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2005.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2006.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2007.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2008.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2009.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2010.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2011.nc",
|
||||||
|
"netcdf4-python/examples/data/rtofs_glo_3dz_f006_6hrly_reg3.nc",
|
||||||
|
"netcdf4-python/test/20171025_2056.Cloud_Top_Height.nc",
|
||||||
|
"netcdf4-python/test/issue1152.nc",
|
||||||
|
"netcdf4-python/test/issue671.nc",
|
||||||
|
"netcdf4-python/test/issue672.nc",
|
||||||
|
"netcdf4-python/test/test_gold.nc",
|
||||||
|
"usnistgov_h5wasm/test/array.h5",
|
||||||
|
"usnistgov_h5wasm/test/compressed.h5",
|
||||||
|
"usnistgov_h5wasm/test/empty.h5",
|
||||||
|
"usnistgov_h5wasm/test/float16.h5",
|
||||||
|
"usnistgov_h5wasm/test/vlen.h5",
|
||||||
|
"xarray-data/ROMS_example.nc",
|
||||||
|
"xarray-data/basin_mask.nc",
|
||||||
|
"xarray-data/imerghh_730.hdf5",
|
||||||
|
"xarray-data/precipitation.nc4"
|
||||||
|
]
|
||||||
|
}
|
||||||
Executable
+88
@@ -0,0 +1,88 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""check.py <results_dir> <baseline.json> [--update]
|
||||||
|
|
||||||
|
The conformance gate. Fails (exit 1) when
|
||||||
|
* clawhdf5 panicked, hung, crashed or ran out of memory on any file, or
|
||||||
|
* the ok count fell below the baseline's, or
|
||||||
|
* a file the baseline lists as ok is no longer ok (even if another file
|
||||||
|
became ok and the total held).
|
||||||
|
New ok files are reported so the baseline can be raised (--update rewrites it
|
||||||
|
from the results).
|
||||||
|
"""
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
|
||||||
|
FATAL = ("panic", "hang", "crash", "oom")
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
args = [a for a in sys.argv[1:] if not a.startswith("--")]
|
||||||
|
update = "--update" in sys.argv
|
||||||
|
res_dir, base_path = args
|
||||||
|
res = json.load(open(os.path.join(res_dir, "results.json")))
|
||||||
|
rows = res["rows"]
|
||||||
|
counts = {}
|
||||||
|
per_corpus = {}
|
||||||
|
for r in rows:
|
||||||
|
counts[r["class"]] = counts.get(r["class"], 0) + 1
|
||||||
|
pc = per_corpus.setdefault(r["corpus"], {})
|
||||||
|
pc[r["class"]] = pc.get(r["class"], 0) + 1
|
||||||
|
ok_files = sorted(r["file"] for r in rows if r["class"] == "ok")
|
||||||
|
|
||||||
|
if update:
|
||||||
|
meta = {}
|
||||||
|
mp = os.path.join(res_dir, "report-meta.json")
|
||||||
|
if os.path.exists(mp):
|
||||||
|
meta = json.load(open(mp))
|
||||||
|
base = {
|
||||||
|
"comment": "conformance/run.sh fails if the ok count drops below `ok` or a file in `ok_files` stops being ok. "
|
||||||
|
"Regenerate with `conformance/run.sh --update-baseline` after an intended change.",
|
||||||
|
"commit": meta.get("commit", ""),
|
||||||
|
"date": meta.get("date", ""),
|
||||||
|
"reference": meta.get("reference", ""),
|
||||||
|
"files": len(rows),
|
||||||
|
"ok": len(ok_files),
|
||||||
|
"counts": dict(sorted(counts.items())),
|
||||||
|
"per_corpus": {k: dict(sorted(v.items())) for k, v in sorted(per_corpus.items())},
|
||||||
|
"ok_files": ok_files,
|
||||||
|
}
|
||||||
|
with open(base_path, "w") as fh:
|
||||||
|
json.dump(base, fh, indent=1)
|
||||||
|
fh.write("\n")
|
||||||
|
print(f"baseline updated: {len(ok_files)} ok of {len(rows)} files -> {base_path}")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
base = json.load(open(base_path))
|
||||||
|
failures = []
|
||||||
|
fatal = [r for r in rows if r["class"] in FATAL]
|
||||||
|
for r in fatal:
|
||||||
|
failures.append(f"{r['class']}: {r['file']}: {r['ours_detail'][:200]}")
|
||||||
|
if len(ok_files) < base["ok"]:
|
||||||
|
failures.append(f"ok count dropped: {len(ok_files)} < baseline {base['ok']}")
|
||||||
|
now_ok = set(ok_files)
|
||||||
|
by_file = {r["file"]: r for r in rows}
|
||||||
|
for f in base["ok_files"]:
|
||||||
|
if f not in now_ok:
|
||||||
|
r = by_file.get(f)
|
||||||
|
why = f"now {r['class']}: {(r['ours_detail'] or r['first_issue'])[:200]}" if r else "no longer in the corpus"
|
||||||
|
failures.append(f"regressed: {f}: {why}")
|
||||||
|
gained = sorted(now_ok - set(base["ok_files"]))
|
||||||
|
|
||||||
|
print(f"conformance: {len(ok_files)} ok of {len(rows)} files (baseline {base['ok']} of {base['files']}); "
|
||||||
|
+ ", ".join(f"{k} {v}" for k, v in sorted(counts.items())))
|
||||||
|
if gained:
|
||||||
|
print(f"{len(gained)} file(s) newly ok — raise the baseline with `conformance/run.sh --update-baseline`:")
|
||||||
|
for f in gained:
|
||||||
|
print(f" + {f}")
|
||||||
|
if failures:
|
||||||
|
print(f"CONFORMANCE GATE FAILED ({len(failures)}):")
|
||||||
|
for f in failures:
|
||||||
|
print(f" - {f}")
|
||||||
|
return 1
|
||||||
|
print("conformance gate passed")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
sys.exit(main())
|
||||||
Executable
+322
@@ -0,0 +1,322 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""compare.py <results_dir>: classify each file and group failures by root cause.
|
||||||
|
|
||||||
|
Writes <results_dir>/results.csv, results.json and summary.md.
|
||||||
|
File classes (first match wins):
|
||||||
|
hang, oom, crash, panic ours: timeout / allocation failure / signal / any panic (caught or not)
|
||||||
|
h5py-cannot-read libhdf5/h5py failed to open the file (or crashed/hung)
|
||||||
|
ref-bug every issue is an object we refuse that h5py reads only through a
|
||||||
|
libhdf5 bug, confirmed in this run by ref_bugs.py (its values
|
||||||
|
change with the reading process's heap)
|
||||||
|
our-error we fail to open, list, or read something h5py reads
|
||||||
|
mismatch we read something with different shape/values, or a different object set
|
||||||
|
ok
|
||||||
|
"""
|
||||||
|
import collections
|
||||||
|
import csv
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
|
||||||
|
R = sys.argv[1]
|
||||||
|
RUNS = os.path.join(R, "runs")
|
||||||
|
|
||||||
|
# Objects ref_bugs.py confirmed in this run: h5py's values for them come from
|
||||||
|
# libhdf5 reading memory the file does not determine.
|
||||||
|
try:
|
||||||
|
REF_BUGS = {(b["file"], b["object"])
|
||||||
|
for b in json.load(open(os.path.join(R, "ref_bugs.json")))["read_bugs"] if b.get("confirmed")}
|
||||||
|
except (OSError, ValueError, KeyError):
|
||||||
|
REF_BUGS = set()
|
||||||
|
|
||||||
|
|
||||||
|
def is_ref_bug(rel, issue):
|
||||||
|
"""An our-error on reading an object that ref_bugs.py confirmed."""
|
||||||
|
kind, detail = issue[0], issue[1]
|
||||||
|
return kind == "our-error" and any(f == rel and detail.startswith(obj + ": error: ") for f, obj in REF_BUGS)
|
||||||
|
|
||||||
|
|
||||||
|
def load(d, name):
|
||||||
|
rc_p = os.path.join(d, name + ".rc")
|
||||||
|
if not os.path.exists(rc_p):
|
||||||
|
return None
|
||||||
|
rc = int(open(rc_p).read().strip() or -1)
|
||||||
|
err = open(os.path.join(d, name + ".err"), errors="replace").read()
|
||||||
|
js = None
|
||||||
|
try:
|
||||||
|
js = json.load(open(os.path.join(d, name + ".json")))
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
pass
|
||||||
|
return {"rc": rc, "err": err, "json": js}
|
||||||
|
|
||||||
|
|
||||||
|
def proc_status(p):
|
||||||
|
"""-> (status, detail)"""
|
||||||
|
if p is None:
|
||||||
|
return "missing", ""
|
||||||
|
rc, err = p["rc"], p["err"]
|
||||||
|
first_panic = next((ln for ln in err.splitlines() if ln.startswith("PANIC:") or "panicked at" in ln), "")
|
||||||
|
if rc == 0 and p["json"] is not None:
|
||||||
|
return "ok", ""
|
||||||
|
if rc == 137 or rc == 124:
|
||||||
|
return "hang", f"timeout ({os.environ.get('TMO', '20')} s)"
|
||||||
|
if "memory allocation of" in err or "MemoryError" in err or "std::bad_alloc" in err:
|
||||||
|
m = re.search(r"memory allocation of \d+ bytes failed", err)
|
||||||
|
return "oom", m.group(0) if m else "allocation failure"
|
||||||
|
if "overflowed its stack" in err:
|
||||||
|
return "crash", "stack overflow"
|
||||||
|
if rc == 101:
|
||||||
|
return "panic", first_panic or (err.strip().splitlines() or [""])[-1]
|
||||||
|
if rc in (134, 139, 136, 135, 132) or rc > 128:
|
||||||
|
sig = {134: "SIGABRT", 139: "SIGSEGV", 136: "SIGFPE", 135: "SIGBUS", 132: "SIGILL"}.get(rc, f"signal {rc - 128}")
|
||||||
|
tail = [ln for ln in err.strip().splitlines() if ln.strip()][-1:]
|
||||||
|
return "crash", f"{sig}: {tail[0][:200] if tail else ''}"
|
||||||
|
tail = [ln for ln in err.strip().splitlines() if ln.strip()][-1:]
|
||||||
|
return "crash", f"rc={rc}: {tail[0][:200] if tail else ''}"
|
||||||
|
|
||||||
|
|
||||||
|
def norm(msg):
|
||||||
|
m = msg.split("\n")[0]
|
||||||
|
m = re.sub(r"0x[0-9a-fA-F]+", "X", m)
|
||||||
|
m = re.sub(r'"[^"]*"', '"…"', m)
|
||||||
|
m = re.sub(r"'[^']*'", "'…'", m)
|
||||||
|
m = re.sub(r"\d+", "N", m)
|
||||||
|
return m[:160]
|
||||||
|
|
||||||
|
|
||||||
|
def panic_head(msg):
|
||||||
|
"""First line + first clawhdf5 frame of a PANIC record."""
|
||||||
|
lines = msg.split("\n")
|
||||||
|
frame = next((ln.strip() for ln in lines[1:] if "clawhdf5_format" in ln), "")
|
||||||
|
return lines[0][:300], frame[:300]
|
||||||
|
|
||||||
|
|
||||||
|
def eq_shape(a, b):
|
||||||
|
return a == b
|
||||||
|
|
||||||
|
|
||||||
|
rows = []
|
||||||
|
issues_by_file = {}
|
||||||
|
root_causes = collections.defaultdict(lambda: {"files": set(), "count": 0, "examples": []})
|
||||||
|
mismatch_causes = collections.defaultdict(lambda: {"files": set(), "count": 0, "examples": []})
|
||||||
|
panics = []
|
||||||
|
ref_only_errors = collections.Counter()
|
||||||
|
incomparable = collections.Counter()
|
||||||
|
# Values ref.py corrected for a known h5py bug: (file, object, fixes, same as ours)
|
||||||
|
ref_fixes = []
|
||||||
|
|
||||||
|
|
||||||
|
def add(bucket, key, file, example):
|
||||||
|
b = bucket[key]
|
||||||
|
b["count"] += 1
|
||||||
|
if file not in b["files"] and len(b["examples"]) < 6:
|
||||||
|
b["examples"].append(example)
|
||||||
|
b["files"].add(file)
|
||||||
|
|
||||||
|
|
||||||
|
files = [ln.strip() for ln in open(os.path.join(R, "files.txt")) if ln.strip()]
|
||||||
|
for rel in files:
|
||||||
|
d = os.path.join(RUNS, rel.replace("/", "__"))
|
||||||
|
corpus = rel.split("/")[0]
|
||||||
|
ours, ref = load(d, "ours"), load(d, "ref")
|
||||||
|
h5dump = load(d, "h5dump")
|
||||||
|
os_, od = proc_status(ours)
|
||||||
|
rs, rd = proc_status(ref)
|
||||||
|
oj = ours["json"] if ours else None
|
||||||
|
rj = ref["json"] if ref else None
|
||||||
|
issues = [] # (kind, detail)
|
||||||
|
caught_panics = []
|
||||||
|
|
||||||
|
def scan_err(path, what, msg):
|
||||||
|
if msg.startswith("PANIC:"):
|
||||||
|
caught_panics.append((path, what, msg))
|
||||||
|
|
||||||
|
if oj:
|
||||||
|
for o in oj.get("objects", []):
|
||||||
|
for k in ("error", "attrs_error", "list_error"):
|
||||||
|
if k in o:
|
||||||
|
scan_err(o["path"], k, o[k])
|
||||||
|
for an, av in (o.get("attrs") or {}).items():
|
||||||
|
if "error" in av:
|
||||||
|
scan_err(o["path"], f"attr {an}", av["error"])
|
||||||
|
if oj.get("open_error", "").startswith("PANIC:"):
|
||||||
|
caught_panics.append(("<open>", "open", oj["open_error"]))
|
||||||
|
|
||||||
|
ref_open_fail = rs != "ok" or (rj is not None and "open_error" in rj)
|
||||||
|
ours_open_err = oj.get("open_error") if oj else None
|
||||||
|
n_obj = n_ok = 0
|
||||||
|
if os_ == "ok" and rj and not ref_open_fail and not ours_open_err:
|
||||||
|
ro = {x["path"]: x for x in rj.get("objects", [])}
|
||||||
|
oo = {x["path"]: x for x in oj.get("objects", [])}
|
||||||
|
our_list_errors = [x for x in oo.values() if "list_error" in x]
|
||||||
|
for p in sorted(set(ro) | set(oo)):
|
||||||
|
a, b = ro.get(p), oo.get(p)
|
||||||
|
n_obj += 1
|
||||||
|
if a is None:
|
||||||
|
issues.append(("mismatch", f"extra object {p} (kind={b.get('kind')})", "extra-object", b))
|
||||||
|
continue
|
||||||
|
if b is None:
|
||||||
|
if our_list_errors:
|
||||||
|
continue # accounted for by the list_error
|
||||||
|
issues.append(("mismatch", f"missing object {p} (kind={a.get('kind')})", "missing-object", a))
|
||||||
|
continue
|
||||||
|
ok = True
|
||||||
|
if a.get("kind") != b.get("kind") and "error" not in b and "error" not in a:
|
||||||
|
issues.append(("mismatch", f"{p}: kind {a.get('kind')} vs ours {b.get('kind')}", "kind", b))
|
||||||
|
ok = False
|
||||||
|
# h5py could not open the object at all: it read none of its
|
||||||
|
# attributes or links, so there is nothing to compare ours with
|
||||||
|
# (the object's own error is compared above and below).
|
||||||
|
ref_unopened = a.get("kind") == "unknown" and "error" in a
|
||||||
|
for k in ("error", "list_error", "attrs_error"):
|
||||||
|
if ref_unopened and k != "error":
|
||||||
|
continue
|
||||||
|
if k in b and k not in a:
|
||||||
|
issues.append(("our-error", f"{p}: {k}: {b[k]}", b[k], b))
|
||||||
|
ok = False
|
||||||
|
elif k in a and k not in b and k == "error":
|
||||||
|
ref_only_errors[norm(a[k])] += 1
|
||||||
|
if a.get("kind") == "dataset" and "error" not in a and "error" not in b:
|
||||||
|
if "skipped" in a or "skipped" in b:
|
||||||
|
pass
|
||||||
|
elif a.get("converted"):
|
||||||
|
incomparable[f"dataset {a['converted']}"] += 1
|
||||||
|
elif a.get("shape") != b.get("shape"):
|
||||||
|
issues.append(("mismatch", f"{p}: shape {a.get('shape')} vs ours {b.get('shape')}", "shape", b))
|
||||||
|
ok = False
|
||||||
|
elif a.get("hash") != b.get("hash"):
|
||||||
|
issues.append(("mismatch", f"{p}: values differ (h5py {a.get('dtype')} vs ours {b.get('dtype')})", "values", b | {"ref_head": a.get("head"), "ref_dtype": a.get("dtype")}))
|
||||||
|
ok = False
|
||||||
|
if a.get("ref_fix") and "hash" in b:
|
||||||
|
ref_fixes.append((rel, p, a["ref_fix"], a.get("hash") == b.get("hash")))
|
||||||
|
ra, oa = a.get("attrs") or {}, b.get("attrs") or {}
|
||||||
|
if "attrs_error" not in b and "attrs_error" not in a and not ref_unopened:
|
||||||
|
for an in sorted(set(ra) | set(oa)):
|
||||||
|
x, y = ra.get(an), oa.get(an)
|
||||||
|
if x is None:
|
||||||
|
issues.append(("mismatch", f"{p}@{an}: extra attribute", "extra-attr", y or {}))
|
||||||
|
elif y is None:
|
||||||
|
issues.append(("mismatch", f"{p}@{an}: missing attribute", "missing-attr", x))
|
||||||
|
elif "error" in y and "error" not in x:
|
||||||
|
issues.append(("our-error", f"{p}@{an}: {y['error']}", y["error"], y))
|
||||||
|
elif "error" in x:
|
||||||
|
continue
|
||||||
|
elif x.get("converted"):
|
||||||
|
incomparable[f"attr {x['converted']}"] += 1
|
||||||
|
elif x.get("shape") != y.get("shape"):
|
||||||
|
issues.append(("mismatch", f"{p}@{an}: attr shape {x.get('shape')} vs ours {y.get('shape')}", "attr-shape", y | {"ref_dtype": x.get("dtype")}))
|
||||||
|
elif x.get("hash") != y.get("hash"):
|
||||||
|
issues.append(("mismatch", f"{p}@{an}: attr values differ (h5py {x.get('dtype')} vs ours {y.get('dtype')})", "attr-values", y | {"ref_head": x.get("head"), "ref_dtype": x.get("dtype")}))
|
||||||
|
if x.get("ref_fix") and "hash" in y:
|
||||||
|
ref_fixes.append((rel, f"{p}@{an}", x["ref_fix"], x.get("hash") == y.get("hash")))
|
||||||
|
if ok:
|
||||||
|
n_ok += 1
|
||||||
|
|
||||||
|
# classify
|
||||||
|
if os_ in ("hang", "oom", "crash", "panic"):
|
||||||
|
cls = os_
|
||||||
|
elif caught_panics:
|
||||||
|
cls = "panic"
|
||||||
|
elif ref_open_fail:
|
||||||
|
cls = "h5py-cannot-read"
|
||||||
|
elif issues and all(is_ref_bug(rel, i) for i in issues):
|
||||||
|
cls = "ref-bug"
|
||||||
|
elif ours_open_err:
|
||||||
|
cls = "our-error"
|
||||||
|
issues.append(("our-error", f"open: {ours_open_err}", ours_open_err, {}))
|
||||||
|
elif any(i[0] == "our-error" for i in issues):
|
||||||
|
cls = "our-error"
|
||||||
|
elif issues:
|
||||||
|
cls = "mismatch"
|
||||||
|
else:
|
||||||
|
cls = "ok"
|
||||||
|
|
||||||
|
if os_ in ("hang", "oom", "crash", "panic") or caught_panics:
|
||||||
|
panics.append({
|
||||||
|
"file": rel, "class": cls, "detail": od,
|
||||||
|
"stderr": (ours["err"] if ours else "")[:3000],
|
||||||
|
"caught": [(p, w, m[:2500]) for p, w, m in caught_panics[:3]],
|
||||||
|
"n_caught": len(caught_panics),
|
||||||
|
})
|
||||||
|
# A ref-bug file's differences are listed with the evidence instead.
|
||||||
|
for kind, detail, key, rec in (issues if cls != "ref-bug" else []):
|
||||||
|
if kind == "our-error":
|
||||||
|
add(root_causes, norm(key), rel, detail[:300])
|
||||||
|
else:
|
||||||
|
if key in ("values", "attr-values", "shape", "attr-shape"):
|
||||||
|
mk = f"{key}: ours={rec.get('dtype')} h5py={rec.get('ref_dtype')} layout={rec.get('layout','-')} filters={rec.get('filters','-')}"
|
||||||
|
else:
|
||||||
|
mk = key
|
||||||
|
add(mismatch_causes, mk, rel, detail[:300] + (f" | ref_head={rec.get('ref_head')} our_head={rec.get('head')}" if rec.get("ref_head") else ""))
|
||||||
|
ref_detail = rd if rs != "ok" else ((rj or {}).get("open_error") or "")
|
||||||
|
h5d = ""
|
||||||
|
if h5dump:
|
||||||
|
rc = h5dump["rc"]
|
||||||
|
h5d = {0: "ok", 1: "error", 137: "hang", 124: "hang", 134: "SIGABRT", 139: "SIGSEGV", 136: "SIGFPE", 135: "SIGBUS"}.get(rc, f"rc={rc}")
|
||||||
|
if "memory allocation" in h5dump["err"] or "Cannot allocate" in h5dump["err"]:
|
||||||
|
h5d += "(oom)"
|
||||||
|
rows.append({
|
||||||
|
"file": rel, "corpus": corpus, "class": cls,
|
||||||
|
"ours": os_ if os_ != "ok" else ("open-error" if ours_open_err else ("panic" if caught_panics else "ok")),
|
||||||
|
"ours_detail": (od or ours_open_err or (caught_panics[0][2].split("\n")[0] if caught_panics else ""))[:300],
|
||||||
|
"ref": rs if rs != "ok" else ("open-error" if (rj or {}).get("open_error") else "ok"),
|
||||||
|
"ref_detail": ref_detail[:300],
|
||||||
|
"h5dump_1_14_6": h5d,
|
||||||
|
"h5dump_detail": ([ln for ln in h5dump["err"].splitlines() if ln.strip()][-1:] or [""])[0][:200] if h5dump else "",
|
||||||
|
"objects": n_obj, "objects_ok": n_ok,
|
||||||
|
"issues": len(issues), "first_issue": issues[0][1][:300] if issues else "",
|
||||||
|
"superblock": (oj or {}).get("superblock_version", ""),
|
||||||
|
})
|
||||||
|
# the first issues of each file, for report.py's known-cause matching
|
||||||
|
issues_by_file[rel] = [
|
||||||
|
{"kind": k, "key": key, "detail": det[:300], "ours_dtype": rec.get("dtype"), "ref_dtype": rec.get("ref_dtype")}
|
||||||
|
for k, det, key, rec in issues[:50]
|
||||||
|
]
|
||||||
|
|
||||||
|
with open(os.path.join(R, "results.csv"), "w", newline="") as fh:
|
||||||
|
w = csv.DictWriter(fh, fieldnames=list(rows[0].keys()))
|
||||||
|
w.writeheader()
|
||||||
|
w.writerows(rows)
|
||||||
|
|
||||||
|
|
||||||
|
def ser(b):
|
||||||
|
return {k: {"files": len(v["files"]), "count": v["count"], "examples": v["examples"], "file_list": sorted(v["files"])} for k, v in sorted(b.items(), key=lambda kv: -len(kv[1]["files"]))}
|
||||||
|
|
||||||
|
|
||||||
|
json.dump({"rows": rows, "issues": issues_by_file, "root_causes": ser(root_causes), "mismatch_causes": ser(mismatch_causes),
|
||||||
|
"panics": panics, "incomparable": incomparable.most_common(), "ref_only_errors": ref_only_errors.most_common(),
|
||||||
|
"ref_fixes": ref_fixes, "ref_bugs_confirmed": sorted(REF_BUGS)},
|
||||||
|
open(os.path.join(R, "results.json"), "w"), indent=1)
|
||||||
|
|
||||||
|
classes = ["ok", "our-error", "mismatch", "h5py-cannot-read", "ref-bug", "hang", "panic", "crash", "oom"]
|
||||||
|
by_corpus = collections.defaultdict(collections.Counter)
|
||||||
|
for r in rows:
|
||||||
|
by_corpus[r["corpus"]][r["class"]] += 1
|
||||||
|
by_corpus["ALL"][r["class"]] += 1
|
||||||
|
lines = ["# Conformance sweep summary", "", "| corpus | files | " + " | ".join(classes) + " |", "|---" * (len(classes) + 2) + "|"]
|
||||||
|
for c in sorted(by_corpus, key=lambda k: (k == "ALL", k)):
|
||||||
|
cnt = by_corpus[c]
|
||||||
|
lines.append(f"| {c} | {sum(cnt.values())} | " + " | ".join(str(cnt.get(k, 0)) for k in classes) + " |")
|
||||||
|
lines += ["", "## Panics / hangs / crashes / OOM", ""]
|
||||||
|
for p in panics:
|
||||||
|
lines.append(f"- **{p['file']}** [{p['class']}] {p['detail']}")
|
||||||
|
for path, what, m in p["caught"][:1]:
|
||||||
|
lines.append(" ```\n " + f"{path} ({what}): " + m.replace("\n", "\n ")[:1500] + "\n ```")
|
||||||
|
if not p["caught"] and p["stderr"]:
|
||||||
|
lines.append(" ```\n " + p["stderr"].strip()[:1500].replace("\n", "\n ") + "\n ```")
|
||||||
|
lines += ["", "## Our-error root causes (files affected)", ""]
|
||||||
|
for k, v in ser(root_causes).items():
|
||||||
|
lines.append(f"- [{v['files']} files, {v['count']} objs] `{k}`")
|
||||||
|
for ex in v["examples"][:3]:
|
||||||
|
lines.append(f" - {ex}")
|
||||||
|
lines += ["", "## Mismatch root causes", ""]
|
||||||
|
for k, v in ser(mismatch_causes).items():
|
||||||
|
lines.append(f"- [{v['files']} files, {v['count']} objs] `{k}`")
|
||||||
|
for ex in v["examples"][:3]:
|
||||||
|
lines.append(f" - {ex}")
|
||||||
|
lines += ["", "## Objects h5py fails on but we read (top)", ""]
|
||||||
|
for k, n in ref_only_errors.most_common(15):
|
||||||
|
lines.append(f"- {n} x `{k}`")
|
||||||
|
open(os.path.join(R, "summary.md"), "w").write("\n".join(lines) + "\n")
|
||||||
|
print("\n".join(lines[:4 + len(by_corpus)]))
|
||||||
@@ -0,0 +1,19 @@
|
|||||||
|
# Conformance corpora, pinned by commit. fetch-corpus.sh reads this file.
|
||||||
|
#
|
||||||
|
# name git-url commit root [sparse-checkout patterns...]
|
||||||
|
#
|
||||||
|
# `root` is the directory inside the checkout that is swept ("." = all of it).
|
||||||
|
# Patterns are git non-cone sparse-checkout patterns; none = whole repository.
|
||||||
|
# Every file under <root> with an HDF5/netCDF-4 extension is probed; for
|
||||||
|
# cve_hdf5 the extension-less files in cvefiles/ and fuzzerfiles/ are too.
|
||||||
|
# Licences: each corpus keeps its upstream licence; nothing here is committed
|
||||||
|
# to this repository — the files are downloaded into the gitignored cache.
|
||||||
|
hdf5 https://github.com/HDFGroup/hdf5.git a3cf1ea82cc7a66e50029a688121e1b105a7ce88 . *.h5 *.he5 *.nc *.hdf5 *.h5f
|
||||||
|
cve_hdf5 https://github.com/HDFGroup/cve_hdf5.git 3fd1f5ae3869e01b8ae02b41d7108de7ffb1a374 .
|
||||||
|
netcdf-c https://github.com/Unidata/netcdf-c.git beb7b9585273c1548386231a59b809d906359033 . /nc_test4/*.nc /ncdump/*.nc /nc_test4/*.h5 /ncdump/*.h5 /h5_test/*.h5 /hdf5_test/*.h5
|
||||||
|
NCAS-CMS_pyfive https://github.com/NCAS-CMS/pyfive.git 8cf07b8749133f41c5e30b8a4c604486f687fe74 . *.h5 *.hdf5 *.hdf *.nc *.he5
|
||||||
|
usnistgov_h5wasm https://github.com/usnistgov/h5wasm.git 02f6336527d2812783fcedabfbf42127ec8d06d2 . *.h5 *.hdf5 *.hdf *.nc *.he5
|
||||||
|
netcdf4-python https://github.com/Unidata/netcdf4-python.git 6e67576d39aef8091fb20bd767b4f1a52ddc1bec . *.nc *.h5
|
||||||
|
xarray-data https://github.com/pydata/xarray-data.git a35297e9da2cc99c811014f0c8a4297345a5c28d . /basin_mask.nc /precipitation.nc4 /imerghh_730.hdf5 /eraint_uvz.nc /ROMS_example.nc /tiny.nc
|
||||||
|
# h5py 3.16.0 (tag 3.16.0), its test data files.
|
||||||
|
h5py_data https://github.com/h5py/h5py.git b2f0347c4200333acd89b43733f1caa0c115162f h5py/tests/data_files /h5py/tests/data_files/*
|
||||||
Executable
+39
@@ -0,0 +1,39 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# fetch-corpus.sh [cache_dir]
|
||||||
|
#
|
||||||
|
# Download the corpora pinned in conformance/corpus.txt into the (gitignored)
|
||||||
|
# cache: <cache>/src/<name> is a shallow, sparse, blob-filtered checkout of the
|
||||||
|
# pinned commit and <cache>/corpus/<name> links to the swept root inside it.
|
||||||
|
# A corpus already checked out at its pinned commit is left alone, so a second
|
||||||
|
# run costs nothing and needs no network.
|
||||||
|
set -euo pipefail
|
||||||
|
HERE="$(cd "$(dirname "$0")" && pwd)"
|
||||||
|
CACHE="${1:-${CONFORMANCE_CACHE:-$HERE/.cache}}"
|
||||||
|
mkdir -p "$CACHE/src" "$CACHE/corpus"
|
||||||
|
CACHE="$(cd "$CACHE" && pwd)"
|
||||||
|
|
||||||
|
retry() { local i; for i in 1 2 3 4; do "$@" && return 0; sleep $((i * 5)); done; return 1; }
|
||||||
|
|
||||||
|
grep -v '^[[:space:]]*\(#\|$\)' "$HERE/corpus.txt" | while read -r name url commit root patterns; do
|
||||||
|
src="$CACHE/src/$name"
|
||||||
|
if [ -d "$src/.git" ] && [ "$(git -C "$src" rev-parse HEAD 2>/dev/null)" = "$commit" ]; then
|
||||||
|
echo "cached $name @ ${commit:0:12}"
|
||||||
|
else
|
||||||
|
echo "fetching $name @ ${commit:0:12} from $url"
|
||||||
|
rm -rf "$src"
|
||||||
|
git init -q "$src"
|
||||||
|
git -C "$src" remote add origin "$url"
|
||||||
|
git -C "$src" config advice.detachedHead false
|
||||||
|
if [ -n "$patterns" ]; then
|
||||||
|
git -C "$src" config core.sparseCheckout true
|
||||||
|
# no-cone patterns (globs); `set -f` keeps the shell from expanding them
|
||||||
|
(set -f; printf '%s\n' $patterns) > "$src/.git/info/sparse-checkout"
|
||||||
|
fi
|
||||||
|
retry git -C "$src" fetch -q --depth 1 --filter=blob:none origin "$commit"
|
||||||
|
retry git -C "$src" checkout -q FETCH_HEAD
|
||||||
|
got="$(git -C "$src" rev-parse HEAD)"
|
||||||
|
[ "$got" = "$commit" ] || { echo "error: $name checked out $got, expected $commit" >&2; exit 1; }
|
||||||
|
fi
|
||||||
|
ln -sfn "$src/$root" "$CACHE/corpus/$name"
|
||||||
|
done
|
||||||
|
echo "corpus ready in $CACHE/corpus"
|
||||||
@@ -0,0 +1,48 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""list_files.py <corpus_dir>: print the files the sweep probes, one per line,
|
||||||
|
as <corpus>/<path> in byte order.
|
||||||
|
|
||||||
|
* every file named *.h5 *.hdf5 *.he5 *.nc *.nc4 *.hdf *.h5f in each corpus,
|
||||||
|
except netCDF classic / 64-bit-offset / CDF5 files (magic "CDF"): they are
|
||||||
|
not HDF5, so neither side can read them and they say nothing;
|
||||||
|
* plus, for cve_hdf5, every file in cvefiles/ and fuzzerfiles/ except
|
||||||
|
.md/.c sources — the reproducers are mostly extension-less, and they are
|
||||||
|
kept whatever their bytes look like (that is their point).
|
||||||
|
"""
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
|
||||||
|
EXTS = (".h5", ".hdf5", ".he5", ".nc", ".nc4", ".hdf", ".h5f")
|
||||||
|
|
||||||
|
|
||||||
|
def walk(top):
|
||||||
|
for dirpath, dirnames, filenames in os.walk(top):
|
||||||
|
dirnames[:] = [d for d in dirnames if d != ".git"]
|
||||||
|
for fn in filenames:
|
||||||
|
p = os.path.join(dirpath, fn)
|
||||||
|
if os.path.isfile(p) and not os.path.islink(p):
|
||||||
|
yield os.path.relpath(p, top)
|
||||||
|
|
||||||
|
|
||||||
|
def main(root):
|
||||||
|
out = set()
|
||||||
|
for corpus in sorted(os.listdir(root)):
|
||||||
|
top = os.path.join(root, corpus)
|
||||||
|
if not os.path.isdir(top):
|
||||||
|
continue
|
||||||
|
for rel in walk(top):
|
||||||
|
path = os.path.join(top, rel)
|
||||||
|
if rel.lower().endswith(EXTS):
|
||||||
|
with open(path, "rb") as fh:
|
||||||
|
if fh.read(3) == b"CDF":
|
||||||
|
continue
|
||||||
|
out.add(f"{corpus}/{rel}")
|
||||||
|
elif corpus == "cve_hdf5" and rel.split(os.sep)[0] in ("cvefiles", "fuzzerfiles") \
|
||||||
|
and not rel.endswith((".md", ".c")):
|
||||||
|
out.add(f"{corpus}/{rel}")
|
||||||
|
for f in sorted(out, key=lambda s: s.encode()):
|
||||||
|
print(f)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main(sys.argv[1])
|
||||||
Generated
+492
@@ -0,0 +1,492 @@
|
|||||||
|
# This file is automatically @generated by Cargo.
|
||||||
|
# It is not intended for manual editing.
|
||||||
|
version = 4
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "adler2"
|
||||||
|
version = "2.0.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "320119579fcad9c21884f5c4861d16174d0e06250625266f50fe6898340abefa"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "better_io"
|
||||||
|
version = "0.2.0"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "ef0a3155e943e341e557863e69a708999c94ede624e37865c8e2a91b94efa78f"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "block-buffer"
|
||||||
|
version = "0.10.4"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "3078c7629b62d3f0439517fa394996acacc5cbc91c5a20d8c658e77abd503a71"
|
||||||
|
dependencies = [
|
||||||
|
"generic-array",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "byteorder"
|
||||||
|
version = "1.5.0"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "1fd0f2584146f6f2ef48085050886acf353beff7305ebd1ae69500e27c67f64b"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "bzip2"
|
||||||
|
version = "0.6.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "f3a53fac24f34a81bc9954b5d6cfce0c21e18ec6959f44f56e8e90e4bb7c346c"
|
||||||
|
dependencies = [
|
||||||
|
"libbz2-rs-sys",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "cc"
|
||||||
|
version = "1.5.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "f360145194ee8e21db5ee7f3fcd4fe52210864c75c985dae33218202c8bbe040"
|
||||||
|
dependencies = [
|
||||||
|
"find-msvc-tools",
|
||||||
|
"jobserver",
|
||||||
|
"libc",
|
||||||
|
"shlex",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "cfg-if"
|
||||||
|
version = "1.0.5"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "4e7648175b45a9a48536d676f68d918270699102aa8dab5496df06904c914600"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "clawhdf5-format"
|
||||||
|
version = "2.7.0"
|
||||||
|
dependencies = [
|
||||||
|
"byteorder",
|
||||||
|
"bzip2",
|
||||||
|
"flate2",
|
||||||
|
"libaec-sys",
|
||||||
|
"libc",
|
||||||
|
"lz4_flex",
|
||||||
|
"pco",
|
||||||
|
"portable-atomic",
|
||||||
|
"ruzstd",
|
||||||
|
"sha2",
|
||||||
|
"snap",
|
||||||
|
"zstd",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "conformance-probe"
|
||||||
|
version = "0.1.0"
|
||||||
|
dependencies = [
|
||||||
|
"clawhdf5-format",
|
||||||
|
"serde_json",
|
||||||
|
"sha2",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "cpufeatures"
|
||||||
|
version = "0.2.17"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "59ed5838eebb26a2bb2e58f6d5b5316989ae9d08bab10e0e6d103e656d1b0280"
|
||||||
|
dependencies = [
|
||||||
|
"libc",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "crc32fast"
|
||||||
|
version = "1.5.2"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "01a7799fd6b852db0e61728dde9a204c423b44d689dbd432522543614b490e78"
|
||||||
|
dependencies = [
|
||||||
|
"cfg-if",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "crunchy"
|
||||||
|
version = "0.2.4"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "460fbee9c2c2f33933d720630a6a0bac33ba7053db5344fac858d4b8952d77d5"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "crypto-common"
|
||||||
|
version = "0.1.7"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "78c8292055d1c1df0cce5d180393dc8cce0abec0a7102adb6c7b1eef6016d60a"
|
||||||
|
dependencies = [
|
||||||
|
"generic-array",
|
||||||
|
"typenum",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "digest"
|
||||||
|
version = "0.10.7"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292"
|
||||||
|
dependencies = [
|
||||||
|
"block-buffer",
|
||||||
|
"crypto-common",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "dtype_dispatch"
|
||||||
|
version = "0.2.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "ab23e69df104e2fd85ee63a533a22d2132ef5975dc6b36f9f3e5a7305e4a8ed7"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "find-msvc-tools"
|
||||||
|
version = "0.1.14"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "aedcfb3409746eddb02b9e19ebda1c3394f759a152e48ee875a0844d1b955484"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "flate2"
|
||||||
|
version = "1.1.10"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "6e634e2e0ebac1ee034020da1ca582e17ffe4e0f5e985823721e168928136dcb"
|
||||||
|
dependencies = [
|
||||||
|
"crc32fast",
|
||||||
|
"miniz_oxide",
|
||||||
|
"zlib-rs",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "generic-array"
|
||||||
|
version = "0.14.7"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "85649ca51fd72272d7821adaf274ad91c288277713d9c18820d8499a7ff69e9a"
|
||||||
|
dependencies = [
|
||||||
|
"typenum",
|
||||||
|
"version_check",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "getrandom"
|
||||||
|
version = "0.4.3"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "300e883d756b2e4ec94e02791f39b04b522276138852cfc41d9fb7e904106099"
|
||||||
|
dependencies = [
|
||||||
|
"cfg-if",
|
||||||
|
"libc",
|
||||||
|
"r-efi",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "half"
|
||||||
|
version = "2.7.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "6ea2d84b969582b4b1864a92dc5d27cd2b77b622a8d79306834f1be5ba20d84b"
|
||||||
|
dependencies = [
|
||||||
|
"cfg-if",
|
||||||
|
"crunchy",
|
||||||
|
"zerocopy",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "itoa"
|
||||||
|
version = "1.0.18"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "jobserver"
|
||||||
|
version = "0.1.35"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "1c00acbd29eabad4a2392fa0e921c874934dbbf4194312ad20f04a0ed67a3cb3"
|
||||||
|
dependencies = [
|
||||||
|
"getrandom",
|
||||||
|
"libc",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "libaec-sys"
|
||||||
|
version = "0.1.0"
|
||||||
|
dependencies = [
|
||||||
|
"pkg-config",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "libbz2-rs-sys"
|
||||||
|
version = "0.2.5"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "34b357333733e8260735ba5894eb928c02ecc69c78715f01a8019e7fa7f2db4c"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "libc"
|
||||||
|
version = "0.2.189"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "lz4_flex"
|
||||||
|
version = "0.11.6"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "373f5eceeeab7925e0c1098212f2fbc4d416adec9d35051a6ab251e824c1854a"
|
||||||
|
dependencies = [
|
||||||
|
"twox-hash",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "memchr"
|
||||||
|
version = "2.8.3"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "miniz_oxide"
|
||||||
|
version = "0.9.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "b63fbc4a50860e98e7b2aa7804ded1db5cbc3aff9193adaff57a6931bf7c4b4c"
|
||||||
|
dependencies = [
|
||||||
|
"adler2",
|
||||||
|
"simd-adler32",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "pco"
|
||||||
|
version = "1.0.3"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "386342cad4c6e97f081568e5d910ea7d871314c843aa8fc564f2a6b64cab9456"
|
||||||
|
dependencies = [
|
||||||
|
"better_io",
|
||||||
|
"dtype_dispatch",
|
||||||
|
"half",
|
||||||
|
"rand_xoshiro",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "pkg-config"
|
||||||
|
version = "0.3.34"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "f6b464fbc74e149a392436b17d523f769e057cb6877f6a5c4618bc6f11800548"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "portable-atomic"
|
||||||
|
version = "1.15.0"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "05c8b63e8d9609db387f0324918f81d68fe27748f084ef092fb35954d0539a85"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "proc-macro2"
|
||||||
|
version = "1.0.107"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9"
|
||||||
|
dependencies = [
|
||||||
|
"unicode-ident",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "quote"
|
||||||
|
version = "1.0.47"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001"
|
||||||
|
dependencies = [
|
||||||
|
"proc-macro2",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "r-efi"
|
||||||
|
version = "6.0.0"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "f8dcc9c7d52a811697d2151c701e0d08956f92b0e24136cf4cf27b57a6a0d9bf"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "rand_core"
|
||||||
|
version = "0.6.4"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "ec0be4795e2f6a28069bec0b5ff3e2ac9bafc99e6a9a7dc3547996c5c816922c"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "rand_xoshiro"
|
||||||
|
version = "0.6.0"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "6f97cdb2a36ed4183de61b2f824cc45c9f1037f28afe0a322e9fff4c108b5aaa"
|
||||||
|
dependencies = [
|
||||||
|
"rand_core",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "ruzstd"
|
||||||
|
version = "0.9.0"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "a252f5e20f038fe7b4ea53e073e65398d652c864cc162fc77c56c2f13717b888"
|
||||||
|
dependencies = [
|
||||||
|
"twox-hash",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "serde"
|
||||||
|
version = "1.0.229"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba"
|
||||||
|
dependencies = [
|
||||||
|
"serde_core",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "serde_core"
|
||||||
|
version = "1.0.229"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48"
|
||||||
|
dependencies = [
|
||||||
|
"serde_derive",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "serde_derive"
|
||||||
|
version = "1.0.229"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348"
|
||||||
|
dependencies = [
|
||||||
|
"proc-macro2",
|
||||||
|
"quote",
|
||||||
|
"syn 3.0.6",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "serde_json"
|
||||||
|
version = "1.0.151"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14"
|
||||||
|
dependencies = [
|
||||||
|
"itoa",
|
||||||
|
"memchr",
|
||||||
|
"serde",
|
||||||
|
"serde_core",
|
||||||
|
"zmij",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "sha2"
|
||||||
|
version = "0.10.9"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "a7507d819769d01a365ab707794a4084392c824f54a7a6a7862f8c3d0892b283"
|
||||||
|
dependencies = [
|
||||||
|
"cfg-if",
|
||||||
|
"cpufeatures",
|
||||||
|
"digest",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "shlex"
|
||||||
|
version = "2.0.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "simd-adler32"
|
||||||
|
version = "0.3.10"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "3a219298ac11a56ea9a6d2120044824d6f01aeb034955e7af7bc16858527deea"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "snap"
|
||||||
|
version = "1.1.2"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "199905e6153d6405f9728fe44daace35f8f837bbf830bb6e85fbd5828709a886"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "syn"
|
||||||
|
version = "2.0.119"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "872831b642d1a07999a962a351ed35b955ea2cfc8f3862091e2a240a84f17297"
|
||||||
|
dependencies = [
|
||||||
|
"proc-macro2",
|
||||||
|
"quote",
|
||||||
|
"unicode-ident",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "syn"
|
||||||
|
version = "3.0.6"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "8593e8e72159ed2257d083c7a454a85cbf854f37a0966d8d483aff8c8a3ebcee"
|
||||||
|
dependencies = [
|
||||||
|
"proc-macro2",
|
||||||
|
"quote",
|
||||||
|
"unicode-ident",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "twox-hash"
|
||||||
|
version = "2.1.4"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "5283634e518fe9e82c7b20520bb4bc209009fd16c82077c802f8111ecbb0117a"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "typenum"
|
||||||
|
version = "1.20.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "b6f5e870be6c3b371b77fe0ee0bafb859fa4964b4404c27de1d380043c4dda20"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "unicode-ident"
|
||||||
|
version = "1.0.26"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "d245f478577f809a851594d02313b640fb437e0bb33866753cff937863096954"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "version_check"
|
||||||
|
version = "0.9.5"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "zerocopy"
|
||||||
|
version = "0.8.59"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "6df92bf3d9227be3d53173901ddbffac2babc27ae50f397776ffd6dc33f800cb"
|
||||||
|
dependencies = [
|
||||||
|
"zerocopy-derive",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "zerocopy-derive"
|
||||||
|
version = "0.8.59"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "ac4f328cf2f05d084e496c3e9c3f33ed0a183656a16e1fcec4d464d8373aec82"
|
||||||
|
dependencies = [
|
||||||
|
"proc-macro2",
|
||||||
|
"quote",
|
||||||
|
"syn 2.0.119",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "zlib-rs"
|
||||||
|
version = "0.6.8"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "b268e58e7c693d7c271f93ffc4ba3b380412554231c85bf61ca7af91042a4112"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "zmij"
|
||||||
|
version = "1.0.23"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "zstd"
|
||||||
|
version = "0.13.3"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "e91ee311a569c327171651566e07972200e76fcfe2242a4fa446149a3881c08a"
|
||||||
|
dependencies = [
|
||||||
|
"zstd-safe",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "zstd-safe"
|
||||||
|
version = "7.3.0"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "64d80649ab6db9d9f6f9c80a40becd948eda4714a0a5ac8c4d157a32231c7882"
|
||||||
|
dependencies = [
|
||||||
|
"zstd-sys",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "zstd-sys"
|
||||||
|
version = "2.1.0+zstd.1.5.7"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "0ef0a8027ec3ee71300ab3bcbcd0393f434aa72b91ca6d635a39941deae8eea0"
|
||||||
|
dependencies = [
|
||||||
|
"cc",
|
||||||
|
"pkg-config",
|
||||||
|
]
|
||||||
@@ -0,0 +1,25 @@
|
|||||||
|
[package]
|
||||||
|
name = "conformance-probe"
|
||||||
|
version = "0.1.0"
|
||||||
|
edition = "2024"
|
||||||
|
rust-version = "1.92"
|
||||||
|
publish = false
|
||||||
|
description = "Walks an HDF5 file with clawhdf5-format and prints a canonical JSON description (see conformance/README.md)"
|
||||||
|
|
||||||
|
# Deliberately outside the main workspace: `cargo test --workspace` never
|
||||||
|
# builds it, and it links the optional C codecs (zstd, libaec) that the core
|
||||||
|
# crates' default build must not.
|
||||||
|
[workspace]
|
||||||
|
|
||||||
|
[dependencies]
|
||||||
|
clawhdf5-format = { path = "../../crates/clawhdf5-format", features = ["lz4", "zstd", "szip", "pcodec", "plugin-filters"] }
|
||||||
|
serde_json = "1"
|
||||||
|
sha2 = "0.10"
|
||||||
|
|
||||||
|
[profile.release]
|
||||||
|
# Keep panics catchable (the probe records them per object) and turn integer
|
||||||
|
# overflow into a reported panic instead of silent wraparound.
|
||||||
|
debug = 1
|
||||||
|
overflow-checks = true
|
||||||
|
debug-assertions = true
|
||||||
|
panic = "unwind"
|
||||||
@@ -0,0 +1,960 @@
|
|||||||
|
//! Conformance probe: walks an HDF5 file with clawhdf5-format (the same calls
|
||||||
|
//! the `clawhdf5` facade makes) and prints a canonical JSON description:
|
||||||
|
//! every hard-linked object (sorted-name DFS, deduplicated by header address),
|
||||||
|
//! and for each dataset / attribute its shape plus the SHA-256 of its values
|
||||||
|
//! in a canonical encoding shared with `ref.py`.
|
||||||
|
//!
|
||||||
|
//! Canonical value encoding (per element, concatenated, row-major):
|
||||||
|
//! int / float / bitfield / enum / time : element bytes, little-endian
|
||||||
|
//! non-IEEE-layout float (e.g. N-Bit) : the IEEE float of the same size it converts to
|
||||||
|
//! int with bit offset / short precision: the full-width integer it converts to
|
||||||
|
//! opaque : raw bytes
|
||||||
|
//! compound : members in declaration order (padding dropped)
|
||||||
|
//! array : base elements row-major
|
||||||
|
//! string (fixed or VL) : b'S' + u32le len + bytes (cut at first NUL, trailing spaces stripped)
|
||||||
|
//! VL sequence : b'V' + u32le count + base elements
|
||||||
|
//! reference : b'R' (payload not compared)
|
||||||
|
//!
|
||||||
|
//! Every object is processed inside catch_unwind; a caught panic is recorded
|
||||||
|
//! with its message, location and the clawhdf5 frames of its backtrace.
|
||||||
|
|
||||||
|
use std::cell::RefCell;
|
||||||
|
use std::collections::HashSet;
|
||||||
|
use std::panic::{self, AssertUnwindSafe};
|
||||||
|
|
||||||
|
use clawhdf5_format::attribute::extract_attributes_full;
|
||||||
|
use clawhdf5_format::data_layout::DataLayout;
|
||||||
|
use clawhdf5_format::data_read;
|
||||||
|
use clawhdf5_format::dataspace::{Dataspace, DataspaceType};
|
||||||
|
use clawhdf5_format::datatype::{Datatype, DatatypeByteOrder};
|
||||||
|
use clawhdf5_format::filter_pipeline::FilterPipeline;
|
||||||
|
use clawhdf5_format::group_v1::{self, GroupEntry};
|
||||||
|
use clawhdf5_format::group_v2;
|
||||||
|
use clawhdf5_format::message_type::MessageType;
|
||||||
|
use clawhdf5_format::object_header::{ObjectClass, ObjectHeader};
|
||||||
|
use clawhdf5_format::signature;
|
||||||
|
use clawhdf5_format::superblock::Superblock;
|
||||||
|
use clawhdf5_format::symbol_table::SymbolTableMessage;
|
||||||
|
use clawhdf5_format::vl_data::{VlResolver, check_element_size};
|
||||||
|
use serde_json::{Map, Value, json};
|
||||||
|
use sha2::{Digest, Sha256};
|
||||||
|
|
||||||
|
const MAX_BYTES: u64 = 200 * 1024 * 1024;
|
||||||
|
const MAX_OBJECTS: usize = 200_000;
|
||||||
|
|
||||||
|
thread_local! {
|
||||||
|
static LAST_PANIC: RefCell<Option<String>> = const { RefCell::new(None) };
|
||||||
|
}
|
||||||
|
|
||||||
|
fn install_hook() {
|
||||||
|
panic::set_hook(Box::new(|info| {
|
||||||
|
let msg = if let Some(s) = info.payload().downcast_ref::<&str>() {
|
||||||
|
s.to_string()
|
||||||
|
} else if let Some(s) = info.payload().downcast_ref::<String>() {
|
||||||
|
s.clone()
|
||||||
|
} else {
|
||||||
|
"<non-string panic>".into()
|
||||||
|
};
|
||||||
|
let loc = info
|
||||||
|
.location()
|
||||||
|
.map(|l| format!("{}:{}", l.file(), l.line()))
|
||||||
|
.unwrap_or_default();
|
||||||
|
let bt = std::backtrace::Backtrace::force_capture().to_string();
|
||||||
|
// keep only frames from clawhdf5 code
|
||||||
|
let mut frames = Vec::new();
|
||||||
|
let lines: Vec<&str> = bt.lines().collect();
|
||||||
|
for (i, l) in lines.iter().enumerate() {
|
||||||
|
let t = l.trim();
|
||||||
|
if t.contains("clawhdf5_format::") || t.contains("conformance_probe::") {
|
||||||
|
let at = lines
|
||||||
|
.get(i + 1)
|
||||||
|
.map(|n| n.trim())
|
||||||
|
.filter(|n| n.starts_with("at "))
|
||||||
|
.map(|n| {
|
||||||
|
let n = n.trim_start_matches("at ");
|
||||||
|
match n.find("/crates/") {
|
||||||
|
Some(p) => n[p + 1..].to_string(),
|
||||||
|
None => n.to_string(),
|
||||||
|
}
|
||||||
|
})
|
||||||
|
.unwrap_or_default();
|
||||||
|
let name = t.split_once(": ").map(|x| x.1).unwrap_or(t);
|
||||||
|
frames.push(format!("{name} ({at})"));
|
||||||
|
if frames.len() >= 12 {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let full = format!("PANIC: {msg} @ {loc}\n {}", frames.join("\n "));
|
||||||
|
eprintln!("{full}");
|
||||||
|
LAST_PANIC.with(|p| *p.borrow_mut() = Some(full));
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Run `f`, turning a panic into Err("PANIC: ...").
|
||||||
|
fn guarded<T>(f: impl FnOnce() -> Result<T, String>) -> Result<T, String> {
|
||||||
|
match panic::catch_unwind(AssertUnwindSafe(f)) {
|
||||||
|
Ok(r) => r,
|
||||||
|
Err(_) => Err(LAST_PANIC
|
||||||
|
.with(|p| p.borrow_mut().take())
|
||||||
|
.unwrap_or_else(|| "PANIC: <unknown>".into())),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn e<E: std::fmt::Debug>(x: E) -> String {
|
||||||
|
format!("{x:?}")
|
||||||
|
}
|
||||||
|
|
||||||
|
struct Ctx<'a> {
|
||||||
|
data: &'a [u8],
|
||||||
|
os: u8,
|
||||||
|
ls: u8,
|
||||||
|
base_dir: std::path::PathBuf,
|
||||||
|
/// Resolves variable-length elements as the library does (null
|
||||||
|
/// elements, strings cut at a NUL, heap objects of the wrong size
|
||||||
|
/// refused), caching each heap collection.
|
||||||
|
vl: RefCell<VlResolver<'a>>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl<'a> Ctx<'a> {
|
||||||
|
fn header(&self, addr: u64) -> Result<ObjectHeader, String> {
|
||||||
|
ObjectHeader::parse(self.data, addr as usize, self.os, self.ls).map_err(e)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn payload(&self, h: &ObjectHeader, t: MessageType) -> Result<Option<Vec<u8>>, String> {
|
||||||
|
match h.messages.iter().find(|m| m.msg_type == t) {
|
||||||
|
None => Ok(None),
|
||||||
|
Some(m) => {
|
||||||
|
clawhdf5_format::shared_message::message_data(self.data, m, self.os, self.ls)
|
||||||
|
.map(|c| Some(c.into_owned()))
|
||||||
|
.map_err(e)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn canon(&self, dt: &Datatype, b: &[u8], out: &mut Vec<u8>) -> Result<(), String> {
|
||||||
|
let size = dt.type_size() as usize;
|
||||||
|
if b.len() < size {
|
||||||
|
return Err(format!(
|
||||||
|
"canon: element slice {} < type size {size}",
|
||||||
|
b.len()
|
||||||
|
));
|
||||||
|
}
|
||||||
|
match dt {
|
||||||
|
Datatype::FloatingPoint { .. } if !ieee_layout(dt) => {
|
||||||
|
canon_custom_float(dt, &b[..size], out)?
|
||||||
|
}
|
||||||
|
Datatype::FixedPoint { .. } if partial_int(dt) => {
|
||||||
|
canon_partial_int(dt, &b[..size], out)?
|
||||||
|
}
|
||||||
|
Datatype::FixedPoint { byte_order, .. }
|
||||||
|
| Datatype::BitField { byte_order, .. }
|
||||||
|
| Datatype::FloatingPoint { byte_order, .. } => match byte_order {
|
||||||
|
DatatypeByteOrder::LittleEndian => out.extend_from_slice(&b[..size]),
|
||||||
|
DatatypeByteOrder::BigEndian => out.extend(b[..size].iter().rev()),
|
||||||
|
DatatypeByteOrder::Vax => return Err("canon: VAX byte order".into()),
|
||||||
|
},
|
||||||
|
Datatype::Time { .. } | Datatype::Opaque { .. } => out.extend_from_slice(&b[..size]),
|
||||||
|
Datatype::String { .. } => canon_str(&b[..size], out),
|
||||||
|
Datatype::Compound { members, .. } => {
|
||||||
|
for m in members {
|
||||||
|
let off = m.byte_offset as usize;
|
||||||
|
let ms = m.datatype.type_size() as usize;
|
||||||
|
if off.checked_add(ms).is_none_or(|end| end > size) {
|
||||||
|
return Err(format!("canon: member {} out of bounds", m.name));
|
||||||
|
}
|
||||||
|
self.canon(&m.datatype, &b[off..off + ms], out)?;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Datatype::Reference { .. } => out.push(b'R'),
|
||||||
|
Datatype::Enumeration { base_type, .. } => self.canon(base_type, b, out)?,
|
||||||
|
Datatype::Array {
|
||||||
|
base_type,
|
||||||
|
dimensions,
|
||||||
|
} => {
|
||||||
|
let n: usize = dimensions.iter().map(|d| *d as usize).product();
|
||||||
|
let bs = base_type.type_size() as usize;
|
||||||
|
for i in 0..n {
|
||||||
|
self.canon(base_type, &b[i * bs..], out)?;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Datatype::VariableLength {
|
||||||
|
size: vl_size,
|
||||||
|
is_string,
|
||||||
|
base_type,
|
||||||
|
..
|
||||||
|
} => {
|
||||||
|
check_element_size(*vl_size, self.os).map_err(e)?;
|
||||||
|
let el = &b[..size];
|
||||||
|
if *is_string {
|
||||||
|
let s = self.vl.borrow_mut().string_bytes(el).map_err(e)?;
|
||||||
|
canon_str(&s[0], out);
|
||||||
|
} else {
|
||||||
|
let bs = base_type.type_size() as usize;
|
||||||
|
// The borrow ends here: the base type may itself be
|
||||||
|
// variable-length.
|
||||||
|
let seq = self.vl.borrow_mut().sequences(el, bs).map_err(e)?;
|
||||||
|
let seq = &seq[0];
|
||||||
|
let len = seq.len() / bs;
|
||||||
|
out.push(b'V');
|
||||||
|
out.extend_from_slice(&(len as u32).to_le_bytes());
|
||||||
|
for i in 0..len {
|
||||||
|
self.canon(base_type, &seq[i * bs..], out)?;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Returns (shape json, n_elements)
|
||||||
|
fn shape(ds: &Dataspace) -> (Value, u64) {
|
||||||
|
match ds.space_type {
|
||||||
|
DataspaceType::Null => (Value::String("null".into()), 0),
|
||||||
|
DataspaceType::Scalar => (json!([]), 1),
|
||||||
|
DataspaceType::Simple => {
|
||||||
|
let n = ds.dimensions.iter().fold(1u64, |a, d| a.saturating_mul(*d));
|
||||||
|
(json!(ds.dimensions), n)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn hash_values(
|
||||||
|
&self,
|
||||||
|
dt: &Datatype,
|
||||||
|
raw: &[u8],
|
||||||
|
n: u64,
|
||||||
|
rec: &mut Map<String, Value>,
|
||||||
|
) -> Result<(), String> {
|
||||||
|
let size = dt.type_size() as usize;
|
||||||
|
let need = (n as usize).checked_mul(size).ok_or("n*size overflow")?;
|
||||||
|
if raw.len() != need {
|
||||||
|
return Err(format!(
|
||||||
|
"raw length {} != n_elements {n} * type_size {size}",
|
||||||
|
raw.len()
|
||||||
|
));
|
||||||
|
}
|
||||||
|
let mut canon = Vec::with_capacity(need);
|
||||||
|
for i in 0..n as usize {
|
||||||
|
self.canon(dt, &raw[i * size..(i + 1) * size], &mut canon)?;
|
||||||
|
}
|
||||||
|
let h = Sha256::digest(&canon);
|
||||||
|
rec.insert("hash".into(), Value::String(hex(&h)));
|
||||||
|
rec.insert(
|
||||||
|
"head".into(),
|
||||||
|
Value::String(hex(&canon[..canon.len().min(48)])),
|
||||||
|
);
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// VDS source files resolve next to the virtual file; like the library,
|
||||||
|
/// refuse absolute paths and `..`.
|
||||||
|
fn vds_resolver(
|
||||||
|
&self,
|
||||||
|
) -> impl Fn(&str) -> Result<Option<Vec<u8>>, clawhdf5_format::error::FormatError> + use<> {
|
||||||
|
let base = self.base_dir.clone();
|
||||||
|
move |name: &str| {
|
||||||
|
use clawhdf5_format::error::FormatError;
|
||||||
|
let p = std::path::Path::new(name);
|
||||||
|
if p.is_absolute()
|
||||||
|
|| p.components()
|
||||||
|
.any(|c| matches!(c, std::path::Component::ParentDir))
|
||||||
|
{
|
||||||
|
return Err(FormatError::ChunkedReadError(format!("refused {name}")));
|
||||||
|
}
|
||||||
|
match std::fs::read(base.join(p)) {
|
||||||
|
Ok(b) => Ok(Some(b)),
|
||||||
|
Err(err) if err.kind() == std::io::ErrorKind::NotFound => Ok(None),
|
||||||
|
Err(err) => Err(FormatError::ChunkedReadError(err.to_string())),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn read_named_datatype(&self, h: &ObjectHeader) -> Result<(), String> {
|
||||||
|
let dtb = self
|
||||||
|
.payload(h, MessageType::Datatype)?
|
||||||
|
.ok_or("MissingMessage(Datatype)")?;
|
||||||
|
Datatype::parse_in_header(&dtb, h.version).map_err(e)?;
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn read_dataset(&self, h: &ObjectHeader, rec: &mut Map<String, Value>) -> Result<(), String> {
|
||||||
|
let dtb = self
|
||||||
|
.payload(h, MessageType::Datatype)?
|
||||||
|
.ok_or("MissingMessage(Datatype)")?;
|
||||||
|
let (dt, _) = Datatype::parse_in_header(&dtb, h.version).map_err(e)?;
|
||||||
|
rec.insert("dtype".into(), Value::String(dtype_str(&dt)));
|
||||||
|
let dsb = self
|
||||||
|
.payload(h, MessageType::Dataspace)?
|
||||||
|
.ok_or("MissingMessage(Dataspace)")?;
|
||||||
|
let mut ds = Dataspace::parse(&dsb, self.ls).map_err(e)?;
|
||||||
|
// A virtual dataset's extent can come from its sources (unlimited /
|
||||||
|
// printf mappings), as h5py reports it, rather than the stored one.
|
||||||
|
if let Some(lm) = h
|
||||||
|
.messages
|
||||||
|
.iter()
|
||||||
|
.find(|m| m.msg_type == MessageType::DataLayout)
|
||||||
|
&& let Ok(dl @ DataLayout::Virtual { .. }) =
|
||||||
|
DataLayout::parse(&lm.data, self.os, self.ls)
|
||||||
|
{
|
||||||
|
let resolver = self.vds_resolver();
|
||||||
|
ds.dimensions = clawhdf5_format::vds::virtual_dataset_extent(
|
||||||
|
self.data,
|
||||||
|
&dl,
|
||||||
|
&ds,
|
||||||
|
self.os,
|
||||||
|
self.ls,
|
||||||
|
Some(&resolver),
|
||||||
|
)
|
||||||
|
.map_err(e)?;
|
||||||
|
}
|
||||||
|
let lm = h
|
||||||
|
.messages
|
||||||
|
.iter()
|
||||||
|
.find(|m| m.msg_type == MessageType::DataLayout)
|
||||||
|
.ok_or("MissingMessage(DataLayout)")?;
|
||||||
|
let dl = DataLayout::parse(&lm.data, self.os, self.ls).map_err(e)?;
|
||||||
|
// What libhdf5 checks when it opens the dataset (as File::dataset).
|
||||||
|
data_read::check_dataset_storage(&dl, &ds, &dt, self.data.len() as u64).map_err(e)?;
|
||||||
|
let (shape, n) = Self::shape(&ds);
|
||||||
|
rec.insert("shape".into(), shape);
|
||||||
|
if n.saturating_mul(dt.type_size() as u64) > MAX_BYTES {
|
||||||
|
rec.insert("skipped".into(), Value::String("too large".into()));
|
||||||
|
return Ok(());
|
||||||
|
}
|
||||||
|
rec.insert(
|
||||||
|
"layout".into(),
|
||||||
|
Value::String(
|
||||||
|
match &dl {
|
||||||
|
DataLayout::Compact { .. } => "compact",
|
||||||
|
DataLayout::Contiguous { .. } => "contiguous",
|
||||||
|
DataLayout::Chunked { .. } => "chunked",
|
||||||
|
DataLayout::Virtual { .. } => "virtual",
|
||||||
|
}
|
||||||
|
.into(),
|
||||||
|
),
|
||||||
|
);
|
||||||
|
let pipeline = match self.payload(h, MessageType::FilterPipeline)? {
|
||||||
|
Some(p) => Some(FilterPipeline::parse(&p).map_err(e)?),
|
||||||
|
None => None,
|
||||||
|
};
|
||||||
|
if let Some(p) = &pipeline {
|
||||||
|
rec.insert(
|
||||||
|
"filters".into(),
|
||||||
|
json!(p.filters.iter().map(|f| f.filter_id).collect::<Vec<_>>()),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
let raw = if matches!(dl, DataLayout::Virtual { .. }) {
|
||||||
|
let resolver = self.vds_resolver();
|
||||||
|
let fill = clawhdf5_format::fill_value::dataset_fill_value_in(
|
||||||
|
self.data,
|
||||||
|
&h.messages,
|
||||||
|
self.os,
|
||||||
|
self.ls,
|
||||||
|
)
|
||||||
|
.map_err(e)?;
|
||||||
|
clawhdf5_format::vds::read_virtual_dataset(
|
||||||
|
self.data,
|
||||||
|
&dl,
|
||||||
|
&ds,
|
||||||
|
&dt,
|
||||||
|
fill.as_deref(),
|
||||||
|
self.os,
|
||||||
|
self.ls,
|
||||||
|
Some(&resolver),
|
||||||
|
)
|
||||||
|
.map_err(e)?
|
||||||
|
.data
|
||||||
|
} else {
|
||||||
|
let cache = clawhdf5_format::chunk_cache::ChunkCache::new();
|
||||||
|
clawhdf5_format::fill_value::read_full_with_fill::<clawhdf5_format::error::FormatError>(
|
||||||
|
&h.messages,
|
||||||
|
self.data,
|
||||||
|
&dl,
|
||||||
|
&ds,
|
||||||
|
dt.type_size() as usize,
|
||||||
|
self.os,
|
||||||
|
self.ls,
|
||||||
|
|| {
|
||||||
|
data_read::read_raw_data_cached(
|
||||||
|
self.data,
|
||||||
|
&dl,
|
||||||
|
&ds,
|
||||||
|
&dt,
|
||||||
|
pipeline.as_ref(),
|
||||||
|
self.os,
|
||||||
|
self.ls,
|
||||||
|
&cache,
|
||||||
|
)
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.map_err(e)?
|
||||||
|
};
|
||||||
|
self.hash_values(&dt, &raw, n, rec)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn attrs(&self, h: &ObjectHeader) -> Result<Map<String, Value>, String> {
|
||||||
|
let msgs = extract_attributes_full(self.data, h, self.os, self.ls).map_err(e)?;
|
||||||
|
let mut out = Map::new();
|
||||||
|
for a in &msgs {
|
||||||
|
let r = guarded(|| {
|
||||||
|
let mut rec = Map::new();
|
||||||
|
rec.insert("dtype".into(), Value::String(dtype_str(&a.datatype)));
|
||||||
|
let (shape, n) = Self::shape(&a.dataspace);
|
||||||
|
rec.insert("shape".into(), shape);
|
||||||
|
self.hash_values(&a.datatype, &a.raw_data, n, &mut rec)?;
|
||||||
|
Ok(rec)
|
||||||
|
});
|
||||||
|
let v = match r {
|
||||||
|
Ok(rec) => Value::Object(rec),
|
||||||
|
Err(msg) => json!({ "error": msg }),
|
||||||
|
};
|
||||||
|
out.insert(a.name.clone(), v);
|
||||||
|
}
|
||||||
|
Ok(out)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn entries(&self, h: &ObjectHeader) -> Result<Vec<GroupEntry>, String> {
|
||||||
|
let v1 = h
|
||||||
|
.messages
|
||||||
|
.iter()
|
||||||
|
.find(|m| m.msg_type == MessageType::SymbolTable);
|
||||||
|
if let Some(m) = v1 {
|
||||||
|
let stm = SymbolTableMessage::parse(&m.data, self.os).map_err(e)?;
|
||||||
|
group_v1::resolve_v1_group_entries(self.data, &stm, self.os, self.ls).map_err(e)
|
||||||
|
} else if h
|
||||||
|
.messages
|
||||||
|
.iter()
|
||||||
|
.any(|m| m.msg_type == MessageType::LinkInfo || m.msg_type == MessageType::Link)
|
||||||
|
{
|
||||||
|
group_v2::resolve_v2_group_entries(self.data, h, self.os, self.ls).map_err(e)
|
||||||
|
} else {
|
||||||
|
Ok(Vec::new())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Element bytes as an unsigned integer (at most 16 bytes), honouring byte order.
|
||||||
|
fn element_bits(b: &[u8], byte_order: &DatatypeByteOrder) -> Result<u128, String> {
|
||||||
|
if b.len() > 16 {
|
||||||
|
return Err(format!("canon: {}-byte numeric element", b.len()));
|
||||||
|
}
|
||||||
|
let mut v = 0u128;
|
||||||
|
match byte_order {
|
||||||
|
DatatypeByteOrder::LittleEndian => {
|
||||||
|
for (i, x) in b.iter().enumerate() {
|
||||||
|
v |= u128::from(*x) << (8 * i);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
DatatypeByteOrder::BigEndian => {
|
||||||
|
for x in b {
|
||||||
|
v = (v << 8) | u128::from(*x);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
DatatypeByteOrder::Vax => return Err("canon: VAX byte order".into()),
|
||||||
|
}
|
||||||
|
Ok(v)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn field(v: u128, pos: u32, len: u32) -> u128 {
|
||||||
|
if len == 0 || pos >= 128 {
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
let v = v >> pos;
|
||||||
|
if len >= 128 {
|
||||||
|
v
|
||||||
|
} else {
|
||||||
|
v & ((1u128 << len) - 1)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// True when a float's bit fields are exactly IEEE 754 binary16/32/64 for its
|
||||||
|
/// size. h5py hands back such a type's bytes untouched; any other layout (an
|
||||||
|
/// N-Bit `H5Tset_precision` float, say) is *converted* by libhdf5 into the
|
||||||
|
/// numpy float of the same size, so comparing raw bytes would be meaningless.
|
||||||
|
fn ieee_layout(dt: &Datatype) -> bool {
|
||||||
|
let Datatype::FloatingPoint {
|
||||||
|
size,
|
||||||
|
bit_offset,
|
||||||
|
bit_precision,
|
||||||
|
exponent_location,
|
||||||
|
exponent_size,
|
||||||
|
mantissa_location,
|
||||||
|
mantissa_size,
|
||||||
|
exponent_bias,
|
||||||
|
..
|
||||||
|
} = dt
|
||||||
|
else {
|
||||||
|
return true;
|
||||||
|
};
|
||||||
|
let std = match size {
|
||||||
|
2 => (16, 10, 5, 10, 15),
|
||||||
|
4 => (32, 23, 8, 23, 127),
|
||||||
|
8 => (64, 52, 11, 52, 1023),
|
||||||
|
_ => return true, // no same-size numpy float to convert to: compare raw
|
||||||
|
};
|
||||||
|
*bit_offset == 0
|
||||||
|
&& (
|
||||||
|
*bit_precision,
|
||||||
|
*exponent_location,
|
||||||
|
*exponent_size,
|
||||||
|
*mantissa_size,
|
||||||
|
*exponent_bias,
|
||||||
|
) == (std.0, std.1, std.2, std.3, std.4)
|
||||||
|
&& *mantissa_location == 0
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Canonicalise a non-IEEE-layout float the way libhdf5's float->float
|
||||||
|
/// conversion presents it to h5py: as the IEEE float of the same size.
|
||||||
|
/// Assumes the implied-leading-one normalisation and the sign bit at the top
|
||||||
|
/// of the precision (what `H5Tset_precision` produces; the parser does not
|
||||||
|
/// keep either field).
|
||||||
|
fn canon_custom_float(dt: &Datatype, b: &[u8], out: &mut Vec<u8>) -> Result<(), String> {
|
||||||
|
let Datatype::FloatingPoint {
|
||||||
|
size,
|
||||||
|
byte_order,
|
||||||
|
bit_offset,
|
||||||
|
bit_precision,
|
||||||
|
exponent_location,
|
||||||
|
exponent_size,
|
||||||
|
mantissa_location,
|
||||||
|
mantissa_size,
|
||||||
|
exponent_bias,
|
||||||
|
} = dt
|
||||||
|
else {
|
||||||
|
unreachable!()
|
||||||
|
};
|
||||||
|
let (esize, msize) = (u32::from(*exponent_size), u32::from(*mantissa_size));
|
||||||
|
if esize == 0 || esize > 30 || msize > 64 {
|
||||||
|
return Err(format!("canon: unsupported float layout e{esize} m{msize}"));
|
||||||
|
}
|
||||||
|
let v = element_bits(b, byte_order)?;
|
||||||
|
let sign_pos = (u32::from(*bit_offset) + u32::from(*bit_precision)).saturating_sub(1);
|
||||||
|
let neg = field(v, sign_pos, 1) == 1;
|
||||||
|
let e = field(v, u32::from(*exponent_location), esize) as i64;
|
||||||
|
let m = field(v, u32::from(*mantissa_location), msize);
|
||||||
|
let emax = (1i64 << esize) - 1;
|
||||||
|
let bias = i64::from(*exponent_bias);
|
||||||
|
let mag = if e == emax {
|
||||||
|
if m == 0 { f64::INFINITY } else { f64::NAN }
|
||||||
|
} else if e == 0 {
|
||||||
|
(m as f64) * 2f64.powi((1 - bias - msize as i64) as i32)
|
||||||
|
} else {
|
||||||
|
((1u128 << msize) as f64 + m as f64) * 2f64.powi((e - bias - msize as i64) as i32)
|
||||||
|
};
|
||||||
|
let x = if neg { -mag } else { mag };
|
||||||
|
match size {
|
||||||
|
2 => out
|
||||||
|
.extend_from_slice(&clawhdf5_format::float16::f32_to_f16_bits(x as f32).to_le_bytes()),
|
||||||
|
4 => out.extend_from_slice(&(x as f32).to_le_bytes()),
|
||||||
|
8 => out.extend_from_slice(&x.to_le_bytes()),
|
||||||
|
_ => unreachable!("ieee_layout keeps other sizes raw"),
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Integers stored with a bit offset or reduced precision (N-Bit): libhdf5
|
||||||
|
/// converts them to the full-width integer of the same size, shifting the
|
||||||
|
/// value down and sign-extending from the top precision bit.
|
||||||
|
fn canon_partial_int(dt: &Datatype, b: &[u8], out: &mut Vec<u8>) -> Result<(), String> {
|
||||||
|
let Datatype::FixedPoint {
|
||||||
|
size,
|
||||||
|
byte_order,
|
||||||
|
signed,
|
||||||
|
bit_offset,
|
||||||
|
bit_precision,
|
||||||
|
} = dt
|
||||||
|
else {
|
||||||
|
unreachable!()
|
||||||
|
};
|
||||||
|
let prec = u32::from(*bit_precision);
|
||||||
|
let v = element_bits(b, byte_order)?;
|
||||||
|
let mut x = field(v, u32::from(*bit_offset), prec);
|
||||||
|
if *signed && prec > 0 && prec < 128 && field(x, prec - 1, 1) == 1 {
|
||||||
|
x |= !0u128 << prec;
|
||||||
|
}
|
||||||
|
out.extend_from_slice(&x.to_le_bytes()[..*size as usize]);
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn partial_int(dt: &Datatype) -> bool {
|
||||||
|
matches!(dt, Datatype::FixedPoint { size, bit_offset, bit_precision, .. }
|
||||||
|
if *bit_offset != 0 || u32::from(*bit_precision) != size * 8)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn canon_str(b: &[u8], out: &mut Vec<u8>) {
|
||||||
|
let cut = b.iter().position(|&c| c == 0).unwrap_or(b.len());
|
||||||
|
let mut s = &b[..cut];
|
||||||
|
while let [rest @ .., b' '] = s {
|
||||||
|
s = rest;
|
||||||
|
}
|
||||||
|
out.push(b'S');
|
||||||
|
out.extend_from_slice(&(s.len() as u32).to_le_bytes());
|
||||||
|
out.extend_from_slice(s);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn hex(b: &[u8]) -> String {
|
||||||
|
b.iter().map(|x| format!("{x:02x}")).collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn dtype_str(dt: &Datatype) -> String {
|
||||||
|
match dt {
|
||||||
|
Datatype::FixedPoint {
|
||||||
|
size,
|
||||||
|
signed,
|
||||||
|
byte_order,
|
||||||
|
..
|
||||||
|
} => {
|
||||||
|
format!(
|
||||||
|
"{}{}{}",
|
||||||
|
bo(byte_order),
|
||||||
|
if *signed { "i" } else { "u" },
|
||||||
|
size
|
||||||
|
)
|
||||||
|
}
|
||||||
|
Datatype::FloatingPoint {
|
||||||
|
size, byte_order, ..
|
||||||
|
} => format!("{}f{}", bo(byte_order), size),
|
||||||
|
Datatype::BitField {
|
||||||
|
size, byte_order, ..
|
||||||
|
} => format!("{}b{}", bo(byte_order), size),
|
||||||
|
Datatype::Time { size, .. } => format!("time{size}"),
|
||||||
|
Datatype::String { size, .. } => format!("S{size}"),
|
||||||
|
Datatype::Opaque { size, .. } => format!("V{size}"),
|
||||||
|
Datatype::Compound { size, members } => format!(
|
||||||
|
"{{{}}}{size}",
|
||||||
|
members
|
||||||
|
.iter()
|
||||||
|
.map(|m| format!("{}:{}", m.name, dtype_str(&m.datatype)))
|
||||||
|
.collect::<Vec<_>>()
|
||||||
|
.join(",")
|
||||||
|
),
|
||||||
|
Datatype::Reference { ref_type, .. } => format!("ref({ref_type:?})"),
|
||||||
|
Datatype::Enumeration { base_type, .. } => format!("enum({})", dtype_str(base_type)),
|
||||||
|
Datatype::VariableLength {
|
||||||
|
is_string: true, ..
|
||||||
|
} => "vlstr".into(),
|
||||||
|
Datatype::VariableLength { base_type, .. } => format!("vlen({})", dtype_str(base_type)),
|
||||||
|
Datatype::Array {
|
||||||
|
base_type,
|
||||||
|
dimensions,
|
||||||
|
} => format!("({}){dimensions:?}", dtype_str(base_type)),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn bo(b: &DatatypeByteOrder) -> &'static str {
|
||||||
|
match b {
|
||||||
|
DatatypeByteOrder::LittleEndian => "<",
|
||||||
|
DatatypeByteOrder::BigEndian => ">",
|
||||||
|
DatatypeByteOrder::Vax => "vax",
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn is_group(h: &ObjectHeader) -> bool {
|
||||||
|
h.messages.iter().any(|m| {
|
||||||
|
matches!(
|
||||||
|
m.msg_type,
|
||||||
|
MessageType::LinkInfo | MessageType::Link | MessageType::SymbolTable
|
||||||
|
)
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The probe's kind for an object header: libhdf5's object class
|
||||||
|
/// ([`ObjectHeader::object_class`]: group, then dataset — a datatype *and* a
|
||||||
|
/// dataspace — then named datatype), which is what h5py opens the object as.
|
||||||
|
/// The root group, and a header with only link messages, count as groups.
|
||||||
|
fn kind_of(h: &ObjectHeader, is_root: bool) -> &'static str {
|
||||||
|
match h.object_class() {
|
||||||
|
Some(ObjectClass::Group) => "group",
|
||||||
|
Some(ObjectClass::Dataset) => "dataset",
|
||||||
|
_ if is_root || is_group(h) => "group",
|
||||||
|
Some(ObjectClass::NamedDatatype) => "datatype",
|
||||||
|
None => "unknown",
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn main() {
|
||||||
|
install_hook();
|
||||||
|
let path = std::env::args().nth(1).expect("usage: probe <file>");
|
||||||
|
let mut top = Map::new();
|
||||||
|
top.insert("file".into(), Value::String(path.clone()));
|
||||||
|
let data = match std::fs::read(&path) {
|
||||||
|
Ok(d) => d,
|
||||||
|
Err(err) => {
|
||||||
|
top.insert("open_error".into(), Value::String(format!("Io({err})")));
|
||||||
|
println!("{}", Value::Object(top));
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
// Every address is relative to the superblock: look at the file from
|
||||||
|
// there on (past any user block), as libhdf5 does.
|
||||||
|
let hdf5: &[u8] = match signature::find_signature(&data) {
|
||||||
|
Ok(off) => &data[off..],
|
||||||
|
Err(_) => &data,
|
||||||
|
};
|
||||||
|
let sb = guarded(|| Superblock::parse(hdf5, 0).map_err(e));
|
||||||
|
let sb = match sb {
|
||||||
|
Ok(sb) => sb,
|
||||||
|
Err(msg) => {
|
||||||
|
top.insert("open_error".into(), Value::String(msg));
|
||||||
|
println!("{}", Value::Object(top));
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
// libhdf5 refuses a truncated file and reads nothing past the recorded
|
||||||
|
// end of file.
|
||||||
|
let base = (data.len() - hdf5.len()) as u64;
|
||||||
|
let hdf5 = match sb.data_end(base, data.len() as u64) {
|
||||||
|
Ok(end) => &hdf5[..end as usize],
|
||||||
|
Err(err) => {
|
||||||
|
top.insert("open_error".into(), Value::String(e(err)));
|
||||||
|
println!("{}", Value::Object(top));
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
// libhdf5 decodes the superblock extension at open (an error refuses
|
||||||
|
// the file), and loads a metadata cache image over the file's own
|
||||||
|
// metadata. It loads the image only when it first reads metadata — the
|
||||||
|
// root group — so a file whose image it cannot load still opens and
|
||||||
|
// that read fails. The library decides all three cases with the same
|
||||||
|
// `cache_image_state`: `File` and `MmapFile` open such a file and fail
|
||||||
|
// every object lookup with the image's error, which is what the probe
|
||||||
|
// records here (on the root group, where libhdf5 reports it).
|
||||||
|
use clawhdf5_format::superblock_ext::{self, CacheImageState};
|
||||||
|
let state = match guarded(|| superblock_ext::cache_image_state(hdf5, &sb).map_err(e)) {
|
||||||
|
Ok(x) => x,
|
||||||
|
Err(msg) => {
|
||||||
|
top.insert("open_error".into(), Value::String(msg));
|
||||||
|
println!("{}", Value::Object(top));
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let mut image_error = None;
|
||||||
|
let view = match state {
|
||||||
|
CacheImageState::Absent => None,
|
||||||
|
CacheImageState::Unloadable(err) => {
|
||||||
|
image_error = Some(e(err));
|
||||||
|
None
|
||||||
|
}
|
||||||
|
CacheImageState::Loaded(image) => {
|
||||||
|
let mut v = hdf5.to_vec();
|
||||||
|
match image.block(hdf5).and_then(|b| image.apply(b, &mut v)) {
|
||||||
|
Ok(()) => Some(v),
|
||||||
|
Err(err) => {
|
||||||
|
image_error = Some(e(err));
|
||||||
|
None
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let hdf5: &[u8] = view.as_deref().unwrap_or(hdf5);
|
||||||
|
top.insert("superblock_version".into(), json!(sb.version));
|
||||||
|
let ctx = Ctx {
|
||||||
|
data: hdf5,
|
||||||
|
os: sb.offset_size,
|
||||||
|
ls: sb.length_size,
|
||||||
|
base_dir: std::path::Path::new(&path)
|
||||||
|
.parent()
|
||||||
|
.map(|p| p.to_path_buf())
|
||||||
|
.unwrap_or_default(),
|
||||||
|
vl: RefCell::new(VlResolver::new(hdf5, sb.offset_size, sb.length_size)),
|
||||||
|
};
|
||||||
|
let mut objects: Vec<Value> = Vec::new();
|
||||||
|
let mut visited = HashSet::new();
|
||||||
|
let mut soft_v1 = 0u64;
|
||||||
|
// explicit DFS stack: (address, path)
|
||||||
|
let mut stack: Vec<(u64, String)> = vec![(sb.root_group_address, "/".to_string())];
|
||||||
|
while let Some((addr, p)) = stack.pop() {
|
||||||
|
if objects.len() >= MAX_OBJECTS {
|
||||||
|
top.insert("truncated".into(), json!(true));
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
if !visited.insert(addr) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let mut rec = Map::new();
|
||||||
|
rec.insert("path".into(), Value::String(p.clone()));
|
||||||
|
let r = guarded(|| {
|
||||||
|
if let Some(msg) = &image_error {
|
||||||
|
return Err(msg.clone());
|
||||||
|
}
|
||||||
|
let h = ctx.header(addr)?;
|
||||||
|
Ok(h)
|
||||||
|
});
|
||||||
|
let h = match r {
|
||||||
|
Ok(h) => h,
|
||||||
|
Err(msg) => {
|
||||||
|
rec.insert("kind".into(), Value::String("unknown".into()));
|
||||||
|
rec.insert("error".into(), Value::String(msg));
|
||||||
|
objects.push(Value::Object(rec));
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let kind = kind_of(&h, addr == sb.root_group_address);
|
||||||
|
rec.insert("kind".into(), Value::String(kind.into()));
|
||||||
|
if kind == "dataset"
|
||||||
|
&& let Err(msg) = guarded(|| ctx.read_dataset(&h, &mut rec))
|
||||||
|
{
|
||||||
|
rec.insert("error".into(), Value::String(msg));
|
||||||
|
}
|
||||||
|
// Opening a committed datatype decodes it (h5py's `f[name]` fails on
|
||||||
|
// one libhdf5 cannot decode), so decode it here too.
|
||||||
|
if kind == "datatype"
|
||||||
|
&& let Err(msg) = guarded(|| ctx.read_named_datatype(&h))
|
||||||
|
{
|
||||||
|
rec.insert("error".into(), Value::String(msg));
|
||||||
|
}
|
||||||
|
if kind != "datatype" {
|
||||||
|
match guarded(|| ctx.attrs(&h)) {
|
||||||
|
Ok(m) => {
|
||||||
|
rec.insert("attrs".into(), Value::Object(m));
|
||||||
|
}
|
||||||
|
Err(msg) => {
|
||||||
|
rec.insert("attrs_error".into(), Value::String(msg));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if kind == "group" {
|
||||||
|
match guarded(|| ctx.entries(&h)) {
|
||||||
|
Ok(mut ents) => {
|
||||||
|
ents.retain(|en| {
|
||||||
|
if en.cache_type == 2 {
|
||||||
|
soft_v1 += 1;
|
||||||
|
false
|
||||||
|
} else {
|
||||||
|
true
|
||||||
|
}
|
||||||
|
});
|
||||||
|
ents.sort_by(|a, b| a.name.cmp(&b.name));
|
||||||
|
let base = if p == "/" { String::new() } else { p.clone() };
|
||||||
|
for en in ents.into_iter().rev() {
|
||||||
|
stack.push((en.object_header_address, format!("{base}/{}", en.name)));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Err(msg) => {
|
||||||
|
rec.insert("list_error".into(), Value::String(msg));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
objects.push(Value::Object(rec));
|
||||||
|
}
|
||||||
|
if soft_v1 > 0 {
|
||||||
|
top.insert("v1_soft_link_entries".into(), json!(soft_v1));
|
||||||
|
}
|
||||||
|
top.insert("objects".into(), Value::Array(objects));
|
||||||
|
println!("{}", Value::Object(top));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
/// The N-Bit float of libhdf5's `test/testfiles/le_data.h5`
|
||||||
|
/// (`Nbit_float_data_le`): offset 7, precision 20, sign bit 26, exponent
|
||||||
|
/// 20+6 (bias 31), mantissa 7+13.
|
||||||
|
fn nbit_f32(byte_order: DatatypeByteOrder) -> Datatype {
|
||||||
|
Datatype::FloatingPoint {
|
||||||
|
size: 4,
|
||||||
|
byte_order,
|
||||||
|
bit_offset: 7,
|
||||||
|
bit_precision: 20,
|
||||||
|
exponent_location: 20,
|
||||||
|
exponent_size: 6,
|
||||||
|
mantissa_location: 7,
|
||||||
|
mantissa_size: 13,
|
||||||
|
exponent_bias: 31,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn canon_one(dt: &Datatype, bytes: &[u8]) -> Vec<u8> {
|
||||||
|
let mut out = Vec::new();
|
||||||
|
canon_custom_float(dt, bytes, &mut out).unwrap();
|
||||||
|
out
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn nbit_float_canonicalises_to_the_value_libhdf5_returns() {
|
||||||
|
let le = nbit_f32(DatatypeByteOrder::LittleEndian);
|
||||||
|
let be = nbit_f32(DatatypeByteOrder::BigEndian);
|
||||||
|
assert!(!ieee_layout(&le));
|
||||||
|
// 1.0: exponent = bias, mantissa 0
|
||||||
|
let one: u32 = 31 << 20;
|
||||||
|
assert_eq!(canon_one(&le, &one.to_le_bytes()), 1.0f32.to_le_bytes());
|
||||||
|
assert_eq!(canon_one(&be, &one.to_be_bytes()), 1.0f32.to_le_bytes());
|
||||||
|
// -2.1999512 (h5py's reading of the file's -2.2): sign, e = 32, m = 819
|
||||||
|
let v: u32 = (1 << 26) | (32 << 20) | (819 << 7);
|
||||||
|
assert_eq!(
|
||||||
|
canon_one(&le, &v.to_le_bytes()),
|
||||||
|
(-2.199_951_2f32).to_le_bytes()
|
||||||
|
);
|
||||||
|
assert_eq!(canon_one(&le, &[0; 4]), 0.0f32.to_le_bytes());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn ieee_floats_keep_their_raw_bytes() {
|
||||||
|
let f32le = Datatype::FloatingPoint {
|
||||||
|
size: 4,
|
||||||
|
byte_order: DatatypeByteOrder::LittleEndian,
|
||||||
|
bit_offset: 0,
|
||||||
|
bit_precision: 32,
|
||||||
|
exponent_location: 23,
|
||||||
|
exponent_size: 8,
|
||||||
|
mantissa_location: 0,
|
||||||
|
mantissa_size: 23,
|
||||||
|
exponent_bias: 127,
|
||||||
|
};
|
||||||
|
assert!(ieee_layout(&f32le));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn kind_follows_libhdf5_object_class() {
|
||||||
|
use clawhdf5_format::object_header::HeaderMessage;
|
||||||
|
let header = |types: &[MessageType]| ObjectHeader {
|
||||||
|
version: 2,
|
||||||
|
messages: types
|
||||||
|
.iter()
|
||||||
|
.map(|&msg_type| HeaderMessage {
|
||||||
|
msg_type,
|
||||||
|
size: 0,
|
||||||
|
flags: 0,
|
||||||
|
creation_order: None,
|
||||||
|
data: Vec::new(),
|
||||||
|
})
|
||||||
|
.collect(),
|
||||||
|
reference_count: None,
|
||||||
|
flags: 0,
|
||||||
|
access_time: None,
|
||||||
|
modification_time: None,
|
||||||
|
change_time: None,
|
||||||
|
birth_time: None,
|
||||||
|
};
|
||||||
|
use MessageType::*;
|
||||||
|
// cve-2024-33874 `/Dset1`: a datatype and a layout but no dataspace
|
||||||
|
// is a named datatype to libhdf5 (h5py opens it as one).
|
||||||
|
assert_eq!(kind_of(&header(&[Datatype, DataLayout]), false), "datatype");
|
||||||
|
assert_eq!(
|
||||||
|
kind_of(&header(&[Datatype, Dataspace, DataLayout]), false),
|
||||||
|
"dataset"
|
||||||
|
);
|
||||||
|
assert_eq!(kind_of(&header(&[SymbolTable]), false), "group");
|
||||||
|
assert_eq!(kind_of(&header(&[Link]), false), "group");
|
||||||
|
assert_eq!(kind_of(&header(&[]), true), "group");
|
||||||
|
assert_eq!(kind_of(&header(&[]), false), "unknown");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn partial_precision_int_is_shifted_and_sign_extended() {
|
||||||
|
let dt = Datatype::FixedPoint {
|
||||||
|
size: 4,
|
||||||
|
byte_order: DatatypeByteOrder::BigEndian,
|
||||||
|
signed: true,
|
||||||
|
bit_offset: 4,
|
||||||
|
bit_precision: 17,
|
||||||
|
};
|
||||||
|
assert!(partial_int(&dt));
|
||||||
|
let stored = (((-5i32) as u32) & 0x1_FFFF) << 4;
|
||||||
|
let mut out = Vec::new();
|
||||||
|
canon_partial_int(&dt, &stored.to_be_bytes(), &mut out).unwrap();
|
||||||
|
assert_eq!(out, (-5i32).to_le_bytes());
|
||||||
|
}
|
||||||
|
}
|
||||||
Executable
+325
@@ -0,0 +1,325 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Reference probe: same JSON as the Rust `conformance-probe`, produced with h5py.
|
||||||
|
|
||||||
|
Walk: iterative DFS from '/', children in sorted (UTF-8 byte) name order, hard
|
||||||
|
links only, each object once (first path wins, deduplicated by object identity).
|
||||||
|
Canonical value encoding: see harness/src/main.rs.
|
||||||
|
"""
|
||||||
|
import hashlib
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import struct
|
||||||
|
import sys
|
||||||
|
|
||||||
|
import numpy as np
|
||||||
|
import h5py
|
||||||
|
|
||||||
|
try:
|
||||||
|
import hdf5plugin # noqa: F401 registers blosc/lz4/zstd/bzip2/... filters
|
||||||
|
except Exception: # pragma: no cover
|
||||||
|
pass
|
||||||
|
|
||||||
|
MAX_BYTES = 200 * 1024 * 1024
|
||||||
|
MAX_OBJECTS = 200_000
|
||||||
|
|
||||||
|
|
||||||
|
def canon_str(b, out):
|
||||||
|
if isinstance(b, str):
|
||||||
|
b = b.encode("utf-8", "surrogateescape")
|
||||||
|
b = bytes(b)
|
||||||
|
cut = b.find(b"\x00")
|
||||||
|
if cut >= 0:
|
||||||
|
b = b[:cut]
|
||||||
|
b = b.rstrip(b" ")
|
||||||
|
out += b"S" + struct.pack("<I", len(b)) + b
|
||||||
|
|
||||||
|
|
||||||
|
def simple(dt):
|
||||||
|
if dt.fields:
|
||||||
|
return all(simple(dt.fields[n][0]) for n in dt.names)
|
||||||
|
if dt.subdtype:
|
||||||
|
return simple(dt.subdtype[0])
|
||||||
|
return dt.kind in "iufcbV"
|
||||||
|
|
||||||
|
|
||||||
|
def packed(dt):
|
||||||
|
if dt.fields:
|
||||||
|
return np.dtype([(n, packed(dt.fields[n][0])) for n in dt.names])
|
||||||
|
if dt.subdtype:
|
||||||
|
base, shape = dt.subdtype
|
||||||
|
return np.dtype((packed(base), shape))
|
||||||
|
if dt.kind in "iufcb":
|
||||||
|
return dt.newbyteorder("<")
|
||||||
|
return dt
|
||||||
|
|
||||||
|
|
||||||
|
# --- reference corrections ---------------------------------------------------
|
||||||
|
# Where h5py is known to return values the file does not hold, and the right
|
||||||
|
# values follow from what it returned, ref.py corrects them and records the
|
||||||
|
# correction on the object ("ref_fix"), so the comparison is still a real
|
||||||
|
# comparison and CONFORMANCE.md lists every corrected object. Each correction
|
||||||
|
# first checks that the installed h5py still has the bug.
|
||||||
|
|
||||||
|
# Corrections applied while encoding the current object.
|
||||||
|
FIXES = set()
|
||||||
|
_BE_VLEN_BUG = None
|
||||||
|
|
||||||
|
|
||||||
|
def be_vlen_bug():
|
||||||
|
"""h5py (3.16 / HDF5 2.0 at least) returns the elements of a
|
||||||
|
variable-length sequence whose base type is big-endian with the file's
|
||||||
|
big-endian bytes under a native (little-endian) dtype: a
|
||||||
|
`vlen_dtype('>f4')` dataset holding [1.0, 2.0] reads back as
|
||||||
|
[4.6e-41, 9.0e-44]. `h5dump` prints the file's values. Checked once per
|
||||||
|
process by writing and reading exactly that dataset in memory."""
|
||||||
|
global _BE_VLEN_BUG
|
||||||
|
if _BE_VLEN_BUG is None:
|
||||||
|
import io
|
||||||
|
try:
|
||||||
|
bio = io.BytesIO()
|
||||||
|
with h5py.File(bio, "w") as f:
|
||||||
|
d = f.create_dataset("v", (1,), dtype=h5py.vlen_dtype(np.dtype(">f4")))
|
||||||
|
d[0] = np.array([1.0, 2.0], dtype=">f4")
|
||||||
|
with h5py.File(bio, "r") as f:
|
||||||
|
got = np.asarray(f["v"][0])
|
||||||
|
_BE_VLEN_BUG = (got.dtype == np.dtype("<f4")
|
||||||
|
and got.view(">f4").tolist() == [1.0, 2.0]
|
||||||
|
and got.tolist() != [1.0, 2.0])
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
_BE_VLEN_BUG = False
|
||||||
|
return _BE_VLEN_BUG
|
||||||
|
|
||||||
|
|
||||||
|
def unswapped(got, base):
|
||||||
|
"""`got` is `base` (big-endian somewhere) with every field in native
|
||||||
|
little-endian order instead: the shape of h5py's big-endian VL bug."""
|
||||||
|
return base.newbyteorder("<") == got and base != got
|
||||||
|
|
||||||
|
|
||||||
|
def canon_el(dt, val, out):
|
||||||
|
if dt.fields:
|
||||||
|
for n in dt.names:
|
||||||
|
canon_el(dt.fields[n][0], val[n], out)
|
||||||
|
return
|
||||||
|
if dt.subdtype:
|
||||||
|
base, _ = dt.subdtype
|
||||||
|
for x in np.asarray(val).reshape(-1):
|
||||||
|
canon_el(base, x, out)
|
||||||
|
return
|
||||||
|
k = dt.kind
|
||||||
|
if k in "iufcb":
|
||||||
|
out += np.asarray(val, dtype=dt).astype(dt.newbyteorder("<")).tobytes()
|
||||||
|
elif k == "V":
|
||||||
|
out += np.asarray(val, dtype=dt).tobytes()
|
||||||
|
elif k == "S":
|
||||||
|
canon_str(val, out)
|
||||||
|
elif k == "O":
|
||||||
|
if h5py.check_string_dtype(dt) is not None:
|
||||||
|
canon_str(val if val is not None else b"", out)
|
||||||
|
elif h5py.check_ref_dtype(dt) is not None:
|
||||||
|
out += b"R"
|
||||||
|
else:
|
||||||
|
base = h5py.check_vlen_dtype(dt)
|
||||||
|
if base is None:
|
||||||
|
raise TypeError(f"unhandled object dtype {dt!r}")
|
||||||
|
arr = np.asarray(val if val is not None else [])
|
||||||
|
if arr.dtype != base and be_vlen_bug() and unswapped(arr.dtype, base):
|
||||||
|
# h5py's big-endian VL bug (see be_vlen_bug): the bytes are
|
||||||
|
# the file's, the dtype label is wrong. Relabel, don't convert.
|
||||||
|
arr = arr.view(base)
|
||||||
|
FIXES.add("h5py-be-vlen")
|
||||||
|
arr = np.asarray(arr, dtype=base).reshape(-1)
|
||||||
|
out += b"V" + struct.pack("<I", arr.shape[0])
|
||||||
|
if simple(base):
|
||||||
|
out += arr.astype(packed(base)).tobytes()
|
||||||
|
else:
|
||||||
|
for x in arr:
|
||||||
|
canon_el(base, x, out)
|
||||||
|
elif k == "U":
|
||||||
|
canon_str(str(val), out)
|
||||||
|
else:
|
||||||
|
raise TypeError(f"unhandled dtype kind {k} ({dt!r})")
|
||||||
|
|
||||||
|
|
||||||
|
def has_obj(dt):
|
||||||
|
if dt.fields:
|
||||||
|
return any(has_obj(dt.fields[n][0]) for n in dt.names)
|
||||||
|
if dt.subdtype:
|
||||||
|
return has_obj(dt.subdtype[0])
|
||||||
|
return dt.kind == "O"
|
||||||
|
|
||||||
|
|
||||||
|
def note_conversion(tid, dt, rec):
|
||||||
|
"""h5py converts some file types (FP8, bfloat16, x87 long double, ...) to a
|
||||||
|
different-sized numpy type; then value bytes are not comparable."""
|
||||||
|
try:
|
||||||
|
if not has_obj(dt) and tid.get_size() != dt.itemsize:
|
||||||
|
rec["converted"] = f"file type size {tid.get_size()} -> numpy {dt} ({dt.itemsize})"
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
def hash_values(arr, dt, rec):
|
||||||
|
# h5py expands an HDF5 array element type into trailing array dims, a
|
||||||
|
# nested array type (an array of arrays) into all of them. Converting the
|
||||||
|
# expanded array back to the inner subarray type would broadcast every
|
||||||
|
# element into a whole subarray, so strip every level.
|
||||||
|
while dt.subdtype is not None:
|
||||||
|
dt = dt.subdtype[0]
|
||||||
|
arr = np.asarray(arr, dtype=dt)
|
||||||
|
FIXES.clear()
|
||||||
|
if simple(dt):
|
||||||
|
c = np.ascontiguousarray(arr).astype(packed(dt)).tobytes()
|
||||||
|
else:
|
||||||
|
out = bytearray()
|
||||||
|
for x in arr.reshape(-1):
|
||||||
|
canon_el(dt, x, out)
|
||||||
|
c = bytes(out)
|
||||||
|
if FIXES:
|
||||||
|
rec["ref_fix"] = sorted(FIXES)
|
||||||
|
rec["hash"] = hashlib.sha256(c).hexdigest()
|
||||||
|
rec["head"] = c[:48].hex()
|
||||||
|
|
||||||
|
|
||||||
|
def err(e):
|
||||||
|
s = f"{type(e).__name__}: {e}"
|
||||||
|
return s.splitlines()[0][:400] if s else type(e).__name__
|
||||||
|
|
||||||
|
|
||||||
|
def shape_of(s):
|
||||||
|
return "null" if s is None else list(s)
|
||||||
|
|
||||||
|
|
||||||
|
def n_bytes(shape, tid):
|
||||||
|
n = 1
|
||||||
|
for d in shape or ():
|
||||||
|
n *= d
|
||||||
|
return n * tid.get_size()
|
||||||
|
|
||||||
|
|
||||||
|
def read_attrs(obj):
|
||||||
|
out = {}
|
||||||
|
names = sorted(obj.attrs.keys(), key=lambda s: s.encode("utf-8", "surrogateescape"))
|
||||||
|
for name in names:
|
||||||
|
rec = {}
|
||||||
|
try:
|
||||||
|
aid = obj.attrs.get_id(name)
|
||||||
|
rec["dtype"] = str(aid.dtype)
|
||||||
|
rec["shape"] = shape_of(aid.shape)
|
||||||
|
note_conversion(aid.get_type(), aid.dtype, rec)
|
||||||
|
if aid.shape is None:
|
||||||
|
hash_values(np.empty((0,), dtype=aid.dtype), aid.dtype, rec)
|
||||||
|
else:
|
||||||
|
val = obj.attrs[name]
|
||||||
|
hash_values(val, aid.dtype, rec)
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
rec = {"error": err(e)}
|
||||||
|
out[name] = rec
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def main(path):
|
||||||
|
top = {"file": path}
|
||||||
|
try:
|
||||||
|
f = h5py.File(path, "r")
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
top["open_error"] = err(e)
|
||||||
|
print(json.dumps(top))
|
||||||
|
return
|
||||||
|
objects = []
|
||||||
|
seen = set()
|
||||||
|
# Objects h5py cannot open have no ObjectID to deduplicate by; they are
|
||||||
|
# deduplicated by the address their hard link points at instead, as the
|
||||||
|
# probe deduplicates every object by header address.
|
||||||
|
seen_unopenable = set()
|
||||||
|
stack = [("/", None, None)]
|
||||||
|
while stack:
|
||||||
|
p, obj, link_addr = stack.pop()
|
||||||
|
if len(objects) >= MAX_OBJECTS:
|
||||||
|
top["truncated"] = True
|
||||||
|
break
|
||||||
|
rec = {"path": p}
|
||||||
|
try:
|
||||||
|
if obj is None:
|
||||||
|
obj = f[p]
|
||||||
|
key = hash(obj.id) # h5py ObjectID hash = (fileno, object address/token)
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
if link_addr is not None:
|
||||||
|
if link_addr in seen_unopenable:
|
||||||
|
continue
|
||||||
|
seen_unopenable.add(link_addr)
|
||||||
|
rec["kind"] = "unknown"
|
||||||
|
rec["error"] = err(e)
|
||||||
|
objects.append(rec)
|
||||||
|
continue
|
||||||
|
if key in seen:
|
||||||
|
continue
|
||||||
|
seen.add(key)
|
||||||
|
if isinstance(obj, h5py.Dataset):
|
||||||
|
kind = "dataset"
|
||||||
|
elif isinstance(obj, h5py.Group):
|
||||||
|
kind = "group"
|
||||||
|
elif isinstance(obj, h5py.Datatype):
|
||||||
|
kind = "datatype"
|
||||||
|
else:
|
||||||
|
kind = "unknown"
|
||||||
|
rec["kind"] = kind
|
||||||
|
if kind == "dataset":
|
||||||
|
try:
|
||||||
|
dt = obj.dtype
|
||||||
|
rec["dtype"] = str(dt)
|
||||||
|
rec["shape"] = shape_of(obj.shape)
|
||||||
|
note_conversion(obj.id.get_type(), dt, rec)
|
||||||
|
if obj.shape is None:
|
||||||
|
hash_values(np.empty((0,), dtype=dt), dt, rec)
|
||||||
|
elif n_bytes(obj.shape, obj.id.get_type()) > MAX_BYTES:
|
||||||
|
rec["skipped"] = "too large"
|
||||||
|
else:
|
||||||
|
arr = np.empty(obj.shape, dtype=dt)
|
||||||
|
if arr.size:
|
||||||
|
try:
|
||||||
|
obj.read_direct(arr)
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
arr = obj[()]
|
||||||
|
hash_values(arr, dt, rec)
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
rec["error"] = err(e)
|
||||||
|
if kind != "datatype":
|
||||||
|
try:
|
||||||
|
rec["attrs"] = read_attrs(obj)
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
rec["attrs_error"] = err(e)
|
||||||
|
if kind == "group":
|
||||||
|
try:
|
||||||
|
names = sorted(obj.keys(), key=lambda s: s.encode("utf-8", "surrogateescape"))
|
||||||
|
base = "" if p == "/" else p
|
||||||
|
kids = []
|
||||||
|
for n in names:
|
||||||
|
# The link's own type: `obj.get(n, getlink=True)` reports
|
||||||
|
# a user-defined link (type 64-255) as a HardLink.
|
||||||
|
try:
|
||||||
|
info = obj.id.links.get_info(n.encode("utf-8", "surrogateescape"))
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
info = None
|
||||||
|
if info is not None and info.type != h5py.h5l.TYPE_HARD:
|
||||||
|
continue
|
||||||
|
addr = info.u if info is not None else None
|
||||||
|
kids.append((f"{base}/{n}", addr))
|
||||||
|
for k, addr in reversed(kids):
|
||||||
|
stack.append((k, None, addr))
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
rec["list_error"] = err(e)
|
||||||
|
objects.append(rec)
|
||||||
|
top["objects"] = objects
|
||||||
|
print(json.dumps(top), flush=True)
|
||||||
|
# Exit without tearing down the h5py objects: freeing them for some files
|
||||||
|
# that hold references (hdf5's h5repack_attr_refs.h5, cve-2024-32623.h5)
|
||||||
|
# makes libhdf5 2.0 abort with "free(): chunks in smallbin corrupted"
|
||||||
|
# about half the time. That happens after the reading is done, so it says
|
||||||
|
# nothing about what h5py read, but it flipped those files between ok and
|
||||||
|
# h5py-cannot-read from one run to the next.
|
||||||
|
os._exit(0)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main(sys.argv[1])
|
||||||
@@ -0,0 +1,105 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""ref_bugs.py <corpus_dir>: re-check the objects h5py reads only through a
|
||||||
|
libhdf5 bug.
|
||||||
|
|
||||||
|
For each object of READ_BUGS (below), h5py reads it in several fresh
|
||||||
|
processes whose heaps differ: h5py imported before numpy (three runs, plus
|
||||||
|
two with glibc's MALLOC_PERTURB_, which fills newly allocated and freed heap
|
||||||
|
blocks with a byte pattern) and numpy imported first. Values the file
|
||||||
|
determines come out the same every time. An object whose values differ
|
||||||
|
between those runs is read from memory the file does not determine — an
|
||||||
|
over-read or an uninitialised buffer in libhdf5 — so the values h5py reports
|
||||||
|
for it are not the file's, and clawhdf5 refusing the object is not a
|
||||||
|
clawhdf5 error. compare.py classifies a file as `ref-bug` only on objects
|
||||||
|
confirmed that way in the same run (`$OUT/ref_bugs.json`); an object whose
|
||||||
|
reading turns out stable stays an our-error.
|
||||||
|
|
||||||
|
Run by conformance/run.sh; on its own it is the reproducer (JSON on stdout).
|
||||||
|
"""
|
||||||
|
import concurrent.futures
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import subprocess
|
||||||
|
import sys
|
||||||
|
|
||||||
|
# (file, object) -> what goes wrong. Checked 2026-09-27 against HDF5 2.0.0
|
||||||
|
# (h5py 3.16), h5dump 1.14.6 and the HDFGroup/hdf5 sources (tag hdf5_1_14_6
|
||||||
|
# and develop); see docs/known-issues.md, "Conformance: the last non-ok files".
|
||||||
|
READ_BUGS = {
|
||||||
|
("cve_hdf5/cvefiles/cve-2025-2308.h5", "/Scale_offset_long_long_data_le"):
|
||||||
|
"the first chunk records minbits 11: its 12 values need 17 bytes of codes, and the "
|
||||||
|
"26-byte chunk holds 5 after its 21-byte header; libhdf5's scale-offset decoder reads "
|
||||||
|
"past its buffer, and develop refuses the chunk (\"Buffer too short\")",
|
||||||
|
("cve_hdf5/cvefiles/cve-2025-44904.h5", "/Scale_offset_float_data_le"):
|
||||||
|
"unfiltered chunks stored as 38 and 37 bytes for 48-byte chunks: 1.14/2.0 read the "
|
||||||
|
"stored bytes into a buffer of that size and use it as the whole chunk "
|
||||||
|
"(H5D__chunk_lock), so the rest is heap memory; develop refuses them (\"incorrect chunk "
|
||||||
|
"size returned from index for unfiltered chunk\")",
|
||||||
|
("hdf5/test/testfiles/bad_nbit_parms_walk.h5", "/Nbit_int_data_le"):
|
||||||
|
"the N-Bit parameter list holds 7 values (cd_values[0] = 7) where an integer needs 8: "
|
||||||
|
"the decoder takes the bit offset from cd_values[7], past the list; libhdf5's own test "
|
||||||
|
"(`test_filter_bad_params`, test/dsets.c on develop) requires the read to fail",
|
||||||
|
}
|
||||||
|
|
||||||
|
# (which module is imported first, MALLOC_PERTURB_)
|
||||||
|
RUNS = [("h5py", None), ("h5py", None), ("h5py", None), ("h5py", "170"), ("h5py", "255"),
|
||||||
|
("numpy", None)]
|
||||||
|
|
||||||
|
READ = r"""
|
||||||
|
import hashlib, sys
|
||||||
|
if sys.argv[3] == "h5py":
|
||||||
|
import h5py, numpy as np
|
||||||
|
else:
|
||||||
|
import numpy as np, h5py
|
||||||
|
try:
|
||||||
|
import hdf5plugin # noqa: F401
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
try:
|
||||||
|
with h5py.File(sys.argv[1], "r") as f:
|
||||||
|
a = np.ascontiguousarray(f[sys.argv[2]][()])
|
||||||
|
print("values " + hashlib.sha256(a.tobytes()).hexdigest()[:16])
|
||||||
|
except Exception as e:
|
||||||
|
print("error " + (str(e).splitlines() or [type(e).__name__])[0][:120])
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
def read_once(path, obj, first, perturb):
|
||||||
|
env = dict(os.environ)
|
||||||
|
env.pop("MALLOC_PERTURB_", None)
|
||||||
|
if perturb:
|
||||||
|
env["MALLOC_PERTURB_"] = perturb
|
||||||
|
try:
|
||||||
|
p = subprocess.run([sys.executable, "-c", READ, path, obj, first], env=env,
|
||||||
|
capture_output=True, text=True, timeout=60)
|
||||||
|
out = p.stdout.strip().splitlines()
|
||||||
|
return out[-1] if out else f"exit {p.returncode}"
|
||||||
|
except subprocess.TimeoutExpired:
|
||||||
|
return "timeout"
|
||||||
|
|
||||||
|
|
||||||
|
def check(corpus, key):
|
||||||
|
f, obj = key
|
||||||
|
path = os.path.join(corpus, f)
|
||||||
|
rec = {"file": f, "object": obj, "why": READ_BUGS[key]}
|
||||||
|
if not os.path.exists(path):
|
||||||
|
return rec | {"missing": True, "confirmed": False}
|
||||||
|
runs = [{"first": a, "malloc_perturb": p, "outcome": read_once(path, obj, a, p)} for a, p in RUNS]
|
||||||
|
distinct = sorted({r["outcome"] for r in runs})
|
||||||
|
return rec | {
|
||||||
|
"runs": runs,
|
||||||
|
"distinct": len(distinct),
|
||||||
|
"confirmed": len(distinct) > 1 and any(o.startswith("values ") for o in distinct),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
corpus = sys.argv[1]
|
||||||
|
keys = list(READ_BUGS)
|
||||||
|
with concurrent.futures.ThreadPoolExecutor(max_workers=len(keys)) as ex:
|
||||||
|
out = list(ex.map(lambda k: check(corpus, k), keys))
|
||||||
|
print(json.dumps({"read_bugs": out}, indent=1))
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
@@ -0,0 +1,361 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""report.py <results_dir> <CONFORMANCE.md> <corpus_dir>
|
||||||
|
|
||||||
|
Render the sweep's results (compare.py's results.json plus the raw per-side
|
||||||
|
runs) as CONFORMANCE.md, and write <results_dir>/report-meta.json (commit,
|
||||||
|
date, versions) for check.py --update.
|
||||||
|
"""
|
||||||
|
import collections
|
||||||
|
import datetime
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import platform
|
||||||
|
|
||||||
|
import subprocess
|
||||||
|
import sys
|
||||||
|
|
||||||
|
import h5py
|
||||||
|
import numpy
|
||||||
|
|
||||||
|
try:
|
||||||
|
import hdf5plugin
|
||||||
|
HDF5PLUGIN = hdf5plugin.version
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
HDF5PLUGIN = "not installed"
|
||||||
|
|
||||||
|
R, OUT_MD, CORPUS = sys.argv[1], sys.argv[2], sys.argv[3]
|
||||||
|
HERE = os.path.dirname(os.path.abspath(__file__))
|
||||||
|
ROOT = os.path.dirname(HERE)
|
||||||
|
CLASSES = ["ok", "our-error", "mismatch", "h5py-cannot-read", "ref-bug", "panic", "hang", "crash", "oom"]
|
||||||
|
|
||||||
|
|
||||||
|
def sh(*cmd, cwd=ROOT):
|
||||||
|
try:
|
||||||
|
return subprocess.run(cmd, cwd=cwd, capture_output=True, text=True, timeout=30).stdout.strip()
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
return ""
|
||||||
|
|
||||||
|
|
||||||
|
def cpu_model():
|
||||||
|
try:
|
||||||
|
for ln in open("/proc/cpuinfo"):
|
||||||
|
if ln.startswith(("model name", "Model")):
|
||||||
|
return ln.split(":", 1)[1].strip()
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
return platform.processor() or "unknown"
|
||||||
|
|
||||||
|
|
||||||
|
def mem_gib():
|
||||||
|
try:
|
||||||
|
for ln in open("/proc/meminfo"):
|
||||||
|
if ln.startswith("MemTotal:"):
|
||||||
|
return f"{int(ln.split()[1]) / 1048576:.0f} GiB"
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
return "?"
|
||||||
|
|
||||||
|
|
||||||
|
res = json.load(open(os.path.join(R, "results.json")))
|
||||||
|
meta_run = json.load(open(os.path.join(R, "meta.json"))) if os.path.exists(os.path.join(R, "meta.json")) else {}
|
||||||
|
rows = res["rows"]
|
||||||
|
issues = res.get("issues", {})
|
||||||
|
|
||||||
|
# safe.directory: a checkout owned by another user (a container) is still ours to read
|
||||||
|
commit = sh("git", "-c", "safe.directory=*", "rev-parse", "HEAD") or os.environ.get("GITHUB_SHA", "unknown")
|
||||||
|
lib_dirty = sh("git", "-c", "safe.directory=*", "status", "--porcelain", "--", "crates", "Cargo.toml")
|
||||||
|
h5dump_v = sh("h5dump", "--version").replace("h5dump: ", "")
|
||||||
|
meta = {
|
||||||
|
"date": datetime.datetime.now(datetime.timezone.utc).strftime("%Y-%m-%d %H:%M UTC"),
|
||||||
|
"commit": commit + (" (library sources modified)" if lib_dirty else ""),
|
||||||
|
"reference": f"h5py {h5py.__version__} / HDF5 {h5py.version.hdf5_version}",
|
||||||
|
}
|
||||||
|
json.dump(meta, open(os.path.join(R, "report-meta.json"), "w"), indent=1)
|
||||||
|
|
||||||
|
pins = []
|
||||||
|
for ln in open(os.path.join(HERE, "corpus.txt")):
|
||||||
|
if ln.strip() and not ln.lstrip().startswith("#"):
|
||||||
|
name, url, rev, root, *_ = ln.split()
|
||||||
|
pins.append((name, url, rev, root))
|
||||||
|
|
||||||
|
by_corpus = collections.defaultdict(collections.Counter)
|
||||||
|
for r in rows:
|
||||||
|
by_corpus[r["corpus"]][r["class"]] += 1
|
||||||
|
total = collections.Counter(r["class"] for r in rows)
|
||||||
|
|
||||||
|
|
||||||
|
def ex_list(files, n=3):
|
||||||
|
s = ", ".join(f"`{f}`" for f in files[:n])
|
||||||
|
return s + (f" (+{len(files) - n} more)" if len(files) > n else "")
|
||||||
|
|
||||||
|
|
||||||
|
# --- reference bugs ---------------------------------------------------------
|
||||||
|
# ref_bugs.py's re-check of the objects h5py reads only through a libhdf5 bug
|
||||||
|
# (compare.py classifies on the confirmed ones), and the objects whose h5py
|
||||||
|
# values ref.py corrected (compare.py's ref_fixes).
|
||||||
|
try:
|
||||||
|
ref_bugs = json.load(open(os.path.join(R, "ref_bugs.json")))["read_bugs"]
|
||||||
|
except (OSError, ValueError, KeyError):
|
||||||
|
ref_bugs = []
|
||||||
|
ref_fixes = res.get("ref_fixes", [])
|
||||||
|
|
||||||
|
|
||||||
|
# --- the CVE corpus: clawhdf5 vs h5dump vs h5py ------------------------------
|
||||||
|
def side(run, name):
|
||||||
|
p = os.path.join(R, "runs", run, name)
|
||||||
|
if not os.path.exists(p + ".rc"):
|
||||||
|
return None
|
||||||
|
rc = int(open(p + ".rc").read().strip() or -1)
|
||||||
|
err = open(p + ".err", errors="replace").read()
|
||||||
|
try:
|
||||||
|
j = json.load(open(p + ".json"))
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
j = None
|
||||||
|
return rc, err, j
|
||||||
|
|
||||||
|
|
||||||
|
def outcome(s, rust=False):
|
||||||
|
"""-> (bucket, text). bucket in read / error / panic / crash / hang / oom."""
|
||||||
|
if s is None:
|
||||||
|
return "missing", "not run"
|
||||||
|
rc, err, j = s
|
||||||
|
if rc in (137, 124):
|
||||||
|
return "hang", "hang (killed at timeout)"
|
||||||
|
if "memory allocation of" in err or "MemoryError" in err or "bad_alloc" in err or "Cannot allocate" in err:
|
||||||
|
return "oom", "out of memory"
|
||||||
|
if rust and (rc == 101 or "PANIC:" in err):
|
||||||
|
return "panic", "panic"
|
||||||
|
if "overflowed its stack" in err:
|
||||||
|
return "crash", "stack overflow"
|
||||||
|
if rc == 139:
|
||||||
|
return "crash", "SIGSEGV"
|
||||||
|
if rc == 134:
|
||||||
|
return "crash", "SIGABRT" + (" (heap corruption)" if ("corrupted" in err or "free()" in err) else "")
|
||||||
|
if rc > 128:
|
||||||
|
return "crash", f"signal {rc - 128}"
|
||||||
|
if j is None:
|
||||||
|
return ("error", "error exit") if rc in (0, 1) else ("crash", f"exit {rc}")
|
||||||
|
if "open_error" in j:
|
||||||
|
return "error", "open error"
|
||||||
|
objs = j.get("objects", [])
|
||||||
|
ne = sum(1 for o in objs for k in ("error", "attrs_error", "list_error") if k in o)
|
||||||
|
ne += sum(1 for o in objs for a in (o.get("attrs") or {}).values() if "error" in a)
|
||||||
|
return "read", f"read {len(objs)} obj" + (f", {ne} errors" if ne else "")
|
||||||
|
|
||||||
|
|
||||||
|
def h5dump_outcome(s):
|
||||||
|
if s is None:
|
||||||
|
return "missing", "not run"
|
||||||
|
rc, err, _ = s
|
||||||
|
if rc in (137, 124):
|
||||||
|
return "hang", "hang (killed at timeout)"
|
||||||
|
if "memory allocation" in err or "Cannot allocate" in err:
|
||||||
|
return "oom", "out of memory"
|
||||||
|
if rc == 139:
|
||||||
|
return "crash", "SIGSEGV"
|
||||||
|
if rc == 134:
|
||||||
|
return "crash", "SIGABRT" + (" (heap corruption)" if ("corrupted" in err or "free()" in err) else "")
|
||||||
|
if rc > 128:
|
||||||
|
return "crash", f"signal {rc - 128}"
|
||||||
|
return ("read", "ok") if rc == 0 else ("error", "error exit")
|
||||||
|
|
||||||
|
|
||||||
|
cve_rows = []
|
||||||
|
buckets = {"clawhdf5": collections.Counter(), "h5dump": collections.Counter(), "h5py": collections.Counter()}
|
||||||
|
ours_panic = {r["file"] for r in rows if r["class"] == "panic"}
|
||||||
|
for r in rows:
|
||||||
|
if r["corpus"] != "cve_hdf5":
|
||||||
|
continue
|
||||||
|
run = r["file"].replace("/", "__")
|
||||||
|
o = outcome(side(run, "ours"), rust=True)
|
||||||
|
if o[0] == "read" and r["file"] in ours_panic:
|
||||||
|
o = ("panic", "caught panic")
|
||||||
|
p = outcome(side(run, "ref"))
|
||||||
|
d = h5dump_outcome(side(run, "h5dump"))
|
||||||
|
buckets["clawhdf5"][o[0]] += 1
|
||||||
|
buckets["h5py"][p[0]] += 1
|
||||||
|
buckets["h5dump"][d[0]] += 1
|
||||||
|
cve_rows.append((r["file"].split("/", 1)[1], d[1], p[1], o[1], r["class"]))
|
||||||
|
|
||||||
|
# --- render -----------------------------------------------------------------
|
||||||
|
L = []
|
||||||
|
w = L.append
|
||||||
|
w("# clawhdf5 conformance report")
|
||||||
|
w("")
|
||||||
|
w("Every HDF5 file of eight public corpora (pinned by commit) is read twice — by")
|
||||||
|
w("clawhdf5 (`conformance/probe`, the same `clawhdf5-format` calls the facade")
|
||||||
|
w("makes) and by h5py/libhdf5 (`conformance/ref.py`) — and the two readings are")
|
||||||
|
w("compared object by object: the set of hard-linked objects, each dataset's and")
|
||||||
|
w("attribute's shape, and a SHA-256 of its values in a canonical encoding. The")
|
||||||
|
w("CVE corpus is also run through `h5dump`. Each side runs under a timeout and an")
|
||||||
|
w("address-space limit, so a hang, crash or runaway allocation is recorded, not")
|
||||||
|
w("fatal. This file is generated by `conformance/run.sh`; do not edit it by hand.")
|
||||||
|
w("")
|
||||||
|
w("## Run")
|
||||||
|
w("")
|
||||||
|
w("| | |")
|
||||||
|
w("|---|---|")
|
||||||
|
w(f"| date | {meta['date']} |")
|
||||||
|
w(f"| clawhdf5 commit | `{meta['commit']}` |")
|
||||||
|
w(f"| machine | `{platform.node()}`: {cpu_model()}, {os.cpu_count()} CPUs, {mem_gib()}, {platform.system()} {platform.release()} {platform.machine()} |")
|
||||||
|
w(f"| command | `{os.environ.get('CONFORMANCE_CMD', 'conformance/run.sh')}` |")
|
||||||
|
w(f"| rustc | {sh('rustc', '-V')} |")
|
||||||
|
w(f"| reference | h5py {h5py.__version__}, HDF5 {h5py.version.hdf5_version}, numpy {numpy.__version__}, hdf5plugin {HDF5PLUGIN}, Python {platform.python_version()} |")
|
||||||
|
w(f"| h5dump | {h5dump_v} (CVE corpus only) |")
|
||||||
|
if meta_run:
|
||||||
|
w(f"| limits | {meta_run.get('timeout_s')} s timeout (SIGKILL), {int(meta_run.get('mem_kb', 0)) // 1024} MiB address space, per process; {meta_run.get('jobs')} files in parallel |")
|
||||||
|
w(f"| runtime | {meta_run.get('probe_seconds')} s probing + comparing ({meta_run.get('build_seconds')} s fetch/build before it) |")
|
||||||
|
w("")
|
||||||
|
w("## Results")
|
||||||
|
w("")
|
||||||
|
w("A file's class is the first that applies:")
|
||||||
|
w("")
|
||||||
|
w("- **panic / hang / crash / oom** — clawhdf5 panicked (caught per object or not), hit the timeout, died on a signal, or failed an allocation. The CI gate fails on any of these.")
|
||||||
|
w("- **h5py-cannot-read** — libhdf5 could not open the file (or itself crashed or hung). Nothing to compare against; most are the deliberately malformed CVE reproducers.")
|
||||||
|
w("- **ref-bug** — every difference is an object clawhdf5 refuses that h5py reads only through a libhdf5 bug: the values h5py returns for it change with the reading process's heap, re-checked in every run (see *Reference bugs*).")
|
||||||
|
w("- **our-error** — clawhdf5 returned an error for something h5py reads.")
|
||||||
|
w("- **mismatch** — both read it, but the shapes, values, object set or attribute set differ.")
|
||||||
|
w("- **ok** — every object h5py reads, clawhdf5 reads identically.")
|
||||||
|
w("")
|
||||||
|
w("| corpus | files | " + " | ".join(CLASSES) + " |")
|
||||||
|
w("|---" * (len(CLASSES) + 2) + "|")
|
||||||
|
for c in sorted(by_corpus):
|
||||||
|
cnt = by_corpus[c]
|
||||||
|
w(f"| {c} | {sum(cnt.values())} | " + " | ".join(str(cnt.get(k, 0)) for k in CLASSES) + " |")
|
||||||
|
w(f"| **all** | **{len(rows)}** | " + " | ".join(f"**{total.get(k, 0)}**" for k in CLASSES) + " |")
|
||||||
|
w("")
|
||||||
|
nonok = total.get("our-error", 0) + total.get("mismatch", 0)
|
||||||
|
w(f"**Our errors and mismatches: {nonok}.** Files not ok: "
|
||||||
|
+ (", ".join(f"{total[c]} {c}" for c in CLASSES if c != "ok" and total.get(c)) or "none") + "."
|
||||||
|
+ (f" {len(ref_fixes)} object(s) were compared against h5py's values corrected for a known h5py bug"
|
||||||
|
f" ({sum(1 for x in ref_fixes if x[3])} identical to clawhdf5's; see *Reference bugs*)." if ref_fixes else ""))
|
||||||
|
w("")
|
||||||
|
w("Corpora (fetched by `conformance/fetch-corpus.sh` into the gitignored `conformance/.cache/`):")
|
||||||
|
w("")
|
||||||
|
w("| corpus | source | commit |")
|
||||||
|
w("|---|---|---|")
|
||||||
|
for name, url, rev, root in pins:
|
||||||
|
w(f"| {name} | {url.removesuffix('.git')}" + ("" if root == "." else f" (`{root}`)") + f" | `{rev[:12]}` |")
|
||||||
|
w("")
|
||||||
|
|
||||||
|
w("## Panics, hangs, crashes, out-of-memory")
|
||||||
|
w("")
|
||||||
|
if not res["panics"]:
|
||||||
|
w("None.")
|
||||||
|
else:
|
||||||
|
for p in res["panics"]:
|
||||||
|
w(f"- `{p['file']}` [{p['class']}] {p['detail']}")
|
||||||
|
w("")
|
||||||
|
|
||||||
|
w("## Our-error root causes")
|
||||||
|
w("")
|
||||||
|
if res["root_causes"]:
|
||||||
|
w("Grouped by normalised error message. *files* counts files whose class this cause affects.")
|
||||||
|
w("")
|
||||||
|
w("| files | objects | error | examples |")
|
||||||
|
w("|---:|---:|---|---|")
|
||||||
|
for k, v in res["root_causes"].items():
|
||||||
|
w(f"| {v['files']} | {v['count']} | `{k.replace('|', '/')}` | {ex_list(v['file_list'])} |")
|
||||||
|
else:
|
||||||
|
w("None.")
|
||||||
|
w("")
|
||||||
|
w("## Mismatch root causes")
|
||||||
|
w("")
|
||||||
|
if res["mismatch_causes"]:
|
||||||
|
w("| files | objects | cause | examples |")
|
||||||
|
w("|---:|---:|---|---|")
|
||||||
|
for k, v in res["mismatch_causes"].items():
|
||||||
|
w(f"| {v['files']} | {v['count']} | `{k.replace('|', '/')}` | {ex_list(v['file_list'])} |")
|
||||||
|
else:
|
||||||
|
w("None.")
|
||||||
|
w("")
|
||||||
|
|
||||||
|
w("## CVE corpus: clawhdf5 vs h5dump vs h5py")
|
||||||
|
w("")
|
||||||
|
w(f"The {len(cve_rows)} files of [HDFGroup/cve_hdf5](https://github.com/HDFGroup/cve_hdf5) — reproducers for")
|
||||||
|
w("published libhdf5 CVEs and fuzzer finds. *read* = produced output (possibly with per-object")
|
||||||
|
w("errors), *error* = refused cleanly. h5dump exits non-zero on any error anywhere in a file, so")
|
||||||
|
w("its read/error split is not comparable with the other two rows; the panic, crash, hang and oom")
|
||||||
|
w("columns are.")
|
||||||
|
w("")
|
||||||
|
w("| tool | read | error | panic | crash | hang | oom |")
|
||||||
|
w("|---|---:|---:|---:|---:|---:|---:|")
|
||||||
|
for tool, label in (("clawhdf5", "clawhdf5"), ("h5dump", f"h5dump {h5dump_v.split()[-1] if h5dump_v else ''}"),
|
||||||
|
("h5py", f"h5py {h5py.__version__} / HDF5 {h5py.version.hdf5_version}")):
|
||||||
|
b = buckets[tool]
|
||||||
|
w(f"| {label} | " + " | ".join(str(b.get(k, 0)) for k in ("read", "error", "panic", "crash", "hang", "oom")) + " |")
|
||||||
|
w("")
|
||||||
|
w("<details><summary>Per-file outcomes</summary>")
|
||||||
|
w("")
|
||||||
|
w("| file | h5dump | h5py | clawhdf5 | class |")
|
||||||
|
w("|---|---|---|---|---|")
|
||||||
|
for f, d, p, o, cls in cve_rows:
|
||||||
|
w(f"| {f} | {d} | {p} | {o} | {cls} |")
|
||||||
|
w("")
|
||||||
|
w("</details>")
|
||||||
|
w("")
|
||||||
|
|
||||||
|
w("## Reference bugs")
|
||||||
|
w("")
|
||||||
|
w("### Objects h5py reads only through a libhdf5 bug (*ref-bug*)")
|
||||||
|
w("")
|
||||||
|
w("clawhdf5 refuses these objects; h5py 3.16 / HDF5 2.0 returns values for them. `conformance/ref_bugs.py`")
|
||||||
|
w("re-reads each with h5py in six fresh processes whose heaps differ (h5py imported before numpy, three")
|
||||||
|
w("times and twice more with `MALLOC_PERTURB_`, and numpy imported first). Values the file determines")
|
||||||
|
w("come out the same every time; these do not, so they are memory libhdf5 over-reads, not the file's")
|
||||||
|
w("data. A file is *ref-bug* only while every one of its differences is such an object confirmed in")
|
||||||
|
w("the same run; an object that reads the same every time goes back to *our-error*. Reproducer:")
|
||||||
|
w("`python conformance/ref_bugs.py conformance/.cache/corpus` (prints every read's outcome).")
|
||||||
|
w("")
|
||||||
|
w("| file | object | distinct results in 6 reads | confirmed | what goes wrong |")
|
||||||
|
w("|---|---|---:|---|---|")
|
||||||
|
for b in ref_bugs:
|
||||||
|
n = "missing" if b.get("missing") else b.get("distinct", "?")
|
||||||
|
w(f"| `{b['file']}` | `{b['object']}` | {n} | {'yes' if b.get('confirmed') else '**no**'} | {b['why']} |")
|
||||||
|
w("")
|
||||||
|
w("### Values corrected for a known h5py bug")
|
||||||
|
w("")
|
||||||
|
w("- **h5py big-endian variable-length sequences.** h5py returns the elements of a VL sequence")
|
||||||
|
w(" whose base type is big-endian with the file's big-endian bytes but a native (little-endian)")
|
||||||
|
w(" numpy dtype: a `h5py.vlen_dtype(np.dtype('>f4'))` dataset holding `[1.0, 2.0]` reads back as")
|
||||||
|
w(" `[4.6e-41, 9.0e-44]`; `h5dump` prints the file's values. `ref.py` checks that the installed")
|
||||||
|
w(" h5py still does this (by writing and reading exactly that dataset in memory) and, if so,")
|
||||||
|
w(" relabels such elements with the file's byte order before hashing, so the values are still")
|
||||||
|
w(" compared. Corrected objects: "
|
||||||
|
+ (", ".join(f"`{f}` `{p}` ({'same as clawhdf5' if same else '**differs from clawhdf5**'})"
|
||||||
|
for f, p, _, same in ref_fixes) if ref_fixes else "none") + ".")
|
||||||
|
w("")
|
||||||
|
w("## Other comparison rules")
|
||||||
|
w("")
|
||||||
|
w("- **Non-IEEE floats and partial-precision integers (N-Bit).** libhdf5 converts a float whose")
|
||||||
|
w(" bit layout is not IEEE (e.g. `H5Tset_precision` for the N-Bit filter) or an integer with a")
|
||||||
|
w(" bit offset / reduced precision into the plain numpy type of the same size. The probe")
|
||||||
|
w(" compares such values as converted numbers, not raw file bytes (before 2026-09-25 it compared")
|
||||||
|
w(" raw bytes, which reported every N-Bit float dataset as a mismatch).")
|
||||||
|
if res["incomparable"]:
|
||||||
|
w("- **Types h5py widens.** Where h5py reads a type into a numpy type of a different size")
|
||||||
|
w(" (FP8 -> float16, bfloat16 -> float32, x87 long double -> float128) the values are not")
|
||||||
|
w(" compared (shape and presence still are): "
|
||||||
|
+ ", ".join(f"{k} ({n}x)" for k, n in res["incomparable"]) + ".")
|
||||||
|
w("- **References** are compared by presence only (`R`), not by target.")
|
||||||
|
w("")
|
||||||
|
if res.get("ref_only_errors"):
|
||||||
|
w("## Objects h5py fails on but clawhdf5 reads")
|
||||||
|
w("")
|
||||||
|
for k, n in res["ref_only_errors"][:15]:
|
||||||
|
w(f"- {n} x `{k}`")
|
||||||
|
w("")
|
||||||
|
w("## Reproduce")
|
||||||
|
w("")
|
||||||
|
w("```sh")
|
||||||
|
w("# needs: Rust, python3 with h5py numpy hdf5plugin (conformance/requirements.txt), h5dump (hdf5-tools), git")
|
||||||
|
w("CLAWHDF5_PYTHON=/path/to/venv/bin/python conformance/run.sh")
|
||||||
|
w("```")
|
||||||
|
w("")
|
||||||
|
w("The corpus (about 450 MB of sparse checkouts) is cached in `conformance/.cache/`; results for")
|
||||||
|
w("every file, both sides' raw JSON and stderr, are in `conformance/.cache/results/`.")
|
||||||
|
w("`conformance/baseline.json` holds the ok files the nightly CI job (`.gitea/workflows/conformance.yml`)")
|
||||||
|
w("must keep; `conformance/run.sh --update-baseline` rewrites it.")
|
||||||
|
|
||||||
|
with open(OUT_MD, "w") as fh:
|
||||||
|
fh.write("\n".join(L) + "\n")
|
||||||
@@ -0,0 +1,6 @@
|
|||||||
|
# The reference side of the conformance sweep. Pinned so the nightly job and a
|
||||||
|
# local run compare against the same libhdf5 (h5py wheels bundle it).
|
||||||
|
h5py==3.16.0
|
||||||
|
numpy==2.5.3
|
||||||
|
hdf5plugin==7.1.0
|
||||||
|
netCDF4==1.7.4
|
||||||
Executable
+90
@@ -0,0 +1,90 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# conformance/run.sh — the clawhdf5 conformance sweep, end to end.
|
||||||
|
#
|
||||||
|
# fetch the pinned corpora (cached) -> build the probe -> probe every file
|
||||||
|
# with clawhdf5 and with h5py (and h5dump for the CVE corpus), each under a
|
||||||
|
# timeout and a memory limit -> compare -> write CONFORMANCE.md -> check the
|
||||||
|
# result against conformance/baseline.json.
|
||||||
|
#
|
||||||
|
# Usage: conformance/run.sh [--no-fetch] [--no-report] [--update-baseline]
|
||||||
|
#
|
||||||
|
# Environment:
|
||||||
|
# CLAWHDF5_PYTHON python with h5py, numpy, hdf5plugin (default: repo .venv, then python3)
|
||||||
|
# CONFORMANCE_CACHE corpus / build / results cache (default: conformance/.cache)
|
||||||
|
# CONFORMANCE_OUT results directory (default: $CONFORMANCE_CACHE/results)
|
||||||
|
# CONFORMANCE_REPORT report path (default: CONFORMANCE.md at the repo root)
|
||||||
|
# JOBS parallel files (default: nproc)
|
||||||
|
# CONFORMANCE_PROBE use this prebuilt probe binary instead of building one
|
||||||
|
# TMO / MEM_KB per-process timeout in seconds (20) / address-space limit in KiB (4 GiB)
|
||||||
|
#
|
||||||
|
# Exit status: 0 = gate passed; 1 = a panic/hang/crash/oom in clawhdf5, or the
|
||||||
|
# ok count fell below the baseline, or a baseline-ok file regressed; 2 = setup error.
|
||||||
|
set -euo pipefail
|
||||||
|
HERE="$(cd "$(dirname "$0")" && pwd)"
|
||||||
|
ROOT="$(cd "$HERE/.." && pwd)"
|
||||||
|
FETCH=1 REPORT=1 UPDATE=0
|
||||||
|
for a in "$@"; do
|
||||||
|
case "$a" in
|
||||||
|
--no-fetch) FETCH=0 ;;
|
||||||
|
--no-report) REPORT=0 ;;
|
||||||
|
--update-baseline) UPDATE=1 ;;
|
||||||
|
-h|--help) sed -n '2,23p' "$0"; exit 0 ;;
|
||||||
|
*) echo "unknown argument: $a" >&2; exit 2 ;;
|
||||||
|
esac
|
||||||
|
done
|
||||||
|
|
||||||
|
export PATH="$HOME/.cargo/bin:$PATH"
|
||||||
|
CACHE="${CONFORMANCE_CACHE:-$HERE/.cache}"
|
||||||
|
mkdir -p "$CACHE"; CACHE="$(cd "$CACHE" && pwd)"
|
||||||
|
OUT="${CONFORMANCE_OUT:-$CACHE/results}"
|
||||||
|
REPORT_PATH="${CONFORMANCE_REPORT:-$ROOT/CONFORMANCE.md}"
|
||||||
|
JOBS="${JOBS:-$(nproc 2>/dev/null || echo 4)}"
|
||||||
|
if [ -n "${CLAWHDF5_PYTHON:-}" ]; then PY="$CLAWHDF5_PYTHON"
|
||||||
|
elif [ -x "$ROOT/.venv/bin/python" ]; then PY="$ROOT/.venv/bin/python"
|
||||||
|
else PY="$(command -v python3)"; fi
|
||||||
|
export PY TMO="${TMO:-20}" MEM_KB="${MEM_KB:-4194304}"
|
||||||
|
command -v h5dump >/dev/null || { echo "error: h5dump not found (install hdf5-tools)" >&2; exit 2; }
|
||||||
|
"$PY" -c 'import h5py, numpy, hdf5plugin' || { echo "error: $PY lacks h5py/numpy/hdf5plugin" >&2; exit 2; }
|
||||||
|
|
||||||
|
t0=$(date +%s)
|
||||||
|
[ "$FETCH" = 1 ] && bash "$HERE/fetch-corpus.sh" "$CACHE"
|
||||||
|
C="$CACHE/corpus"
|
||||||
|
[ -d "$C" ] || { echo "error: no corpus in $C (run without --no-fetch)" >&2; exit 2; }
|
||||||
|
|
||||||
|
if [ -n "${CONFORMANCE_PROBE:-}" ]; then
|
||||||
|
export PROBE="$CONFORMANCE_PROBE" # a prebuilt probe, e.g. an older one for a before/after
|
||||||
|
else
|
||||||
|
echo "== building the probe"
|
||||||
|
CARGO_TARGET_DIR="${CARGO_TARGET_DIR:-$CACHE/target}" \
|
||||||
|
cargo build -q --release --manifest-path "$HERE/probe/Cargo.toml"
|
||||||
|
export PROBE="${CARGO_TARGET_DIR:-$CACHE/target}/release/conformance-probe"
|
||||||
|
fi
|
||||||
|
t1=$(date +%s)
|
||||||
|
|
||||||
|
rm -rf "$OUT"; mkdir -p "$OUT"
|
||||||
|
"$PY" "$HERE/list_files.py" "$C" > "$OUT/files.txt"
|
||||||
|
echo "== probing $(wc -l <"$OUT/files.txt") files, $JOBS at a time (timeout ${TMO}s, limit $((MEM_KB / 1024)) MiB)"
|
||||||
|
export C OUT HERE
|
||||||
|
# The shell's "Segmentation fault (core dumped)" notices go to probe.log; the
|
||||||
|
# signals themselves are recorded in each side's .rc.
|
||||||
|
xargs -a "$OUT/files.txt" -d '\n' -P "$JOBS" -I{} bash -c '
|
||||||
|
f="$1"; d="$OUT/runs/${f//\//__}"
|
||||||
|
case "$f" in cve_hdf5/*) export WITH_H5DUMP=1 ;; esac
|
||||||
|
"$HERE/run_one.sh" "$C/$f" "$d"' _ {} 2>"$OUT/probe.log"
|
||||||
|
echo "== re-checking the objects h5py reads only through a libhdf5 bug"
|
||||||
|
"$PY" "$HERE/ref_bugs.py" "$C" > "$OUT/ref_bugs.json" 2> "$OUT/ref_bugs.err" || true
|
||||||
|
echo "== comparing"
|
||||||
|
"$PY" "$HERE/compare.py" "$OUT" >/dev/null
|
||||||
|
t2=$(date +%s)
|
||||||
|
cat > "$OUT/meta.json" <<EOF
|
||||||
|
{"build_seconds": $((t1 - t0)), "probe_seconds": $((t2 - t1)), "jobs": $JOBS, "timeout_s": $TMO, "mem_kb": $MEM_KB}
|
||||||
|
EOF
|
||||||
|
export CONFORMANCE_CMD="${CONFORMANCE_CMD:-conformance/run.sh${*:+ $*}}"
|
||||||
|
if [ "$REPORT" = 1 ]; then
|
||||||
|
"$PY" "$HERE/report.py" "$OUT" "$REPORT_PATH" "$C"
|
||||||
|
echo "== wrote $REPORT_PATH"
|
||||||
|
fi
|
||||||
|
if [ "$UPDATE" = 1 ]; then
|
||||||
|
"$PY" "$HERE/check.py" "$OUT" "$HERE/baseline.json" --update
|
||||||
|
fi
|
||||||
|
"$PY" "$HERE/check.py" "$OUT" "$HERE/baseline.json"
|
||||||
Executable
+27
@@ -0,0 +1,27 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# run_one.sh <file> <outdir>
|
||||||
|
#
|
||||||
|
# Probe one file with clawhdf5 (PROBE) and with h5py (PY ref.py), and with
|
||||||
|
# h5dump too when WITH_H5DUMP is set. Each side runs under a timeout (TMO
|
||||||
|
# seconds, SIGKILL) and an address-space limit (MEM_KB), with core dumps off.
|
||||||
|
# Writes <outdir>/<side>.{json,err,rc}; rc 137 = killed by the timeout.
|
||||||
|
set -u
|
||||||
|
f="$1"; out="$2"; mkdir -p "$out"
|
||||||
|
HERE="$(cd "$(dirname "$0")" && pwd)"
|
||||||
|
: "${PROBE:?PROBE must name the conformance-probe binary}"
|
||||||
|
: "${PY:?PY must name a python with h5py}"
|
||||||
|
TMO="${TMO:-20}"
|
||||||
|
MEM_KB="${MEM_KB:-4194304}"
|
||||||
|
run() { # name cmd...
|
||||||
|
local name=$1; shift
|
||||||
|
( ulimit -v "$MEM_KB"; ulimit -c 0; RUST_BACKTRACE=1 exec timeout -s KILL "$TMO" "$@" ) \
|
||||||
|
>"$out/$name.json" 2>"$out/$name.err"
|
||||||
|
echo $? >"$out/$name.rc"
|
||||||
|
}
|
||||||
|
run ours "$PROBE" "$f"
|
||||||
|
run ref "$PY" "$HERE/ref.py" "$f"
|
||||||
|
if [ -n "${WITH_H5DUMP:-}" ]; then
|
||||||
|
run h5dump h5dump "$f"
|
||||||
|
: >"$out/h5dump.json" # h5dump's text dump is not compared, only its exit status
|
||||||
|
fi
|
||||||
|
exit 0
|
||||||
@@ -0,0 +1,92 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Tests of the reference side's corrections: `python conformance/test_ref.py`.
|
||||||
|
|
||||||
|
- ref.py compares a big-endian VL sequence by the file's values even though
|
||||||
|
h5py returns them byte-swapped (and records that it corrected them);
|
||||||
|
- ref_bugs.py confirms an object only when its reads disagree.
|
||||||
|
"""
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import subprocess
|
||||||
|
import sys
|
||||||
|
import tempfile
|
||||||
|
import unittest
|
||||||
|
|
||||||
|
import h5py
|
||||||
|
import numpy as np
|
||||||
|
|
||||||
|
HERE = os.path.dirname(os.path.abspath(__file__))
|
||||||
|
sys.path.insert(0, HERE)
|
||||||
|
import ref_bugs # noqa: E402
|
||||||
|
|
||||||
|
|
||||||
|
def ref_objects(path):
|
||||||
|
out = subprocess.run([sys.executable, os.path.join(HERE, "ref.py"), path],
|
||||||
|
capture_output=True, text=True, check=True).stdout
|
||||||
|
return {o["path"]: o for o in json.loads(out)["objects"]}
|
||||||
|
|
||||||
|
|
||||||
|
class BigEndianVlen(unittest.TestCase):
|
||||||
|
def test_be_vlen_compared_by_file_values(self):
|
||||||
|
with tempfile.TemporaryDirectory() as d:
|
||||||
|
path = os.path.join(d, "v.h5")
|
||||||
|
with h5py.File(path, "w") as f:
|
||||||
|
for name, order in (("be", ">"), ("le", "<")):
|
||||||
|
t = np.dtype(order + "f4")
|
||||||
|
ds = f.create_dataset(name, (2,), dtype=h5py.vlen_dtype(t))
|
||||||
|
ds[0] = np.array([1.0, 2.0], dtype=t)
|
||||||
|
ds[1] = np.array([3.0], dtype=t)
|
||||||
|
u = np.dtype(order + "u8")
|
||||||
|
f.attrs.create(name, [np.array([1, 2], dtype=u), np.array([42], dtype=u)],
|
||||||
|
dtype=h5py.vlen_dtype(u))
|
||||||
|
objs = ref_objects(path)
|
||||||
|
be, le = objs["/be"], objs["/le"]
|
||||||
|
# Same values, so the same canonical hash whatever the file's byte order.
|
||||||
|
self.assertEqual(be["hash"], le["hash"])
|
||||||
|
self.assertEqual(objs["/"]["attrs"]["be"]["hash"], objs["/"]["attrs"]["le"]["hash"])
|
||||||
|
self.assertNotIn("ref_fix", le)
|
||||||
|
# And the correction is recorded wherever h5py needed it.
|
||||||
|
import ref
|
||||||
|
if ref.be_vlen_bug():
|
||||||
|
self.assertEqual(be.get("ref_fix"), ["h5py-be-vlen"])
|
||||||
|
self.assertEqual(objs["/"]["attrs"]["be"].get("ref_fix"), ["h5py-be-vlen"])
|
||||||
|
|
||||||
|
|
||||||
|
class RefBugsConfirmation(unittest.TestCase):
|
||||||
|
def run_check(self, outcomes):
|
||||||
|
seq = iter(outcomes)
|
||||||
|
saved = ref_bugs.read_once
|
||||||
|
ref_bugs.read_once = lambda *a: next(seq)
|
||||||
|
try:
|
||||||
|
key = next(iter(ref_bugs.READ_BUGS))
|
||||||
|
with tempfile.TemporaryDirectory() as d:
|
||||||
|
p = os.path.join(d, key[0])
|
||||||
|
os.makedirs(os.path.dirname(p))
|
||||||
|
open(p, "wb").close()
|
||||||
|
return ref_bugs.check(d, key)
|
||||||
|
finally:
|
||||||
|
ref_bugs.read_once = saved
|
||||||
|
|
||||||
|
def test_stable_values_are_not_confirmed(self):
|
||||||
|
r = self.run_check(["values a"] * len(ref_bugs.RUNS))
|
||||||
|
self.assertFalse(r["confirmed"])
|
||||||
|
|
||||||
|
def test_changing_values_are_confirmed(self):
|
||||||
|
r = self.run_check(["values a"] * (len(ref_bugs.RUNS) - 1) + ["values b"])
|
||||||
|
self.assertTrue(r["confirmed"])
|
||||||
|
r = self.run_check(["values a"] * (len(ref_bugs.RUNS) - 1) + ["error filter failed"])
|
||||||
|
self.assertTrue(r["confirmed"])
|
||||||
|
|
||||||
|
def test_errors_only_are_not_confirmed(self):
|
||||||
|
# h5py cannot read it at all: nothing it reads, nothing to excuse.
|
||||||
|
r = self.run_check(["error x"] * (len(ref_bugs.RUNS) - 1) + ["error y"])
|
||||||
|
self.assertFalse(r["confirmed"])
|
||||||
|
|
||||||
|
def test_missing_file_is_not_confirmed(self):
|
||||||
|
key = next(iter(ref_bugs.READ_BUGS))
|
||||||
|
with tempfile.TemporaryDirectory() as d:
|
||||||
|
self.assertFalse(ref_bugs.check(d, key)["confirmed"])
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
@@ -3,7 +3,7 @@ name = "clawhdf5-accel"
|
|||||||
version = "2.7.0"
|
version = "2.7.0"
|
||||||
edition = "2024"
|
edition = "2024"
|
||||||
rust-version.workspace = true
|
rust-version.workspace = true
|
||||||
description = "SIMD-accelerated operations for rustyhdf5"
|
description = "SIMD kernels (AVX2, NEON) used by clawhdf5 — pure Rust"
|
||||||
license = "MIT"
|
license = "MIT"
|
||||||
repository = "https://git.redclaw.dev/quantumclaw/clawhdf5"
|
repository = "https://git.redclaw.dev/quantumclaw/clawhdf5"
|
||||||
readme = "README.md"
|
readme = "README.md"
|
||||||
|
|||||||
@@ -1,24 +1,62 @@
|
|||||||
# clawhdf5-accel
|
# clawhdf5-accel
|
||||||
|
|
||||||
[](https://crates.io/crates/clawhdf5-accel)
|
CPU SIMD kernels for vector search: dot products, cosine similarity, L2
|
||||||
[](https://docs.rs/clawhdf5-accel)
|
distance, norms and int8 dot products, dispatched at run time to the best
|
||||||
|
backend the CPU has, with a portable scalar fallback for every operation.
|
||||||
|
[`clawhdf5-ann`](../clawhdf5-ann/README.md) and
|
||||||
|
[`clawhdf5-agent`](../clawhdf5-agent/README.md) use it in their distance
|
||||||
|
loops; it has nothing to do with HDF5 file I/O.
|
||||||
|
|
||||||
SIMD-accelerated operations for clawhdf5.
|
Not on crates.io yet; depend on it from git:
|
||||||
|
|
||||||
|
```toml
|
||||||
|
[dependencies]
|
||||||
|
clawhdf5-accel = { git = "https://git.redclaw.dev/quantumclaw/clawhdf5" }
|
||||||
|
```
|
||||||
|
|
||||||
|
## API
|
||||||
|
|
||||||
|
```rust
|
||||||
|
use clawhdf5_accel::{cosine_similarity, detect_backend, dot_i8, dot_product, l2_distance};
|
||||||
|
|
||||||
|
let a = [1.0f32, 2.0, 3.0, 4.0];
|
||||||
|
let b = [4.0f32, 3.0, 2.0, 1.0];
|
||||||
|
assert_eq!(dot_product(&a, &b), 20.0);
|
||||||
|
let _cos = cosine_similarity(&a, &b);
|
||||||
|
let _l2 = l2_distance(&a, &b);
|
||||||
|
assert_eq!(dot_i8(&[1, -2, 3], &[4, 5, -6]), -24);
|
||||||
|
println!("{:?}", detect_backend()); // e.g. Avx2 on x86-64, Neon on aarch64
|
||||||
|
```
|
||||||
|
|
||||||
|
Also `vector_norm`, `batch_norms`, `batch_cosine`, `batch_cosine_prenorm`,
|
||||||
|
`f16_to_f32_batch`, `checksum_fletcher32` and `align_to_cache_line`.
|
||||||
|
|
||||||
|
## Backends
|
||||||
|
|
||||||
|
`detect_backend()` picks once per process: `Avx512` (with the `avx512`
|
||||||
|
feature), `Avx2` (AVX2 + FMA), `Neon` (every aarch64 CPU), or `Scalar`.
|
||||||
|
`Sse4` and `WasmSimd128` are reported when detected but run the scalar
|
||||||
|
kernels.
|
||||||
|
`dot_i8`, used by the agent's quantised (int8) HNSW index, runs on
|
||||||
|
AVX2 and on NEON — with the `SDOT` instruction (through inline assembly,
|
||||||
|
since the intrinsic is unstable) on cores that have dotprod, such as the
|
||||||
|
Raspberry Pi 5, and plain NEON on older ones. At equal recall the int8
|
||||||
|
index answers 1.63x the queries per second of the f32 one on x86-64
|
||||||
|
(AVX2; 2026-09-20, machine not recorded, not re-run) and 1.18x on a
|
||||||
|
Raspberry Pi 5 (2026-09-21) ([`BENCHMARKS.md` § Quantising the index copy](../../BENCHMARKS.md#quantising-the-index-copy-quantized_index)).
|
||||||
|
|
||||||
|
The aarch64 code is compiled out on x86, so only the `test-arm64` CI job
|
||||||
|
builds and tests it.
|
||||||
|
|
||||||
## Features
|
## Features
|
||||||
|
|
||||||
- AVX2 and NEON SIMD acceleration
|
| Feature | Default | What | Builds C |
|
||||||
- AVX-512 support (`avx512` feature)
|
|---|---|---|---|
|
||||||
- Float16 conversion (`float16` feature)
|
| `avx512` | no | AVX-512F kernels | no |
|
||||||
- CRC32 checksum acceleration
|
| `float16` | no | `f16_to_f32_batch` through the `half` crate (a software conversion otherwise) | no |
|
||||||
|
|
||||||
## Usage
|
The half-precision conversion used for stored embeddings is
|
||||||
|
`clawhdf5_format::float16`, not this crate's.
|
||||||
```rust
|
|
||||||
use clawhdf5_accel::checksum::crc32_simd;
|
|
||||||
|
|
||||||
let crc = crc32_simd(&data);
|
|
||||||
```
|
|
||||||
|
|
||||||
## License
|
## License
|
||||||
|
|
||||||
|
|||||||
@@ -237,7 +237,10 @@ pub fn f16_to_f32_batch(input: &[u16], output: &mut [f32]) {
|
|||||||
convert::f16_to_f32_batch(input, output);
|
convert::f16_to_f32_batch(input, output);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Compute Fletcher-32 checksum.
|
/// Compute a textbook Fletcher-32 checksum (both sums start at 0xffff).
|
||||||
|
///
|
||||||
|
/// This is not HDF5's checksum; the Fletcher-32 I/O filter uses
|
||||||
|
/// `clawhdf5_format::checksum::fletcher32`.
|
||||||
pub fn checksum_fletcher32(data: &[u8]) -> u32 {
|
pub fn checksum_fletcher32(data: &[u8]) -> u32 {
|
||||||
checksum::checksum_fletcher32(data)
|
checksum::checksum_fletcher32(data)
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -19,6 +19,10 @@ clawhdf5-ann = { path = "../clawhdf5-ann", version = "2.7.0", optional = true }
|
|||||||
clawhdf5-gpu = { path = "../clawhdf5-gpu", version = "2.7.0", optional = true, default-features = false }
|
clawhdf5-gpu = { path = "../clawhdf5-gpu", version = "2.7.0", optional = true, default-features = false }
|
||||||
serde = { workspace = true }
|
serde = { workspace = true }
|
||||||
byteorder = "1"
|
byteorder = "1"
|
||||||
|
# Signed checkpoints (MemoryConfig-independent; see `signing`). Pure Rust.
|
||||||
|
ed25519-dalek = { version = "2", features = ["rand_core"] }
|
||||||
|
sha2 = "0.10"
|
||||||
|
rand_core = { version = "0.6", features = ["getrandom"] }
|
||||||
half = { workspace = true, optional = true }
|
half = { workspace = true, optional = true }
|
||||||
rayon = { version = "1", optional = true }
|
rayon = { version = "1", optional = true }
|
||||||
matrixmultiply = { version = "0.3", optional = true }
|
matrixmultiply = { version = "0.3", optional = true }
|
||||||
|
|||||||
+112
-16
@@ -1,28 +1,124 @@
|
|||||||
# clawhdf5-agent
|
# clawhdf5-agent
|
||||||
|
|
||||||
[](https://crates.io/crates/clawhdf5-agent)
|
Persistent memory for AI agents in a single HDF5 file: text chunks with
|
||||||
[](https://docs.rs/clawhdf5-agent)
|
embeddings and metadata, hybrid search (HNSW vector search + BM25 keyword
|
||||||
|
search, fused), sessions, a knowledge graph, a write-ahead log for crash
|
||||||
|
safety, and optionally Ed25519-signed checkpoints. Stores open in h5py like
|
||||||
|
any other HDF5 file. Built on [`clawhdf5`](../clawhdf5/README.md),
|
||||||
|
[`clawhdf5-ann`](../clawhdf5-ann/README.md) and
|
||||||
|
[`clawhdf5-accel`](../clawhdf5-accel/README.md).
|
||||||
|
|
||||||
HDF5-backed persistent memory store for on-device AI agents.
|
It is a library: no agent framework integrates it (OpenClaw and ZeroClaw
|
||||||
|
integration claims were withdrawn on 2026-09-25; see
|
||||||
|
[`docs/openclaw.md`](../../docs/openclaw.md)). The command-line front end
|
||||||
|
is [`clawhdf5-cli`](../clawhdf5-cli/README.md).
|
||||||
|
|
||||||
Built on [clawhdf5](https://crates.io/crates/clawhdf5), clawhdf5-agent provides a vector-searchable memory backend optimized for edge AI workloads. Store embeddings, text chunks, and metadata in a single HDF5 file with SIMD-accelerated similarity search.
|
Not on crates.io yet; depend on it from git:
|
||||||
|
|
||||||
## Features
|
|
||||||
|
|
||||||
- Persistent vector store in HDF5 format
|
|
||||||
- Cosine similarity and L2 distance search
|
|
||||||
- SIMD-accelerated via clawhdf5-accel (AVX2, NEON)
|
|
||||||
- Optional GPU acceleration via clawhdf5-gpu
|
|
||||||
- Memory-mapped access for large stores
|
|
||||||
- f16 storage support for compact embeddings
|
|
||||||
|
|
||||||
## Usage
|
|
||||||
|
|
||||||
```toml
|
```toml
|
||||||
[dependencies]
|
[dependencies]
|
||||||
clawhdf5-agent = "2.1.0"
|
clawhdf5-agent = { git = "https://git.redclaw.dev/quantumclaw/clawhdf5" }
|
||||||
```
|
```
|
||||||
|
|
||||||
|
## Usage
|
||||||
|
|
||||||
|
```rust,no_run
|
||||||
|
use std::path::PathBuf;
|
||||||
|
use clawhdf5_agent::{AgentMemory, HDF5Memory, MemoryConfig, MemoryEntry, SearchOptions};
|
||||||
|
|
||||||
|
let config = MemoryConfig::new(PathBuf::from("agent.h5"), "my-agent", 384);
|
||||||
|
let mut mem = HDF5Memory::create(config)?;
|
||||||
|
|
||||||
|
mem.save(MemoryEntry {
|
||||||
|
chunk: "The deploy key rotates every Monday.".into(),
|
||||||
|
embedding: vec![0.01; 384], // from your embedding model
|
||||||
|
source_channel: "chat".into(),
|
||||||
|
timestamp: 1_790_000_000.0,
|
||||||
|
session_id: "s1".into(),
|
||||||
|
tags: "ops".into(),
|
||||||
|
})?;
|
||||||
|
|
||||||
|
let query = vec![0.01f32; 384];
|
||||||
|
let hits = mem.search(&query, "deploy key", &SearchOptions::new(5).with_sources(["chat"]));
|
||||||
|
for h in &hits {
|
||||||
|
println!("{:.3} {}", h.score, h.chunk);
|
||||||
|
}
|
||||||
|
mem.flush_wal()?; // checkpoint now; otherwise one is made once the WAL holds more than 500 entries (wal_max_entries)
|
||||||
|
# Ok::<(), clawhdf5_agent::MemoryError>(())
|
||||||
|
```
|
||||||
|
|
||||||
|
## What is in it
|
||||||
|
|
||||||
|
- **`HDF5Memory`** — `create`, `open` (single writer: an exclusive lock on
|
||||||
|
`<store>.h5.lock`, a second opener gets `MemoryError::Locked`),
|
||||||
|
`open_read_only` (no lock, never writes). Through the `AgentMemory`
|
||||||
|
trait: `save`, `save_batch`, `delete`, `compact`, `count`, `snapshot`,
|
||||||
|
sessions; also `save_or_update`, `delete_batch`, `flush_wal`.
|
||||||
|
- **Search** — `search(query_embedding, text, &SearchOptions)`: optional
|
||||||
|
source-channel filter applied before ranking, vector + BM25 fusion
|
||||||
|
(weighted or RRF), Hebbian activation scaling, optional re-ranking
|
||||||
|
(`reranker::ReRankConfig`) and confidence rejection
|
||||||
|
(`confidence::ConfidenceConfig`). `hybrid_search` and
|
||||||
|
`hybrid_search_with` are thin wrappers. The vector stage uses the HNSW
|
||||||
|
index (`hnsw` feature); its graph is saved to `<store>.h5.ann` at each
|
||||||
|
checkpoint and reloaded on open (rebuilt if stale or damaged).
|
||||||
|
- **Storage settings** (`MemoryConfig`, persisted with the store):
|
||||||
|
`float16` embeddings (on by default for new stores; 48% smaller file at
|
||||||
|
100K records, same retrieval on LongMemEval), `quantized_index` (int8
|
||||||
|
copy of the vectors in the index, on by default; re-scored against the
|
||||||
|
exact embeddings), `compression` (off by default), HNSW `m`/`ef`
|
||||||
|
parameters, WAL settings (`wal_enabled`, on by default; `wal_max_entries`,
|
||||||
|
500: the WAL is checkpointed into the `.h5` once it holds more).
|
||||||
|
- **WAL** (`wal`) — every write is appended to `<store>.h5.wal` with a
|
||||||
|
chained CRC32 per entry, so a corrupted, reordered or spliced entry stops
|
||||||
|
replay. Recovers from a process crash at any point, including between a
|
||||||
|
checkpoint and the WAL truncate. WAL appends are not fsynced: saves since
|
||||||
|
the last checkpoint can be lost on power failure. An unreadable WAL is
|
||||||
|
quarantined to `<store>.h5.wal.corrupt-<ts>`.
|
||||||
|
- **Signed checkpoints** (`signing`) — `set_signing_key` signs a manifest
|
||||||
|
(SHA-256 Merkle tree over records, plus settings, sessions and graph) at
|
||||||
|
every checkpoint; `HDF5Memory::verify(path, &public_key)` checks it and
|
||||||
|
locates edits. WAL entries after the checkpoint are not covered.
|
||||||
|
- **Knowledge graph** (`knowledge`, `entity_extract`) — `add_entity`,
|
||||||
|
`add_entity_alias`, `add_relation`, `extract_and_store_entities`,
|
||||||
|
traversal and spreading activation.
|
||||||
|
- **Also:** sessions (`session`), temporal index (`temporal`),
|
||||||
|
consolidation tiers (`consolidation`), an in-memory TTL tier
|
||||||
|
(`ephemeral`), multi-modal embeddings (`multimodal`), `AGENTS.md`
|
||||||
|
generation (`agents_md`), query expansion, and a session-scoped
|
||||||
|
provenance ledger and write-anomaly detector on every save
|
||||||
|
(`take_anomaly_alerts`; alerts never block a save, and the source is
|
||||||
|
inferred from `source_channel`, not authenticated).
|
||||||
|
- `openclaw::ClawhdfBackend` is `search` with re-ranking and confidence
|
||||||
|
on, plus Markdown import/export. The module name is historical: it is not
|
||||||
|
an OpenClaw plugin.
|
||||||
|
|
||||||
|
## Features
|
||||||
|
|
||||||
|
| Feature | Default | What | Builds C |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `hnsw` | yes | HNSW vector index (`clawhdf5-ann`); without it the vector stage is an exact linear cosine scan | no |
|
||||||
|
| `parallel` | yes | build the HNSW index on a rayon pool (same graph either way) | no |
|
||||||
|
| `float16` | yes | f16 helpers in `vector_search` (`half`). Stores' `MemoryConfig::float16` works without it. | no |
|
||||||
|
| `fast-math` | no | `matrixmultiply` batch distances in `strategy` | no |
|
||||||
|
| `accelerate` | no | Apple Accelerate BLAS in `strategy` (macOS) | links a system framework |
|
||||||
|
| `openblas` | no | OpenBLAS in `strategy` | yes (`openblas-src`) |
|
||||||
|
| `gpu` | no | `gpu_search` through [`clawhdf5-gpu`](../clawhdf5-gpu/README.md) (wgpu), used by `strategy`, not by `HDF5Memory::search` | no, but needs GPU drivers |
|
||||||
|
| `zstd` | no | Zstd instead of deflate when `MemoryConfig::compression` is on | yes (libzstd) |
|
||||||
|
| `async` | no | `async_memory` wrapper on tokio | no |
|
||||||
|
|
||||||
|
`--no-default-features --features float16` forces the exact linear scan.
|
||||||
|
|
||||||
|
## Measurements and limits
|
||||||
|
|
||||||
|
- Search recall and latency, file size, LongMemEval and MemoryArena
|
||||||
|
retrieval numbers: [`BENCHMARKS.md`](../../BENCHMARKS.md), measured with
|
||||||
|
the `clawhdf5-bench` binaries (`search_harness`, `longmemeval_bench`,
|
||||||
|
`footprint_bench`, ...).
|
||||||
|
- Known issues and their history: [`docs/known-issues.md`](../../docs/known-issues.md).
|
||||||
|
- Migrating a SQLite memory database:
|
||||||
|
[`clawhdf5-migrate`](../clawhdf5-migrate/README.md).
|
||||||
|
|
||||||
## License
|
## License
|
||||||
|
|
||||||
MIT
|
MIT
|
||||||
|
|||||||
@@ -203,7 +203,7 @@ impl ImportanceScorer {
|
|||||||
/// Novelty score: 1.0 − max cosine similarity against all existing records.
|
/// Novelty score: 1.0 − max cosine similarity against all existing records.
|
||||||
/// Returns 1.0 when there are no existing memories.
|
/// Returns 1.0 when there are no existing memories.
|
||||||
///
|
///
|
||||||
/// Same result as [`Self::cosine_similarity`] against each record, but the
|
/// Same result as the reference cosine similarity against each record, but the
|
||||||
/// new embedding's norm is computed once rather than per record, each
|
/// new embedding's norm is computed once rather than per record, each
|
||||||
/// record costs one fused pass (dot product and its norm together) rather
|
/// record costs one fused pass (dot product and its norm together) rather
|
||||||
/// than three, and a large working set is scored in parallel. Every insert
|
/// than three, and a large working set is scored in parallel. Every insert
|
||||||
|
|||||||
@@ -1,4 +1,4 @@
|
|||||||
//! ZeroClaw agent memory HDF5 backend.
|
//! Agent memory stored in a single HDF5 file.
|
||||||
//!
|
//!
|
||||||
//! Provides persistent memory storage for AI agents using HDF5 files.
|
//! Provides persistent memory storage for AI agents using HDF5 files.
|
||||||
//! All data is cached in-memory for fast access and flushed to disk
|
//! All data is cached in-memory for fast access and flushed to disk
|
||||||
@@ -36,6 +36,7 @@ pub mod reranker;
|
|||||||
pub mod schema;
|
pub mod schema;
|
||||||
pub mod search;
|
pub mod search;
|
||||||
pub mod session;
|
pub mod session;
|
||||||
|
pub mod signing;
|
||||||
pub mod storage;
|
pub mod storage;
|
||||||
mod store_lock;
|
mod store_lock;
|
||||||
pub mod temporal;
|
pub mod temporal;
|
||||||
@@ -78,6 +79,7 @@ pub use session::{SessionCache, SessionEntry};
|
|||||||
// --- Error type ---
|
// --- Error type ---
|
||||||
|
|
||||||
#[derive(Debug)]
|
#[derive(Debug)]
|
||||||
|
#[non_exhaustive]
|
||||||
pub enum MemoryError {
|
pub enum MemoryError {
|
||||||
Io(std::io::Error),
|
Io(std::io::Error),
|
||||||
Hdf5(String),
|
Hdf5(String),
|
||||||
@@ -88,6 +90,11 @@ pub enum MemoryError {
|
|||||||
/// A record the store cannot hold as given, e.g. an embedding value
|
/// A record the store cannot hold as given, e.g. an embedding value
|
||||||
/// outside the half-precision range of a `float16` store.
|
/// outside the half-precision range of a `float16` store.
|
||||||
InvalidEntry(String),
|
InvalidEntry(String),
|
||||||
|
/// The store's checkpoints are signed and no signing key is set, so a
|
||||||
|
/// checkpoint would leave it unsigned. Set the key with
|
||||||
|
/// [`HDF5Memory::set_signing_key`], or drop the signature on purpose with
|
||||||
|
/// [`HDF5Memory::remove_signature`].
|
||||||
|
SigningKeyRequired(String),
|
||||||
}
|
}
|
||||||
|
|
||||||
impl std::fmt::Display for MemoryError {
|
impl std::fmt::Display for MemoryError {
|
||||||
@@ -99,6 +106,7 @@ impl std::fmt::Display for MemoryError {
|
|||||||
MemoryError::NotFound(e) => write!(f, "not found: {e}"),
|
MemoryError::NotFound(e) => write!(f, "not found: {e}"),
|
||||||
MemoryError::Locked(e) => write!(f, "store is locked: {e}"),
|
MemoryError::Locked(e) => write!(f, "store is locked: {e}"),
|
||||||
MemoryError::InvalidEntry(e) => write!(f, "invalid entry: {e}"),
|
MemoryError::InvalidEntry(e) => write!(f, "invalid entry: {e}"),
|
||||||
|
MemoryError::SigningKeyRequired(e) => write!(f, "signing key required: {e}"),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -317,6 +325,12 @@ pub struct HDF5Memory {
|
|||||||
activations_dirty: bool,
|
activations_dirty: bool,
|
||||||
/// Opened with [`HDF5Memory::open_read_only`]: nothing may reach the disk.
|
/// Opened with [`HDF5Memory::open_read_only`]: nothing may reach the disk.
|
||||||
read_only: bool,
|
read_only: bool,
|
||||||
|
/// Key that signs every checkpoint; never persisted. See
|
||||||
|
/// [`HDF5Memory::set_signing_key`].
|
||||||
|
signing_key: Option<signing::SigningKey>,
|
||||||
|
/// Checkpoints of this store are signed: the file on disk is, or a key
|
||||||
|
/// has been set. A checkpoint without a key is then refused.
|
||||||
|
signed: bool,
|
||||||
/// A WAL that `open()` could not read and moved aside; see
|
/// A WAL that `open()` could not read and moved aside; see
|
||||||
/// [`HDF5Memory::quarantined_wal`].
|
/// [`HDF5Memory::quarantined_wal`].
|
||||||
quarantined_wal: Option<PathBuf>,
|
quarantined_wal: Option<PathBuf>,
|
||||||
@@ -372,6 +386,8 @@ impl HDF5Memory {
|
|||||||
bm25_filter: bm25::TokenFilter::default(),
|
bm25_filter: bm25::TokenFilter::default(),
|
||||||
activations_dirty: false,
|
activations_dirty: false,
|
||||||
read_only: false,
|
read_only: false,
|
||||||
|
signing_key: None,
|
||||||
|
signed: false,
|
||||||
quarantined_wal: None,
|
quarantined_wal: None,
|
||||||
_lock: Some(lock),
|
_lock: Some(lock),
|
||||||
})
|
})
|
||||||
@@ -550,6 +566,8 @@ impl HDF5Memory {
|
|||||||
bm25_filter: bm25::TokenFilter::default(),
|
bm25_filter: bm25::TokenFilter::default(),
|
||||||
activations_dirty: false,
|
activations_dirty: false,
|
||||||
read_only,
|
read_only,
|
||||||
|
signing_key: None,
|
||||||
|
signed: checkpoint.signed,
|
||||||
quarantined_wal,
|
quarantined_wal,
|
||||||
_lock: lock,
|
_lock: lock,
|
||||||
})
|
})
|
||||||
@@ -710,6 +728,39 @@ impl HDF5Memory {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Sign every checkpoint from now on with `key` (Ed25519). The key is
|
||||||
|
/// never written anywhere; set it again after every `open`. Once a store
|
||||||
|
/// is signed, a checkpoint without the key is refused
|
||||||
|
/// ([`MemoryError::SigningKeyRequired`]) rather than silently leaving it
|
||||||
|
/// unsigned. Setting a different key re-signs the store under that key
|
||||||
|
/// from the next checkpoint; a verifier trusting the old key will then
|
||||||
|
/// reject it, which is the point. Call [`AgentMemory::flush_wal`] to sign
|
||||||
|
/// right away.
|
||||||
|
pub fn set_signing_key(&mut self, key: signing::SigningKey) {
|
||||||
|
self.signing_key = Some(key);
|
||||||
|
self.signed = true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Stop signing: the next checkpoint writes the store unsigned. The
|
||||||
|
/// deliberate way out of [`MemoryError::SigningKeyRequired`].
|
||||||
|
pub fn remove_signature(&mut self) {
|
||||||
|
self.signing_key = None;
|
||||||
|
self.signed = false;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Checkpoints of this store are signed (on disk, or from the next
|
||||||
|
/// checkpoint because a key has been set).
|
||||||
|
pub fn is_signed(&self) -> bool {
|
||||||
|
self.signed
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Check the checkpoint at `path` against the public key the caller
|
||||||
|
/// trusts; see [`signing::verify_store`]. Reads the file only: it works
|
||||||
|
/// on a store another process has open.
|
||||||
|
pub fn verify(path: &Path, trusted: &signing::VerifyingKey) -> Result<signing::VerifyReport> {
|
||||||
|
signing::verify_store(path, trusted)
|
||||||
|
}
|
||||||
|
|
||||||
/// Flush current state to disk and truncate the WAL.
|
/// Flush current state to disk and truncate the WAL.
|
||||||
///
|
///
|
||||||
/// Every code path that persists the full cache to the .h5 file must
|
/// Every code path that persists the full cache to the .h5 file must
|
||||||
@@ -725,10 +776,28 @@ impl HDF5Memory {
|
|||||||
// Record which WAL prefix this checkpoint contains, so a crash before
|
// Record which WAL prefix this checkpoint contains, so a crash before
|
||||||
// the truncate below can't replay those entries a second time.
|
// the truncate below can't replay those entries a second time.
|
||||||
let wal_applied = self.wal.as_ref().map(|w| w.mark());
|
let wal_applied = self.wal.as_ref().map(|w| w.mark());
|
||||||
|
let signature = match &self.signing_key {
|
||||||
|
Some(key) => Some(signing::sign(
|
||||||
|
key,
|
||||||
|
&self.config,
|
||||||
|
&self.cache,
|
||||||
|
&self.sessions,
|
||||||
|
&self.knowledge,
|
||||||
|
wal_applied,
|
||||||
|
)),
|
||||||
|
None if self.signed => {
|
||||||
|
return Err(MemoryError::SigningKeyRequired(format!(
|
||||||
|
"{} is signed; set its signing key before a checkpoint \
|
||||||
|
(saves so far are held in the WAL or in memory)",
|
||||||
|
self.config.path.display()
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
None => None,
|
||||||
|
};
|
||||||
// Written before the .h5 so a crash in between leaves a sidecar whose
|
// Written before the .h5 so a crash in between leaves a sidecar whose
|
||||||
// generation matches no checkpoint (ignored), never the reverse.
|
// generation matches no checkpoint (ignored), never the reverse.
|
||||||
let ann_generation = self.persist_vector_index();
|
let ann_generation = self.persist_vector_index();
|
||||||
storage::write_to_disk_with_meta(
|
storage::write_to_disk_signed(
|
||||||
&self.config.path,
|
&self.config.path,
|
||||||
&self.config,
|
&self.config,
|
||||||
&self.cache,
|
&self.cache,
|
||||||
@@ -737,7 +806,9 @@ impl HDF5Memory {
|
|||||||
&schema::CheckpointMeta {
|
&schema::CheckpointMeta {
|
||||||
wal_applied,
|
wal_applied,
|
||||||
ann_generation,
|
ann_generation,
|
||||||
|
signed: signature.is_some(),
|
||||||
},
|
},
|
||||||
|
signature.as_ref(),
|
||||||
)?;
|
)?;
|
||||||
if let Some(ref mut w) = self.wal {
|
if let Some(ref mut w) = self.wal {
|
||||||
w.truncate()?;
|
w.truncate()?;
|
||||||
|
|||||||
@@ -1,10 +1,12 @@
|
|||||||
//! OpenClaw Integration Layer.
|
//! A Markdown-oriented memory backend over [`crate::HDF5Memory`].
|
||||||
//!
|
//!
|
||||||
//! Bridge between OpenClaw agent gateway (Markdown + sqlite-vec) and the
|
//! Named for OpenClaw, whose workspace memory is Markdown, but **not an
|
||||||
//! clawhdf5 HDF5-backed memory backend. Provides:
|
//! OpenClaw plugin**: nothing here registers with OpenClaw, and the
|
||||||
|
//! integration it was written for never worked (see `docs/openclaw.md`).
|
||||||
|
//! Provides:
|
||||||
//!
|
//!
|
||||||
//! - [`MemoryBackend`] — the trait OpenClaw implements against.
|
//! - [`MemoryBackend`] — search / read back / write / ingest / export.
|
||||||
//! - [`ClawhdfBackend`] — concrete HDF5-backed implementation.
|
//! - [`ClawhdfBackend`] — the HDF5-backed implementation.
|
||||||
//! - [`MarkdownParser`] — splits Markdown into [`MarkdownSection`] records.
|
//! - [`MarkdownParser`] — splits Markdown into [`MarkdownSection`] records.
|
||||||
//! - [`MarkdownExporter`] — renders sections back to Markdown text.
|
//! - [`MarkdownExporter`] — renders sections back to Markdown text.
|
||||||
|
|
||||||
@@ -61,7 +63,8 @@ pub struct BackendStats {
|
|||||||
// MemoryBackend trait
|
// MemoryBackend trait
|
||||||
// ─────────────────────────────────────────────────────────────────────────────
|
// ─────────────────────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
/// Interface that OpenClaw uses to interact with a memory backend.
|
/// A Markdown-oriented memory backend: search, read back by path, write,
|
||||||
|
/// ingest and export.
|
||||||
///
|
///
|
||||||
/// Implementors provide persistent storage, full-text + vector search,
|
/// Implementors provide persistent storage, full-text + vector search,
|
||||||
/// Markdown ingestion / export, and statistics.
|
/// Markdown ingestion / export, and statistics.
|
||||||
@@ -318,7 +321,7 @@ impl MarkdownExporter {
|
|||||||
///
|
///
|
||||||
/// # Path mapping
|
/// # Path mapping
|
||||||
///
|
///
|
||||||
/// OpenClaw addresses memories by file path (e.g. `"memory/user.md"`).
|
/// Memories are addressed by file path (e.g. `"memory/user.md"`).
|
||||||
/// Internally every [`MemoryEntry`] stores the originating path as its
|
/// Internally every [`MemoryEntry`] stores the originating path as its
|
||||||
/// `source_channel`. Section sub-paths are stored as
|
/// `source_channel`. Section sub-paths are stored as
|
||||||
/// `"<path>::<heading>"`.
|
/// `"<path>::<heading>"`.
|
||||||
@@ -421,7 +424,7 @@ impl ClawhdfBackend {
|
|||||||
|
|
||||||
// ── Compaction & Consolidation hooks (7.6) ────────────────────────────
|
// ── Compaction & Consolidation hooks (7.6) ────────────────────────────
|
||||||
|
|
||||||
/// Run a compaction cycle — called by OpenClaw during session compaction.
|
/// Run a compaction cycle (decay, compaction, WAL flush).
|
||||||
///
|
///
|
||||||
/// Sequence:
|
/// Sequence:
|
||||||
/// 1. `tick_session()` — apply Hebbian decay to all activation weights.
|
/// 1. `tick_session()` — apply Hebbian decay to all activation weights.
|
||||||
|
|||||||
@@ -2,7 +2,7 @@
|
|||||||
//!
|
//!
|
||||||
//! Records the origin, authorship, and a content hash of every memory chunk
|
//! Records the origin, authorship, and a content hash of every memory chunk
|
||||||
//! so the system can detect *accidental* corruption and trace data lineage.
|
//! so the system can detect *accidental* corruption and trace data lineage.
|
||||||
//! The hash is unkeyed (see [`fnv1a_64`]) — this is not a tamper-evidence or
|
//! The hash is unkeyed (FNV-1a) — this is not a tamper-evidence or
|
||||||
//! authenticity guarantee.
|
//! authenticity guarantee.
|
||||||
|
|
||||||
use std::collections::HashMap;
|
use std::collections::HashMap;
|
||||||
|
|||||||
@@ -15,6 +15,9 @@ use crate::session::SessionCache;
|
|||||||
use crate::wal::WalMark;
|
use crate::wal::WalMark;
|
||||||
|
|
||||||
pub const SCHEMA_VERSION: &str = "1.0";
|
pub const SCHEMA_VERSION: &str = "1.0";
|
||||||
|
/// Writer-version tag stored in `/meta` as `edgehdf5_version`. Kept for file
|
||||||
|
/// compatibility; despite the name it has nothing to do with ZeroClaw, which
|
||||||
|
/// does not use clawhdf5.
|
||||||
pub const ZEROCLAW_VERSION: &str = "0.8.0";
|
pub const ZEROCLAW_VERSION: &str = "0.8.0";
|
||||||
|
|
||||||
/// `/meta` attributes holding the [`WalMark`] of the WAL prefix already folded
|
/// `/meta` attributes holding the [`WalMark`] of the WAL prefix already folded
|
||||||
@@ -23,6 +26,7 @@ pub const ZEROCLAW_VERSION: &str = "0.8.0";
|
|||||||
const WAL_APPLIED_LEN_ATTR: &str = "wal_applied_len";
|
const WAL_APPLIED_LEN_ATTR: &str = "wal_applied_len";
|
||||||
const WAL_APPLIED_CRC_ATTR: &str = "wal_applied_crc";
|
const WAL_APPLIED_CRC_ATTR: &str = "wal_applied_crc";
|
||||||
const ANN_GENERATION_ATTR: &str = "ann_generation";
|
const ANN_GENERATION_ATTR: &str = "ann_generation";
|
||||||
|
const SIG_VERSION_ATTR: &str = "sig_version";
|
||||||
|
|
||||||
/// Build a complete HDF5 file from the in-memory state.
|
/// Build a complete HDF5 file from the in-memory state.
|
||||||
pub fn build_hdf5_file(
|
pub fn build_hdf5_file(
|
||||||
@@ -46,7 +50,7 @@ pub fn build_hdf5_file_with_mark(
|
|||||||
) -> Result<Vec<u8>, MemoryError> {
|
) -> Result<Vec<u8>, MemoryError> {
|
||||||
let meta = CheckpointMeta {
|
let meta = CheckpointMeta {
|
||||||
wal_applied,
|
wal_applied,
|
||||||
ann_generation: None,
|
..CheckpointMeta::default()
|
||||||
};
|
};
|
||||||
build_hdf5_file_with_meta(config, cache, sessions, knowledge, &meta)
|
build_hdf5_file_with_meta(config, cache, sessions, knowledge, &meta)
|
||||||
}
|
}
|
||||||
@@ -61,6 +65,10 @@ pub struct CheckpointMeta {
|
|||||||
/// one left over from another checkpoint can never be attached to records
|
/// one left over from another checkpoint can never be attached to records
|
||||||
/// it wasn't built from.
|
/// it wasn't built from.
|
||||||
pub ann_generation: Option<u64>,
|
pub ann_generation: Option<u64>,
|
||||||
|
/// The checkpoint carries an Ed25519 signature (see [`crate::signing`]).
|
||||||
|
/// Read-only: whether a checkpoint is *written* signed is decided by the
|
||||||
|
/// signature passed to [`build_hdf5_file_signed`].
|
||||||
|
pub signed: bool,
|
||||||
}
|
}
|
||||||
|
|
||||||
/// [`build_hdf5_file`] with checkpoint bookkeeping.
|
/// [`build_hdf5_file`] with checkpoint bookkeeping.
|
||||||
@@ -70,6 +78,19 @@ pub fn build_hdf5_file_with_meta(
|
|||||||
sessions: &SessionCache,
|
sessions: &SessionCache,
|
||||||
knowledge: &KnowledgeCache,
|
knowledge: &KnowledgeCache,
|
||||||
checkpoint: &CheckpointMeta,
|
checkpoint: &CheckpointMeta,
|
||||||
|
) -> Result<Vec<u8>, MemoryError> {
|
||||||
|
build_hdf5_file_signed(config, cache, sessions, knowledge, checkpoint, None)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`build_hdf5_file_with_meta`], plus a signed manifest of the contents
|
||||||
|
/// (see [`crate::signing`]).
|
||||||
|
pub fn build_hdf5_file_signed(
|
||||||
|
config: &MemoryConfig,
|
||||||
|
cache: &MemoryCache,
|
||||||
|
sessions: &SessionCache,
|
||||||
|
knowledge: &KnowledgeCache,
|
||||||
|
checkpoint: &CheckpointMeta,
|
||||||
|
signature: Option<&crate::signing::StoredSignature>,
|
||||||
) -> Result<Vec<u8>, MemoryError> {
|
) -> Result<Vec<u8>, MemoryError> {
|
||||||
let wal_applied = checkpoint.wal_applied;
|
let wal_applied = checkpoint.wal_applied;
|
||||||
let mut builder = clawhdf5::FileBuilder::new();
|
let mut builder = clawhdf5::FileBuilder::new();
|
||||||
@@ -130,11 +151,42 @@ pub fn build_hdf5_file_with_meta(
|
|||||||
// round trip through every reader.
|
// round trip through every reader.
|
||||||
meta.set_attr(ANN_GENERATION_ATTR, AttrValue::I64(generation as i64));
|
meta.set_attr(ANN_GENERATION_ATTR, AttrValue::I64(generation as i64));
|
||||||
}
|
}
|
||||||
|
if let Some(sig) = signature {
|
||||||
|
use crate::signing::to_hex;
|
||||||
|
let m = &sig.manifest;
|
||||||
|
meta.set_attr(
|
||||||
|
SIG_VERSION_ATTR,
|
||||||
|
AttrValue::I64(crate::signing::MANIFEST_VERSION),
|
||||||
|
);
|
||||||
|
meta.set_attr("sig_algorithm", AttrValue::String("ed25519".into()));
|
||||||
|
meta.set_attr("sig_public_key", AttrValue::String(to_hex(&sig.public_key)));
|
||||||
|
meta.set_attr("sig_signature", AttrValue::String(to_hex(&sig.signature)));
|
||||||
|
meta.set_attr("sig_record_count", AttrValue::I64(m.record_count as i64));
|
||||||
|
meta.set_attr(
|
||||||
|
"sig_records_root",
|
||||||
|
AttrValue::String(to_hex(&m.records_root)),
|
||||||
|
);
|
||||||
|
meta.set_attr("sig_settings", AttrValue::String(to_hex(&m.settings)));
|
||||||
|
meta.set_attr("sig_sessions", AttrValue::String(to_hex(&m.sessions)));
|
||||||
|
meta.set_attr("sig_graph", AttrValue::String(to_hex(&m.graph)));
|
||||||
|
}
|
||||||
// Need at least one dataset in the group for it to be a proper group
|
// Need at least one dataset in the group for it to be a proper group
|
||||||
meta.create_dataset("_marker").with_u8_data(&[1]).compact();
|
meta.create_dataset("_marker").with_u8_data(&[1]).compact();
|
||||||
let finished_meta = meta.finish();
|
let finished_meta = meta.finish();
|
||||||
builder.add_group(finished_meta);
|
builder.add_group(finished_meta);
|
||||||
|
|
||||||
|
// /integrity: the signed per-record hashes, so verification can say
|
||||||
|
// which records changed.
|
||||||
|
if let Some(sig) = signature {
|
||||||
|
let mut group = builder.create_group("integrity");
|
||||||
|
let flat: Vec<u8> = sig.record_hashes.iter().flatten().copied().collect();
|
||||||
|
group
|
||||||
|
.create_dataset("record_hashes")
|
||||||
|
.with_u8_data(&flat)
|
||||||
|
.with_shape(&[sig.record_hashes.len() as u64, 32]);
|
||||||
|
builder.add_group(group.finish());
|
||||||
|
}
|
||||||
|
|
||||||
// /memory group
|
// /memory group
|
||||||
build_memory_group(&mut builder, config, cache)?;
|
build_memory_group(&mut builder, config, cache)?;
|
||||||
|
|
||||||
@@ -425,10 +477,34 @@ fn write_string_dataset(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// `/meta`'s attributes, failing if any of them cannot be read.
|
||||||
|
///
|
||||||
|
/// `Group::attrs` leaves out an attribute it cannot decode. For the store's
|
||||||
|
/// settings that would silently fall back to defaults (e.g. `float16`, the
|
||||||
|
/// WAL mark), so an unreadable attribute is an error here, as it was before
|
||||||
|
/// `attrs` became tolerant.
|
||||||
|
fn meta_attrs(
|
||||||
|
file: &clawhdf5::File,
|
||||||
|
) -> Result<std::collections::HashMap<String, AttrValue>, MemoryError> {
|
||||||
|
let meta = file
|
||||||
|
.group("meta")
|
||||||
|
.map_err(|e| MemoryError::Schema(format!("missing /meta group: {e}")))?;
|
||||||
|
let (attrs, errors) = meta
|
||||||
|
.attrs_with_errors()
|
||||||
|
.map_err(|e| MemoryError::Schema(format!("cannot read /meta attrs: {e}")))?;
|
||||||
|
if let Some(e) = errors.first() {
|
||||||
|
return Err(MemoryError::Schema(format!(
|
||||||
|
"cannot read /meta attrs: {} unreadable, first: {e}",
|
||||||
|
errors.len()
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
Ok(attrs)
|
||||||
|
}
|
||||||
|
|
||||||
/// Validate an HDF5 file has the correct schema and load all data.
|
/// Validate an HDF5 file has the correct schema and load all data.
|
||||||
/// Read the checkpoint's [`WalMark`] from `/meta`, if it has one.
|
/// Read the checkpoint's [`WalMark`] from `/meta`, if it has one.
|
||||||
pub fn read_wal_mark(file: &clawhdf5::File) -> Option<WalMark> {
|
pub fn read_wal_mark(file: &clawhdf5::File) -> Option<WalMark> {
|
||||||
let attrs = file.group("meta").ok()?.attrs().ok()?;
|
let attrs = meta_attrs(file).ok()?;
|
||||||
let len = match attrs.get(WAL_APPLIED_LEN_ATTR)? {
|
let len = match attrs.get(WAL_APPLIED_LEN_ATTR)? {
|
||||||
AttrValue::I64(v) => u64::try_from(*v).ok()?,
|
AttrValue::I64(v) => u64::try_from(*v).ok()?,
|
||||||
_ => return None,
|
_ => return None,
|
||||||
@@ -440,19 +516,75 @@ pub fn read_wal_mark(file: &clawhdf5::File) -> Option<WalMark> {
|
|||||||
Some(WalMark { len, crc })
|
Some(WalMark { len, crc })
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Read a checkpoint's signature, if it has one. A signature whose
|
||||||
|
/// attributes are present but malformed is an error, not "unsigned".
|
||||||
|
pub fn read_signature(
|
||||||
|
file: &clawhdf5::File,
|
||||||
|
) -> Result<Option<crate::signing::StoredSignature>, MemoryError> {
|
||||||
|
use crate::signing::{Manifest, StoredSignature, from_hex};
|
||||||
|
let attrs = meta_attrs(file)?;
|
||||||
|
let version = match attrs.get(SIG_VERSION_ATTR) {
|
||||||
|
None => return Ok(None),
|
||||||
|
Some(AttrValue::I64(v)) => *v,
|
||||||
|
Some(_) => return Err(MemoryError::Schema("malformed sig_version".into())),
|
||||||
|
};
|
||||||
|
if version != crate::signing::MANIFEST_VERSION {
|
||||||
|
return Err(MemoryError::Schema(format!(
|
||||||
|
"unsupported signature version {version}"
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
fn hex<const N: usize>(
|
||||||
|
attrs: &std::collections::HashMap<String, AttrValue>,
|
||||||
|
name: &str,
|
||||||
|
) -> Result<[u8; N], MemoryError> {
|
||||||
|
match attrs.get(name) {
|
||||||
|
Some(AttrValue::String(s)) => from_hex::<N>(s),
|
||||||
|
_ => None,
|
||||||
|
}
|
||||||
|
.ok_or_else(|| MemoryError::Schema(format!("malformed or missing {name}")))
|
||||||
|
}
|
||||||
|
let record_count = match attrs.get("sig_record_count") {
|
||||||
|
Some(AttrValue::I64(v)) if *v >= 0 => *v as u64,
|
||||||
|
_ => return Err(MemoryError::Schema("malformed sig_record_count".into())),
|
||||||
|
};
|
||||||
|
let group = file
|
||||||
|
.group("integrity")
|
||||||
|
.map_err(|e| MemoryError::Schema(format!("signed checkpoint without /integrity: {e}")))?;
|
||||||
|
let flat = read_u8_dataset(&group, "record_hashes")?;
|
||||||
|
if flat.len() % 32 != 0 {
|
||||||
|
return Err(MemoryError::Schema(
|
||||||
|
"/integrity/record_hashes is not a whole number of hashes".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
let record_hashes = flat.as_chunks::<32>().0.to_vec();
|
||||||
|
Ok(Some(StoredSignature {
|
||||||
|
manifest: Manifest {
|
||||||
|
record_count,
|
||||||
|
records_root: hex::<32>(&attrs, "sig_records_root")?,
|
||||||
|
settings: hex::<32>(&attrs, "sig_settings")?,
|
||||||
|
sessions: hex::<32>(&attrs, "sig_sessions")?,
|
||||||
|
graph: hex::<32>(&attrs, "sig_graph")?,
|
||||||
|
},
|
||||||
|
record_hashes,
|
||||||
|
public_key: hex::<32>(&attrs, "sig_public_key")?,
|
||||||
|
signature: hex::<64>(&attrs, "sig_signature")?,
|
||||||
|
}))
|
||||||
|
}
|
||||||
|
|
||||||
/// Read the checkpoint bookkeeping from `/meta`.
|
/// Read the checkpoint bookkeeping from `/meta`.
|
||||||
pub fn read_checkpoint_meta(file: &clawhdf5::File) -> CheckpointMeta {
|
pub fn read_checkpoint_meta(file: &clawhdf5::File) -> CheckpointMeta {
|
||||||
let ann_generation = file
|
let ann_generation =
|
||||||
.group("meta")
|
meta_attrs(file)
|
||||||
.ok()
|
.ok()
|
||||||
.and_then(|g| g.attrs().ok())
|
.and_then(|attrs| match attrs.get(ANN_GENERATION_ATTR) {
|
||||||
.and_then(|attrs| match attrs.get(ANN_GENERATION_ATTR) {
|
Some(AttrValue::I64(v)) => Some(*v as u64),
|
||||||
Some(AttrValue::I64(v)) => Some(*v as u64),
|
_ => None,
|
||||||
_ => None,
|
});
|
||||||
});
|
let signed = meta_attrs(file).is_ok_and(|attrs| attrs.contains_key(SIG_VERSION_ATTR));
|
||||||
CheckpointMeta {
|
CheckpointMeta {
|
||||||
wal_applied: read_wal_mark(file),
|
wal_applied: read_wal_mark(file),
|
||||||
ann_generation,
|
ann_generation,
|
||||||
|
signed,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -460,12 +592,7 @@ pub fn validate_and_load(
|
|||||||
file: &clawhdf5::File,
|
file: &clawhdf5::File,
|
||||||
) -> Result<(MemoryConfig, MemoryCache, SessionCache, KnowledgeCache), MemoryError> {
|
) -> Result<(MemoryConfig, MemoryCache, SessionCache, KnowledgeCache), MemoryError> {
|
||||||
// Read /meta group attributes
|
// Read /meta group attributes
|
||||||
let meta = file
|
let attrs = meta_attrs(file)?;
|
||||||
.group("meta")
|
|
||||||
.map_err(|e| MemoryError::Schema(format!("missing /meta group: {e}")))?;
|
|
||||||
let attrs = meta
|
|
||||||
.attrs()
|
|
||||||
.map_err(|e| MemoryError::Schema(format!("cannot read /meta attrs: {e}")))?;
|
|
||||||
|
|
||||||
let schema_version = match attrs.get("schema_version") {
|
let schema_version = match attrs.get("schema_version") {
|
||||||
Some(AttrValue::String(s)) => s.clone(),
|
Some(AttrValue::String(s)) => s.clone(),
|
||||||
|
|||||||
@@ -0,0 +1,419 @@
|
|||||||
|
//! Ed25519-signed checkpoints.
|
||||||
|
//!
|
||||||
|
//! When a signing key is set ([`crate::HDF5Memory::set_signing_key`]), every
|
||||||
|
//! checkpoint writes a signed manifest of the store: a SHA-256 per memory
|
||||||
|
//! record rolled into a Merkle root, plus hashes of the store's settings, its
|
||||||
|
//! sessions and its knowledge graph. [`verify_store`] recomputes all of it from
|
||||||
|
//! the file and checks the signature against a public key the caller trusts,
|
||||||
|
//! so any change to the checkpointed file — a record's text or embedding, a
|
||||||
|
//! setting, a session, a graph edge, made through this crate or any other HDF5
|
||||||
|
//! tool — is detected, and the per-record hashes say which records changed.
|
||||||
|
//!
|
||||||
|
//! What it does not cover: saves still only in the WAL (made since the last
|
||||||
|
//! checkpoint). [`VerifyReport::wal_entries_unsigned`] counts them.
|
||||||
|
//!
|
||||||
|
//! The hashes cover exactly what the file persists, in the form the loader
|
||||||
|
//! returns it, so a store verifies after any number of reopen/checkpoint
|
||||||
|
//! cycles. Derived data (L2 norms, the vector index) is not covered; it is
|
||||||
|
//! recomputed from covered data.
|
||||||
|
|
||||||
|
use ed25519_dalek::{Signature, Signer, Verifier};
|
||||||
|
pub use ed25519_dalek::{SigningKey, VerifyingKey};
|
||||||
|
use sha2::{Digest, Sha256};
|
||||||
|
|
||||||
|
use crate::MemoryConfig;
|
||||||
|
use crate::cache::MemoryCache;
|
||||||
|
use crate::knowledge::KnowledgeCache;
|
||||||
|
use crate::session::SessionCache;
|
||||||
|
use crate::wal::WalMark;
|
||||||
|
|
||||||
|
/// Version of the manifest encoding; part of what is signed.
|
||||||
|
pub const MANIFEST_VERSION: i64 = 1;
|
||||||
|
|
||||||
|
type Hash = [u8; 32];
|
||||||
|
|
||||||
|
/// The hashes a signature covers.
|
||||||
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||||
|
pub struct Manifest {
|
||||||
|
pub record_count: u64,
|
||||||
|
/// Merkle root over the per-record hashes.
|
||||||
|
pub records_root: Hash,
|
||||||
|
/// Settings persisted in `/meta`, plus the checkpoint's WAL mark.
|
||||||
|
pub settings: Hash,
|
||||||
|
pub sessions: Hash,
|
||||||
|
pub graph: Hash,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Manifest {
|
||||||
|
/// The exact bytes that are signed.
|
||||||
|
pub fn signed_bytes(&self) -> Vec<u8> {
|
||||||
|
let mut m = Vec::with_capacity(160);
|
||||||
|
m.extend_from_slice(b"clawhdf5-agent signed checkpoint\0");
|
||||||
|
m.extend_from_slice(&MANIFEST_VERSION.to_le_bytes());
|
||||||
|
m.extend_from_slice(&self.record_count.to_le_bytes());
|
||||||
|
m.extend_from_slice(&self.records_root);
|
||||||
|
m.extend_from_slice(&self.settings);
|
||||||
|
m.extend_from_slice(&self.sessions);
|
||||||
|
m.extend_from_slice(&self.graph);
|
||||||
|
m
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A signature as stored in a checkpoint.
|
||||||
|
#[derive(Debug, Clone)]
|
||||||
|
pub struct StoredSignature {
|
||||||
|
pub manifest: Manifest,
|
||||||
|
pub record_hashes: Vec<Hash>,
|
||||||
|
pub public_key: [u8; 32],
|
||||||
|
pub signature: [u8; 64],
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Build the manifest (and per-record hashes) for the state about to be
|
||||||
|
/// checkpointed, and sign it.
|
||||||
|
pub fn sign(
|
||||||
|
key: &SigningKey,
|
||||||
|
config: &MemoryConfig,
|
||||||
|
cache: &MemoryCache,
|
||||||
|
sessions: &SessionCache,
|
||||||
|
knowledge: &KnowledgeCache,
|
||||||
|
wal_applied: Option<WalMark>,
|
||||||
|
) -> StoredSignature {
|
||||||
|
let (manifest, record_hashes) = manifest(config, cache, sessions, knowledge, wal_applied);
|
||||||
|
let signature = key.sign(&manifest.signed_bytes()).to_bytes();
|
||||||
|
StoredSignature {
|
||||||
|
manifest,
|
||||||
|
record_hashes,
|
||||||
|
public_key: key.verifying_key().to_bytes(),
|
||||||
|
signature,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Compute the manifest of a store's state.
|
||||||
|
pub fn manifest(
|
||||||
|
config: &MemoryConfig,
|
||||||
|
cache: &MemoryCache,
|
||||||
|
sessions: &SessionCache,
|
||||||
|
knowledge: &KnowledgeCache,
|
||||||
|
wal_applied: Option<WalMark>,
|
||||||
|
) -> (Manifest, Vec<Hash>) {
|
||||||
|
let record_hashes: Vec<Hash> = (0..cache.len()).map(|i| record_hash(cache, i)).collect();
|
||||||
|
let manifest = Manifest {
|
||||||
|
record_count: cache.len() as u64,
|
||||||
|
records_root: merkle_root(&record_hashes),
|
||||||
|
settings: settings_hash(config, wal_applied),
|
||||||
|
sessions: sessions_hash(sessions),
|
||||||
|
graph: graph_hash(knowledge),
|
||||||
|
};
|
||||||
|
(manifest, record_hashes)
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Canonical encoding
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// A SHA-256 over length-prefixed fields, so no two different field lists
|
||||||
|
/// hash the same bytes.
|
||||||
|
struct Fields(Sha256);
|
||||||
|
|
||||||
|
impl Fields {
|
||||||
|
fn new(domain: &str) -> Self {
|
||||||
|
let mut h = Sha256::new();
|
||||||
|
h.update((domain.len() as u64).to_le_bytes());
|
||||||
|
h.update(domain.as_bytes());
|
||||||
|
Self(h)
|
||||||
|
}
|
||||||
|
fn bytes(&mut self, b: &[u8]) -> &mut Self {
|
||||||
|
self.0.update((b.len() as u64).to_le_bytes());
|
||||||
|
self.0.update(b);
|
||||||
|
self
|
||||||
|
}
|
||||||
|
/// Strings as the loader returns them: stored null-padded, so a trailing
|
||||||
|
/// NUL cannot survive a round trip and must not be part of the hash.
|
||||||
|
fn str(&mut self, s: &str) -> &mut Self {
|
||||||
|
self.bytes(s.trim_end_matches('\0').as_bytes())
|
||||||
|
}
|
||||||
|
fn u64(&mut self, v: u64) -> &mut Self {
|
||||||
|
self.0.update(v.to_le_bytes());
|
||||||
|
self
|
||||||
|
}
|
||||||
|
fn f64(&mut self, v: f64) -> &mut Self {
|
||||||
|
self.0.update(v.to_bits().to_le_bytes());
|
||||||
|
self
|
||||||
|
}
|
||||||
|
fn f32(&mut self, v: f32) -> &mut Self {
|
||||||
|
self.0.update(v.to_bits().to_le_bytes());
|
||||||
|
self
|
||||||
|
}
|
||||||
|
fn finish(self) -> Hash {
|
||||||
|
self.0.finalize().into()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Everything persisted about record `i`, including its position. The
|
||||||
|
/// embedding is hashed as the cache holds it — for a `float16` store that is
|
||||||
|
/// the half-rounded value the file holds.
|
||||||
|
fn record_hash(cache: &MemoryCache, i: usize) -> Hash {
|
||||||
|
let mut f = Fields::new("clawhdf5-agent/record");
|
||||||
|
f.u64(i as u64).str(&cache.chunks[i]);
|
||||||
|
let emb: Vec<u8> = cache.embeddings[i]
|
||||||
|
.iter()
|
||||||
|
.flat_map(|v| v.to_bits().to_le_bytes())
|
||||||
|
.collect();
|
||||||
|
f.bytes(&emb)
|
||||||
|
.str(&cache.source_channels[i])
|
||||||
|
.f64(cache.timestamps[i])
|
||||||
|
.str(&cache.session_ids[i])
|
||||||
|
.str(&cache.tags[i])
|
||||||
|
.u64(u64::from(cache.tombstones[i]))
|
||||||
|
.f32(cache.activation_weights[i]);
|
||||||
|
f.finish()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Binary Merkle tree: leaves are the record hashes; a parent hashes its two
|
||||||
|
/// children with a node prefix; an odd node is carried up unchanged.
|
||||||
|
fn merkle_root(leaves: &[Hash]) -> Hash {
|
||||||
|
if leaves.is_empty() {
|
||||||
|
return Fields::new("clawhdf5-agent/merkle-empty").finish();
|
||||||
|
}
|
||||||
|
let mut level: Vec<Hash> = leaves.to_vec();
|
||||||
|
while level.len() > 1 {
|
||||||
|
level = level
|
||||||
|
.chunks(2)
|
||||||
|
.map(|pair| match pair {
|
||||||
|
[l, r] => {
|
||||||
|
let mut h = Sha256::new();
|
||||||
|
h.update([1u8]);
|
||||||
|
h.update(l);
|
||||||
|
h.update(r);
|
||||||
|
h.finalize().into()
|
||||||
|
}
|
||||||
|
[only] => *only,
|
||||||
|
_ => unreachable!(),
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
}
|
||||||
|
level[0]
|
||||||
|
}
|
||||||
|
|
||||||
|
fn settings_hash(c: &MemoryConfig, wal_applied: Option<WalMark>) -> Hash {
|
||||||
|
let mut f = Fields::new("clawhdf5-agent/settings");
|
||||||
|
f.str(crate::schema::SCHEMA_VERSION)
|
||||||
|
.str(&c.created_at)
|
||||||
|
.str(&c.agent_id)
|
||||||
|
.str(&c.embedder)
|
||||||
|
.u64(c.embedding_dim as u64)
|
||||||
|
.u64(c.chunk_size as u64)
|
||||||
|
.u64(c.overlap as u64)
|
||||||
|
.u64(u64::from(c.float16))
|
||||||
|
.u64(u64::from(c.compression))
|
||||||
|
.u64(u64::from(c.compression_level))
|
||||||
|
.f32(c.compact_threshold)
|
||||||
|
.f32(c.hebbian_boost)
|
||||||
|
.f32(c.decay_factor)
|
||||||
|
.u64(u64::from(c.wal_enabled))
|
||||||
|
.u64(c.wal_max_entries as u64)
|
||||||
|
.u64(u64::from(c.quantized_index))
|
||||||
|
.u64(c.hnsw_m as u64)
|
||||||
|
.u64(c.hnsw_ef_construction as u64)
|
||||||
|
.u64(c.hnsw_ef_search as u64);
|
||||||
|
// An empty mark is not written to the file, so it must hash as none.
|
||||||
|
match wal_applied.filter(|m| m.len > 0) {
|
||||||
|
Some(m) => f.u64(1).u64(m.len).u64(u64::from(m.crc)),
|
||||||
|
None => f.u64(0),
|
||||||
|
};
|
||||||
|
f.finish()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn sessions_hash(s: &SessionCache) -> Hash {
|
||||||
|
let mut f = Fields::new("clawhdf5-agent/sessions");
|
||||||
|
f.u64(s.entries.len() as u64);
|
||||||
|
for (i, e) in s.entries.iter().enumerate() {
|
||||||
|
f.str(&e.id)
|
||||||
|
.u64(e.start_idx)
|
||||||
|
.u64(e.end_idx)
|
||||||
|
.str(&e.channel)
|
||||||
|
.f64(e.ts)
|
||||||
|
.str(s.summaries.get(i).map(String::as_str).unwrap_or(""));
|
||||||
|
}
|
||||||
|
f.finish()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn graph_hash(k: &KnowledgeCache) -> Hash {
|
||||||
|
let mut f = Fields::new("clawhdf5-agent/graph");
|
||||||
|
f.u64(k.entities.len() as u64);
|
||||||
|
for e in &k.entities {
|
||||||
|
f.u64(e.id)
|
||||||
|
.str(&e.name)
|
||||||
|
.str(&e.entity_type)
|
||||||
|
.u64(e.embedding_idx as u64);
|
||||||
|
}
|
||||||
|
f.u64(k.relations.len() as u64);
|
||||||
|
for r in &k.relations {
|
||||||
|
f.u64(r.src)
|
||||||
|
.u64(r.tgt)
|
||||||
|
.str(&r.relation)
|
||||||
|
.f32(r.weight)
|
||||||
|
.f64(r.ts);
|
||||||
|
}
|
||||||
|
f.u64(k.alias_strings.len() as u64);
|
||||||
|
for (s, id) in k.alias_strings.iter().zip(&k.alias_entity_ids) {
|
||||||
|
f.str(s).u64(*id as u64);
|
||||||
|
}
|
||||||
|
f.finish()
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Verification
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// The outcome of [`verify_store`].
|
||||||
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||||
|
pub struct VerifyReport {
|
||||||
|
/// The checkpoint carries a signature.
|
||||||
|
pub signed: bool,
|
||||||
|
/// The signature was made by the key the caller trusts.
|
||||||
|
pub key_matches: bool,
|
||||||
|
/// The signature over the stored manifest is valid.
|
||||||
|
pub signature_valid: bool,
|
||||||
|
/// The file's current contents match the signed manifest.
|
||||||
|
pub records_match: bool,
|
||||||
|
pub settings_match: bool,
|
||||||
|
pub sessions_match: bool,
|
||||||
|
pub graph_match: bool,
|
||||||
|
/// Records whose contents differ from what was signed (by position),
|
||||||
|
/// when the stored per-record hashes are themselves authentic.
|
||||||
|
pub changed_records: Vec<usize>,
|
||||||
|
/// Records in the file versus in the signed manifest.
|
||||||
|
pub record_count: u64,
|
||||||
|
pub signed_record_count: u64,
|
||||||
|
/// The public key the checkpoint claims to be signed by.
|
||||||
|
pub public_key: Option<[u8; 32]>,
|
||||||
|
/// Saves in the WAL after the checkpoint: not covered by the signature.
|
||||||
|
pub wal_entries_unsigned: usize,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl VerifyReport {
|
||||||
|
/// Signed by the trusted key, signature valid, and every part of the
|
||||||
|
/// file unchanged since it was signed.
|
||||||
|
pub fn is_valid(&self) -> bool {
|
||||||
|
self.signed
|
||||||
|
&& self.key_matches
|
||||||
|
&& self.signature_valid
|
||||||
|
&& self.records_match
|
||||||
|
&& self.settings_match
|
||||||
|
&& self.sessions_match
|
||||||
|
&& self.graph_match
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Check a store file against the public key the caller trusts.
|
||||||
|
///
|
||||||
|
/// Reads the checkpoint (not the WAL), recomputes every hash from its
|
||||||
|
/// contents and checks the signature. Never writes.
|
||||||
|
pub fn verify_store(
|
||||||
|
path: &std::path::Path,
|
||||||
|
trusted: &VerifyingKey,
|
||||||
|
) -> Result<VerifyReport, crate::MemoryError> {
|
||||||
|
let file = clawhdf5::File::open(path)
|
||||||
|
.map_err(|e| crate::MemoryError::Hdf5(format!("cannot open {}: {e}", path.display())))?;
|
||||||
|
let (config, cache, sessions, knowledge) = crate::schema::validate_and_load(&file)?;
|
||||||
|
let checkpoint = crate::schema::read_checkpoint_meta(&file);
|
||||||
|
let stored = crate::schema::read_signature(&file)?;
|
||||||
|
let wal_entries_unsigned = count_wal_entries_after(path, checkpoint.wal_applied);
|
||||||
|
|
||||||
|
let (current, current_hashes) = manifest(
|
||||||
|
&config,
|
||||||
|
&cache,
|
||||||
|
&sessions,
|
||||||
|
&knowledge,
|
||||||
|
checkpoint.wal_applied,
|
||||||
|
);
|
||||||
|
|
||||||
|
let Some(stored) = stored else {
|
||||||
|
return Ok(VerifyReport {
|
||||||
|
signed: false,
|
||||||
|
key_matches: false,
|
||||||
|
signature_valid: false,
|
||||||
|
records_match: false,
|
||||||
|
settings_match: false,
|
||||||
|
sessions_match: false,
|
||||||
|
graph_match: false,
|
||||||
|
changed_records: Vec::new(),
|
||||||
|
record_count: current.record_count,
|
||||||
|
signed_record_count: 0,
|
||||||
|
public_key: None,
|
||||||
|
wal_entries_unsigned,
|
||||||
|
});
|
||||||
|
};
|
||||||
|
|
||||||
|
let key_matches = stored.public_key == trusted.to_bytes();
|
||||||
|
let signature_valid = trusted
|
||||||
|
.verify(
|
||||||
|
&stored.manifest.signed_bytes(),
|
||||||
|
&Signature::from_bytes(&stored.signature),
|
||||||
|
)
|
||||||
|
.is_ok();
|
||||||
|
// The stored per-record hashes can localise a change only if they are
|
||||||
|
// the ones that were signed.
|
||||||
|
let hashes_authentic = signature_valid
|
||||||
|
&& stored.record_hashes.len() as u64 == stored.manifest.record_count
|
||||||
|
&& merkle_root(&stored.record_hashes) == stored.manifest.records_root;
|
||||||
|
let changed_records = if hashes_authentic {
|
||||||
|
let n = current_hashes.len().max(stored.record_hashes.len());
|
||||||
|
(0..n)
|
||||||
|
.filter(|&i| current_hashes.get(i) != stored.record_hashes.get(i))
|
||||||
|
.collect()
|
||||||
|
} else {
|
||||||
|
Vec::new()
|
||||||
|
};
|
||||||
|
|
||||||
|
Ok(VerifyReport {
|
||||||
|
signed: true,
|
||||||
|
key_matches,
|
||||||
|
signature_valid,
|
||||||
|
records_match: signature_valid
|
||||||
|
&& current.record_count == stored.manifest.record_count
|
||||||
|
&& current.records_root == stored.manifest.records_root,
|
||||||
|
settings_match: signature_valid && current.settings == stored.manifest.settings,
|
||||||
|
sessions_match: signature_valid && current.sessions == stored.manifest.sessions,
|
||||||
|
graph_match: signature_valid && current.graph == stored.manifest.graph,
|
||||||
|
changed_records,
|
||||||
|
record_count: current.record_count,
|
||||||
|
signed_record_count: stored.manifest.record_count,
|
||||||
|
public_key: Some(stored.public_key),
|
||||||
|
wal_entries_unsigned,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
fn count_wal_entries_after(store: &std::path::Path, mark: Option<WalMark>) -> usize {
|
||||||
|
let wal = store.with_extension("h5.wal");
|
||||||
|
if !wal.exists() {
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
crate::wal::WalFile::read_entries_for_migration(&wal, mark)
|
||||||
|
.map(|e| e.len())
|
||||||
|
.unwrap_or(0)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A new random signing key from the operating system's RNG.
|
||||||
|
pub fn generate_key() -> SigningKey {
|
||||||
|
SigningKey::generate(&mut rand_core::OsRng)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Hex encoding for keys and signatures in attributes and the CLI.
|
||||||
|
pub fn to_hex(bytes: &[u8]) -> String {
|
||||||
|
bytes.iter().map(|b| format!("{b:02x}")).collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Parse hex into exactly `N` bytes.
|
||||||
|
pub fn from_hex<const N: usize>(s: &str) -> Option<[u8; N]> {
|
||||||
|
let s = s.trim();
|
||||||
|
if s.len() != 2 * N {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
let mut out = [0u8; N];
|
||||||
|
for (i, byte) in out.iter_mut().enumerate() {
|
||||||
|
*byte = u8::from_str_radix(&s[2 * i..2 * i + 2], 16).ok()?;
|
||||||
|
}
|
||||||
|
Some(out)
|
||||||
|
}
|
||||||
@@ -36,7 +36,7 @@ pub fn write_to_disk_with_mark(
|
|||||||
) -> Result<(), MemoryError> {
|
) -> Result<(), MemoryError> {
|
||||||
let meta = schema::CheckpointMeta {
|
let meta = schema::CheckpointMeta {
|
||||||
wal_applied,
|
wal_applied,
|
||||||
ann_generation: None,
|
..schema::CheckpointMeta::default()
|
||||||
};
|
};
|
||||||
write_to_disk_with_meta(path, config, cache, sessions, knowledge, &meta)
|
write_to_disk_with_meta(path, config, cache, sessions, knowledge, &meta)
|
||||||
}
|
}
|
||||||
@@ -50,7 +50,21 @@ pub fn write_to_disk_with_meta(
|
|||||||
knowledge: &KnowledgeCache,
|
knowledge: &KnowledgeCache,
|
||||||
checkpoint: &schema::CheckpointMeta,
|
checkpoint: &schema::CheckpointMeta,
|
||||||
) -> Result<(), MemoryError> {
|
) -> Result<(), MemoryError> {
|
||||||
let bytes = schema::build_hdf5_file_with_meta(config, cache, sessions, knowledge, checkpoint)?;
|
write_to_disk_signed(path, config, cache, sessions, knowledge, checkpoint, None)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`write_to_disk_with_meta`] with a signed manifest of the contents.
|
||||||
|
pub fn write_to_disk_signed(
|
||||||
|
path: &Path,
|
||||||
|
config: &MemoryConfig,
|
||||||
|
cache: &MemoryCache,
|
||||||
|
sessions: &SessionCache,
|
||||||
|
knowledge: &KnowledgeCache,
|
||||||
|
checkpoint: &schema::CheckpointMeta,
|
||||||
|
signature: Option<&crate::signing::StoredSignature>,
|
||||||
|
) -> Result<(), MemoryError> {
|
||||||
|
let bytes =
|
||||||
|
schema::build_hdf5_file_signed(config, cache, sessions, knowledge, checkpoint, signature)?;
|
||||||
|
|
||||||
if bytes.is_empty() {
|
if bytes.is_empty() {
|
||||||
return Err(MemoryError::Hdf5("build_hdf5_file produced 0 bytes".into()));
|
return Err(MemoryError::Hdf5("build_hdf5_file produced 0 bytes".into()));
|
||||||
|
|||||||
@@ -20,7 +20,7 @@ const LOCK_RETRY_DELAY: std::time::Duration = std::time::Duration::from_millis(1
|
|||||||
/// never leaves a stale lock behind; the empty lock file itself is harmless).
|
/// never leaves a stale lock behind; the empty lock file itself is harmless).
|
||||||
#[derive(Debug)]
|
#[derive(Debug)]
|
||||||
pub(crate) struct StoreLock {
|
pub(crate) struct StoreLock {
|
||||||
_file: File,
|
file: File,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl StoreLock {
|
impl StoreLock {
|
||||||
@@ -42,7 +42,7 @@ impl StoreLock {
|
|||||||
let mut attempts_left = LOCK_RETRIES;
|
let mut attempts_left = LOCK_RETRIES;
|
||||||
loop {
|
loop {
|
||||||
match file.try_lock() {
|
match file.try_lock() {
|
||||||
Ok(()) => return Ok(Self { _file: file }),
|
Ok(()) => return Ok(Self { file }),
|
||||||
Err(TryLockError::WouldBlock) if attempts_left > 0 => {
|
Err(TryLockError::WouldBlock) if attempts_left > 0 => {
|
||||||
attempts_left -= 1;
|
attempts_left -= 1;
|
||||||
std::thread::sleep(LOCK_RETRY_DELAY);
|
std::thread::sleep(LOCK_RETRY_DELAY);
|
||||||
@@ -60,6 +60,16 @@ impl StoreLock {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
impl Drop for StoreLock {
|
||||||
|
/// Unlocks before the file is closed: a process another thread forks
|
||||||
|
/// inherits the descriptor until it execs, and a `flock` lasts while any
|
||||||
|
/// descriptor of the open file does, so closing alone could keep the
|
||||||
|
/// store locked for a moment after the drop (see `FileEditor`'s `Drop`).
|
||||||
|
fn drop(&mut self) {
|
||||||
|
let _ = self.file.unlock();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
mod tests {
|
mod tests {
|
||||||
use super::*;
|
use super::*;
|
||||||
|
|||||||
@@ -258,3 +258,43 @@ fn an_existing_f32_store_stays_f32() {
|
|||||||
assert_eq!(&values[..before.1.len()], before.1.as_slice());
|
assert_eq!(&values[..before.1.len()], before.1.as_slice());
|
||||||
assert_eq!(&values[before.1.len()..], odd.as_slice());
|
assert_eq!(&values[before.1.len()..], odd.as_slice());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// `Group::attrs` leaves out an attribute it cannot decode. A store whose
|
||||||
|
/// `float16` setting is unreadable must not open as `float16 = false` (or with
|
||||||
|
/// any other default in place of a setting it has): it is an error.
|
||||||
|
#[test]
|
||||||
|
fn unreadable_meta_attribute_fails_open_instead_of_defaulting() {
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let path = dir.path().join("store.h5");
|
||||||
|
{
|
||||||
|
let mut m = HDF5Memory::create(config(&dir, "store.h5", true)).unwrap();
|
||||||
|
m.save(entry(1)).unwrap();
|
||||||
|
m.flush_wal().unwrap();
|
||||||
|
}
|
||||||
|
assert!(HDF5Memory::open_read_only(&path).is_ok());
|
||||||
|
|
||||||
|
// Give the `float16` attribute message an unknown version (the name is
|
||||||
|
// at +8 in a version-1 message and +9 in a version-3 one).
|
||||||
|
let mut bytes = std::fs::read(&path).unwrap();
|
||||||
|
let name = b"float16\0";
|
||||||
|
let mut hit = false;
|
||||||
|
let positions: Vec<usize> = (9..bytes.len() - name.len())
|
||||||
|
.filter(|&p| &bytes[p..p + name.len()] == name)
|
||||||
|
.collect();
|
||||||
|
for pos in positions {
|
||||||
|
for (back, version) in [(8, 1u8), (9, 3u8)] {
|
||||||
|
if bytes[pos - back] == version {
|
||||||
|
bytes[pos - back] = 0x7f;
|
||||||
|
hit = true;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
assert!(hit, "float16 attribute message not found");
|
||||||
|
std::fs::write(&path, &bytes).unwrap();
|
||||||
|
|
||||||
|
match HDF5Memory::open_read_only(&path) {
|
||||||
|
Err(MemoryError::Schema(msg)) => assert!(msg.contains("/meta"), "{msg}"),
|
||||||
|
Err(e) => panic!("unexpected error: {e}"),
|
||||||
|
Ok(_) => panic!("store opened with an unreadable float16 setting"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -92,3 +92,64 @@ print(len(names))
|
|||||||
assert!(n >= 10, "only {n} datasets");
|
assert!(n >= 10, "only {n} datasets");
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn an_edit_made_with_h5py_breaks_the_signature_and_names_the_record() {
|
||||||
|
if !h5py_available() {
|
||||||
|
assert!(
|
||||||
|
std::env::var("CLAWHDF5_REQUIRE_INTEROP").as_deref() != Ok("1"),
|
||||||
|
"CLAWHDF5_REQUIRE_INTEROP=1 but python3 with h5py is not available"
|
||||||
|
);
|
||||||
|
eprintln!("SKIP: python3 with h5py not available");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
use clawhdf5_agent::signing::SigningKey;
|
||||||
|
let dir = tempfile::tempdir().unwrap();
|
||||||
|
let path = dir.path().join("signed.h5");
|
||||||
|
let key = SigningKey::from_bytes(&[42; 32]);
|
||||||
|
let mut m = HDF5Memory::create(MemoryConfig::new(path.clone(), "agent", 8)).unwrap();
|
||||||
|
m.set_signing_key(key.clone());
|
||||||
|
m.save_batch(
|
||||||
|
(0..10)
|
||||||
|
.map(|i| MemoryEntry {
|
||||||
|
chunk: format!("memory {i}"),
|
||||||
|
embedding: (0..8).map(|j| ((i * 8 + j) as f32).cos()).collect(),
|
||||||
|
source_channel: "test".into(),
|
||||||
|
timestamp: i as f64,
|
||||||
|
session_id: "s".into(),
|
||||||
|
tags: String::new(),
|
||||||
|
})
|
||||||
|
.collect(),
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
drop(m);
|
||||||
|
assert!(
|
||||||
|
HDF5Memory::verify(&path, &key.verifying_key())
|
||||||
|
.unwrap()
|
||||||
|
.is_valid()
|
||||||
|
);
|
||||||
|
|
||||||
|
// Someone edits one timestamp in place with h5py.
|
||||||
|
let script = format!(
|
||||||
|
r#"
|
||||||
|
import h5py
|
||||||
|
with h5py.File("{}", "r+") as f:
|
||||||
|
ts = f["memory/timestamps"]
|
||||||
|
ts[3] = 12345.0
|
||||||
|
"#,
|
||||||
|
path.display()
|
||||||
|
);
|
||||||
|
let out = Command::new(python())
|
||||||
|
.args(["-c", &script])
|
||||||
|
.output()
|
||||||
|
.unwrap();
|
||||||
|
assert!(
|
||||||
|
out.status.success(),
|
||||||
|
"{}",
|
||||||
|
String::from_utf8_lossy(&out.stderr)
|
||||||
|
);
|
||||||
|
|
||||||
|
let r = HDF5Memory::verify(&path, &key.verifying_key()).unwrap();
|
||||||
|
assert!(r.signature_valid && !r.is_valid(), "{r:?}");
|
||||||
|
assert_eq!(r.changed_records, vec![3]);
|
||||||
|
}
|
||||||
|
|||||||
@@ -0,0 +1,330 @@
|
|||||||
|
//! Ed25519-signed checkpoints: `HDF5Memory::set_signing_key` and
|
||||||
|
//! `HDF5Memory::verify`.
|
||||||
|
|
||||||
|
use std::path::Path;
|
||||||
|
|
||||||
|
use clawhdf5_agent::signing::{SigningKey, VerifyReport, VerifyingKey};
|
||||||
|
use clawhdf5_agent::storage;
|
||||||
|
use clawhdf5_agent::{AgentMemory, HDF5Memory, MemoryConfig, MemoryEntry, MemoryError, schema};
|
||||||
|
use tempfile::TempDir;
|
||||||
|
|
||||||
|
const DIM: usize = 16;
|
||||||
|
|
||||||
|
fn key(seed: u8) -> SigningKey {
|
||||||
|
SigningKey::from_bytes(&[seed; 32])
|
||||||
|
}
|
||||||
|
|
||||||
|
fn entry(i: usize, chunk: &str) -> MemoryEntry {
|
||||||
|
MemoryEntry {
|
||||||
|
chunk: chunk.to_string(),
|
||||||
|
embedding: (0..DIM)
|
||||||
|
.map(|j| ((i * DIM + j) as f32 * 0.37).sin())
|
||||||
|
.collect(),
|
||||||
|
source_channel: "chat".into(),
|
||||||
|
timestamp: 1_700_000_000.0 + i as f64,
|
||||||
|
session_id: format!("s{}", i % 3),
|
||||||
|
tags: format!("t{i}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Awkward strings on purpose: they must hash the same after a round trip.
|
||||||
|
const TEXTS: [&str; 6] = [
|
||||||
|
"plain text",
|
||||||
|
"ünïcödé — 日本語 🙂",
|
||||||
|
"",
|
||||||
|
"trailing spaces ",
|
||||||
|
"tab\tand\nnewline",
|
||||||
|
"x",
|
||||||
|
];
|
||||||
|
|
||||||
|
fn signed_store(dir: &TempDir, float16: bool, k: &SigningKey) -> std::path::PathBuf {
|
||||||
|
let mut cfg = MemoryConfig::new(dir.path().join("s.h5"), "agent", DIM);
|
||||||
|
cfg.float16 = float16;
|
||||||
|
let path = cfg.path.clone();
|
||||||
|
let mut m = HDF5Memory::create(cfg).unwrap();
|
||||||
|
m.set_signing_key(k.clone());
|
||||||
|
let entries = (0..30).map(|i| entry(i, TEXTS[i % TEXTS.len()])).collect();
|
||||||
|
m.save_batch(entries).unwrap();
|
||||||
|
// Some graph and a deleted record, so every part of the manifest is used.
|
||||||
|
let a = m.knowledge_mut().add_entity("Alice", "person", 0);
|
||||||
|
let b = m.knowledge_mut().add_entity("Acme", "org", -1);
|
||||||
|
m.knowledge_mut().add_relation(a, b, "works_at", 0.75);
|
||||||
|
m.sessions_mut()
|
||||||
|
.add_at("s0", 0, 9, "chat", "first session", 1_700_000_000.0);
|
||||||
|
m.delete(4).unwrap();
|
||||||
|
m.flush_wal().unwrap();
|
||||||
|
path
|
||||||
|
}
|
||||||
|
|
||||||
|
fn verify(path: &Path, k: &SigningKey) -> VerifyReport {
|
||||||
|
HDF5Memory::verify(path, &k.verifying_key()).unwrap()
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn a_signed_store_verifies_through_reopen_and_checkpoint_cycles() {
|
||||||
|
for float16 in [true, false] {
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let k = key(7);
|
||||||
|
let path = signed_store(&dir, float16, &k);
|
||||||
|
let r = verify(&path, &k);
|
||||||
|
assert!(r.is_valid(), "float16={float16}: {r:?}");
|
||||||
|
assert_eq!(r.public_key, Some(k.verifying_key().to_bytes()));
|
||||||
|
assert_eq!(r.record_count, 30);
|
||||||
|
assert!(r.changed_records.is_empty());
|
||||||
|
|
||||||
|
// Reopen, change nothing, checkpoint again (with the key): still valid.
|
||||||
|
for _ in 0..3 {
|
||||||
|
let mut m = HDF5Memory::open(&path).unwrap();
|
||||||
|
assert!(m.is_signed());
|
||||||
|
m.set_signing_key(k.clone());
|
||||||
|
m.flush_wal().unwrap();
|
||||||
|
drop(m);
|
||||||
|
assert!(verify(&path, &k).is_valid());
|
||||||
|
}
|
||||||
|
// And after real changes, re-signed.
|
||||||
|
let mut m = HDF5Memory::open(&path).unwrap();
|
||||||
|
m.set_signing_key(k.clone());
|
||||||
|
m.save(entry(99, "added later")).unwrap();
|
||||||
|
m.hybrid_search(&entry(1, "").embedding, "text", 0.4, 0.6, 5);
|
||||||
|
m.flush_wal().unwrap();
|
||||||
|
drop(m);
|
||||||
|
let r = verify(&path, &k);
|
||||||
|
assert!(r.is_valid(), "{r:?}");
|
||||||
|
assert_eq!(r.record_count, 31);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn a_signed_store_refuses_to_checkpoint_without_its_key() {
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let k = key(1);
|
||||||
|
let path = signed_store(&dir, true, &k);
|
||||||
|
|
||||||
|
let mut m = HDF5Memory::open(&path).unwrap();
|
||||||
|
m.save(entry(50, "pending")).unwrap();
|
||||||
|
match m.flush_wal() {
|
||||||
|
Err(MemoryError::SigningKeyRequired(msg)) => assert!(msg.contains("signed"), "{msg}"),
|
||||||
|
other => panic!("expected SigningKeyRequired, got {other:?}"),
|
||||||
|
}
|
||||||
|
// The file is untouched and still valid; the save is still in the WAL.
|
||||||
|
let r = verify(&path, &k);
|
||||||
|
assert!(r.is_valid());
|
||||||
|
assert_eq!(r.wal_entries_unsigned, 1);
|
||||||
|
|
||||||
|
// Supplying the key lets the checkpoint through, signed.
|
||||||
|
m.set_signing_key(k.clone());
|
||||||
|
m.flush_wal().unwrap();
|
||||||
|
drop(m);
|
||||||
|
let r = verify(&path, &k);
|
||||||
|
assert!(r.is_valid());
|
||||||
|
assert_eq!((r.record_count, r.wal_entries_unsigned), (31, 0));
|
||||||
|
|
||||||
|
// Removing the signature on purpose writes it unsigned.
|
||||||
|
let mut m = HDF5Memory::open(&path).unwrap();
|
||||||
|
m.remove_signature();
|
||||||
|
m.flush_wal().unwrap();
|
||||||
|
drop(m);
|
||||||
|
let r = verify(&path, &k);
|
||||||
|
assert!(!r.signed && !r.is_valid());
|
||||||
|
assert!(!HDF5Memory::open(&path).unwrap().is_signed());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn the_wrong_key_does_not_verify_and_a_new_key_re_signs() {
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let (a, b) = (key(1), key(2));
|
||||||
|
let path = signed_store(&dir, true, &a);
|
||||||
|
let r = verify(&path, &b);
|
||||||
|
assert!(r.signed && !r.key_matches && !r.signature_valid && !r.is_valid());
|
||||||
|
|
||||||
|
let mut m = HDF5Memory::open(&path).unwrap();
|
||||||
|
m.set_signing_key(b.clone());
|
||||||
|
m.flush_wal().unwrap();
|
||||||
|
drop(m);
|
||||||
|
assert!(verify(&path, &b).is_valid());
|
||||||
|
assert!(!verify(&path, &a).is_valid());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Rewrite the store with changed contents but the *old* signature — what
|
||||||
|
/// someone with write access to the file, but not the key, can do.
|
||||||
|
fn tamper(path: &Path, change: impl FnOnce(&mut Tampered)) {
|
||||||
|
let file = clawhdf5::File::open(path).unwrap();
|
||||||
|
let (config, cache, sessions, knowledge) = schema::validate_and_load(&file).unwrap();
|
||||||
|
let checkpoint = schema::read_checkpoint_meta(&file);
|
||||||
|
let signature = schema::read_signature(&file).unwrap().unwrap();
|
||||||
|
drop(file);
|
||||||
|
let mut t = Tampered {
|
||||||
|
config,
|
||||||
|
cache,
|
||||||
|
sessions,
|
||||||
|
knowledge,
|
||||||
|
};
|
||||||
|
change(&mut t);
|
||||||
|
storage::write_to_disk_signed(
|
||||||
|
path,
|
||||||
|
&t.config,
|
||||||
|
&t.cache,
|
||||||
|
&t.sessions,
|
||||||
|
&t.knowledge,
|
||||||
|
&checkpoint,
|
||||||
|
Some(&signature),
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
}
|
||||||
|
|
||||||
|
struct Tampered {
|
||||||
|
config: MemoryConfig,
|
||||||
|
cache: clawhdf5_agent::cache::MemoryCache,
|
||||||
|
sessions: clawhdf5_agent::SessionCache,
|
||||||
|
knowledge: clawhdf5_agent::knowledge::KnowledgeCache,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn every_kind_of_edit_is_detected_and_located() {
|
||||||
|
let k = key(3);
|
||||||
|
type Edit = Box<dyn FnOnce(&mut Tampered)>;
|
||||||
|
type Case = (&'static str, Edit, fn(&VerifyReport) -> bool);
|
||||||
|
let cases: Vec<Case> = vec![
|
||||||
|
(
|
||||||
|
"record text",
|
||||||
|
Box::new(|t: &mut Tampered| t.cache.chunks[7] = "rewritten".into()),
|
||||||
|
|r| !r.records_match && r.changed_records == vec![7],
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"one embedding value",
|
||||||
|
Box::new(|t: &mut Tampered| {
|
||||||
|
let mut e = t.cache.embeddings[12].to_vec();
|
||||||
|
e[3] = 0.5;
|
||||||
|
t.cache.embeddings.set(12, &e);
|
||||||
|
}),
|
||||||
|
|r| r.changed_records == vec![12],
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"undelete",
|
||||||
|
Box::new(|t: &mut Tampered| t.cache.tombstones[4] = 0),
|
||||||
|
|r| r.changed_records == vec![4],
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"timestamp",
|
||||||
|
Box::new(|t: &mut Tampered| t.cache.timestamps[20] += 1.0),
|
||||||
|
|r| r.changed_records == vec![20],
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"record appended",
|
||||||
|
Box::new(|t: &mut Tampered| {
|
||||||
|
t.cache.push(
|
||||||
|
"new".into(),
|
||||||
|
vec![0.1; DIM],
|
||||||
|
"x".into(),
|
||||||
|
1.0,
|
||||||
|
"s".into(),
|
||||||
|
"".into(),
|
||||||
|
);
|
||||||
|
}),
|
||||||
|
|r| !r.records_match && r.changed_records == vec![30] && r.record_count == 31,
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"setting",
|
||||||
|
Box::new(|t: &mut Tampered| t.config.agent_id = "someone-else".into()),
|
||||||
|
|r| !r.settings_match && r.records_match,
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"session summary",
|
||||||
|
Box::new(|t: &mut Tampered| t.sessions.summaries[0] = "edited".into()),
|
||||||
|
|r| !r.sessions_match && r.records_match,
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"graph edge",
|
||||||
|
Box::new(|t: &mut Tampered| t.knowledge.relations[0].weight = 1.0),
|
||||||
|
|r| !r.graph_match && r.records_match,
|
||||||
|
),
|
||||||
|
];
|
||||||
|
for (name, edit, check) in cases {
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let path = signed_store(&dir, true, &k);
|
||||||
|
tamper(&path, edit);
|
||||||
|
let r = verify(&path, &k);
|
||||||
|
assert!(
|
||||||
|
r.signed && r.key_matches && r.signature_valid,
|
||||||
|
"{name}: {r:?}"
|
||||||
|
);
|
||||||
|
assert!(!r.is_valid(), "{name}: edit not detected: {r:?}");
|
||||||
|
assert!(check(&r), "{name}: {r:?}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn a_forged_manifest_fails_the_signature() {
|
||||||
|
// Recomputing the hashes for tampered contents does not help without the
|
||||||
|
// key: the signature no longer matches the manifest.
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let k = key(5);
|
||||||
|
let path = signed_store(&dir, true, &k);
|
||||||
|
let file = clawhdf5::File::open(&path).unwrap();
|
||||||
|
let (config, mut cache, sessions, knowledge) = schema::validate_and_load(&file).unwrap();
|
||||||
|
let checkpoint = schema::read_checkpoint_meta(&file);
|
||||||
|
let mut sig = schema::read_signature(&file).unwrap().unwrap();
|
||||||
|
drop(file);
|
||||||
|
cache.chunks[0] = "forged".into();
|
||||||
|
// Re-sign with an attacker key, then splice the victim's public key back.
|
||||||
|
let forged = clawhdf5_agent::signing::sign(
|
||||||
|
&key(66),
|
||||||
|
&config,
|
||||||
|
&cache,
|
||||||
|
&sessions,
|
||||||
|
&knowledge,
|
||||||
|
checkpoint.wal_applied,
|
||||||
|
);
|
||||||
|
sig.manifest = forged.manifest;
|
||||||
|
sig.record_hashes = forged.record_hashes;
|
||||||
|
storage::write_to_disk_signed(
|
||||||
|
&path,
|
||||||
|
&config,
|
||||||
|
&cache,
|
||||||
|
&sessions,
|
||||||
|
&knowledge,
|
||||||
|
&checkpoint,
|
||||||
|
Some(&sig),
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
let r = verify(&path, &k);
|
||||||
|
assert!(
|
||||||
|
r.key_matches && !r.signature_valid && !r.is_valid(),
|
||||||
|
"{r:?}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn an_unsigned_store_reports_unsigned() {
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let mut m = HDF5Memory::create(MemoryConfig::new(dir.path().join("u.h5"), "a", DIM)).unwrap();
|
||||||
|
m.save_batch(vec![entry(0, "hello")]).unwrap();
|
||||||
|
drop(m);
|
||||||
|
let r = HDF5Memory::verify(&dir.path().join("u.h5"), &VerifyingKey::from(&key(1))).unwrap();
|
||||||
|
assert!(!r.signed && !r.is_valid());
|
||||||
|
assert_eq!(r.record_count, 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn nul_bytes_in_text_still_verify() {
|
||||||
|
// Strings are stored null-padded; the hash must follow what a reopened
|
||||||
|
// store actually holds, or an untouched store would fail to verify.
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let k = key(9);
|
||||||
|
let mut m = HDF5Memory::create(MemoryConfig::new(dir.path().join("n.h5"), "a", DIM)).unwrap();
|
||||||
|
m.set_signing_key(k.clone());
|
||||||
|
m.save_batch(vec![
|
||||||
|
entry(0, "inner\0nul"),
|
||||||
|
entry(1, "trailing nul\0"),
|
||||||
|
entry(2, "\0leading"),
|
||||||
|
])
|
||||||
|
.unwrap();
|
||||||
|
drop(m);
|
||||||
|
let r = verify(&dir.path().join("n.h5"), &k);
|
||||||
|
assert!(r.is_valid(), "{r:?}");
|
||||||
|
let m = HDF5Memory::open(&dir.path().join("n.h5")).unwrap();
|
||||||
|
eprintln!(
|
||||||
|
"reloaded: {:?}",
|
||||||
|
(0..3).map(|i| m.get_chunk(i)).collect::<Vec<_>>()
|
||||||
|
);
|
||||||
|
}
|
||||||
@@ -3,7 +3,7 @@ name = "clawhdf5-android"
|
|||||||
version = "2.7.0"
|
version = "2.7.0"
|
||||||
edition = "2024"
|
edition = "2024"
|
||||||
rust-version.workspace = true
|
rust-version.workspace = true
|
||||||
description = "Android JNI bridge for edgehdf5-memory HDF5 backend"
|
description = "Android JNI bindings for clawhdf5 agent memory"
|
||||||
license = "MIT"
|
license = "MIT"
|
||||||
|
|
||||||
[lib]
|
[lib]
|
||||||
|
|||||||
@@ -0,0 +1,44 @@
|
|||||||
|
# clawhdf5-android
|
||||||
|
|
||||||
|
A C ABI over [`clawhdf5-agent`](../clawhdf5-agent/README.md) for Android
|
||||||
|
apps: a `cdylib` exporting `extern "C"` functions (`edgehdf5_*`, a name
|
||||||
|
kept from the project's earlier "edgehdf5" days) that manage an
|
||||||
|
`HDF5Memory` through an opaque handle.
|
||||||
|
|
||||||
|
The functions are plain C symbols, not JNI-mangled `Java_...` entry points:
|
||||||
|
a Kotlin/Java app calls them through a thin JNI shim or JNA of its own. No
|
||||||
|
such shim, Gradle project or AAR is in this repository, and the crate is
|
||||||
|
not built for an Android target in CI (only its host-side unit tests run
|
||||||
|
with the workspace).
|
||||||
|
|
||||||
|
## Functions
|
||||||
|
|
||||||
|
| Function | What |
|
||||||
|
|---|---|
|
||||||
|
| `edgehdf5_create(path, agent_id, embedding_dim)` / `edgehdf5_open(path)` | a handle, or null on failure |
|
||||||
|
| `edgehdf5_close(handle)` | drop the store; what is not yet checkpointed stays in its WAL, as with any `HDF5Memory` |
|
||||||
|
| `edgehdf5_save(handle, ...)` | save one entry; the embedding length is checked against the store's dimension before the pointer is read |
|
||||||
|
| `edgehdf5_delete`, `edgehdf5_count`, `edgehdf5_count_active` | |
|
||||||
|
| `edgehdf5_hybrid_search(handle, query, len, text, vector_weight, keyword_weight, max_results, out_indices, out_scores, out_chunks)` | results into caller-provided arrays; returns the number written |
|
||||||
|
| `edgehdf5_add_session`, `edgehdf5_get_session_summary` | sessions |
|
||||||
|
| `edgehdf5_add_entity`, `edgehdf5_add_relation` | knowledge graph |
|
||||||
|
| `edgehdf5_free_string` | free a string this library returned |
|
||||||
|
|
||||||
|
Every function is `unsafe`: the caller guarantees valid, NUL-terminated
|
||||||
|
strings and correctly sized buffers (see each function's `# Safety`
|
||||||
|
section), and serialises access to a handle; separate handles are
|
||||||
|
independent.
|
||||||
|
|
||||||
|
## Build
|
||||||
|
|
||||||
|
```bash
|
||||||
|
cargo build --release -p clawhdf5-android # host build; for a device, add --target aarch64-linux-android with the NDK's linker configured
|
||||||
|
```
|
||||||
|
|
||||||
|
It depends on `clawhdf5-agent` with **default features off**, so there is
|
||||||
|
no HNSW index (the vector stage is an exact linear scan) and no rayon
|
||||||
|
pool. No C is compiled.
|
||||||
|
|
||||||
|
## License
|
||||||
|
|
||||||
|
MIT
|
||||||
@@ -1,25 +1,70 @@
|
|||||||
# clawhdf5-ann
|
# clawhdf5-ann
|
||||||
|
|
||||||
[](https://crates.io/crates/clawhdf5-ann)
|
An HNSW (Hierarchical Navigable Small World) approximate nearest-neighbour
|
||||||
[](https://docs.rs/clawhdf5-ann)
|
index in pure Rust, with cosine or L2 distance, optional int8 storage of
|
||||||
|
the vectors, deletions, and persistence as an HDF5 file. It is the vector
|
||||||
|
stage of [`clawhdf5-agent`](../clawhdf5-agent/README.md)'s search (the
|
||||||
|
agent's `hnsw` feature, on by default); distances run on
|
||||||
|
[`clawhdf5-accel`](../clawhdf5-accel/README.md)'s SIMD kernels.
|
||||||
|
|
||||||
HNSW approximate nearest neighbor index stored as HDF5.
|
Neighbours are chosen with the HNSW paper's diversity heuristic, not plain
|
||||||
|
closest-M (which capped recall on clustered data at 0.31 recall@10 at 100K
|
||||||
|
vectors).
|
||||||
|
|
||||||
## Features
|
Not on crates.io yet; depend on it from git:
|
||||||
|
|
||||||
- Build and query HNSW indexes persisted in HDF5 format
|
```toml
|
||||||
- Pure Rust, no C dependencies
|
[dependencies]
|
||||||
- Efficient similarity search for high-dimensional vectors
|
clawhdf5-ann = { git = "https://git.redclaw.dev/quantumclaw/clawhdf5" }
|
||||||
|
```
|
||||||
|
|
||||||
## Usage
|
## Usage
|
||||||
|
|
||||||
```rust
|
```rust
|
||||||
use clawhdf5_ann::HnswIndex;
|
use clawhdf5_ann::{DistanceMetric, HnswIndex, Storage};
|
||||||
|
|
||||||
let index = HnswIndex::from_hdf5("vectors.h5").unwrap();
|
let vectors: Vec<Vec<f32>> = (0..500)
|
||||||
let neighbors = index.search(&query, 10);
|
.map(|i| (0..16).map(|j| ((i * 31 + j * 7) % 97) as f32 / 97.0).collect())
|
||||||
|
.collect();
|
||||||
|
|
||||||
|
// m = 16 connections per node, ef_construction = 200
|
||||||
|
let mut index = HnswIndex::build_with(&vectors, 16, 200, DistanceMetric::Cosine, Storage::Int8);
|
||||||
|
let hits = index.search(&vectors[42], 10, 64); // (id, distance), closest first; ef >= k
|
||||||
|
assert!(hits[0].1 < 1e-3); // vector 42 itself (or an identical one)
|
||||||
|
|
||||||
|
let id = index.insert(vec![0.5; 16]);
|
||||||
|
index.mark_deleted(id);
|
||||||
|
|
||||||
|
// Persist as HDF5 (a self-contained file: graph and vectors) and load it back
|
||||||
|
let bytes = index.to_hdf5_bytes().unwrap();
|
||||||
|
let loaded = HnswIndex::load_from_hdf5(&bytes).unwrap();
|
||||||
|
assert_eq!(loaded.len(), index.len());
|
||||||
```
|
```
|
||||||
|
|
||||||
|
- `HnswIndex::build` (L2), `build_with_metric`, `build_with` (metric and
|
||||||
|
storage); `new`/`new_with` plus `insert` for an index built
|
||||||
|
incrementally.
|
||||||
|
- `Storage::Int8` keeps each vector as `i8`, a quarter of the memory; it
|
||||||
|
applies to `Cosine` only (an L2 index keeps `Float32`). Distances are then
|
||||||
|
approximate, so a caller that needs exact ranking re-scores the
|
||||||
|
candidates, as the agent does.
|
||||||
|
- `mark_deleted`, `is_deleted`, `deleted_count`, `active_len`, `compact`
|
||||||
|
(returns the old-to-new id map).
|
||||||
|
- `save_to_hdf5(&mut writer)` / `to_hdf5_bytes` / `load_from_hdf5` store
|
||||||
|
the whole index; `graph_to_bytes` / `from_graph_bytes` store only the
|
||||||
|
graph (with a CRC32) for a caller that keeps the vectors elsewhere — the
|
||||||
|
agent's `<store>.h5.ann` sidecar.
|
||||||
|
|
||||||
|
## Features
|
||||||
|
|
||||||
|
| Feature | Default | What | Builds C |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `parallel` | no | build the graph on a rayon pool; the graph is identical with or without it | no |
|
||||||
|
|
||||||
|
Recall and speed against exact search, for the index alone and in the
|
||||||
|
agent: [`BENCHMARKS.md`](../../BENCHMARKS.md), measured with
|
||||||
|
`cargo run --release -p clawhdf5-bench --bin search_harness`.
|
||||||
|
|
||||||
## License
|
## License
|
||||||
|
|
||||||
MIT
|
MIT
|
||||||
|
|||||||
@@ -13,7 +13,7 @@ use clawhdf5_format::filter_pipeline::FilterPipeline;
|
|||||||
use clawhdf5_format::group_v2::resolve_path_any;
|
use clawhdf5_format::group_v2::resolve_path_any;
|
||||||
use clawhdf5_format::message_type::MessageType;
|
use clawhdf5_format::message_type::MessageType;
|
||||||
use clawhdf5_format::object_header::ObjectHeader;
|
use clawhdf5_format::object_header::ObjectHeader;
|
||||||
use clawhdf5_format::signature::find_signature;
|
use clawhdf5_format::signature::split_user_block;
|
||||||
use clawhdf5_format::superblock::Superblock;
|
use clawhdf5_format::superblock::Superblock;
|
||||||
use clawhdf5_io::FileWriter as IoFileWriter;
|
use clawhdf5_io::FileWriter as IoFileWriter;
|
||||||
|
|
||||||
@@ -861,8 +861,9 @@ impl HnswIndex {
|
|||||||
/// The HDF5 data must contain the `/ann/vectors`, `/ann/graph_layer_*`,
|
/// The HDF5 data must contain the `/ann/vectors`, `/ann/graph_layer_*`,
|
||||||
/// and `/ann/config` datasets as produced by [`to_hdf5_bytes`].
|
/// and `/ann/config` datasets as produced by [`to_hdf5_bytes`].
|
||||||
pub fn load_from_hdf5(data: &[u8]) -> Result<Self, FormatError> {
|
pub fn load_from_hdf5(data: &[u8]) -> Result<Self, FormatError> {
|
||||||
let sig_offset = find_signature(data)?;
|
// Addresses are relative to the superblock: skip any user block.
|
||||||
let sb = Superblock::parse(data, sig_offset)?;
|
let (_, data) = split_user_block(data)?;
|
||||||
|
let sb = Superblock::parse(data, 0)?;
|
||||||
|
|
||||||
// Read config dataset and its attributes
|
// Read config dataset and its attributes
|
||||||
let config_attrs = read_dataset_attrs(data, &sb, "ann/config")?;
|
let config_attrs = read_dataset_attrs(data, &sb, "ann/config")?;
|
||||||
|
|||||||
@@ -34,6 +34,10 @@ path = "src/bin/consolidation_efficiency.rs"
|
|||||||
name = "ephemeral_perf"
|
name = "ephemeral_perf"
|
||||||
path = "src/bin/ephemeral_perf.rs"
|
path = "src/bin/ephemeral_perf.rs"
|
||||||
|
|
||||||
|
[[bin]]
|
||||||
|
name = "concurrent_read"
|
||||||
|
path = "src/bin/concurrent_read.rs"
|
||||||
|
|
||||||
[[bin]]
|
[[bin]]
|
||||||
name = "mpi_io_bench"
|
name = "mpi_io_bench"
|
||||||
path = "src/bin/mpi_io_bench.rs"
|
path = "src/bin/mpi_io_bench.rs"
|
||||||
@@ -64,6 +68,10 @@ clawhdf5-io = { path = "../clawhdf5-io" }
|
|||||||
mpi = { version = "0.8", optional = true }
|
mpi = { version = "0.8", optional = true }
|
||||||
serde = { workspace = true }
|
serde = { workspace = true }
|
||||||
serde_json = "1"
|
serde_json = "1"
|
||||||
|
# concurrent_read: size the decode pool (--decode-threads) and evict files
|
||||||
|
# from the page cache (--cold, posix_fadvise). Both pure Rust / bindings only.
|
||||||
|
rayon = "1"
|
||||||
|
libc = "0.2"
|
||||||
tempfile = { workspace = true }
|
tempfile = { workspace = true }
|
||||||
# Optional: libhdf5 C wrapper for side-by-side comparison (requires system libhdf5).
|
# Optional: libhdf5 C wrapper for side-by-side comparison (requires system libhdf5).
|
||||||
# Enable with: cargo bench -p clawhdf5-bench --features libhdf5-compare
|
# Enable with: cargo bench -p clawhdf5-bench --features libhdf5-compare
|
||||||
|
|||||||
@@ -0,0 +1,49 @@
|
|||||||
|
# clawhdf5-bench
|
||||||
|
|
||||||
|
The measurement harnesses behind [`BENCHMARKS.md`](../../BENCHMARKS.md):
|
||||||
|
HDF5 read and write speed (against libhdf5 and h5py where noted) and the
|
||||||
|
agent store's search, footprint and retrieval quality. Not meant for
|
||||||
|
publishing; nothing else in the workspace depends on it. Run everything with
|
||||||
|
`--release`, and quote numbers with the machine, date and command, as
|
||||||
|
`BENCHMARKS.md` does.
|
||||||
|
|
||||||
|
## Binaries
|
||||||
|
|
||||||
|
| Binary | Measures |
|
||||||
|
|---|---|
|
||||||
|
| `read_harness` | full reads vs hyperslab selections of a chunked 2-D dataset (compressed and not) and a contiguous one: does a selection cost scale with the selection or the dataset? (`-- --large` for 512 MB) |
|
||||||
|
| `concurrent_read` | decoded read throughput vs threads on one open `File`; `scripts/concurrent_read_h5py.py` runs the same workload with h5py (threads and processes) and `scripts/compare_concurrent_read.py` tabulates both |
|
||||||
|
| `search_harness` | HNSW recall@10 vs exact search, QPS and latency per `ef`, and end-to-end `HDF5Memory` ingest/checkpoint/open/search at 1K–100K (`--full`); studies: `--float16-study`, `--options-study`, `--signing-study`, `--ann-only --uniform` |
|
||||||
|
| `longmemeval_bench` | LongMemEval retrieval recall (turn and session Hit@k, MRR) — **retrieval, not QA accuracy**. Oracle or full `longmemeval_s` haystack; `--features embeddings` (or `embeddings-cuda`) embeds with MiniLM, otherwise the vector stage is inert and the run is BM25-only |
|
||||||
|
| `memory_arena` | a deterministic multi-session retrieval benchmark (BM25-only) |
|
||||||
|
| `footprint_bench` | file size and bytes per record at 100–100K records, float16 or `--f32`, WAL on/off, compressed or not |
|
||||||
|
| `consolidation_efficiency` | retrieval before and after consolidation on signal + noise records |
|
||||||
|
| `ephemeral_perf` | the in-memory ephemeral tier's set/get latency |
|
||||||
|
| `mpi_io_bench` | `clawhdf5-io`'s `MpiVol` (root-read + broadcast, not collective I/O); needs `--features mpi-io` and `mpirun` |
|
||||||
|
|
||||||
|
```bash
|
||||||
|
cargo run --release -p clawhdf5-bench --bin search_harness -- --full
|
||||||
|
cargo run --release -p clawhdf5-bench --bin read_harness
|
||||||
|
```
|
||||||
|
|
||||||
|
## Criterion benches and example
|
||||||
|
|
||||||
|
- `cargo bench -p clawhdf5-bench` runs `h5bench_write`, `h5bench_read` and
|
||||||
|
`h5bench_meta` (h5bench-style sequential, chunked, strided and metadata
|
||||||
|
workloads). `--features libhdf5-compare` adds the same workloads through
|
||||||
|
libhdf5 (the `hdf5-metno` crate; needs a system libhdf5 1.14).
|
||||||
|
- `examples/worldmodel_sampling.rs`: shuffled per-frame reads of a
|
||||||
|
`(N, H, W, C)` `uint8` dataset, clawhdf5 against h5py on the same file.
|
||||||
|
|
||||||
|
## Features
|
||||||
|
|
||||||
|
| Feature | What | Builds C |
|
||||||
|
|---|---|---|
|
||||||
|
| `libhdf5-compare` | libhdf5 variants of the Criterion benches | links the system libhdf5 |
|
||||||
|
| `mpi-io` | `mpi_io_bench` | yes (`mpi-sys`; needs an MPI installation) |
|
||||||
|
| `embeddings` | MiniLM embeddings for `longmemeval_bench` (candle) | yes (a `cc` build dependency in the candle/tokenizers tree) |
|
||||||
|
| `embeddings-cuda` | the same on a CUDA GPU (minutes instead of hours on the full haystack) | yes (CUDA) |
|
||||||
|
|
||||||
|
## License
|
||||||
|
|
||||||
|
MIT
|
||||||
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,70 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Tabulate concurrent_read JSON results (clawhdf5, h5py threads/processes).
|
||||||
|
|
||||||
|
python compare_concurrent_read.py clawhdf5.json h5py-threads.json h5py-procs.json
|
||||||
|
|
||||||
|
Prints one Markdown table: for each layout, mode and thread count, every
|
||||||
|
tool's MB/s and scaling efficiency, and the first file's MB/s relative to each
|
||||||
|
of the others. Refuses to compare runs whose workload parameters differ.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import json
|
||||||
|
import sys
|
||||||
|
|
||||||
|
COMPARED = ("datasets", "rows", "cols", "chunk", "deflate_level", "slab", "slabs", "seed")
|
||||||
|
|
||||||
|
|
||||||
|
def main(paths):
|
||||||
|
if len(paths) < 2:
|
||||||
|
sys.exit(__doc__)
|
||||||
|
docs = []
|
||||||
|
for p in paths:
|
||||||
|
with open(p) as fh:
|
||||||
|
docs.append(json.load(fh))
|
||||||
|
ref = docs[0]
|
||||||
|
for d, p in zip(docs[1:], paths[1:]):
|
||||||
|
diff = [k for k in COMPARED if d["params"].get(k) != ref["params"].get(k)]
|
||||||
|
if diff:
|
||||||
|
sys.exit(f"{p}: workload differs from {paths[0]} in {', '.join(diff)}")
|
||||||
|
if d["cache"] != ref["cache"]:
|
||||||
|
print(f"warning: {p} ran {d['cache']!r}, {paths[0]} ran {ref['cache']!r}",
|
||||||
|
file=sys.stderr)
|
||||||
|
if d.get("host") != ref.get("host"):
|
||||||
|
print(f"warning: {p} ran on {d.get('host')}, {paths[0]} on {ref.get('host')}",
|
||||||
|
file=sys.stderr)
|
||||||
|
|
||||||
|
names = [d["tool"] for d in docs]
|
||||||
|
for d in docs:
|
||||||
|
extra = f", HDF5 {d['hdf5_version']}" if "hdf5_version" in d else ""
|
||||||
|
print(f"- {d['tool']} {d['version']}{extra}: host {d.get('host')}, "
|
||||||
|
f"{d.get('cpus')} CPUs, cache {d['cache']}, decode threads per read "
|
||||||
|
f"{d.get('decode_threads')}")
|
||||||
|
p = ref["params"]
|
||||||
|
print(f"\n{p['datasets']} datasets of {p['rows']} x {p['cols']} f32, chunks "
|
||||||
|
f"{p['chunk'][0]} x {p['chunk'][1]} (deflate {p['deflate_level']}); "
|
||||||
|
f"`same`: {p['slabs']} slabs of {p['slab']} x {p['slab']}\n")
|
||||||
|
|
||||||
|
index = [{(r["layout"], r["mode"], r["threads"]): r for r in d["results"]} for d in docs]
|
||||||
|
keys = [(r["layout"], r["mode"], r["threads"]) for r in ref["results"]]
|
||||||
|
|
||||||
|
head = ["layout", "mode", "threads"]
|
||||||
|
head += [f"{n} MB/s (eff)" for n in names]
|
||||||
|
head += [f"{names[0]} / {n}" for n in names[1:]]
|
||||||
|
print("| " + " | ".join(head) + " |")
|
||||||
|
print("|---|---|" + "---:|" * (len(head) - 2))
|
||||||
|
for key in keys:
|
||||||
|
cells = [key[0], key[1], str(key[2])]
|
||||||
|
rs = [ix.get(key) for ix in index]
|
||||||
|
for r in rs:
|
||||||
|
if r is None:
|
||||||
|
cells.append("-")
|
||||||
|
else:
|
||||||
|
eff = "-" if r["efficiency"] is None else f"{r['efficiency']:.2f}"
|
||||||
|
cells.append(f"{r['mb_s']:.0f} ({eff})")
|
||||||
|
for r in rs[1:]:
|
||||||
|
cells.append("-" if r is None else f"{rs[0]['mb_s'] / r['mb_s']:.2f}x")
|
||||||
|
print("| " + " | ".join(cells) + " |")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main(sys.argv[1:])
|
||||||
@@ -0,0 +1,265 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""The concurrent_read workload with h5py, on the files concurrent_read wrote.
|
||||||
|
|
||||||
|
libhdf5 serialises every API call under one global lock, and h5py holds its
|
||||||
|
own global lock around every call as well, so h5py *threads* cannot decode in
|
||||||
|
parallel. h5py users scale with *processes* instead; ``--executor processes``
|
||||||
|
measures that (each worker opens the file itself).
|
||||||
|
|
||||||
|
The workload mirrors ``crates/clawhdf5-bench/src/bin/concurrent_read.rs``:
|
||||||
|
|
||||||
|
* ``distinct``: every dataset read in full once per repetition; worker ``t``
|
||||||
|
of ``T`` reads datasets ``t, t + T, ...``.
|
||||||
|
* ``same``: ``--slabs`` random ``--slab`` x ``--slab`` hyperslabs of ``d00``
|
||||||
|
(slab ``j`` to worker ``j % T``), offsets from the same splitmix64 stream.
|
||||||
|
|
||||||
|
Each worker times itself from a start barrier; a repetition spans the earliest
|
||||||
|
start to the latest finish (CLOCK_MONOTONIC, comparable across processes).
|
||||||
|
Threads share one ``h5py.File`` per repetition; process workers open the file
|
||||||
|
inside the timed region (a few ms against reads of many MiB).
|
||||||
|
|
||||||
|
Generate the files first with the Rust harness (it writes ``manifest.json``),
|
||||||
|
then, for example::
|
||||||
|
|
||||||
|
python concurrent_read_h5py.py --dir DIR --executor threads --json h5py-threads.json
|
||||||
|
python concurrent_read_h5py.py --dir DIR --executor processes --json h5py-procs.json
|
||||||
|
"""
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import json
|
||||||
|
import multiprocessing as mp
|
||||||
|
import os
|
||||||
|
import platform
|
||||||
|
import socket
|
||||||
|
import sys
|
||||||
|
import threading
|
||||||
|
import time
|
||||||
|
|
||||||
|
import h5py
|
||||||
|
import numpy as np
|
||||||
|
|
||||||
|
M64 = (1 << 64) - 1
|
||||||
|
|
||||||
|
|
||||||
|
def splitmix64(state):
|
||||||
|
"""Return (new_state, value); the same stream as the Rust harness."""
|
||||||
|
state = (state + 0x9E3779B97F4A7C15) & M64
|
||||||
|
z = state
|
||||||
|
z = ((z ^ (z >> 30)) * 0xBF58476D1CE4E5B9) & M64
|
||||||
|
z = ((z ^ (z >> 27)) * 0x94D049BB133111EB) & M64
|
||||||
|
return state, z ^ (z >> 31)
|
||||||
|
|
||||||
|
|
||||||
|
def value(k, i):
|
||||||
|
"""Element i (row-major) of dataset k, exactly as concurrent_read writes it."""
|
||||||
|
_, noise = splitmix64(i ^ (k << 40))
|
||||||
|
return np.float32((((i >> 6) % 16384) + k) + (noise & 0xFF) / 256.0)
|
||||||
|
|
||||||
|
|
||||||
|
def slab_offsets(seed, count, rows, cols, slab):
|
||||||
|
s = seed
|
||||||
|
out = []
|
||||||
|
for _ in range(count):
|
||||||
|
s, r = splitmix64(s)
|
||||||
|
s, c = splitmix64(s)
|
||||||
|
out.append((r % (rows - slab + 1), c % (cols - slab + 1)))
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def now():
|
||||||
|
return time.clock_gettime(time.CLOCK_MONOTONIC)
|
||||||
|
|
||||||
|
|
||||||
|
def work(f, mode, t, threads, m, slabs, slab, verify):
|
||||||
|
"""Worker t's share of one repetition on an open h5py.File."""
|
||||||
|
n = m["rows"] * m["cols"]
|
||||||
|
if mode == "distinct":
|
||||||
|
for k in range(t, m["datasets"], threads):
|
||||||
|
got = f[f"d{k:02d}"][...]
|
||||||
|
assert got.size == n
|
||||||
|
if verify:
|
||||||
|
flat = got.reshape(-1)
|
||||||
|
for i in (0, n // 3, n - 1):
|
||||||
|
assert flat[i] == value(k, i), f"d{k:02d}[{i}]"
|
||||||
|
else:
|
||||||
|
ds = f["d00"]
|
||||||
|
cols = m["cols"]
|
||||||
|
for r, c in slabs[t::threads]:
|
||||||
|
got = ds[r : r + slab, c : c + slab]
|
||||||
|
assert got.shape == (slab, slab)
|
||||||
|
if verify:
|
||||||
|
assert got[0, 0] == value(0, r * cols + c)
|
||||||
|
last = (r + slab - 1) * cols + c + slab - 1
|
||||||
|
assert got[-1, -1] == value(0, last)
|
||||||
|
|
||||||
|
|
||||||
|
# ----- process workers ------------------------------------------------------
|
||||||
|
|
||||||
|
_barrier = None
|
||||||
|
|
||||||
|
|
||||||
|
def _init(barrier):
|
||||||
|
global _barrier
|
||||||
|
_barrier = barrier
|
||||||
|
|
||||||
|
|
||||||
|
def _proc_task(task):
|
||||||
|
path, mode, t, threads, m, slabs, slab = task
|
||||||
|
_barrier.wait()
|
||||||
|
start = now()
|
||||||
|
with h5py.File(path, "r") as f:
|
||||||
|
work(f, mode, t, threads, m, slabs, slab, False)
|
||||||
|
return start, now()
|
||||||
|
|
||||||
|
|
||||||
|
def _noop(_):
|
||||||
|
return os.getpid()
|
||||||
|
|
||||||
|
|
||||||
|
def run_threads(path, mode, threads, m, slabs, slab):
|
||||||
|
spans = [None] * threads
|
||||||
|
barrier = threading.Barrier(threads)
|
||||||
|
with h5py.File(path, "r") as f:
|
||||||
|
|
||||||
|
def body(t):
|
||||||
|
barrier.wait()
|
||||||
|
start = now()
|
||||||
|
work(f, mode, t, threads, m, slabs, slab, False)
|
||||||
|
spans[t] = (start, now())
|
||||||
|
|
||||||
|
ts = [threading.Thread(target=body, args=(t,)) for t in range(threads)]
|
||||||
|
for th in ts:
|
||||||
|
th.start()
|
||||||
|
for th in ts:
|
||||||
|
th.join()
|
||||||
|
return max(e for _, e in spans) - min(s for s, _ in spans)
|
||||||
|
|
||||||
|
|
||||||
|
def run_processes(pool, path, mode, threads, m, slabs, slab):
|
||||||
|
tasks = [(path, mode, t, threads, m, slabs, slab) for t in range(threads)]
|
||||||
|
# One task per worker: each blocks in the barrier until all T have
|
||||||
|
# started, so no worker can take a second task.
|
||||||
|
spans = pool.map(_proc_task, tasks, chunksize=1)
|
||||||
|
return max(e for _, e in spans) - min(s for s, _ in spans)
|
||||||
|
|
||||||
|
|
||||||
|
def warm(path):
|
||||||
|
with open(path, "rb") as fh:
|
||||||
|
while fh.read(1 << 24):
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
def evict(path):
|
||||||
|
fd = os.open(path, os.O_RDONLY)
|
||||||
|
try:
|
||||||
|
os.posix_fadvise(fd, 0, 0, os.POSIX_FADV_DONTNEED)
|
||||||
|
finally:
|
||||||
|
os.close(fd)
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
ap = argparse.ArgumentParser(description=__doc__.split("\n\n")[0])
|
||||||
|
ap.add_argument("--dir", default="concurrent-read-data")
|
||||||
|
ap.add_argument("--executor", choices=["threads", "processes"], default="threads")
|
||||||
|
ap.add_argument("--threads", default="1,2,4,8,16")
|
||||||
|
ap.add_argument("--reps", type=int, default=3)
|
||||||
|
ap.add_argument("--slab", type=int, default=256)
|
||||||
|
ap.add_argument("--slabs", type=int, default=1024)
|
||||||
|
ap.add_argument("--seed", type=int, default=42)
|
||||||
|
ap.add_argument("--cold", action="store_true")
|
||||||
|
ap.add_argument("--modes", default="distinct,same")
|
||||||
|
ap.add_argument("--layouts", default="deflate,contiguous")
|
||||||
|
ap.add_argument("--json")
|
||||||
|
a = ap.parse_args()
|
||||||
|
|
||||||
|
# The Rust harness pins this value (splitmix64_reference).
|
||||||
|
assert splitmix64(42)[1] == 0xBDD732262FEB6E95, "splitmix64 port is wrong"
|
||||||
|
|
||||||
|
try:
|
||||||
|
with open(os.path.join(a.dir, "manifest.json")) as fh:
|
||||||
|
m = json.load(fh)
|
||||||
|
except FileNotFoundError:
|
||||||
|
sys.exit(f"{a.dir}/manifest.json not found: generate the files with "
|
||||||
|
"`cargo run --release -p clawhdf5-bench --bin concurrent_read -- --dir ...` first")
|
||||||
|
threads_list = [int(x) for x in a.threads.split(",")]
|
||||||
|
modes = a.modes.split(",")
|
||||||
|
layouts = a.layouts.split(",")
|
||||||
|
if a.slab < 1 or a.slab > min(m["rows"], m["cols"]):
|
||||||
|
sys.exit(f"--slab must be 1..={min(m['rows'], m['cols'])}")
|
||||||
|
files = dict(m["files"])
|
||||||
|
slabs = slab_offsets(a.seed, a.slabs, m["rows"], m["cols"], a.slab)
|
||||||
|
dataset_bytes = m["rows"] * m["cols"] * 4
|
||||||
|
tool = f"h5py-{a.executor}"
|
||||||
|
|
||||||
|
ctx = mp.get_context("spawn") # never fork a process holding HDF5 state
|
||||||
|
pools = {}
|
||||||
|
if a.executor == "processes":
|
||||||
|
for t in threads_list:
|
||||||
|
pool = ctx.Pool(t, initializer=_init, initargs=(ctx.Barrier(t),))
|
||||||
|
pool.map(_noop, range(t)) # start the workers outside the timing
|
||||||
|
pools[t] = pool
|
||||||
|
|
||||||
|
rows = []
|
||||||
|
print("| layout | mode | threads | MB/s | efficiency | median s |")
|
||||||
|
print("|---|---|---:|---:|---:|---:|")
|
||||||
|
try:
|
||||||
|
for layout in layouts:
|
||||||
|
path = os.path.join(a.dir, files[layout])
|
||||||
|
if not a.cold:
|
||||||
|
warm(path)
|
||||||
|
for mode in modes:
|
||||||
|
with h5py.File(path, "r") as f: # untimed, checked pass
|
||||||
|
work(f, mode, 0, 1, m, slabs, a.slab, True)
|
||||||
|
nbytes = (dataset_bytes * m["datasets"] if mode == "distinct"
|
||||||
|
else a.slab * a.slab * 4 * a.slabs)
|
||||||
|
base = None
|
||||||
|
for t in threads_list:
|
||||||
|
times = []
|
||||||
|
for _ in range(a.reps):
|
||||||
|
if a.cold:
|
||||||
|
evict(path)
|
||||||
|
if a.executor == "threads":
|
||||||
|
times.append(run_threads(path, mode, t, m, slabs, a.slab))
|
||||||
|
else:
|
||||||
|
times.append(run_processes(pools[t], path, mode, t, m, slabs, a.slab))
|
||||||
|
med = sorted(times)[len(times) // 2]
|
||||||
|
mb_s = nbytes / (1 << 20) / med
|
||||||
|
if t == 1:
|
||||||
|
base = mb_s
|
||||||
|
eff = mb_s / (t * base) if base else None
|
||||||
|
print(f"| {layout} | {mode} | {t} | {mb_s:.0f} | "
|
||||||
|
f"{'-' if eff is None else f'{eff:.2f}'} | {med:.4f} |")
|
||||||
|
rows.append({
|
||||||
|
"layout": layout, "mode": mode, "threads": t, "bytes": nbytes,
|
||||||
|
"times_s": times, "median_s": med, "mb_s": mb_s, "efficiency": eff,
|
||||||
|
})
|
||||||
|
finally:
|
||||||
|
for pool in pools.values():
|
||||||
|
pool.terminate()
|
||||||
|
|
||||||
|
if a.json:
|
||||||
|
doc = {
|
||||||
|
"tool": tool,
|
||||||
|
"version": h5py.__version__,
|
||||||
|
"hdf5_version": h5py.version.hdf5_version,
|
||||||
|
"python": platform.python_version(),
|
||||||
|
"host": socket.gethostname(),
|
||||||
|
"cpus": os.cpu_count(),
|
||||||
|
"unix_time": int(time.time()),
|
||||||
|
"cache": ("cold (posix_fadvise DONTNEED before each repetition)"
|
||||||
|
if a.cold else "warm"),
|
||||||
|
"decode_threads": 1,
|
||||||
|
"params": {
|
||||||
|
"datasets": m["datasets"], "rows": m["rows"], "cols": m["cols"],
|
||||||
|
"chunk": m["chunk"], "deflate_level": m["deflate_level"],
|
||||||
|
"mib": dataset_bytes // (1 << 20), "slab": a.slab, "slabs": a.slabs,
|
||||||
|
"seed": a.seed, "reps": a.reps, "dir": a.dir,
|
||||||
|
},
|
||||||
|
"results": rows,
|
||||||
|
}
|
||||||
|
with open(a.json, "w") as fh:
|
||||||
|
json.dump(doc, fh, indent=2)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
@@ -0,0 +1,523 @@
|
|||||||
|
//! Concurrent-read harness: how does decoded read throughput scale with the
|
||||||
|
//! number of threads reading one open file?
|
||||||
|
//!
|
||||||
|
//! libhdf5 (threadsafe build) serialises every API call under one global
|
||||||
|
//! mutex, and h5py holds it too, so threads cannot decode in parallel there.
|
||||||
|
//! A clawhdf5 [`File`] is `Send + Sync`; this harness measures what that buys.
|
||||||
|
//! `crates/clawhdf5-bench/scripts/concurrent_read_h5py.py` runs the same
|
||||||
|
//! workload on the same files with h5py (threads, and processes), and
|
||||||
|
//! `compare_concurrent_read.py` tabulates the JSON both write.
|
||||||
|
//!
|
||||||
|
//! Files (generated on first use, reused while `manifest.json` matches):
|
||||||
|
//!
|
||||||
|
//! * `<dir>/deflate.h5`: `--datasets` datasets `d00`, `d01`, ... of `f32`,
|
||||||
|
//! `--mib` MiB decoded each, shape `[mib * 256, 1024]`, chunks `256 x 256`,
|
||||||
|
//! deflate level 4.
|
||||||
|
//! * `<dir>/contiguous.h5`: the same datasets, contiguous.
|
||||||
|
//!
|
||||||
|
//! Modes, for each layout and each thread count `T` (strong scaling: the total
|
||||||
|
//! work per repetition is fixed, split among the threads):
|
||||||
|
//!
|
||||||
|
//! * `distinct`: every dataset is read in full once; thread `t` reads datasets
|
||||||
|
//! `t, t + T, t + 2T, ...`.
|
||||||
|
//! * `same`: all threads read `d00`, `--slabs` random `--slab` x `--slab`
|
||||||
|
//! hyperslabs in total (slab `j` goes to thread `j % T`). The offsets come
|
||||||
|
//! from a splitmix64 stream seeded with `--seed`, identical in the h5py
|
||||||
|
//! script.
|
||||||
|
//!
|
||||||
|
//! One `File` per layout per repetition is shared by all threads (opened
|
||||||
|
//! fresh each repetition, so no chunk cache carries over). Page cache:
|
||||||
|
//! `warm` (default) reads every file once before timing; `--cold` evicts the
|
||||||
|
//! files from the page cache with `posix_fadvise(POSIX_FADV_DONTNEED)` before
|
||||||
|
//! every repetition (no root needed; it only evicts clean, unmapped pages, so
|
||||||
|
//! it is best effort — the JSON says which was used).
|
||||||
|
//!
|
||||||
|
//! Decode inside one read is itself parallel when clawhdf5-format's `parallel`
|
||||||
|
//! feature is on (it is in this binary, via clawhdf5-agent). `--decode-threads
|
||||||
|
//! N` sizes that rayon pool; `--decode-threads 1` measures the API's own
|
||||||
|
//! thread scaling, comparable with h5py where each call decodes on the
|
||||||
|
//! calling thread.
|
||||||
|
//!
|
||||||
|
//! ```text
|
||||||
|
//! cargo run --release -p clawhdf5-bench --bin concurrent_read -- \
|
||||||
|
//! --dir /data/concurrent-read --json clawhdf5.json
|
||||||
|
//! cargo run --release -p clawhdf5-bench --bin concurrent_read -- \
|
||||||
|
//! --dir /tmp/cr --datasets 4 --mib 1 --threads 1,2 --slabs 16 --reps 1 # smoke
|
||||||
|
//! ```
|
||||||
|
|
||||||
|
use std::path::{Path, PathBuf};
|
||||||
|
use std::sync::Barrier;
|
||||||
|
use std::time::Instant;
|
||||||
|
|
||||||
|
use clawhdf5::{File, FileBuilder, Selection};
|
||||||
|
use serde::{Deserialize, Serialize};
|
||||||
|
|
||||||
|
const COLS: u64 = 1024;
|
||||||
|
const ROWS_PER_MIB: u64 = 256; // 256 rows x 1024 cols x 4 bytes = 1 MiB
|
||||||
|
const CHUNK: u64 = 256;
|
||||||
|
const DEFLATE_LEVEL: u32 = 4;
|
||||||
|
const LAYOUTS: [&str; 2] = ["deflate", "contiguous"];
|
||||||
|
const MANIFEST_VERSION: u32 = 1;
|
||||||
|
|
||||||
|
/// splitmix64 — shared with the h5py script, which must produce the same
|
||||||
|
/// stream (both the data and the hyperslab offsets depend on it).
|
||||||
|
fn splitmix64(state: &mut u64) -> u64 {
|
||||||
|
*state = state.wrapping_add(0x9E37_79B9_7F4A_7C15);
|
||||||
|
let mut z = *state;
|
||||||
|
z = (z ^ (z >> 30)).wrapping_mul(0xBF58_476D_1CE4_E5B9);
|
||||||
|
z = (z ^ (z >> 27)).wrapping_mul(0x94D0_49BB_1331_11EB);
|
||||||
|
z ^ (z >> 31)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Element `i` (row-major) of dataset `k`: a slowly varying integer part plus
|
||||||
|
/// 8 bits of noise, so deflate has real work to do (about 3.1x) and every value
|
||||||
|
/// is exact in `f32` (< 2^15 with 8 fraction bits), which lets both harnesses
|
||||||
|
/// check what they read against this formula.
|
||||||
|
fn value(k: u64, i: u64) -> f32 {
|
||||||
|
let mut s = i ^ (k << 40);
|
||||||
|
let noise = splitmix64(&mut s) & 0xff;
|
||||||
|
(((i >> 6) % 16384) + k) as f32 + noise as f32 / 256.0
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Serialize, Deserialize, PartialEq, Debug, Clone)]
|
||||||
|
struct Manifest {
|
||||||
|
version: u32,
|
||||||
|
datasets: u64,
|
||||||
|
rows: u64,
|
||||||
|
cols: u64,
|
||||||
|
chunk: [u64; 2],
|
||||||
|
deflate_level: u32,
|
||||||
|
files: Vec<(String, String)>, // (layout, file name)
|
||||||
|
writer: String,
|
||||||
|
}
|
||||||
|
|
||||||
|
fn manifest_for(datasets: u64, mib: u64) -> Manifest {
|
||||||
|
Manifest {
|
||||||
|
version: MANIFEST_VERSION,
|
||||||
|
datasets,
|
||||||
|
rows: mib * ROWS_PER_MIB,
|
||||||
|
cols: COLS,
|
||||||
|
chunk: [CHUNK, CHUNK],
|
||||||
|
deflate_level: DEFLATE_LEVEL,
|
||||||
|
files: LAYOUTS
|
||||||
|
.iter()
|
||||||
|
.map(|l| (l.to_string(), format!("{l}.h5")))
|
||||||
|
.collect(),
|
||||||
|
writer: format!("clawhdf5 {}", env!("CARGO_PKG_VERSION")),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn dataset_values(k: u64, n: u64) -> Vec<f32> {
|
||||||
|
(0..n).map(|i| value(k, i)).collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write the files unless `dir` already holds ones matching `want`.
|
||||||
|
fn ensure_files(dir: &Path, want: &Manifest) -> std::io::Result<bool> {
|
||||||
|
let manifest_path = dir.join("manifest.json");
|
||||||
|
if let Ok(text) = std::fs::read_to_string(&manifest_path)
|
||||||
|
&& let Ok(have) = serde_json::from_str::<Manifest>(&text)
|
||||||
|
&& have.version == want.version
|
||||||
|
&& have.datasets == want.datasets
|
||||||
|
&& have.rows == want.rows
|
||||||
|
&& have.cols == want.cols
|
||||||
|
&& have.chunk == want.chunk
|
||||||
|
&& have.deflate_level == want.deflate_level
|
||||||
|
&& have.files == want.files
|
||||||
|
&& want.files.iter().all(|(_, f)| dir.join(f).exists())
|
||||||
|
{
|
||||||
|
return Ok(false);
|
||||||
|
}
|
||||||
|
std::fs::create_dir_all(dir)?;
|
||||||
|
// A stale manifest must not survive a half-written regeneration.
|
||||||
|
let _ = std::fs::remove_file(&manifest_path);
|
||||||
|
let n = want.rows * want.cols;
|
||||||
|
for (layout, file) in &want.files {
|
||||||
|
// One layout at a time keeps the peak memory to about twice one
|
||||||
|
// file's decoded size.
|
||||||
|
let mut b = FileBuilder::new();
|
||||||
|
for k in 0..want.datasets {
|
||||||
|
let ds = b.create_dataset(&format!("d{k:02}"));
|
||||||
|
ds.with_f32_data(&dataset_values(k, n))
|
||||||
|
.with_shape(&[want.rows, want.cols]);
|
||||||
|
if layout == "deflate" {
|
||||||
|
ds.with_chunks(&[CHUNK.min(want.rows), CHUNK])
|
||||||
|
.with_deflate(DEFLATE_LEVEL);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
b.write(dir.join(file)).map_err(std::io::Error::other)?;
|
||||||
|
}
|
||||||
|
std::fs::write(
|
||||||
|
&manifest_path,
|
||||||
|
serde_json::to_string_pretty(want).map_err(std::io::Error::other)?,
|
||||||
|
)?;
|
||||||
|
Ok(true)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn slab_offsets(seed: u64, count: usize, rows: u64, cols: u64, slab: u64) -> Vec<(u64, u64)> {
|
||||||
|
let mut s = seed;
|
||||||
|
(0..count)
|
||||||
|
.map(|_| {
|
||||||
|
let r = splitmix64(&mut s) % (rows - slab + 1);
|
||||||
|
let c = splitmix64(&mut s) % (cols - slab + 1);
|
||||||
|
(r, c)
|
||||||
|
})
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Warm the page cache by reading every byte of `path`.
|
||||||
|
fn warm(path: &Path) -> std::io::Result<()> {
|
||||||
|
let mut f = std::fs::File::open(path)?;
|
||||||
|
std::io::copy(&mut f, &mut std::io::sink())?;
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Ask the kernel to drop `path`'s pages from the page cache.
|
||||||
|
fn evict(path: &Path) -> std::io::Result<()> {
|
||||||
|
use std::os::fd::AsRawFd;
|
||||||
|
let f = std::fs::File::open(path)?;
|
||||||
|
// SAFETY: plain syscall on a valid, open file descriptor.
|
||||||
|
let rc = unsafe { libc::posix_fadvise(f.as_raw_fd(), 0, 0, libc::POSIX_FADV_DONTNEED) };
|
||||||
|
if rc != 0 {
|
||||||
|
return Err(std::io::Error::from_raw_os_error(rc));
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Serialize)]
|
||||||
|
struct Row {
|
||||||
|
layout: String,
|
||||||
|
mode: String,
|
||||||
|
threads: usize,
|
||||||
|
/// Decoded (selected) bytes read per repetition.
|
||||||
|
bytes: u64,
|
||||||
|
times_s: Vec<f64>,
|
||||||
|
median_s: f64,
|
||||||
|
mb_s: f64,
|
||||||
|
/// `mb_s / (threads * mb_s at threads = 1)`; null without a 1-thread row.
|
||||||
|
efficiency: Option<f64>,
|
||||||
|
}
|
||||||
|
|
||||||
|
struct Args {
|
||||||
|
dir: PathBuf,
|
||||||
|
datasets: u64,
|
||||||
|
mib: u64,
|
||||||
|
threads: Vec<usize>,
|
||||||
|
reps: usize,
|
||||||
|
slab: u64,
|
||||||
|
slabs: usize,
|
||||||
|
seed: u64,
|
||||||
|
cold: bool,
|
||||||
|
decode_threads: usize,
|
||||||
|
modes: Vec<String>,
|
||||||
|
layouts: Vec<String>,
|
||||||
|
json: Option<PathBuf>,
|
||||||
|
}
|
||||||
|
|
||||||
|
const USAGE: &str = "\
|
||||||
|
usage: concurrent_read [--dir DIR] [--datasets N] [--mib N] [--threads 1,2,4,8,16]
|
||||||
|
[--reps N] [--slab N] [--slabs N] [--seed N] [--cold]
|
||||||
|
[--decode-threads N] [--modes distinct,same]
|
||||||
|
[--layouts deflate,contiguous] [--json FILE]";
|
||||||
|
|
||||||
|
fn parse_list<T: std::str::FromStr>(s: &str) -> Result<Vec<T>, String> {
|
||||||
|
s.split(',')
|
||||||
|
.map(|x| x.trim().parse().map_err(|_| format!("bad list item {x:?}")))
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parse_args() -> Result<Args, String> {
|
||||||
|
let mut a = Args {
|
||||||
|
dir: PathBuf::from("concurrent-read-data"),
|
||||||
|
datasets: 64,
|
||||||
|
mib: 64,
|
||||||
|
threads: vec![1, 2, 4, 8, 16],
|
||||||
|
reps: 3,
|
||||||
|
slab: 256,
|
||||||
|
slabs: 1024,
|
||||||
|
seed: 42,
|
||||||
|
cold: false,
|
||||||
|
decode_threads: 0,
|
||||||
|
modes: vec!["distinct".into(), "same".into()],
|
||||||
|
layouts: LAYOUTS.iter().map(|s| s.to_string()).collect(),
|
||||||
|
json: None,
|
||||||
|
};
|
||||||
|
let mut it = std::env::args().skip(1);
|
||||||
|
while let Some(flag) = it.next() {
|
||||||
|
if flag == "--cold" {
|
||||||
|
a.cold = true;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if flag == "-h" || flag == "--help" {
|
||||||
|
return Err(USAGE.into());
|
||||||
|
}
|
||||||
|
let v = it.next().ok_or(format!("{flag} needs a value\n{USAGE}"))?;
|
||||||
|
let num = |v: &str| {
|
||||||
|
v.parse::<u64>()
|
||||||
|
.map_err(|_| format!("{flag}: bad number {v:?}"))
|
||||||
|
};
|
||||||
|
match flag.as_str() {
|
||||||
|
"--dir" => a.dir = v.into(),
|
||||||
|
"--datasets" => a.datasets = num(&v)?,
|
||||||
|
"--mib" => a.mib = num(&v)?,
|
||||||
|
"--threads" => a.threads = parse_list(&v)?,
|
||||||
|
"--reps" => a.reps = num(&v)? as usize,
|
||||||
|
"--slab" => a.slab = num(&v)?,
|
||||||
|
"--slabs" => a.slabs = num(&v)? as usize,
|
||||||
|
"--seed" => a.seed = num(&v)?,
|
||||||
|
"--decode-threads" => a.decode_threads = num(&v)? as usize,
|
||||||
|
"--modes" => a.modes = parse_list(&v)?,
|
||||||
|
"--layouts" => a.layouts = parse_list(&v)?,
|
||||||
|
"--json" => a.json = Some(v.into()),
|
||||||
|
_ => return Err(format!("unknown flag {flag}\n{USAGE}")),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if a.datasets == 0 || a.datasets > 100 {
|
||||||
|
return Err("--datasets must be 1..=100".into());
|
||||||
|
}
|
||||||
|
if a.mib == 0 || a.reps == 0 || a.slabs == 0 || a.threads.contains(&0) {
|
||||||
|
return Err("--mib, --reps, --slabs and every --threads value must be > 0".into());
|
||||||
|
}
|
||||||
|
if a.slab == 0 || a.slab > COLS || a.slab > a.mib * ROWS_PER_MIB {
|
||||||
|
return Err(format!(
|
||||||
|
"--slab must be 1..={}",
|
||||||
|
COLS.min(a.mib * ROWS_PER_MIB)
|
||||||
|
));
|
||||||
|
}
|
||||||
|
for m in &a.modes {
|
||||||
|
if m != "distinct" && m != "same" {
|
||||||
|
return Err(format!("unknown mode {m:?}"));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for l in &a.layouts {
|
||||||
|
if !LAYOUTS.contains(&l.as_str()) {
|
||||||
|
return Err(format!("unknown layout {l:?}"));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(a)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One timed repetition: `T` threads on one shared `File`. Returns seconds.
|
||||||
|
fn run_once(
|
||||||
|
path: &Path,
|
||||||
|
mode: &str,
|
||||||
|
threads: usize,
|
||||||
|
m: &Manifest,
|
||||||
|
slabs: &[(u64, u64)],
|
||||||
|
slab: u64,
|
||||||
|
verify: bool,
|
||||||
|
) -> f64 {
|
||||||
|
let file = File::open(path).expect("open");
|
||||||
|
let barrier = Barrier::new(threads + 1); // + the spawning thread
|
||||||
|
let n = m.rows * m.cols;
|
||||||
|
// Each thread times itself from the barrier; the repetition spans the
|
||||||
|
// earliest start to the latest finish (timing on the spawning thread
|
||||||
|
// instead undercounts whenever it is scheduled after the workers ran).
|
||||||
|
let spans: Vec<(Instant, Instant)> = std::thread::scope(|s| {
|
||||||
|
let handles: Vec<_> = (0..threads)
|
||||||
|
.map(|t| {
|
||||||
|
let (file, barrier) = (&file, &barrier);
|
||||||
|
s.spawn(move || {
|
||||||
|
barrier.wait();
|
||||||
|
let start = Instant::now();
|
||||||
|
match mode {
|
||||||
|
"distinct" => {
|
||||||
|
for k in (t as u64..m.datasets).step_by(threads) {
|
||||||
|
let got = file.dataset(&format!("d{k:02}")).unwrap().read_f32();
|
||||||
|
let got = got.unwrap();
|
||||||
|
assert_eq!(got.len() as u64, n);
|
||||||
|
if verify {
|
||||||
|
for i in [0, n / 3, n - 1] {
|
||||||
|
assert_eq!(got[i as usize], value(k, i), "d{k:02}[{i}]");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
std::hint::black_box(got);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
_ => {
|
||||||
|
let ds = file.dataset("d00").unwrap();
|
||||||
|
for &(r, c) in slabs.iter().skip(t).step_by(threads) {
|
||||||
|
let sel = Selection::Hyperslab {
|
||||||
|
start: vec![r, c],
|
||||||
|
stride: vec![1, 1],
|
||||||
|
count: vec![slab, slab],
|
||||||
|
block: vec![1, 1],
|
||||||
|
};
|
||||||
|
let got = ds.read_f32_selection(&sel).unwrap();
|
||||||
|
assert_eq!(got.len() as u64, slab * slab);
|
||||||
|
if verify {
|
||||||
|
let last = (r + slab - 1) * m.cols + c + slab - 1;
|
||||||
|
assert_eq!(got[0], value(0, r * m.cols + c));
|
||||||
|
assert_eq!(*got.last().unwrap(), value(0, last));
|
||||||
|
}
|
||||||
|
std::hint::black_box(got);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
(start, Instant::now())
|
||||||
|
})
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
barrier.wait();
|
||||||
|
handles.into_iter().map(|h| h.join().unwrap()).collect()
|
||||||
|
});
|
||||||
|
let start = spans.iter().map(|s| s.0).min().unwrap();
|
||||||
|
let end = spans.iter().map(|s| s.1).max().unwrap();
|
||||||
|
(end - start).as_secs_f64()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn median(v: &[f64]) -> f64 {
|
||||||
|
let mut s = v.to_vec();
|
||||||
|
s.sort_by(f64::total_cmp);
|
||||||
|
s[s.len() / 2]
|
||||||
|
}
|
||||||
|
|
||||||
|
fn hostname() -> String {
|
||||||
|
std::fs::read_to_string("/proc/sys/kernel/hostname")
|
||||||
|
.map(|s| s.trim().to_string())
|
||||||
|
.unwrap_or_else(|_| "unknown".into())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn main() {
|
||||||
|
let args = match parse_args() {
|
||||||
|
Ok(a) => a,
|
||||||
|
Err(e) => {
|
||||||
|
eprintln!("{e}");
|
||||||
|
std::process::exit(2);
|
||||||
|
}
|
||||||
|
};
|
||||||
|
if cfg!(debug_assertions) {
|
||||||
|
eprintln!("warning: debug build — numbers are meaningless. Use --release.");
|
||||||
|
}
|
||||||
|
if args.decode_threads > 0 {
|
||||||
|
rayon::ThreadPoolBuilder::new()
|
||||||
|
.num_threads(args.decode_threads)
|
||||||
|
.build_global()
|
||||||
|
.expect("configure rayon pool");
|
||||||
|
}
|
||||||
|
|
||||||
|
let manifest = manifest_for(args.datasets, args.mib);
|
||||||
|
let t = Instant::now();
|
||||||
|
match ensure_files(&args.dir, &manifest) {
|
||||||
|
Ok(true) => eprintln!(
|
||||||
|
"generated {} in {:.1} s",
|
||||||
|
args.dir.display(),
|
||||||
|
t.elapsed().as_secs_f64()
|
||||||
|
),
|
||||||
|
Ok(false) => eprintln!("reusing {}", args.dir.display()),
|
||||||
|
Err(e) => {
|
||||||
|
eprintln!("cannot write test files in {}: {e}", args.dir.display());
|
||||||
|
std::process::exit(1);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let path_of = |layout: &str| args.dir.join(format!("{layout}.h5"));
|
||||||
|
let slabs = slab_offsets(
|
||||||
|
args.seed,
|
||||||
|
args.slabs,
|
||||||
|
manifest.rows,
|
||||||
|
manifest.cols,
|
||||||
|
args.slab,
|
||||||
|
);
|
||||||
|
let dataset_bytes = manifest.rows * manifest.cols * 4;
|
||||||
|
|
||||||
|
let mut rows: Vec<Row> = Vec::new();
|
||||||
|
println!("| layout | mode | threads | MB/s | efficiency | median s |");
|
||||||
|
println!("|---|---|---:|---:|---:|---:|");
|
||||||
|
for layout in &args.layouts {
|
||||||
|
let path = path_of(layout);
|
||||||
|
// Untimed pass: page cache warm (unless --cold), results checked.
|
||||||
|
if !args.cold {
|
||||||
|
warm(&path).expect("warm page cache");
|
||||||
|
}
|
||||||
|
for mode in &args.modes {
|
||||||
|
run_once(&path, mode, 1, &manifest, &slabs, args.slab, true);
|
||||||
|
let bytes = match mode.as_str() {
|
||||||
|
"distinct" => dataset_bytes * manifest.datasets,
|
||||||
|
_ => args.slab * args.slab * 4 * args.slabs as u64,
|
||||||
|
};
|
||||||
|
let mut base: Option<f64> = None;
|
||||||
|
for &threads in &args.threads {
|
||||||
|
let times: Vec<f64> = (0..args.reps)
|
||||||
|
.map(|_| {
|
||||||
|
if args.cold {
|
||||||
|
evict(&path).expect("posix_fadvise");
|
||||||
|
}
|
||||||
|
run_once(&path, mode, threads, &manifest, &slabs, args.slab, false)
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
let med = median(×);
|
||||||
|
let mb_s = bytes as f64 / (1 << 20) as f64 / med;
|
||||||
|
if threads == 1 {
|
||||||
|
base = Some(mb_s);
|
||||||
|
}
|
||||||
|
let efficiency = base.map(|b| mb_s / (threads as f64 * b));
|
||||||
|
println!(
|
||||||
|
"| {layout} | {mode} | {threads} | {mb_s:.0} | {} | {med:.4} |",
|
||||||
|
efficiency.map_or("-".into(), |e| format!("{e:.2}"))
|
||||||
|
);
|
||||||
|
rows.push(Row {
|
||||||
|
layout: layout.clone(),
|
||||||
|
mode: mode.clone(),
|
||||||
|
threads,
|
||||||
|
bytes,
|
||||||
|
times_s: times,
|
||||||
|
median_s: med,
|
||||||
|
mb_s,
|
||||||
|
efficiency,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if let Some(out) = &args.json {
|
||||||
|
let doc = serde_json::json!({
|
||||||
|
"tool": "clawhdf5",
|
||||||
|
"version": env!("CARGO_PKG_VERSION"),
|
||||||
|
"host": hostname(),
|
||||||
|
"cpus": std::thread::available_parallelism().map_or(0, |n| n.get()),
|
||||||
|
"unix_time": std::time::SystemTime::now()
|
||||||
|
.duration_since(std::time::UNIX_EPOCH)
|
||||||
|
.map_or(0, |d| d.as_secs()),
|
||||||
|
"cache": if args.cold { "cold (posix_fadvise DONTNEED before each repetition)" } else { "warm" },
|
||||||
|
"decode_threads": rayon::current_num_threads(),
|
||||||
|
"params": {
|
||||||
|
"datasets": manifest.datasets,
|
||||||
|
"mib": args.mib,
|
||||||
|
"rows": manifest.rows,
|
||||||
|
"cols": manifest.cols,
|
||||||
|
"chunk": manifest.chunk,
|
||||||
|
"deflate_level": manifest.deflate_level,
|
||||||
|
"slab": args.slab,
|
||||||
|
"slabs": args.slabs,
|
||||||
|
"seed": args.seed,
|
||||||
|
"reps": args.reps,
|
||||||
|
"dir": args.dir,
|
||||||
|
},
|
||||||
|
"results": rows,
|
||||||
|
});
|
||||||
|
std::fs::write(out, serde_json::to_string_pretty(&doc).unwrap()).expect("write json");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn values_are_exact_in_f32() {
|
||||||
|
for k in [0, 7, 63] {
|
||||||
|
for i in [0u64, 1, 4095, 1 << 20, (1 << 24) - 1] {
|
||||||
|
let v = value(k, i);
|
||||||
|
assert_eq!(v, (v as f64) as f32);
|
||||||
|
assert!(v < 32768.0);
|
||||||
|
assert_eq!((v * 256.0).fract(), 0.0);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The h5py script hard-codes this vector to check its splitmix64 port.
|
||||||
|
#[test]
|
||||||
|
fn splitmix64_reference() {
|
||||||
|
let mut s = 42;
|
||||||
|
assert_eq!(splitmix64(&mut s), 0xBDD7_3226_2FEB_6E95);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -21,6 +21,7 @@
|
|||||||
//! cargo run --release -p clawhdf5-bench --bin search_harness -- --ann-only --uniform
|
//! cargo run --release -p clawhdf5-bench --bin search_harness -- --ann-only --uniform
|
||||||
//! cargo run --release -p clawhdf5-bench --bin search_harness -- --float16-study --full
|
//! cargo run --release -p clawhdf5-bench --bin search_harness -- --float16-study --full
|
||||||
//! cargo run --release -p clawhdf5-bench --bin search_harness -- --options-study --full
|
//! cargo run --release -p clawhdf5-bench --bin search_harness -- --options-study --full
|
||||||
|
//! cargo run --release -p clawhdf5-bench --bin search_harness -- --signing-study --full
|
||||||
//! ```
|
//! ```
|
||||||
|
|
||||||
use std::time::{Duration, Instant};
|
use std::time::{Duration, Instant};
|
||||||
@@ -488,6 +489,81 @@ fn bench_end_to_end(n: usize, json: &mut Vec<serde_json::Value>) {
|
|||||||
}));
|
}));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Signing study: what does an Ed25519-signed checkpoint cost?
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// `--signing-study`: checkpoint time unsigned vs signed, `verify` time, and
|
||||||
|
/// the file-size cost of the stored per-record hashes. Default store
|
||||||
|
/// settings (float16, int8 index). Medians of five checkpoints / three
|
||||||
|
/// verifies.
|
||||||
|
fn signing_study(n: usize) {
|
||||||
|
use clawhdf5_agent::signing::SigningKey;
|
||||||
|
let data = make_dataset(n, 0x516 ^ n as u64);
|
||||||
|
let mut rng = Rng(9);
|
||||||
|
let entries: Vec<MemoryEntry> = data
|
||||||
|
.vectors
|
||||||
|
.iter()
|
||||||
|
.enumerate()
|
||||||
|
.map(|(i, v)| MemoryEntry {
|
||||||
|
chunk: text_for(data.cluster_of[i], i, &mut rng),
|
||||||
|
embedding: v.clone(),
|
||||||
|
source_channel: "bench".into(),
|
||||||
|
timestamp: i as f64,
|
||||||
|
session_id: format!("s{}", i % 50),
|
||||||
|
tags: format!("t{i}"),
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
let dir = tempfile::tempdir().unwrap();
|
||||||
|
let path = dir.path().join("sign.h5");
|
||||||
|
let mut mem = HDF5Memory::create(MemoryConfig::new(path.clone(), "bench", DIM)).unwrap();
|
||||||
|
mem.save_batch(entries).unwrap();
|
||||||
|
std::hint::black_box(mem.hybrid_search(&data.queries[0], "", 1.0, 0.0, K));
|
||||||
|
|
||||||
|
let median = |mut v: Vec<Duration>| {
|
||||||
|
v.sort();
|
||||||
|
v[v.len() / 2]
|
||||||
|
};
|
||||||
|
let checkpoint = |mem: &mut HDF5Memory| {
|
||||||
|
median(
|
||||||
|
(0..5)
|
||||||
|
.map(|_| {
|
||||||
|
let t = Instant::now();
|
||||||
|
mem.flush_wal().unwrap();
|
||||||
|
t.elapsed()
|
||||||
|
})
|
||||||
|
.collect(),
|
||||||
|
)
|
||||||
|
};
|
||||||
|
let unsigned = checkpoint(&mut mem);
|
||||||
|
let unsigned_bytes = std::fs::metadata(&path).unwrap().len();
|
||||||
|
let key = SigningKey::from_bytes(&[7; 32]);
|
||||||
|
mem.set_signing_key(key.clone());
|
||||||
|
let signed = checkpoint(&mut mem);
|
||||||
|
let signed_bytes = std::fs::metadata(&path).unwrap().len();
|
||||||
|
drop(mem);
|
||||||
|
let vk = key.verifying_key();
|
||||||
|
let verify = median(
|
||||||
|
(0..3)
|
||||||
|
.map(|_| {
|
||||||
|
let t = Instant::now();
|
||||||
|
let r = HDF5Memory::verify(&path, &vk).unwrap();
|
||||||
|
let d = t.elapsed();
|
||||||
|
assert!(r.is_valid());
|
||||||
|
d
|
||||||
|
})
|
||||||
|
.collect(),
|
||||||
|
);
|
||||||
|
println!(
|
||||||
|
"| {n} | {:.1} | {:.1} | {:+.1} | {:.1} | {:+.2} |",
|
||||||
|
millis(unsigned),
|
||||||
|
millis(signed),
|
||||||
|
millis(signed) - millis(unsigned),
|
||||||
|
millis(verify),
|
||||||
|
(signed_bytes as f64 - unsigned_bytes as f64) / (1024.0 * 1024.0),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// Search options study: source filters, re-ranking, confidence rejection
|
// Search options study: source filters, re-ranking, confidence rejection
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
@@ -960,6 +1036,21 @@ fn main() {
|
|||||||
}
|
}
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
if args.iter().any(|a| a == "--signing-study") {
|
||||||
|
println!("## Signed checkpoints ({DIM}-dim, float16, int8 index)\n");
|
||||||
|
println!(
|
||||||
|
"| N | checkpoint ms, unsigned | checkpoint ms, signed | signing adds ms | verify ms | file MiB added |"
|
||||||
|
);
|
||||||
|
println!("|---:|---:|---:|---:|---:|---:|");
|
||||||
|
for &n in if full {
|
||||||
|
&[1_000, 10_000, 100_000][..]
|
||||||
|
} else {
|
||||||
|
&[1_000, 10_000][..]
|
||||||
|
} {
|
||||||
|
signing_study(n);
|
||||||
|
}
|
||||||
|
return;
|
||||||
|
}
|
||||||
if args.iter().any(|a| a == "--options-study") {
|
if args.iter().any(|a| a == "--options-study") {
|
||||||
println!("## Search options ({DIM}-dim, k = {K}, Hebbian boost off)\n");
|
println!("## Search options ({DIM}-dim, k = {K}, Hebbian boost off)\n");
|
||||||
println!("| N | options | filtered recall@10 | p50 ms | p99 ms |");
|
println!("| N | options | filtered recall@10 | p50 ms | p99 ms |");
|
||||||
|
|||||||
@@ -0,0 +1,148 @@
|
|||||||
|
//! Keeps the concurrent-read harnesses working: runs `concurrent_read`, the
|
||||||
|
//! h5py script (threads and processes) and the comparison script end to end
|
||||||
|
//! on tiny files. h5py reading the files also checks, element by element at
|
||||||
|
//! spot positions, that both harnesses generate the same data and slabs.
|
||||||
|
//!
|
||||||
|
//! The h5py half is skipped when python3 with h5py is unavailable, unless
|
||||||
|
//! `CLAWHDF5_REQUIRE_INTEROP=1`; `CLAWHDF5_PYTHON` picks the interpreter.
|
||||||
|
|
||||||
|
use std::path::{Path, PathBuf};
|
||||||
|
use std::process::Command;
|
||||||
|
|
||||||
|
fn python() -> String {
|
||||||
|
std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn interop_required() -> bool {
|
||||||
|
std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1")
|
||||||
|
}
|
||||||
|
|
||||||
|
fn python_available() -> bool {
|
||||||
|
Command::new(python())
|
||||||
|
.args(["-c", "import h5py, numpy"])
|
||||||
|
.output()
|
||||||
|
.map(|o| o.status.success())
|
||||||
|
.unwrap_or(false)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn scripts() -> PathBuf {
|
||||||
|
Path::new(env!("CARGO_MANIFEST_DIR")).join("scripts")
|
||||||
|
}
|
||||||
|
|
||||||
|
fn run(cmd: &mut Command) -> String {
|
||||||
|
let out = cmd.output().expect("spawn");
|
||||||
|
assert!(
|
||||||
|
out.status.success(),
|
||||||
|
"{cmd:?} failed\nSTDOUT:\n{}\nSTDERR:\n{}",
|
||||||
|
String::from_utf8_lossy(&out.stdout),
|
||||||
|
String::from_utf8_lossy(&out.stderr)
|
||||||
|
);
|
||||||
|
String::from_utf8_lossy(&out.stdout).into_owned()
|
||||||
|
}
|
||||||
|
|
||||||
|
const SMALL: [&str; 8] = [
|
||||||
|
"--threads",
|
||||||
|
"1,2",
|
||||||
|
"--slabs",
|
||||||
|
"8",
|
||||||
|
"--reps",
|
||||||
|
"1",
|
||||||
|
"--slab",
|
||||||
|
"64",
|
||||||
|
];
|
||||||
|
|
||||||
|
fn results(path: &Path) -> serde_json::Value {
|
||||||
|
serde_json::from_str(&std::fs::read_to_string(path).unwrap()).unwrap()
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn harnesses_run_end_to_end_on_tiny_files() {
|
||||||
|
let dir = tempfile::TempDir::new().unwrap();
|
||||||
|
let data = dir.path().join("data");
|
||||||
|
let claw = dir.path().join("claw.json");
|
||||||
|
|
||||||
|
let bin = env!("CARGO_BIN_EXE_concurrent_read");
|
||||||
|
run(Command::new(bin)
|
||||||
|
.arg("--dir")
|
||||||
|
.arg(&data)
|
||||||
|
.args(["--datasets", "3", "--mib", "1"])
|
||||||
|
.args(SMALL)
|
||||||
|
.arg("--json")
|
||||||
|
.arg(&claw));
|
||||||
|
// Second run reuses the files (and exercises --cold).
|
||||||
|
let out = Command::new(bin)
|
||||||
|
.arg("--dir")
|
||||||
|
.arg(&data)
|
||||||
|
.args(["--datasets", "3", "--mib", "1", "--cold"])
|
||||||
|
.args(SMALL)
|
||||||
|
.output()
|
||||||
|
.unwrap();
|
||||||
|
assert!(out.status.success());
|
||||||
|
assert!(String::from_utf8_lossy(&out.stderr).contains("reusing"));
|
||||||
|
|
||||||
|
let doc = results(&claw);
|
||||||
|
assert_eq!(doc["tool"], "clawhdf5");
|
||||||
|
// 2 layouts x 2 modes x 2 thread counts.
|
||||||
|
assert_eq!(doc["results"].as_array().unwrap().len(), 8);
|
||||||
|
for r in doc["results"].as_array().unwrap() {
|
||||||
|
assert!(r["mb_s"].as_f64().unwrap() > 0.0, "{r}");
|
||||||
|
}
|
||||||
|
|
||||||
|
if !python_available() {
|
||||||
|
assert!(
|
||||||
|
!interop_required(),
|
||||||
|
"CLAWHDF5_REQUIRE_INTEROP=1 but {} has no h5py",
|
||||||
|
python()
|
||||||
|
);
|
||||||
|
eprintln!("skipping the h5py half: no h5py in {}", python());
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
let mut jsons = vec![claw];
|
||||||
|
for executor in ["threads", "processes"] {
|
||||||
|
let out = dir.path().join(format!("h5py-{executor}.json"));
|
||||||
|
run(Command::new(python())
|
||||||
|
.arg(scripts().join("concurrent_read_h5py.py"))
|
||||||
|
.arg("--dir")
|
||||||
|
.arg(&data)
|
||||||
|
.args(["--executor", executor])
|
||||||
|
.args(SMALL)
|
||||||
|
.arg("--json")
|
||||||
|
.arg(&out));
|
||||||
|
let doc = results(&out);
|
||||||
|
assert_eq!(doc["tool"], format!("h5py-{executor}"));
|
||||||
|
assert_eq!(doc["results"].as_array().unwrap().len(), 8);
|
||||||
|
jsons.push(out);
|
||||||
|
}
|
||||||
|
let table = run(Command::new(python())
|
||||||
|
.arg(scripts().join("compare_concurrent_read.py"))
|
||||||
|
.args(&jsons));
|
||||||
|
assert!(table.contains("| deflate | same | 2 |"), "{table}");
|
||||||
|
assert!(table.contains("clawhdf5 / h5py-processes"), "{table}");
|
||||||
|
|
||||||
|
// A different workload must not be compared.
|
||||||
|
let other = dir.path().join("other.json");
|
||||||
|
run(Command::new(python())
|
||||||
|
.arg(scripts().join("concurrent_read_h5py.py"))
|
||||||
|
.arg("--dir")
|
||||||
|
.arg(&data)
|
||||||
|
.args([
|
||||||
|
"--threads",
|
||||||
|
"1",
|
||||||
|
"--slabs",
|
||||||
|
"4",
|
||||||
|
"--reps",
|
||||||
|
"1",
|
||||||
|
"--slab",
|
||||||
|
"64",
|
||||||
|
])
|
||||||
|
.arg("--json")
|
||||||
|
.arg(&other));
|
||||||
|
let out = Command::new(python())
|
||||||
|
.arg(scripts().join("compare_concurrent_read.py"))
|
||||||
|
.arg(&jsons[0])
|
||||||
|
.arg(&other)
|
||||||
|
.output()
|
||||||
|
.unwrap();
|
||||||
|
assert!(!out.status.success());
|
||||||
|
assert!(String::from_utf8_lossy(&out.stderr).contains("slabs"));
|
||||||
|
}
|
||||||
@@ -0,0 +1,49 @@
|
|||||||
|
# clawhdf5-cli
|
||||||
|
|
||||||
|
The `clawhdf5` command: create, fill, search and inspect a
|
||||||
|
[`clawhdf5-agent`](../clawhdf5-agent/README.md) memory store from the
|
||||||
|
shell. Output is JSON. (For general HDF5 files use `h5rs` from
|
||||||
|
[`clawhdf5-tools`](../clawhdf5-tools/README.md).)
|
||||||
|
|
||||||
|
```bash
|
||||||
|
cargo install --path crates/clawhdf5-cli # installs `clawhdf5`; not on crates.io yet
|
||||||
|
# or: cargo run -p clawhdf5-cli -- --help
|
||||||
|
```
|
||||||
|
|
||||||
|
No C is compiled.
|
||||||
|
|
||||||
|
## Commands
|
||||||
|
|
||||||
|
The store is `--path FILE` (or `CLAWHDF5_PATH`) before the subcommand.
|
||||||
|
|
||||||
|
| Command | What |
|
||||||
|
|---|---|
|
||||||
|
| `create [--agent-id ID] [--dim N] [--wal] [--f32] [--f32-index]` | a new store (dimension 384 by default); float16 embeddings and an int8 index copy unless `--f32` / `--f32-index`. The WAL is off unless `--wal` (the library's default is on), so each save is checkpointed at once |
|
||||||
|
| `save [--json '{...}']` | save one entry, from `--json` or stdin: `{"chunk", "embedding", "source_channel", "timestamp", "session_id", "tags"}` |
|
||||||
|
| `search --embedding '[...]' [--query TEXT] [-k N] [--vector-weight W] [--keyword-weight W]` | hybrid search (defaults 5 results, weights 0.7 / 0.3) |
|
||||||
|
| `recall INDEX` | one entry by index |
|
||||||
|
| `stats` | counts and configuration |
|
||||||
|
| `flush-wal` | checkpoint the WAL into the `.h5` |
|
||||||
|
| `agents-md [--output FILE]` | generate an `AGENTS.md` from the store |
|
||||||
|
| `export` | every entry as JSON lines |
|
||||||
|
| `snapshot DEST` | a copy of the store's `.h5` file |
|
||||||
|
| `keygen --out FILE` | a new Ed25519 signing key (64 hex characters, created owner-only on Unix) |
|
||||||
|
| `verify --public-key HEX_OR_FILE` | check a signed store; exit status 2 if it does not verify |
|
||||||
|
|
||||||
|
`recall`, `stats`, `agents-md` and `export` open the store read-only
|
||||||
|
(no lock, nothing written), so they work while another process has it
|
||||||
|
open. `save`, `search` (which records activation boosts) and `flush-wal`
|
||||||
|
open it for writing and take the store's lock. With
|
||||||
|
`--signing-key FILE` (or `CLAWHDF5_SIGNING_KEY`) every checkpoint a command
|
||||||
|
makes is signed; a signed store refuses to checkpoint without the key.
|
||||||
|
|
||||||
|
```bash
|
||||||
|
clawhdf5 --path mem.h5 create --agent-id demo --dim 3
|
||||||
|
echo '{"chunk":"hello","embedding":[0.1,0.2,0.3],"source_channel":"cli","timestamp":0,"session_id":"s1","tags":""}' \
|
||||||
|
| clawhdf5 --path mem.h5 save
|
||||||
|
clawhdf5 --path mem.h5 search --embedding '[0.1,0.2,0.3]' --query hello -k 3
|
||||||
|
```
|
||||||
|
|
||||||
|
## License
|
||||||
|
|
||||||
|
MIT
|
||||||
+126
-16
@@ -1,15 +1,22 @@
|
|||||||
use std::path::PathBuf;
|
use std::path::{Path, PathBuf};
|
||||||
|
|
||||||
use clap::{Parser, Subcommand};
|
use clap::{Parser, Subcommand};
|
||||||
|
use clawhdf5_agent::signing::{self, SigningKey, VerifyingKey};
|
||||||
use clawhdf5_agent::{AgentMemory, HDF5Memory, MemoryConfig, MemoryEntry};
|
use clawhdf5_agent::{AgentMemory, HDF5Memory, MemoryConfig, MemoryEntry};
|
||||||
|
|
||||||
/// ClawhDF5 — HDF5-backed cognitive memory for AI agents
|
/// ClawhDF5 — HDF5-backed cognitive memory for AI agents
|
||||||
#[derive(Parser)]
|
#[derive(Parser)]
|
||||||
#[command(name = "clawhdf5", version, about)]
|
#[command(name = "clawhdf5", version, about)]
|
||||||
struct Cli {
|
struct Cli {
|
||||||
/// Path to the .h5 memory file
|
/// Path to the .h5 memory file (not needed for `keygen`)
|
||||||
#[arg(short, long, env = "CLAWHDF5_PATH")]
|
#[arg(short, long, env = "CLAWHDF5_PATH")]
|
||||||
path: PathBuf,
|
path: Option<PathBuf>,
|
||||||
|
|
||||||
|
/// File holding an Ed25519 signing key (64 hex characters, from
|
||||||
|
/// `keygen`). Every checkpoint this command makes is then signed; a
|
||||||
|
/// signed store refuses to checkpoint without it.
|
||||||
|
#[arg(long, env = "CLAWHDF5_SIGNING_KEY", global = true)]
|
||||||
|
signing_key: Option<PathBuf>,
|
||||||
|
|
||||||
#[command(subcommand)]
|
#[command(subcommand)]
|
||||||
command: Commands,
|
command: Commands,
|
||||||
@@ -91,6 +98,38 @@ enum Commands {
|
|||||||
/// Destination path
|
/// Destination path
|
||||||
dest: PathBuf,
|
dest: PathBuf,
|
||||||
},
|
},
|
||||||
|
/// Generate an Ed25519 signing key for signed checkpoints
|
||||||
|
Keygen {
|
||||||
|
/// Where to write the secret key (created new, owner-only on Unix)
|
||||||
|
#[arg(long)]
|
||||||
|
out: PathBuf,
|
||||||
|
},
|
||||||
|
/// Verify a signed store against a public key; exit status 2 if not valid
|
||||||
|
Verify {
|
||||||
|
/// The trusted public key: 64 hex characters, or a file holding them
|
||||||
|
#[arg(long)]
|
||||||
|
public_key: String,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
fn read_signing_key(path: &Path) -> Result<SigningKey, Box<dyn std::error::Error>> {
|
||||||
|
let text = std::fs::read_to_string(path)
|
||||||
|
.map_err(|e| format!("cannot read signing key {}: {e}", path.display()))?;
|
||||||
|
let bytes = signing::from_hex::<32>(&text)
|
||||||
|
.ok_or_else(|| format!("{} is not a 64-hex-character key", path.display()))?;
|
||||||
|
Ok(SigningKey::from_bytes(&bytes))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Open for writing, with the signing key applied if one was given.
|
||||||
|
fn open_writable(
|
||||||
|
path: &Path,
|
||||||
|
key: &Option<SigningKey>,
|
||||||
|
) -> Result<HDF5Memory, Box<dyn std::error::Error>> {
|
||||||
|
let mut mem = HDF5Memory::open(path)?;
|
||||||
|
if let Some(k) = key {
|
||||||
|
mem.set_signing_key(k.clone());
|
||||||
|
}
|
||||||
|
Ok(mem)
|
||||||
}
|
}
|
||||||
|
|
||||||
fn main() {
|
fn main() {
|
||||||
@@ -103,6 +142,37 @@ fn main() {
|
|||||||
}
|
}
|
||||||
|
|
||||||
fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
||||||
|
if let Commands::Keygen { out } = &cli.command {
|
||||||
|
let key = signing::generate_key();
|
||||||
|
let mut opts = std::fs::OpenOptions::new();
|
||||||
|
opts.write(true).create_new(true);
|
||||||
|
#[cfg(unix)]
|
||||||
|
{
|
||||||
|
use std::os::unix::fs::OpenOptionsExt;
|
||||||
|
opts.mode(0o600);
|
||||||
|
}
|
||||||
|
use std::io::Write;
|
||||||
|
let mut f = opts
|
||||||
|
.open(out)
|
||||||
|
.map_err(|e| format!("cannot create {}: {e}", out.display()))?;
|
||||||
|
writeln!(f, "{}", signing::to_hex(&key.to_bytes()))?;
|
||||||
|
let j = serde_json::json!({
|
||||||
|
"status": "generated",
|
||||||
|
"secret_key_file": out.display().to_string(),
|
||||||
|
"public_key": signing::to_hex(&key.verifying_key().to_bytes()),
|
||||||
|
});
|
||||||
|
println!("{}", serde_json::to_string_pretty(&j)?);
|
||||||
|
return Ok(());
|
||||||
|
}
|
||||||
|
let path = cli
|
||||||
|
.path
|
||||||
|
.clone()
|
||||||
|
.ok_or("--path (or CLAWHDF5_PATH) is required")?;
|
||||||
|
let key = cli
|
||||||
|
.signing_key
|
||||||
|
.as_deref()
|
||||||
|
.map(read_signing_key)
|
||||||
|
.transpose()?;
|
||||||
match cli.command {
|
match cli.command {
|
||||||
Commands::Create {
|
Commands::Create {
|
||||||
agent_id,
|
agent_id,
|
||||||
@@ -113,7 +183,7 @@ fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
|||||||
f32,
|
f32,
|
||||||
float16: _,
|
float16: _,
|
||||||
} => {
|
} => {
|
||||||
let mut config = MemoryConfig::new(cli.path.clone(), &agent_id, dim);
|
let mut config = MemoryConfig::new(path.clone(), &agent_id, dim);
|
||||||
config.wal_enabled = wal;
|
config.wal_enabled = wal;
|
||||||
// As with --f32-index: only ever switch the library default off.
|
// As with --f32-index: only ever switch the library default off.
|
||||||
if f32 {
|
if f32 {
|
||||||
@@ -127,15 +197,21 @@ fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
|||||||
config.quantized_index = false;
|
config.quantized_index = false;
|
||||||
}
|
}
|
||||||
let config_quantized = config.quantized_index;
|
let config_quantized = config.quantized_index;
|
||||||
let mem = HDF5Memory::create(config)?;
|
let mut mem = HDF5Memory::create(config)?;
|
||||||
|
// Sign straight away, so the store is never on disk unsigned.
|
||||||
|
if let Some(k) = &key {
|
||||||
|
mem.set_signing_key(k.clone());
|
||||||
|
mem.flush_wal()?;
|
||||||
|
}
|
||||||
let j = serde_json::json!({
|
let j = serde_json::json!({
|
||||||
"status": "created",
|
"status": "created",
|
||||||
"path": cli.path.display().to_string(),
|
"path": path.display().to_string(),
|
||||||
"agent_id": agent_id,
|
"agent_id": agent_id,
|
||||||
"embedding_dim": dim,
|
"embedding_dim": dim,
|
||||||
"wal_enabled": wal,
|
"wal_enabled": wal,
|
||||||
"quantized_index": config_quantized,
|
"quantized_index": config_quantized,
|
||||||
"float16": config_float16,
|
"float16": config_float16,
|
||||||
|
"signed": mem.is_signed(),
|
||||||
"count": mem.count(),
|
"count": mem.count(),
|
||||||
});
|
});
|
||||||
println!("{}", serde_json::to_string_pretty(&j)?);
|
println!("{}", serde_json::to_string_pretty(&j)?);
|
||||||
@@ -152,7 +228,7 @@ fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
let entry: MemoryEntry = serde_json::from_str(&input)?;
|
let entry: MemoryEntry = serde_json::from_str(&input)?;
|
||||||
let mut mem = HDF5Memory::open(&cli.path)?;
|
let mut mem = open_writable(&path, &key)?;
|
||||||
let idx = mem.save(entry)?;
|
let idx = mem.save(entry)?;
|
||||||
let j = serde_json::json!({ "status": "saved", "index": idx, "count": mem.count() });
|
let j = serde_json::json!({ "status": "saved", "index": idx, "count": mem.count() });
|
||||||
println!("{}", serde_json::to_string(&j)?);
|
println!("{}", serde_json::to_string(&j)?);
|
||||||
@@ -166,7 +242,7 @@ fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
|||||||
keyword_weight,
|
keyword_weight,
|
||||||
} => {
|
} => {
|
||||||
let emb: Vec<f32> = serde_json::from_str(&embedding)?;
|
let emb: Vec<f32> = serde_json::from_str(&embedding)?;
|
||||||
let mut mem = HDF5Memory::open(&cli.path)?;
|
let mut mem = open_writable(&path, &key)?;
|
||||||
let results = mem.hybrid_search(&emb, &query, vector_weight, keyword_weight, top_k);
|
let results = mem.hybrid_search(&emb, &query, vector_weight, keyword_weight, top_k);
|
||||||
let j: Vec<serde_json::Value> = results
|
let j: Vec<serde_json::Value> = results
|
||||||
.iter()
|
.iter()
|
||||||
@@ -184,7 +260,7 @@ fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
|||||||
}
|
}
|
||||||
|
|
||||||
Commands::Recall { index } => {
|
Commands::Recall { index } => {
|
||||||
let mem = HDF5Memory::open_read_only(&cli.path)?;
|
let mem = HDF5Memory::open_read_only(&path)?;
|
||||||
match mem.get_chunk(index) {
|
match mem.get_chunk(index) {
|
||||||
Some(content) => {
|
Some(content) => {
|
||||||
let j = serde_json::json!({ "index": index, "chunk": content });
|
let j = serde_json::json!({ "index": index, "chunk": content });
|
||||||
@@ -198,22 +274,23 @@ fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
|||||||
}
|
}
|
||||||
|
|
||||||
Commands::Stats => {
|
Commands::Stats => {
|
||||||
let mem = HDF5Memory::open_read_only(&cli.path)?;
|
let mem = HDF5Memory::open_read_only(&path)?;
|
||||||
let cfg = mem.config();
|
let cfg = mem.config();
|
||||||
let j = serde_json::json!({
|
let j = serde_json::json!({
|
||||||
"path": cli.path.display().to_string(),
|
"path": path.display().to_string(),
|
||||||
"agent_id": cfg.agent_id,
|
"agent_id": cfg.agent_id,
|
||||||
"embedding_dim": cfg.embedding_dim,
|
"embedding_dim": cfg.embedding_dim,
|
||||||
"count": mem.count(),
|
"count": mem.count(),
|
||||||
"active": mem.count_active(),
|
"active": mem.count_active(),
|
||||||
"wal_enabled": cfg.wal_enabled,
|
"wal_enabled": cfg.wal_enabled,
|
||||||
"wal_pending": mem.wal_pending_count(),
|
"wal_pending": mem.wal_pending_count(),
|
||||||
|
"signed": mem.is_signed(),
|
||||||
});
|
});
|
||||||
println!("{}", serde_json::to_string_pretty(&j)?);
|
println!("{}", serde_json::to_string_pretty(&j)?);
|
||||||
}
|
}
|
||||||
|
|
||||||
Commands::FlushWal => {
|
Commands::FlushWal => {
|
||||||
let mut mem = HDF5Memory::open(&cli.path)?;
|
let mut mem = open_writable(&path, &key)?;
|
||||||
let before = mem.wal_pending_count();
|
let before = mem.wal_pending_count();
|
||||||
mem.flush_wal()?;
|
mem.flush_wal()?;
|
||||||
let j = serde_json::json!({
|
let j = serde_json::json!({
|
||||||
@@ -225,7 +302,7 @@ fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
|||||||
}
|
}
|
||||||
|
|
||||||
Commands::AgentsMd { output } => {
|
Commands::AgentsMd { output } => {
|
||||||
let mem = HDF5Memory::open_read_only(&cli.path)?;
|
let mem = HDF5Memory::open_read_only(&path)?;
|
||||||
let md = mem.generate_agents_md();
|
let md = mem.generate_agents_md();
|
||||||
match output {
|
match output {
|
||||||
Some(p) => {
|
Some(p) => {
|
||||||
@@ -237,7 +314,7 @@ fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
|||||||
}
|
}
|
||||||
|
|
||||||
Commands::Export => {
|
Commands::Export => {
|
||||||
let mem = HDF5Memory::open_read_only(&cli.path)?;
|
let mem = HDF5Memory::open_read_only(&path)?;
|
||||||
for i in 0..mem.count() {
|
for i in 0..mem.count() {
|
||||||
if let Some(chunk) = mem.get_chunk(i) {
|
if let Some(chunk) = mem.get_chunk(i) {
|
||||||
let j = serde_json::json!({ "index": i, "chunk": chunk });
|
let j = serde_json::json!({ "index": i, "chunk": chunk });
|
||||||
@@ -246,11 +323,44 @@ fn run(cli: Cli) -> Result<(), Box<dyn std::error::Error>> {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
Commands::Keygen { .. } => unreachable!("handled before opening a store"),
|
||||||
|
|
||||||
|
Commands::Verify { public_key } => {
|
||||||
|
let text = if Path::new(&public_key).is_file() {
|
||||||
|
std::fs::read_to_string(&public_key)?
|
||||||
|
} else {
|
||||||
|
public_key
|
||||||
|
};
|
||||||
|
let bytes = signing::from_hex::<32>(&text)
|
||||||
|
.ok_or("--public-key must be 64 hex characters or a file holding them")?;
|
||||||
|
let trusted = VerifyingKey::from_bytes(&bytes)?;
|
||||||
|
let r = HDF5Memory::verify(&path, &trusted)?;
|
||||||
|
let j = serde_json::json!({
|
||||||
|
"valid": r.is_valid(),
|
||||||
|
"signed": r.signed,
|
||||||
|
"key_matches": r.key_matches,
|
||||||
|
"signature_valid": r.signature_valid,
|
||||||
|
"records_match": r.records_match,
|
||||||
|
"settings_match": r.settings_match,
|
||||||
|
"sessions_match": r.sessions_match,
|
||||||
|
"graph_match": r.graph_match,
|
||||||
|
"changed_records": r.changed_records,
|
||||||
|
"record_count": r.record_count,
|
||||||
|
"signed_record_count": r.signed_record_count,
|
||||||
|
"signed_by": r.public_key.map(|k| signing::to_hex(&k)),
|
||||||
|
"wal_entries_unsigned": r.wal_entries_unsigned,
|
||||||
|
});
|
||||||
|
println!("{}", serde_json::to_string_pretty(&j)?);
|
||||||
|
if !r.is_valid() {
|
||||||
|
std::process::exit(2);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
Commands::Snapshot { dest } => {
|
Commands::Snapshot { dest } => {
|
||||||
let _result = clawhdf5_agent::storage::snapshot_file(&cli.path, &dest)?;
|
let _result = clawhdf5_agent::storage::snapshot_file(&path, &dest)?;
|
||||||
let j = serde_json::json!({
|
let j = serde_json::json!({
|
||||||
"status": "snapshot_created",
|
"status": "snapshot_created",
|
||||||
"source": cli.path.display().to_string(),
|
"source": path.display().to_string(),
|
||||||
"dest": dest.display().to_string(),
|
"dest": dest.display().to_string(),
|
||||||
});
|
});
|
||||||
println!("{}", serde_json::to_string(&j)?);
|
println!("{}", serde_json::to_string(&j)?);
|
||||||
|
|||||||
@@ -3,7 +3,7 @@ name = "clawhdf5-derive"
|
|||||||
version = "2.7.0"
|
version = "2.7.0"
|
||||||
edition = "2024"
|
edition = "2024"
|
||||||
rust-version.workspace = true
|
rust-version.workspace = true
|
||||||
description = "Derive macros for rustyhdf5 HDF5 traits"
|
description = "Derive macro (H5Type) for clawhdf5 compound types"
|
||||||
license = "MIT"
|
license = "MIT"
|
||||||
repository = "https://git.redclaw.dev/quantumclaw/clawhdf5"
|
repository = "https://git.redclaw.dev/quantumclaw/clawhdf5"
|
||||||
readme = "README.md"
|
readme = "README.md"
|
||||||
|
|||||||
@@ -1,28 +1,50 @@
|
|||||||
# clawhdf5-derive
|
# clawhdf5-derive
|
||||||
|
|
||||||
[](https://crates.io/crates/clawhdf5-derive)
|
`#[derive(H5Type)]`: maps a Rust struct with named fields to an HDF5
|
||||||
[](https://docs.rs/clawhdf5-derive)
|
compound datatype. The derive generates three inherent methods:
|
||||||
|
|
||||||
Derive macros for clawhdf5 HDF5 traits.
|
- `hdf5_datatype() -> clawhdf5_format::datatype::Datatype` — the
|
||||||
|
`Datatype::Compound` (members in field order, packed, little-endian);
|
||||||
|
- `to_bytes(&self) -> Vec<u8>` — one element in that layout;
|
||||||
|
- `from_bytes(&[u8]) -> Self` — the reverse (panics if the slice is shorter
|
||||||
|
than the compound).
|
||||||
|
|
||||||
## Features
|
Supported field types: `f32`, `f64`, `i8`–`i64`, `u8`–`u64`, `bool`
|
||||||
|
(stored as `u8`) and fixed-size arrays `[T; N]` of those numeric types.
|
||||||
|
Tuple structs, enums and nested structs are refused at compile time.
|
||||||
|
|
||||||
- `#[derive(HDF5Type)]` for automatic HDF5 datatype mapping
|
The generated code names `clawhdf5_format`, so the crate using the derive
|
||||||
- Struct-to-compound-type derivation
|
must depend on [`clawhdf5-format`](../clawhdf5-format/README.md) too. Not
|
||||||
|
on crates.io yet:
|
||||||
|
|
||||||
## Usage
|
```toml
|
||||||
|
[dependencies]
|
||||||
|
clawhdf5-derive = { git = "https://git.redclaw.dev/quantumclaw/clawhdf5" }
|
||||||
|
clawhdf5-format = { git = "https://git.redclaw.dev/quantumclaw/clawhdf5" }
|
||||||
|
```
|
||||||
|
|
||||||
|
## Example
|
||||||
|
|
||||||
```rust
|
```rust
|
||||||
use clawhdf5_derive::HDF5Type;
|
use clawhdf5_derive::H5Type;
|
||||||
|
use clawhdf5_format::datatype::Datatype;
|
||||||
|
|
||||||
#[derive(HDF5Type)]
|
#[derive(H5Type, Debug, PartialEq)]
|
||||||
struct Point {
|
struct Point {
|
||||||
x: f64,
|
id: u32,
|
||||||
y: f64,
|
pos: [f64; 3],
|
||||||
z: f64,
|
valid: bool,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
let p = Point { id: 7, pos: [1.0, 2.0, 3.0], valid: true };
|
||||||
|
let bytes = p.to_bytes();
|
||||||
|
assert_eq!(bytes.len(), 4 + 24 + 1);
|
||||||
|
assert_eq!(Point::from_bytes(&bytes), p);
|
||||||
|
assert!(matches!(Point::hdf5_datatype(), Datatype::Compound { size: 29, .. }));
|
||||||
```
|
```
|
||||||
|
|
||||||
|
Tests: `crates/clawhdf5-format/tests/derive_tests.rs`.
|
||||||
|
|
||||||
## License
|
## License
|
||||||
|
|
||||||
MIT
|
MIT
|
||||||
|
|||||||
@@ -1,27 +1,56 @@
|
|||||||
# clawhdf5-filters
|
# clawhdf5-filters
|
||||||
|
|
||||||
[](https://crates.io/crates/clawhdf5-filters)
|
Standalone deflate (zlib) compression and decompression with a choice of
|
||||||
[](https://docs.rs/clawhdf5-filters)
|
backend: pure-Rust zlib-rs (default), zlib-ng, Apple's Compression
|
||||||
|
framework, or miniz_oxide.
|
||||||
|
|
||||||
Filter and compression pipeline for clawhdf5.
|
This crate holds **deflate backends only**. The HDF5 filter pipeline, the
|
||||||
|
filter registry and every other codec (shuffle, Fletcher-32, N-Bit,
|
||||||
|
scale-offset, LZ4, Zstd, SZIP, pcodec, LZF, bitshuffle, bzip2, Blosc,
|
||||||
|
Blosc2, ZFP) live in [`clawhdf5-format`](../clawhdf5-format/README.md),
|
||||||
|
which calls flate2 itself and selects its deflate backend with its own
|
||||||
|
features. No library crate of the workspace depends on this one (the
|
||||||
|
`clawhdf5` facade uses it only in tests).
|
||||||
|
|
||||||
## Features
|
Not on crates.io yet; depend on it from git:
|
||||||
|
|
||||||
- DEFLATE compression/decompression
|
```toml
|
||||||
- Pure-Rust deflate via zlib-rs (default, `zlib-rs` feature)
|
[dependencies]
|
||||||
- zlib-ng instead, if you want it (`fast-deflate` feature; C, needs cmake)
|
clawhdf5-filters = { git = "https://git.redclaw.dev/quantumclaw/clawhdf5" }
|
||||||
- Apple Compression framework support (`apple-compression` feature)
|
```
|
||||||
|
|
||||||
## Usage
|
## API
|
||||||
|
|
||||||
```rust
|
```rust
|
||||||
use clawhdf5_filters::{deflate_compress, deflate_decompress};
|
use clawhdf5_filters::{deflate_backend, deflate_compress, deflate_decompress};
|
||||||
|
|
||||||
|
let data: Vec<u8> = (0..10_000u32).map(|i| (i % 251) as u8).collect();
|
||||||
let compressed = deflate_compress(&data, 6).unwrap();
|
let compressed = deflate_compress(&data, 6).unwrap();
|
||||||
// The second argument bounds the output: the expected decompressed size.
|
// The second argument bounds the output: the expected decompressed size.
|
||||||
let decompressed = deflate_decompress(&compressed, data.len()).unwrap();
|
let decompressed = deflate_decompress(&compressed, data.len()).unwrap();
|
||||||
|
assert_eq!(decompressed, data);
|
||||||
|
println!("backend: {}", deflate_backend()); // "zlib-rs" by default
|
||||||
```
|
```
|
||||||
|
|
||||||
|
Also `deflate_compress_miniz`/`deflate_decompress_miniz` (always
|
||||||
|
miniz_oxide) and `fast_deflate::{compress, decompress, active_backend}`.
|
||||||
|
|
||||||
|
## Features
|
||||||
|
|
||||||
|
Backend priority: `apple-compression` (macOS only) > zlib-ng > zlib-rs >
|
||||||
|
miniz_oxide (with none enabled).
|
||||||
|
|
||||||
|
| Feature | Default | Backend | Builds C |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `zlib-rs` | yes | zlib-rs through flate2, with `runtime_detection` (needed for its SIMD) | no |
|
||||||
|
| `fast-deflate` | no | zlib-ng through flate2 | yes (cmake) |
|
||||||
|
| `system-zlib` | no | the system zlib through flate2 | yes (`libz-sys`) |
|
||||||
|
| `apple-compression` | no | Apple Compression framework, macOS only (ignored elsewhere) | no (links a system framework) |
|
||||||
|
|
||||||
|
zlib-rs matches zlib-ng on HDF5 reads and writes and produces
|
||||||
|
byte-identical output: see "Deflate backend" in
|
||||||
|
[`BENCHMARKS.md`](../../BENCHMARKS.md).
|
||||||
|
|
||||||
## License
|
## License
|
||||||
|
|
||||||
MIT
|
MIT
|
||||||
|
|||||||
@@ -22,6 +22,17 @@ zstd = { version = "0.13", optional = true }
|
|||||||
blake3 = { version = "1", optional = true }
|
blake3 = { version = "1", optional = true }
|
||||||
libaec-sys = { path = "../libaec-sys", version = "0.1", optional = true }
|
libaec-sys = { path = "../libaec-sys", version = "0.1", optional = true }
|
||||||
pco = { version = "1.0", optional = true }
|
pco = { version = "1.0", optional = true }
|
||||||
|
# Pure-Rust Zstandard, for the plugin filters that embed zstd (bitshuffle,
|
||||||
|
# blosc). The `zstd` feature (filter 32015) links libzstd instead.
|
||||||
|
ruzstd = { version = "0.9", optional = true }
|
||||||
|
# bzip2 with its default backend, libbz2-rs-sys: a pure-Rust port of
|
||||||
|
# libbzip2 (no C is compiled, despite the -sys name).
|
||||||
|
bzip2 = { version = "0.6", optional = true }
|
||||||
|
snap = { version = "1", optional = true }
|
||||||
|
|
||||||
|
[target.'cfg(target_os = "linux")'.dependencies]
|
||||||
|
# madvise(MADV_HUGEPAGE) for large read buffers (see src/bulk_alloc.rs).
|
||||||
|
libc = { version = "0.2", default-features = false }
|
||||||
|
|
||||||
[dev-dependencies]
|
[dev-dependencies]
|
||||||
half = { workspace = true }
|
half = { workspace = true }
|
||||||
@@ -37,7 +48,7 @@ harness = false
|
|||||||
# Deflate backend: `zlib-rs` (pure Rust) by default. `fast-deflate` selects
|
# Deflate backend: `zlib-rs` (pure Rust) by default. `fast-deflate` selects
|
||||||
# zlib-ng instead (C, built with cmake); flate2 prefers a C zlib whenever one
|
# zlib-ng instead (C, built with cmake); flate2 prefers a C zlib whenever one
|
||||||
# is enabled, so turning it on anywhere in the build overrides the default.
|
# is enabled, so turning it on anywhere in the build overrides the default.
|
||||||
default = ["std", "checksum", "deflate", "provenance", "zlib-rs", "system-zlib-decompress"]
|
default = ["std", "checksum", "deflate", "provenance", "zlib-rs", "system-zlib-decompress", "lzf"]
|
||||||
std = []
|
std = []
|
||||||
checksum = []
|
checksum = []
|
||||||
deflate = ["flate2"]
|
deflate = ["flate2"]
|
||||||
@@ -56,6 +67,25 @@ zstd = ["dep:zstd"]
|
|||||||
blake3_hash = ["blake3"]
|
blake3_hash = ["blake3"]
|
||||||
szip = ["libaec-sys"]
|
szip = ["libaec-sys"]
|
||||||
pcodec = ["dep:pco"]
|
pcodec = ["dep:pco"]
|
||||||
|
# Plugin filters, pure Rust. LZF (32000) is h5py's built-in compression; it
|
||||||
|
# has no dependencies, so it is on by default.
|
||||||
|
lzf = []
|
||||||
|
# Bitshuffle (32008), with its LZ4 and Zstandard modes.
|
||||||
|
bitshuffle = ["lz4_flex", "ruzstd"]
|
||||||
|
# bzip2 (307).
|
||||||
|
bzip2 = ["dep:bzip2", "std"]
|
||||||
|
# Blosc 1 (32001) with its BloscLZ, LZ4, Snappy, Zlib and Zstandard codecs.
|
||||||
|
blosc = ["lz4_flex", "ruzstd", "snap", "deflate", "std"]
|
||||||
|
# Blosc2 (32026), read-only: frames, B2ND arrays, and the Blosc codecs above.
|
||||||
|
blosc2 = ["blosc"]
|
||||||
|
# ZFP (32013, H5Z-ZFP), read-only: every mode, for int32, int64, float and
|
||||||
|
# double fields of 1 to 4 dimensions.
|
||||||
|
zfp = []
|
||||||
|
# Every plugin filter above.
|
||||||
|
plugin-filters = ["lzf", "bitshuffle", "bzip2", "blosc", "blosc2", "zfp"]
|
||||||
|
# Test instrumentation: per-thread counts of heap objects read (see
|
||||||
|
# `lookup_stats`), so tests can bound the cost of a name lookup.
|
||||||
|
lookup-stats = ["std"]
|
||||||
|
|
||||||
[[bench]]
|
[[bench]]
|
||||||
name = "parallel_decompress_bench"
|
name = "parallel_decompress_bench"
|
||||||
|
|||||||
@@ -1,27 +1,106 @@
|
|||||||
# clawhdf5-format
|
# clawhdf5-format
|
||||||
|
|
||||||
[](https://crates.io/crates/clawhdf5-format)
|
The HDF5 file format in pure Rust: parsers and writers for every on-disk
|
||||||
[](https://docs.rs/clawhdf5-format)
|
structure, the filter pipeline and its codecs, and the shared type
|
||||||
|
definitions the other crates use. Most users want the
|
||||||
|
[`clawhdf5`](../clawhdf5/README.md) facade, which wraps this crate in an
|
||||||
|
h5py-like API; use this one directly for low-level access or in `no_std`
|
||||||
|
code.
|
||||||
|
|
||||||
Pure-Rust HDF5 binary format parsing and writing — no C dependencies.
|
Not on crates.io yet; depend on it from git:
|
||||||
|
|
||||||
|
```toml
|
||||||
|
[dependencies]
|
||||||
|
clawhdf5-format = { git = "https://git.redclaw.dev/quantumclaw/clawhdf5" }
|
||||||
|
```
|
||||||
|
|
||||||
|
## What is in it
|
||||||
|
|
||||||
|
- **Parsing:** superblock v0–v3 (`superblock`, with the superblock
|
||||||
|
extension and metadata cache images, `superblock_ext`), object headers v1
|
||||||
|
and v2 (`object_header`), every header message the readers use
|
||||||
|
(`datatype`, `dataspace`, `data_layout` v1–v4 including virtual datasets,
|
||||||
|
`fill_value`, `attribute`, `link_message`, `shared_message`, ...), groups
|
||||||
|
old and new (`group_v1` symbol tables with local heaps, `group_v2` with
|
||||||
|
fractal heaps and v2 B-trees), and every chunk index (v1 B-tree, single
|
||||||
|
chunk, implicit, fixed array, extensible array, v2 B-tree).
|
||||||
|
- **Reading data:** `data_read` (contiguous, compact, chunked),
|
||||||
|
`partial_read` and `selection` (hyperslabs and points), `vl_data`
|
||||||
|
(variable-length strings and sequences through the global heap),
|
||||||
|
`chunk_cache`.
|
||||||
|
- **Storage:** the `storage::Storage` trait (`read_at`, `read_ranges`,
|
||||||
|
`len`, `hint`) that every read path goes through, so a file can be read
|
||||||
|
from memory, a file handle or a remote backend
|
||||||
|
([`clawhdf5-remote`](../clawhdf5-remote/README.md)).
|
||||||
|
- **Writing:** `file_writer::FileWriter` and the builders in
|
||||||
|
`type_builders` (datasets, groups, attributes, compound and enum types,
|
||||||
|
links, virtual datasets, creation-order tracking); chunk indexes and
|
||||||
|
dense-storage B-trees of any size (`chunked_write`, `btree_v2_write`,
|
||||||
|
`ea_writer`). Output is read by h5py and h5dump.
|
||||||
|
- **Filters:** `filter_pipeline` and `filter_registry` (look up by ID; other
|
||||||
|
IDs can be registered at run time with `register_filter`). Built in:
|
||||||
|
deflate, shuffle, Fletcher-32, N-Bit, scale-offset; behind features LZ4,
|
||||||
|
Zstd, SZIP (decode), pcodec, and the plugin filters LZF, bitshuffle,
|
||||||
|
bzip2, Blosc 1 (read and write), Blosc2 and ZFP (read only).
|
||||||
|
- **Shared pieces:** `float16` (the one IEEE half-precision conversion the
|
||||||
|
workspace uses), `provenance` (SHA-256 dataset hashes), `checksum`
|
||||||
|
(Jenkins lookup3 for v2+ structures).
|
||||||
|
|
||||||
|
## Example
|
||||||
|
|
||||||
|
```rust
|
||||||
|
use clawhdf5_format::file_writer::{AttrValue, FileWriter};
|
||||||
|
use clawhdf5_format::{group_v2, object_header, signature, superblock};
|
||||||
|
|
||||||
|
// Write a file to memory
|
||||||
|
let mut fw = FileWriter::new();
|
||||||
|
fw.create_dataset("data")
|
||||||
|
.with_f64_data(&[1.0, 2.0, 3.0])
|
||||||
|
.with_shape(&[3])
|
||||||
|
.set_attr("unit", AttrValue::String("m/s".into()));
|
||||||
|
let bytes = fw.finish().unwrap();
|
||||||
|
|
||||||
|
// Parse it back: superblock -> path -> object header
|
||||||
|
let (_user_block, file) = signature::split_user_block(&bytes).unwrap();
|
||||||
|
let sb = superblock::Superblock::parse(file, 0).unwrap();
|
||||||
|
let addr = group_v2::resolve_path_any(file, &sb, "data").unwrap();
|
||||||
|
let hdr = object_header::ObjectHeader::parse(file, addr as usize, sb.offset_size, sb.length_size)
|
||||||
|
.unwrap();
|
||||||
|
assert!(!hdr.messages.is_empty());
|
||||||
|
```
|
||||||
|
|
||||||
## Features
|
## Features
|
||||||
|
|
||||||
- Zero-copy superblock, object header, and B-tree parsing
|
| Feature | Default | What | Builds C |
|
||||||
- Chunked dataset read/write with filter pipelines
|
|---|---|---|---|
|
||||||
- `no_std` support (disable `std` feature)
|
| `std` | yes | standard library; without it the crate is `no_std` + `alloc` (CI builds it for `thumbv7em-none-eabihf`) | no |
|
||||||
- Optional parallel reads via Rayon
|
| `checksum` | yes | verify Jenkins lookup3 checksums | no |
|
||||||
- SHA-256 provenance tracking
|
| `deflate` | yes | deflate through flate2 | no |
|
||||||
|
| `zlib-rs` | yes | flate2's pure-Rust zlib-rs backend, with `runtime_detection` (without it zlib-rs loses SIMD and inflates 3.5x slower) | no |
|
||||||
|
| `system-zlib-decompress` | yes | macOS only: inflate with the system libz first, falling back to flate2; no effect elsewhere | no (links the system libz on macOS) |
|
||||||
|
| `provenance` | yes | SHA-256 provenance hashes | no |
|
||||||
|
| `lzf` | yes | LZF (32000) | no |
|
||||||
|
| `parallel` | no | rayon-parallel chunk decoding | no |
|
||||||
|
| `fast-checksum` | no | hardware CRC32 through `crc32fast` | no |
|
||||||
|
| `lz4` | no | LZ4 (32004) | no |
|
||||||
|
| `pcodec` | no | pcodec | no |
|
||||||
|
| `bitshuffle`, `bzip2`, `blosc` | no | 32008, 307, 32001, read and write | no |
|
||||||
|
| `blosc2`, `zfp` | no | 32026, 32013, read only | no |
|
||||||
|
| `plugin-filters` | no | all six plugin filters above | no |
|
||||||
|
| `lookup-stats` | no | counters for name-lookup benchmarks | no |
|
||||||
|
| `zstd` | no | Zstandard (32015) | yes (libzstd) |
|
||||||
|
| `szip` | no | SZIP (4) decoding | links the system libaec (`libaec-dev`) |
|
||||||
|
| `fast-deflate` | no | zlib-ng | yes (cmake) |
|
||||||
|
| `system-zlib` | no | the system zlib | yes (`libz-sys`) |
|
||||||
|
| `blake3_hash` | no | `provenance::blake3_hash` | yes (`cc`) |
|
||||||
|
|
||||||
## Usage
|
## Robustness
|
||||||
|
|
||||||
```rust
|
Every parser is meant to return an error, never panic, on hostile input:
|
||||||
use clawhdf5_format::Superblock;
|
nine cargo-fuzz targets live in [`fuzz/`](fuzz/README.md), the conformance
|
||||||
|
sweep includes the HDF Group's CVE corpus
|
||||||
let data = std::fs::read("data.h5").unwrap();
|
([`CONFORMANCE.md`](../../CONFORMANCE.md)), and header checks follow
|
||||||
let sb = Superblock::from_bytes(&data).unwrap();
|
libhdf5's. Open gaps are in [`docs/known-issues.md`](../../docs/known-issues.md).
|
||||||
println!("HDF5 version {}.{}", sb.version_major(), sb.version_minor());
|
|
||||||
```
|
|
||||||
|
|
||||||
## License
|
## License
|
||||||
|
|
||||||
|
|||||||
@@ -51,10 +51,18 @@ done
|
|||||||
|
|
||||||
## CI
|
## CI
|
||||||
|
|
||||||
These targets are **not** run in CI (`.gitea/workflows/ci.yml`) — cargo-fuzz
|
These targets are **not** run by the CI workflows (`.gitea/workflows/ci.yml`)
|
||||||
requires nightly and each meaningful run takes minutes, which doesn't fit a
|
— cargo-fuzz requires nightly and each meaningful run takes minutes, which
|
||||||
per-PR gate. Run them manually on a schedule (e.g. before a release, or after
|
doesn't fit a per-PR gate. Run them by hand before a release or after
|
||||||
touching parser code) instead.
|
touching parser code. `scripts/ci-test.sh` has an opt-in smoke run: with
|
||||||
|
`CLAWHDF5_FUZZ_SECONDS=N` it runs every target of this crate and of
|
||||||
|
`crates/clawhdf5-agent/fuzz` (the WAL parser) for N seconds each.
|
||||||
|
|
||||||
|
Other robustness checks that do run: the nightly conformance sweep reads
|
||||||
|
the HDF Group's CVE reproducers and fails on any panic, hang, crash or
|
||||||
|
out-of-memory ([`conformance/README.md`](../../../conformance/README.md)),
|
||||||
|
and `scripts/h5rs-fuzz.sh` runs every `h5rs` subcommand over them, optionally
|
||||||
|
on byte-flipped copies.
|
||||||
|
|
||||||
## Reproducing Crashes
|
## Reproducing Crashes
|
||||||
|
|
||||||
|
|||||||
Binary file not shown.
@@ -0,0 +1,121 @@
|
|||||||
|
//! File address and length → in-memory index conversion.
|
||||||
|
//!
|
||||||
|
//! HDF5 addresses and lengths are 64-bit; the file is parsed through a
|
||||||
|
//! `&[u8]` indexed by `usize`. On a 64-bit target every `u64` fits, but on a
|
||||||
|
//! 32-bit one (`wasm32`, `i686`, `thumbv7em`) an address past `usize::MAX`
|
||||||
|
//! used to be truncated by an `as usize` cast — silently pointing at another
|
||||||
|
//! part of the file — or to panic. [`to_usize`] is the one conversion the
|
||||||
|
//! parsers use instead: such an address is a clean
|
||||||
|
//! [`FormatError::Overflow`]. It cannot be inside the data anyway: no slice
|
||||||
|
//! is longer than `isize::MAX` bytes.
|
||||||
|
|
||||||
|
#[cfg(not(feature = "std"))]
|
||||||
|
use alloc::format;
|
||||||
|
|
||||||
|
use crate::error::FormatError;
|
||||||
|
|
||||||
|
/// A file address, offset or length from the file as a `usize` index.
|
||||||
|
///
|
||||||
|
/// Fails with [`FormatError::Overflow`] when the value does not fit this
|
||||||
|
/// platform's `usize` (only possible on targets narrower than 64 bits).
|
||||||
|
#[inline]
|
||||||
|
pub fn to_usize(value: u64) -> Result<usize, FormatError> {
|
||||||
|
to_index::<usize>(value)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A file address for a [`crate::storage::Storage`] read, checked as
|
||||||
|
/// [`to_usize`] checks it: the parsers read through 64-bit offsets, but an
|
||||||
|
/// address that could not index an in-memory file on this platform is the
|
||||||
|
/// same [`FormatError::Overflow`] the slice parsers gave for it.
|
||||||
|
#[inline]
|
||||||
|
pub fn checked_addr(value: u64) -> Result<u64, FormatError> {
|
||||||
|
to_usize(value).map(|_| value)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`to_usize`] for an index type of any width. `usize` is 64 bits wide on
|
||||||
|
/// the hosts CI tests on, where the error path cannot be reached through
|
||||||
|
/// `usize`; tests run the same code with `u32` in its place, as on a 32-bit
|
||||||
|
/// target.
|
||||||
|
#[inline]
|
||||||
|
fn to_index<T: TryFrom<u64>>(value: u64) -> Result<T, FormatError> {
|
||||||
|
T::try_from(value).map_err(|_| too_large(value))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A count or offset into an in-memory buffer (a codec's progress counter,
|
||||||
|
/// a size the writer computed from data it holds) as a `usize`, saturating
|
||||||
|
/// at `usize::MAX` instead of truncating.
|
||||||
|
///
|
||||||
|
/// For values that are bounded by the length of something in memory, so
|
||||||
|
/// always fit; if one ever did not, a saturated index fails its bounds check
|
||||||
|
/// or allocation instead of silently addressing the wrong bytes. A value
|
||||||
|
/// read from the file uses [`to_usize`].
|
||||||
|
#[inline]
|
||||||
|
pub fn saturating_usize(value: u64) -> usize {
|
||||||
|
saturating_index(value, usize::MAX)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`saturating_usize`] for an index type of any width, whose largest
|
||||||
|
/// value is `max` (see [`to_index`]).
|
||||||
|
#[inline]
|
||||||
|
fn saturating_index<T: TryFrom<u64>>(value: u64, max: T) -> T {
|
||||||
|
T::try_from(value).unwrap_or(max)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cold]
|
||||||
|
#[inline(never)]
|
||||||
|
fn too_large(value: u64) -> FormatError {
|
||||||
|
FormatError::Overflow(format!(
|
||||||
|
"file address or length {value:#x} exceeds this platform's address space"
|
||||||
|
))
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn values_that_fit_convert_exactly() {
|
||||||
|
assert_eq!(to_usize(0), Ok(0));
|
||||||
|
assert_eq!(to_usize(0x1234), Ok(0x1234));
|
||||||
|
assert_eq!(to_usize(usize::MAX as u64), Ok(usize::MAX));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn saturating_conversion_never_wraps() {
|
||||||
|
assert_eq!(saturating_usize(0), 0);
|
||||||
|
assert_eq!(saturating_usize(0x1234), 0x1234);
|
||||||
|
assert_eq!(saturating_usize(usize::MAX as u64), usize::MAX);
|
||||||
|
// Past usize::MAX (32-bit targets) or at u64::MAX: saturates.
|
||||||
|
assert_eq!(saturating_usize(u64::MAX), usize::MAX);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn values_past_usize_max_are_an_error_not_truncated() {
|
||||||
|
// Reachable through `usize` only where it is narrower than u64 (no
|
||||||
|
// such target runs tests in CI), so the same conversion is run with
|
||||||
|
// u32 standing in for a 32-bit usize.
|
||||||
|
let max = u64::from(u32::MAX);
|
||||||
|
assert_eq!(to_index::<u32>(max), Ok(u32::MAX));
|
||||||
|
for past in [max + 1, max + 0x10, 0x1_0000_1234, u64::MAX] {
|
||||||
|
let err = to_index::<u32>(past).unwrap_err();
|
||||||
|
assert!(
|
||||||
|
matches!(err, FormatError::Overflow(_)),
|
||||||
|
"{past:#x}: {err:?}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
// Where an `as` cast would have wrapped to a small, valid-looking
|
||||||
|
// index, it is not returned.
|
||||||
|
assert_eq!(0x1_0000_1234_u64 as u32, 0x1234);
|
||||||
|
assert!(to_index::<u32>(0x1_0000_1234).is_err());
|
||||||
|
|
||||||
|
assert_eq!(saturating_index(max + 1, u32::MAX), u32::MAX);
|
||||||
|
assert_eq!(saturating_index(0x1_0000_1234, u32::MAX), u32::MAX);
|
||||||
|
assert_eq!(saturating_index(0x1234, u32::MAX), 0x1234);
|
||||||
|
|
||||||
|
// And through `usize` itself, whichever width it has here.
|
||||||
|
match (usize::MAX as u64).checked_add(1) {
|
||||||
|
Some(past) => assert!(matches!(to_usize(past), Err(FormatError::Overflow(_)))),
|
||||||
|
None => assert_eq!(to_usize(u64::MAX), Ok(usize::MAX)),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -5,8 +5,10 @@ use alloc::{borrow::Cow, string::String, vec::Vec};
|
|||||||
#[cfg(feature = "std")]
|
#[cfg(feature = "std")]
|
||||||
use std::borrow::Cow;
|
use std::borrow::Cow;
|
||||||
|
|
||||||
|
use crate::addr::to_usize;
|
||||||
use crate::attribute_info::AttributeInfoMessage;
|
use crate::attribute_info::AttributeInfoMessage;
|
||||||
use crate::btree_v2::{BTreeV2Header, collect_btree_v2_records};
|
use crate::btree_v2::{BTreeV2Header, collect_btree_v2_records_in, find_btree_v2_records_in};
|
||||||
|
use crate::checksum::jenkins_lookup3;
|
||||||
use crate::data_read;
|
use crate::data_read;
|
||||||
use crate::dataspace::Dataspace;
|
use crate::dataspace::Dataspace;
|
||||||
use crate::datatype::Datatype;
|
use crate::datatype::Datatype;
|
||||||
@@ -15,6 +17,7 @@ use crate::fractal_heap::FractalHeapHeader;
|
|||||||
use crate::message_type::MessageType;
|
use crate::message_type::MessageType;
|
||||||
use crate::object_header::ObjectHeader;
|
use crate::object_header::ObjectHeader;
|
||||||
use crate::shared_message;
|
use crate::shared_message;
|
||||||
|
use crate::storage::Storage;
|
||||||
use crate::vl_data;
|
use crate::vl_data;
|
||||||
|
|
||||||
/// A parsed HDF5 attribute message.
|
/// A parsed HDF5 attribute message.
|
||||||
@@ -50,7 +53,7 @@ impl AttributeMessage {
|
|||||||
///
|
///
|
||||||
/// `length_size` is needed for dataspace dimension parsing.
|
/// `length_size` is needed for dataspace dimension parsing.
|
||||||
pub fn parse(data: &[u8], length_size: u8) -> Result<AttributeMessage, FormatError> {
|
pub fn parse(data: &[u8], length_size: u8) -> Result<AttributeMessage, FormatError> {
|
||||||
Self::parse_impl(data, length_size, None)
|
Self::parse_impl(data, length_size, None::<(&[u8], u8)>)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// [`AttributeMessage::parse`] with access to the rest of the file, which
|
/// [`AttributeMessage::parse`] with access to the rest of the file, which
|
||||||
@@ -65,13 +68,24 @@ impl AttributeMessage {
|
|||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
) -> Result<AttributeMessage, FormatError> {
|
) -> Result<AttributeMessage, FormatError> {
|
||||||
Self::parse_impl(data, length_size, Some((file_data, offset_size)))
|
Self::parse_in_storage(data, file_data, offset_size, length_size)
|
||||||
}
|
}
|
||||||
|
|
||||||
fn parse_impl(
|
/// [`AttributeMessage::parse_in_file`] with the file behind any
|
||||||
|
/// [`Storage`].
|
||||||
|
pub fn parse_in_storage<S: Storage + ?Sized>(
|
||||||
|
data: &[u8],
|
||||||
|
file: &S,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<AttributeMessage, FormatError> {
|
||||||
|
Self::parse_impl(data, length_size, Some((file, offset_size)))
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parse_impl<S: Storage + ?Sized>(
|
||||||
data: &[u8],
|
data: &[u8],
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
file: Option<(&[u8], u8)>,
|
file: Option<(&S, u8)>,
|
||||||
) -> Result<AttributeMessage, FormatError> {
|
) -> Result<AttributeMessage, FormatError> {
|
||||||
ensure_len(data, 0, 2)?;
|
ensure_len(data, 0, 2)?;
|
||||||
let version = data[0];
|
let version = data[0];
|
||||||
@@ -86,19 +100,19 @@ impl AttributeMessage {
|
|||||||
|
|
||||||
/// The bytes of an embedded datatype/dataspace message, following the
|
/// The bytes of an embedded datatype/dataspace message, following the
|
||||||
/// shared-message reference when `shared` is set.
|
/// shared-message reference when `shared` is set.
|
||||||
fn embedded_message<'a>(
|
fn embedded_message<'a, S: Storage + ?Sized>(
|
||||||
bytes: &'a [u8],
|
bytes: &'a [u8],
|
||||||
shared: bool,
|
shared: bool,
|
||||||
msg_type: MessageType,
|
msg_type: MessageType,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
file: Option<(&[u8], u8)>,
|
file: Option<(&S, u8)>,
|
||||||
) -> Result<Cow<'a, [u8]>, FormatError> {
|
) -> Result<Cow<'a, [u8]>, FormatError> {
|
||||||
if !shared {
|
if !shared {
|
||||||
return Ok(Cow::Borrowed(bytes));
|
return Ok(Cow::Borrowed(bytes));
|
||||||
}
|
}
|
||||||
let (file_data, offset_size) = file.ok_or(FormatError::UnresolvedSharedMessage)?;
|
let (file_data, offset_size) = file.ok_or(FormatError::UnresolvedSharedMessage)?;
|
||||||
let shared_ref = shared_message::parse_shared_ref(bytes, offset_size)?;
|
let shared_ref = shared_message::parse_shared_ref_sized(bytes, offset_size, length_size)?;
|
||||||
shared_message::resolve_shared_message(
|
shared_message::resolve_shared_message_in(
|
||||||
file_data,
|
file_data,
|
||||||
&shared_ref,
|
&shared_ref,
|
||||||
msg_type,
|
msg_type,
|
||||||
@@ -143,10 +157,10 @@ impl AttributeMessage {
|
|||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
fn parse_v2(
|
fn parse_v2<S: Storage + ?Sized>(
|
||||||
data: &[u8],
|
data: &[u8],
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
file: Option<(&[u8], u8)>,
|
file: Option<(&S, u8)>,
|
||||||
) -> Result<AttributeMessage, FormatError> {
|
) -> Result<AttributeMessage, FormatError> {
|
||||||
// Flags: bit 0 = datatype is shared, bit 1 = dataspace is shared.
|
// Flags: bit 0 = datatype is shared, bit 1 = dataspace is shared.
|
||||||
let flags = data.get(1).copied().unwrap_or(0);
|
let flags = data.get(1).copied().unwrap_or(0);
|
||||||
@@ -197,10 +211,10 @@ impl AttributeMessage {
|
|||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
fn parse_v3(
|
fn parse_v3<S: Storage + ?Sized>(
|
||||||
data: &[u8],
|
data: &[u8],
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
file: Option<(&[u8], u8)>,
|
file: Option<(&S, u8)>,
|
||||||
) -> Result<AttributeMessage, FormatError> {
|
) -> Result<AttributeMessage, FormatError> {
|
||||||
// Flags: bit 0 = datatype is shared, bit 1 = dataspace is shared.
|
// Flags: bit 0 = datatype is shared, bit 1 = dataspace is shared.
|
||||||
let flags = data.get(1).copied().unwrap_or(0);
|
let flags = data.get(1).copied().unwrap_or(0);
|
||||||
@@ -322,9 +336,19 @@ impl AttributeMessage {
|
|||||||
file_data: &[u8],
|
file_data: &[u8],
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
|
) -> Result<Vec<String>, FormatError> {
|
||||||
|
self.read_vl_strings_in(file_data, offset_size, length_size)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`Self::read_vl_strings`] over any [`Storage`].
|
||||||
|
pub fn read_vl_strings_in<S: Storage + ?Sized>(
|
||||||
|
&self,
|
||||||
|
file_data: &S,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
) -> Result<Vec<String>, FormatError> {
|
) -> Result<Vec<String>, FormatError> {
|
||||||
let num_elements = self.dataspace.num_elements();
|
let num_elements = self.dataspace.num_elements();
|
||||||
vl_data::read_vl_strings(
|
vl_data::read_vl_strings_in(
|
||||||
file_data,
|
file_data,
|
||||||
&self.raw_data,
|
&self.raw_data,
|
||||||
num_elements,
|
num_elements,
|
||||||
@@ -341,7 +365,8 @@ fn compute_raw_data(
|
|||||||
dataspace: &Dataspace,
|
dataspace: &Dataspace,
|
||||||
datatype: &Datatype,
|
datatype: &Datatype,
|
||||||
) -> Vec<u8> {
|
) -> Vec<u8> {
|
||||||
let num_elements = dataspace.num_elements() as usize;
|
// Saturating, like the product: the size is capped at what is there.
|
||||||
|
let num_elements = usize::try_from(dataspace.num_elements()).unwrap_or(usize::MAX);
|
||||||
let elem_size = datatype.type_size() as usize;
|
let elem_size = datatype.type_size() as usize;
|
||||||
let expected_size = num_elements.saturating_mul(elem_size);
|
let expected_size = num_elements.saturating_mul(elem_size);
|
||||||
let available = data.len().saturating_sub(pos);
|
let available = data.len().saturating_sub(pos);
|
||||||
@@ -362,6 +387,18 @@ fn extract_name(bytes: &[u8]) -> String {
|
|||||||
String::from_utf8_lossy(&bytes[..end]).into_owned()
|
String::from_utf8_lossy(&bytes[..end]).into_owned()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// An attribute's datatype gets libhdf5's extra check for a header without
|
||||||
|
/// a checksum (see [`Datatype::check_unused_bits`]).
|
||||||
|
fn check_in_header(
|
||||||
|
attr: AttributeMessage,
|
||||||
|
header: &ObjectHeader,
|
||||||
|
) -> Result<AttributeMessage, FormatError> {
|
||||||
|
if header.version == 1 {
|
||||||
|
attr.datatype.check_unused_bits()?;
|
||||||
|
}
|
||||||
|
Ok(attr)
|
||||||
|
}
|
||||||
|
|
||||||
/// Extract all attribute messages from an object header.
|
/// Extract all attribute messages from an object header.
|
||||||
pub fn extract_attributes(
|
pub fn extract_attributes(
|
||||||
header: &ObjectHeader,
|
header: &ObjectHeader,
|
||||||
@@ -371,7 +408,7 @@ pub fn extract_attributes(
|
|||||||
for msg in &header.messages {
|
for msg in &header.messages {
|
||||||
if msg.msg_type == MessageType::Attribute {
|
if msg.msg_type == MessageType::Attribute {
|
||||||
let attr = AttributeMessage::parse(&msg.data, length_size)?;
|
let attr = AttributeMessage::parse(&msg.data, length_size)?;
|
||||||
attrs.push(attr);
|
attrs.push(check_in_header(attr, header)?);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
Ok(attrs)
|
Ok(attrs)
|
||||||
@@ -394,59 +431,339 @@ pub fn find_attribute<'a>(
|
|||||||
///
|
///
|
||||||
/// Use this instead of `extract_attributes` when reading files that may use dense storage
|
/// Use this instead of `extract_attributes` when reading files that may use dense storage
|
||||||
/// (e.g., objects with many attributes, typically >8).
|
/// (e.g., objects with many attributes, typically >8).
|
||||||
|
///
|
||||||
|
/// Fails if any attribute cannot be read; see [`extract_attributes_tolerant`]
|
||||||
|
/// to read the others.
|
||||||
pub fn extract_attributes_full(
|
pub fn extract_attributes_full(
|
||||||
file_data: &[u8],
|
file_data: &[u8],
|
||||||
header: &ObjectHeader,
|
header: &ObjectHeader,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
) -> Result<Vec<AttributeMessage>, FormatError> {
|
) -> Result<Vec<AttributeMessage>, FormatError> {
|
||||||
let mut attrs = Vec::new();
|
extract_attributes_full_in(file_data, header, offset_size, length_size)
|
||||||
|
}
|
||||||
|
|
||||||
// Collect compact attributes (inline in OH)
|
/// [`extract_attributes_full`] over any [`Storage`]. Dense attribute
|
||||||
for msg in &header.messages {
|
/// storage is indexed by a v2 B-tree, which is not read over [`Storage`]
|
||||||
if msg.msg_type == MessageType::Attribute {
|
/// yet: on a backend without the whole file in memory an object with dense
|
||||||
if shared_message::is_shared(msg.flags) {
|
/// attributes is [`FormatError::ContiguousStorageRequired`].
|
||||||
// Shared attribute: resolve the reference to get actual attribute data
|
pub fn extract_attributes_full_in<S: Storage + ?Sized>(
|
||||||
let shared_ref = shared_message::parse_shared_ref(&msg.data, offset_size)?;
|
file: &S,
|
||||||
let resolved_data = shared_message::resolve_shared_message(
|
header: &ObjectHeader,
|
||||||
file_data,
|
offset_size: u8,
|
||||||
&shared_ref,
|
length_size: u8,
|
||||||
MessageType::Attribute,
|
) -> Result<Vec<AttributeMessage>, FormatError> {
|
||||||
offset_size,
|
extract_attributes_with(file, header, offset_size, length_size, &mut Err)
|
||||||
length_size,
|
}
|
||||||
)?;
|
|
||||||
let attr = AttributeMessage::parse_in_file(
|
/// Like [`extract_attributes_full`], but an attribute that cannot be read
|
||||||
&resolved_data,
|
/// (a corrupt or unsupported attribute message, or a heap object that cannot
|
||||||
file_data,
|
/// be located) is left out and its error returned alongside the attributes
|
||||||
offset_size,
|
/// that could be read, instead of failing them all.
|
||||||
length_size,
|
///
|
||||||
)?;
|
/// Errors in the structures that index the attributes (the Attribute Info
|
||||||
attrs.push(attr);
|
/// message, the dense-storage heap header or B-tree) still fail the call:
|
||||||
} else {
|
/// then it is unknown which attributes exist at all.
|
||||||
let attr = AttributeMessage::parse_in_file(
|
pub fn extract_attributes_tolerant(
|
||||||
&msg.data,
|
file_data: &[u8],
|
||||||
file_data,
|
header: &ObjectHeader,
|
||||||
offset_size,
|
offset_size: u8,
|
||||||
length_size,
|
length_size: u8,
|
||||||
)?;
|
) -> Result<(Vec<AttributeMessage>, Vec<FormatError>), FormatError> {
|
||||||
attrs.push(attr);
|
extract_attributes_tolerant_core(file_data, header, offset_size, length_size)
|
||||||
}
|
}
|
||||||
}
|
|
||||||
|
/// [`extract_attributes_tolerant`] over any [`Storage`] (see
|
||||||
|
/// [`extract_attributes_full_in`] for dense storage). One with the whole
|
||||||
|
/// file in memory is read as the slice, by code compiled in this crate (see
|
||||||
|
/// [`crate::storage`], "Slice entry points").
|
||||||
|
#[inline]
|
||||||
|
pub fn extract_attributes_tolerant_in<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
header: &ObjectHeader,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<(Vec<AttributeMessage>, Vec<FormatError>), FormatError> {
|
||||||
|
match file_data.as_contiguous() {
|
||||||
|
Some(all) => extract_attributes_tolerant(all, header, offset_size, length_size),
|
||||||
|
None => extract_attributes_tolerant_core(file_data, header, offset_size, length_size),
|
||||||
}
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn extract_attributes_tolerant_core<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
header: &ObjectHeader,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<(Vec<AttributeMessage>, Vec<FormatError>), FormatError> {
|
||||||
|
let mut errors = Vec::new();
|
||||||
|
let attrs = extract_attributes_with(file_data, header, offset_size, length_size, &mut |e| {
|
||||||
|
errors.push(e);
|
||||||
|
Ok(())
|
||||||
|
})?;
|
||||||
|
Ok((attrs, errors))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read every attribute; each one that fails goes to `on_error`, which
|
||||||
|
/// either stops the read (returns the error) or skips that attribute.
|
||||||
|
fn extract_attributes_with<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
header: &ObjectHeader,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
on_error: &mut dyn FnMut(FormatError) -> Result<(), FormatError>,
|
||||||
|
) -> Result<Vec<AttributeMessage>, FormatError> {
|
||||||
|
let mut attrs = Vec::new();
|
||||||
|
// Each attribute's creation order, where the file records one.
|
||||||
|
let mut orders: Vec<u32> = Vec::new();
|
||||||
|
|
||||||
|
extract_compact_attributes(
|
||||||
|
file_data,
|
||||||
|
header,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
&mut attrs,
|
||||||
|
&mut orders,
|
||||||
|
on_error,
|
||||||
|
)?;
|
||||||
|
|
||||||
// Check for dense attributes via AttributeInfo message
|
// Check for dense attributes via AttributeInfo message
|
||||||
let attr_info = find_attribute_info(header, offset_size)?;
|
let attr_info = find_attribute_info(header, offset_size)?;
|
||||||
if let Some(info) = attr_info
|
if let Some(info) = &attr_info
|
||||||
&& let Some(fh_addr) = info.fractal_heap_address
|
&& let Some(fh_addr) = info.fractal_heap_address
|
||||||
{
|
{
|
||||||
let dense_attrs =
|
extract_dense_attributes(
|
||||||
extract_dense_attributes(file_data, &info, fh_addr, offset_size, length_size)?;
|
file_data,
|
||||||
attrs.extend(dense_attrs);
|
info,
|
||||||
|
fh_addr,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
&mut attrs,
|
||||||
|
&mut orders,
|
||||||
|
on_error,
|
||||||
|
)?;
|
||||||
|
}
|
||||||
|
|
||||||
|
// An object that tracks attribute creation order lists its attributes
|
||||||
|
// in that order (h5py's `track_order=True`), as libhdf5 does; otherwise
|
||||||
|
// they come in storage order.
|
||||||
|
if attr_info.is_some_and(|i| i.max_creation_index.is_some()) {
|
||||||
|
let mut paired: Vec<(u32, AttributeMessage)> = orders.into_iter().zip(attrs).collect();
|
||||||
|
paired.sort_by_key(|(o, _)| *o);
|
||||||
|
attrs = paired.into_iter().map(|(_, a)| a).collect();
|
||||||
}
|
}
|
||||||
|
|
||||||
Ok(attrs)
|
Ok(attrs)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// B-tree v2 record type of dense attribute storage's name index.
|
||||||
|
const ATTRIBUTE_NAME_INDEX: u8 = 8;
|
||||||
|
|
||||||
|
/// The attribute called `name` on the object with header `header`: the
|
||||||
|
/// first one [`extract_attributes_tolerant`] returns under that name, or
|
||||||
|
/// `None` if it returns none (an attribute that cannot be read is not
|
||||||
|
/// returned there either).
|
||||||
|
///
|
||||||
|
/// Compact attributes are in the header and are scanned. Dense attributes
|
||||||
|
/// are found through the name index (a v2 B-tree of lookup3 name hashes,
|
||||||
|
/// record type 8): only the attributes whose names hash like `name` are read
|
||||||
|
/// from the heap, O(log n) instead of all of them. Errors in the structures
|
||||||
|
/// that index the attributes fail the call, as they fail a listing.
|
||||||
|
pub fn find_attribute_in_file(
|
||||||
|
file_data: &[u8],
|
||||||
|
header: &ObjectHeader,
|
||||||
|
name: &str,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<Option<AttributeMessage>, FormatError> {
|
||||||
|
find_attribute_core(
|
||||||
|
file_data,
|
||||||
|
header,
|
||||||
|
name,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
&mut Vec::new(),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`find_attribute_in_file`] over any [`Storage`] (see
|
||||||
|
/// [`extract_attributes_full_in`] for dense storage, whose name index still
|
||||||
|
/// needs the whole file in memory). One with the whole file in memory is
|
||||||
|
/// read as the slice, by code compiled in this crate (see
|
||||||
|
/// [`crate::storage`], "Slice entry points").
|
||||||
|
#[inline]
|
||||||
|
pub fn find_attribute_in<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
header: &ObjectHeader,
|
||||||
|
name: &str,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<Option<AttributeMessage>, FormatError> {
|
||||||
|
match file_data.as_contiguous() {
|
||||||
|
Some(all) => find_attribute_in_file(all, header, name, offset_size, length_size),
|
||||||
|
None => find_attribute_core(
|
||||||
|
file_data,
|
||||||
|
header,
|
||||||
|
name,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
&mut Vec::new(),
|
||||||
|
),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`find_attribute_in`], also returning the errors of the attributes it
|
||||||
|
/// could not read on the way (which it leaves out rather than failing
|
||||||
|
/// the call): the attribute asked for may be one of them. A reader of a
|
||||||
|
/// file that is being written uses them to tell a read that raced the
|
||||||
|
/// writer from an absent attribute.
|
||||||
|
pub fn find_attribute_reporting_in<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
header: &ObjectHeader,
|
||||||
|
name: &str,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<(Option<AttributeMessage>, Vec<FormatError>), FormatError> {
|
||||||
|
let mut errors = Vec::new();
|
||||||
|
let found = find_attribute_core(
|
||||||
|
file_data,
|
||||||
|
header,
|
||||||
|
name,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
&mut errors,
|
||||||
|
)?;
|
||||||
|
Ok((found, errors))
|
||||||
|
}
|
||||||
|
|
||||||
|
fn find_attribute_core<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
header: &ObjectHeader,
|
||||||
|
name: &str,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
errors: &mut Vec<FormatError>,
|
||||||
|
) -> Result<Option<AttributeMessage>, FormatError> {
|
||||||
|
let attr_info = find_attribute_info(header, offset_size)?;
|
||||||
|
let dense = attr_info
|
||||||
|
.as_ref()
|
||||||
|
.and_then(|i| Some((i.fractal_heap_address?, i.btree_name_index_address?)));
|
||||||
|
let Some((fh_addr, btree_addr)) = dense else {
|
||||||
|
// Compact only (or dense storage without a name index, which a
|
||||||
|
// listing reports): as a listing finds it.
|
||||||
|
let (attrs, errs) =
|
||||||
|
extract_attributes_tolerant_in(file_data, header, offset_size, length_size)?;
|
||||||
|
errors.extend(errs);
|
||||||
|
return Ok(attrs.into_iter().find(|a| a.name == name));
|
||||||
|
};
|
||||||
|
let btree_hdr = BTreeV2Header::parse_in(
|
||||||
|
file_data,
|
||||||
|
to_usize(btree_addr)? as u64,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
)?;
|
||||||
|
let fh = FractalHeapHeader::parse_in(file_data, fh_addr, offset_size, length_size)?;
|
||||||
|
if btree_hdr.tree_type != ATTRIBUTE_NAME_INDEX || btree_hdr.record_size < 4 {
|
||||||
|
let (attrs, errs) =
|
||||||
|
extract_attributes_tolerant_in(file_data, header, offset_size, length_size)?;
|
||||||
|
errors.extend(errs);
|
||||||
|
return Ok(attrs.into_iter().find(|a| a.name == name));
|
||||||
|
}
|
||||||
|
|
||||||
|
// A listing has the compact attributes first.
|
||||||
|
let mut compact = Vec::new();
|
||||||
|
extract_compact_attributes(
|
||||||
|
file_data,
|
||||||
|
header,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
&mut compact,
|
||||||
|
&mut Vec::new(),
|
||||||
|
&mut |_| Ok(()),
|
||||||
|
)?;
|
||||||
|
if let Some(a) = compact.into_iter().find(|a| a.name == name) {
|
||||||
|
return Ok(Some(a));
|
||||||
|
}
|
||||||
|
|
||||||
|
// Record: heap ID + message flags(1) + creation order(4) + hash(4); the
|
||||||
|
// hash is the last field.
|
||||||
|
let hash = jenkins_lookup3(name.as_bytes());
|
||||||
|
let hash_at = usize::from(btree_hdr.record_size) - 4;
|
||||||
|
let records = find_btree_v2_records_in(file_data, &btree_hdr, offset_size, &mut |r| match r
|
||||||
|
.get(hash_at..hash_at + 4)
|
||||||
|
{
|
||||||
|
Some(h) => u32::from_le_bytes([h[0], h[1], h[2], h[3]]).cmp(&hash),
|
||||||
|
None => core::cmp::Ordering::Less,
|
||||||
|
})?;
|
||||||
|
let id_len = usize::from(fh.heap_id_length);
|
||||||
|
for record in &records {
|
||||||
|
let Some(id_bytes) = record.data.get(..id_len) else {
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
let attr = fh
|
||||||
|
.read_managed_object_in(file_data, id_bytes, offset_size)
|
||||||
|
.and_then(|d| {
|
||||||
|
AttributeMessage::parse_in_storage(&d, file_data, offset_size, length_size)
|
||||||
|
});
|
||||||
|
// One that cannot be read is left out, as from a listing.
|
||||||
|
match attr {
|
||||||
|
Ok(attr) if attr.name == name => return Ok(Some(attr)),
|
||||||
|
Ok(_) => {}
|
||||||
|
Err(e) => errors.push(e),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(None)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The attributes stored in the object header itself (compact storage), and
|
||||||
|
/// each one's creation order into `orders`.
|
||||||
|
fn extract_compact_attributes<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
header: &ObjectHeader,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
attrs: &mut Vec<AttributeMessage>,
|
||||||
|
orders: &mut Vec<u32>,
|
||||||
|
on_error: &mut dyn FnMut(FormatError) -> Result<(), FormatError>,
|
||||||
|
) -> Result<(), FormatError> {
|
||||||
|
for msg in &header.messages {
|
||||||
|
if msg.msg_type == MessageType::Attribute {
|
||||||
|
let attr = if shared_message::is_shared(msg.flags) {
|
||||||
|
// Shared attribute: resolve the reference to get actual attribute data
|
||||||
|
shared_message::parse_shared_ref_sized(&msg.data, offset_size, length_size)
|
||||||
|
.and_then(|shared_ref| {
|
||||||
|
shared_message::resolve_shared_message_in(
|
||||||
|
file_data,
|
||||||
|
&shared_ref,
|
||||||
|
MessageType::Attribute,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
)
|
||||||
|
})
|
||||||
|
.and_then(|resolved| {
|
||||||
|
AttributeMessage::parse_in_storage(
|
||||||
|
&resolved,
|
||||||
|
file_data,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
)
|
||||||
|
})
|
||||||
|
} else {
|
||||||
|
AttributeMessage::parse_in_storage(&msg.data, file_data, offset_size, length_size)
|
||||||
|
};
|
||||||
|
let attr = attr.and_then(|a| check_in_header(a, header));
|
||||||
|
match attr {
|
||||||
|
Ok(attr) => {
|
||||||
|
attrs.push(attr);
|
||||||
|
orders.push(msg.creation_order.map_or(0, u32::from));
|
||||||
|
}
|
||||||
|
Err(e) => on_error(e)?,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
/// Find and parse the Attribute Info message from an object header.
|
/// Find and parse the Attribute Info message from an object header.
|
||||||
fn find_attribute_info(
|
fn find_attribute_info(
|
||||||
header: &ObjectHeader,
|
header: &ObjectHeader,
|
||||||
@@ -461,16 +778,21 @@ fn find_attribute_info(
|
|||||||
Ok(None)
|
Ok(None)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Extract attributes from dense storage (fractal heap + B-tree v2).
|
/// Extract attributes from dense storage (fractal heap + B-tree v2), and
|
||||||
fn extract_dense_attributes(
|
/// each one's creation order into `orders`.
|
||||||
file_data: &[u8],
|
#[allow(clippy::too_many_arguments)]
|
||||||
|
fn extract_dense_attributes<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
attr_info: &AttributeInfoMessage,
|
attr_info: &AttributeInfoMessage,
|
||||||
fh_addr: u64,
|
fh_addr: u64,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
) -> Result<Vec<AttributeMessage>, FormatError> {
|
attrs: &mut Vec<AttributeMessage>,
|
||||||
|
orders: &mut Vec<u32>,
|
||||||
|
on_error: &mut dyn FnMut(FormatError) -> Result<(), FormatError>,
|
||||||
|
) -> Result<(), FormatError> {
|
||||||
// Parse fractal heap
|
// Parse fractal heap
|
||||||
let fh = FractalHeapHeader::parse(file_data, fh_addr as usize, offset_size, length_size)?;
|
let fh = FractalHeapHeader::parse_in(file_data, fh_addr, offset_size, length_size)?;
|
||||||
|
|
||||||
// Parse B-tree v2 for name index (type 8)
|
// Parse B-tree v2 for name index (type 8)
|
||||||
let btree_addr = attr_info
|
let btree_addr = attr_info
|
||||||
@@ -479,31 +801,47 @@ fn extract_dense_attributes(
|
|||||||
expected: 1,
|
expected: 1,
|
||||||
available: 0,
|
available: 0,
|
||||||
})?;
|
})?;
|
||||||
let btree_hdr = BTreeV2Header::parse(file_data, btree_addr as usize, offset_size, length_size)?;
|
let btree_hdr = BTreeV2Header::parse_in(
|
||||||
let records = collect_btree_v2_records(file_data, &btree_hdr, offset_size, length_size)?;
|
file_data,
|
||||||
|
to_usize(btree_addr)? as u64,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
)?;
|
||||||
|
let records = collect_btree_v2_records_in(file_data, &btree_hdr, offset_size, length_size)?;
|
||||||
|
|
||||||
let mut attrs = Vec::new();
|
|
||||||
for record in &records {
|
for record in &records {
|
||||||
// Per HDF5 spec, both type 8 and type 9 records start with heap_id:
|
// Per HDF5 spec, both type 8 and type 9 records start with heap_id:
|
||||||
// Type 8: heap_id(8) + msg_flags(1) + creation_order(4) + hash(4)
|
// Type 8: heap_id(8) + msg_flags(1) + creation_order(4) + hash(4)
|
||||||
// Type 9: heap_id(8) + msg_flags(1) + creation_order(4)
|
// Type 9: heap_id(8) + msg_flags(1) + creation_order(4)
|
||||||
let id_offset = 0;
|
let id_len = fh.heap_id_length as usize;
|
||||||
|
let Some(id_bytes) = record.data.get(..id_len) else {
|
||||||
if record.data.len() < id_offset + fh.heap_id_length as usize {
|
on_error(FormatError::UnexpectedEof {
|
||||||
|
expected: id_len,
|
||||||
|
available: record.data.len(),
|
||||||
|
})?;
|
||||||
continue;
|
continue;
|
||||||
}
|
};
|
||||||
let id_bytes = &record.data[id_offset..id_offset + fh.heap_id_length as usize];
|
|
||||||
|
|
||||||
// Read attribute message from fractal heap
|
|
||||||
let attr_data = fh.read_managed_object(file_data, id_bytes, offset_size)?;
|
|
||||||
|
|
||||||
// The data in the heap is a complete attribute message
|
// The data in the heap is a complete attribute message
|
||||||
let attr =
|
let attr = fh
|
||||||
AttributeMessage::parse_in_file(&attr_data, file_data, offset_size, length_size)?;
|
.read_managed_object_in(file_data, id_bytes, offset_size)
|
||||||
attrs.push(attr);
|
.and_then(|attr_data| {
|
||||||
|
AttributeMessage::parse_in_storage(&attr_data, file_data, offset_size, length_size)
|
||||||
|
});
|
||||||
|
match attr {
|
||||||
|
Ok(attr) => {
|
||||||
|
attrs.push(attr);
|
||||||
|
let order = record
|
||||||
|
.data
|
||||||
|
.get(id_len + 1..id_len + 5)
|
||||||
|
.map_or(0, |b| u32::from_le_bytes([b[0], b[1], b[2], b[3]]));
|
||||||
|
orders.push(order);
|
||||||
|
}
|
||||||
|
Err(e) => on_error(e)?,
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
Ok(attrs)
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
@@ -523,7 +861,8 @@ mod tests {
|
|||||||
|
|
||||||
/// Build an f64 LE datatype message.
|
/// Build an f64 LE datatype message.
|
||||||
fn build_f64_dt() -> Vec<u8> {
|
fn build_f64_dt() -> Vec<u8> {
|
||||||
let mut buf = build_dt_header(1, 1, [0x00, 0x00, 0x02], 8);
|
// Sign bit 63 (bits 8-15 of the class bits).
|
||||||
|
let mut buf = build_dt_header(1, 1, [0x20, 63, 0x00], 8);
|
||||||
let mut props = [0u8; 12];
|
let mut props = [0u8; 12];
|
||||||
props[2..4].copy_from_slice(&64u16.to_le_bytes()); // bit_precision
|
props[2..4].copy_from_slice(&64u16.to_le_bytes()); // bit_precision
|
||||||
props[4] = 52; // exp_location
|
props[4] = 52; // exp_location
|
||||||
@@ -897,4 +1236,73 @@ mod tests {
|
|||||||
let strs = attr.read_as_strings().unwrap();
|
let strs = attr.read_as_strings().unwrap();
|
||||||
assert_eq!(strs, vec!["abcd", "EFGH"]);
|
assert_eq!(strs, vec!["abcd", "EFGH"]);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Every object's attributes in h5py-written files read identically
|
||||||
|
/// through a read_at-only CountingStorage — compact ones, shared ones,
|
||||||
|
/// those behind an Attribute Info message and dense storage (its v2
|
||||||
|
/// B-tree name index included) — and through a slice as Storage.
|
||||||
|
#[test]
|
||||||
|
fn storage_reads_match_slice_reads() {
|
||||||
|
use crate::storage::CountingStorage;
|
||||||
|
let files: [(&str, &[u8]); 5] = [
|
||||||
|
("attrs", include_bytes!("../tests/fixtures/attrs.h5")),
|
||||||
|
(
|
||||||
|
"mixed_attrs",
|
||||||
|
include_bytes!("../tests/fixtures/mixed_attrs.h5"),
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"dense_attrs",
|
||||||
|
include_bytes!("../tests/fixtures/dense_attrs.h5"),
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"dense_attrs_root",
|
||||||
|
include_bytes!("../tests/fixtures/dense_attrs_root.h5"),
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"shared_fill_value",
|
||||||
|
include_bytes!("../tests/fixtures/shared_fill_value.h5"),
|
||||||
|
),
|
||||||
|
];
|
||||||
|
let (mut same, mut dense, mut attrs) = (0, 0, 0);
|
||||||
|
for (name, file) in files {
|
||||||
|
let sb = crate::superblock::Superblock::parse(file, 0).unwrap();
|
||||||
|
let (os, ls) = (sb.offset_size, sb.length_size);
|
||||||
|
let mut addrs = vec![sb.root_group_address];
|
||||||
|
addrs.extend(
|
||||||
|
crate::group_v2::resolve_group_children(file, &sb, sb.root_group_address)
|
||||||
|
.unwrap()
|
||||||
|
.iter()
|
||||||
|
.map(|e| e.object_header_address),
|
||||||
|
);
|
||||||
|
let storage = CountingStorage::new(file.to_vec());
|
||||||
|
for addr in addrs {
|
||||||
|
let header = ObjectHeader::parse(file, addr as usize, os, ls).unwrap();
|
||||||
|
let want = extract_attributes_full(file, &header, os, ls);
|
||||||
|
let slice_storage = extract_attributes_full_in(&file, &header, os, ls);
|
||||||
|
assert_eq!(format!("{slice_storage:?}"), format!("{want:?}"));
|
||||||
|
let got = extract_attributes_full_in(&storage, &header, os, ls);
|
||||||
|
let got_t = extract_attributes_tolerant_in(&storage, &header, os, ls);
|
||||||
|
let is_dense = find_attribute_info(&header, os)
|
||||||
|
.unwrap()
|
||||||
|
.is_some_and(|i| i.fractal_heap_address.is_some());
|
||||||
|
if is_dense {
|
||||||
|
dense += 1;
|
||||||
|
}
|
||||||
|
attrs += want.as_ref().map_or(0, Vec::len);
|
||||||
|
assert_eq!(format!("{got:?}"), format!("{want:?}"), "{name}");
|
||||||
|
let want_t = extract_attributes_tolerant(file, &header, os, ls);
|
||||||
|
assert_eq!(format!("{got_t:?}"), format!("{want_t:?}"), "{name}");
|
||||||
|
same += 1;
|
||||||
|
for a in want.iter().flatten() {
|
||||||
|
let one = find_attribute_in(&storage, &header, &a.name, os, ls);
|
||||||
|
let want_one = find_attribute_in_file(file, &header, &a.name, os, ls);
|
||||||
|
assert_eq!(format!("{one:?}"), format!("{want_one:?}"), "{name}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
assert!(
|
||||||
|
same >= 5 && dense >= 2 && attrs >= 5,
|
||||||
|
"{same} {dense} {attrs}"
|
||||||
|
);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -4,6 +4,7 @@
|
|||||||
use alloc::vec::Vec;
|
use alloc::vec::Vec;
|
||||||
|
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
|
use crate::storage::{Storage, read_exact_at};
|
||||||
|
|
||||||
/// A parsed B-tree v1 node.
|
/// A parsed B-tree v1 node.
|
||||||
#[derive(Debug, Clone)]
|
#[derive(Debug, Clone)]
|
||||||
@@ -74,13 +75,30 @@ impl BTreeV1Node {
|
|||||||
file_data: &[u8],
|
file_data: &[u8],
|
||||||
offset: usize,
|
offset: usize,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<BTreeV1Node, FormatError> {
|
||||||
|
Self::parse_in(file_data, offset as u64, offset_size, length_size)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`Self::parse`] over any [`Storage`]: one read of the node's header,
|
||||||
|
/// one of its keys and children.
|
||||||
|
pub fn parse_in<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
offset: u64,
|
||||||
|
offset_size: u8,
|
||||||
_length_size: u8,
|
_length_size: u8,
|
||||||
) -> Result<BTreeV1Node, FormatError> {
|
) -> Result<BTreeV1Node, FormatError> {
|
||||||
// signature(4) + node_type(1) + node_level(1) + entries_used(2) = 8
|
// signature(4) + node_type(1) + node_level(1) + entries_used(2) = 8
|
||||||
// + left_sibling(offset_size) + right_sibling(offset_size)
|
// + left_sibling(offset_size) + right_sibling(offset_size)
|
||||||
let os = offset_size as usize;
|
let os = offset_size as usize;
|
||||||
let header_size = 8 + os * 2;
|
let header_size = 8 + os * 2;
|
||||||
ensure_len(file_data, offset, header_size)?;
|
// The body is read once the header says how long it is.
|
||||||
|
file.hint(offset, NODE_HINT_LEN);
|
||||||
|
let header = read_exact_at(file, offset, header_size)?;
|
||||||
|
let file_data: &[u8] = &header;
|
||||||
|
// The header's read checked that `offset + header_size` fits.
|
||||||
|
let body_start = offset + header_size as u64;
|
||||||
|
let offset = 0usize;
|
||||||
|
|
||||||
if &file_data[offset..offset + 4] != b"TREE" {
|
if &file_data[offset..offset + 4] != b"TREE" {
|
||||||
return Err(FormatError::InvalidBTreeSignature);
|
return Err(FormatError::InvalidBTreeSignature);
|
||||||
@@ -102,31 +120,30 @@ impl BTreeV1Node {
|
|||||||
} else {
|
} else {
|
||||||
Some(read_offset(file_data, pos, offset_size)?)
|
Some(read_offset(file_data, pos, offset_size)?)
|
||||||
};
|
};
|
||||||
pos += os;
|
|
||||||
|
|
||||||
// For type 0: keys are offset_size bytes, children are offset_size bytes
|
// For type 0: keys are offset_size bytes, children are offset_size bytes
|
||||||
// Layout: key[0], child[0], key[1], child[1], ..., key[N-1], child[N-1], key[N]
|
// Layout: key[0], child[0], key[1], child[1], ..., key[N-1], child[N-1], key[N]
|
||||||
let eu = entries_used as usize;
|
let eu = entries_used as usize;
|
||||||
let key_size = os; // For type 0, key = offset_size
|
let key_size = os; // For type 0, key = offset_size
|
||||||
let needed = eu * (key_size + os) + key_size; // eu children + (eu+1) keys
|
let needed = eu * (key_size + os) + key_size; // eu children + (eu+1) keys
|
||||||
ensure_len(file_data, pos, needed)?;
|
let body = read_exact_at(file, body_start, needed)?;
|
||||||
|
let file_data: &[u8] = &body;
|
||||||
|
|
||||||
let mut keys = Vec::with_capacity(eu + 1);
|
let mut keys = Vec::with_capacity(eu + 1);
|
||||||
let mut children = Vec::with_capacity(eu);
|
let mut children = Vec::with_capacity(eu);
|
||||||
|
|
||||||
for _i in 0..eu {
|
if os == 0 {
|
||||||
// key[i]
|
// What reading the first key reports (and keeps `chunks_exact`
|
||||||
let key = read_offset(file_data, pos, offset_size)?;
|
// below from being given a zero size).
|
||||||
keys.push(key);
|
return Err(FormatError::InvalidOffsetSize(offset_size));
|
||||||
pos += key_size;
|
|
||||||
// child[i]
|
|
||||||
let child = read_offset(file_data, pos, offset_size)?;
|
|
||||||
children.push(child);
|
|
||||||
pos += os;
|
|
||||||
}
|
}
|
||||||
// final key
|
// `needed` bytes: key[0], child[0], ..., child[eu - 1], key[eu].
|
||||||
let key = read_offset(file_data, pos, offset_size)?;
|
let (pairs, last) = file_data.split_at(eu * (key_size + os));
|
||||||
keys.push(key);
|
for pair in pairs.chunks_exact(key_size + os) {
|
||||||
|
keys.push(read_offset(pair, 0, offset_size)?);
|
||||||
|
children.push(read_offset(pair, key_size, offset_size)?);
|
||||||
|
}
|
||||||
|
keys.push(read_offset(last, 0, offset_size)?);
|
||||||
|
|
||||||
Ok(BTreeV1Node {
|
Ok(BTreeV1Node {
|
||||||
node_type,
|
node_type,
|
||||||
@@ -141,7 +158,18 @@ impl BTreeV1Node {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Maximum recursion depth for B-tree traversal (malformed data protection).
|
/// Maximum recursion depth for B-tree traversal (malformed data protection).
|
||||||
const MAX_BTREE_DEPTH: usize = 64;
|
pub(crate) const MAX_BTREE_DEPTH: usize = 64;
|
||||||
|
|
||||||
|
/// What a symbol table node takes with libhdf5's default group leaf K (4):
|
||||||
|
/// its 8-byte header and 2K entries of 40 bytes (8-byte offsets). Hinted
|
||||||
|
/// before one is read ([`Storage::hint`]); a node of another size is read
|
||||||
|
/// all the same.
|
||||||
|
const SNOD_HINT_LEN: usize = 8 + 8 * 40;
|
||||||
|
|
||||||
|
/// What a group B-tree node takes with libhdf5's default internal K (16):
|
||||||
|
/// its header (24 bytes with 8-byte offsets), 2K + 1 keys and 2K children
|
||||||
|
/// of 8 bytes. Hinted before one is read.
|
||||||
|
const NODE_HINT_LEN: usize = 24 + (2 * 16 + 1 + 2 * 16) * 8;
|
||||||
|
|
||||||
/// Collect all leaf-level child addresses (SNOD addresses) by traversing the B-tree.
|
/// Collect all leaf-level child addresses (SNOD addresses) by traversing the B-tree.
|
||||||
pub fn collect_symbol_table_nodes(
|
pub fn collect_symbol_table_nodes(
|
||||||
@@ -150,11 +178,21 @@ pub fn collect_symbol_table_nodes(
|
|||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
) -> Result<Vec<u64>, FormatError> {
|
) -> Result<Vec<u64>, FormatError> {
|
||||||
collect_symbol_table_nodes_inner(file_data, btree_address, offset_size, length_size, 0)
|
collect_symbol_table_nodes_in(file_data, btree_address, offset_size, length_size)
|
||||||
}
|
}
|
||||||
|
|
||||||
fn collect_symbol_table_nodes_inner(
|
/// [`collect_symbol_table_nodes`] over any [`Storage`]: two reads per node.
|
||||||
file_data: &[u8],
|
pub fn collect_symbol_table_nodes_in<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
btree_address: u64,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<Vec<u64>, FormatError> {
|
||||||
|
collect_symbol_table_nodes_inner(file, btree_address, offset_size, length_size, 0)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn collect_symbol_table_nodes_inner<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
btree_address: u64,
|
btree_address: u64,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
@@ -164,29 +202,47 @@ fn collect_symbol_table_nodes_inner(
|
|||||||
return Err(FormatError::NestingDepthExceeded);
|
return Err(FormatError::NestingDepthExceeded);
|
||||||
}
|
}
|
||||||
|
|
||||||
let node = BTreeV1Node::parse(file_data, btree_address as usize, offset_size, length_size)?;
|
let node = BTreeV1Node::parse_in(file, btree_address, offset_size, length_size)?;
|
||||||
|
|
||||||
if node.node_type != 0 {
|
if node.node_type != 0 {
|
||||||
return Err(FormatError::InvalidBTreeNodeType(node.node_type));
|
return Err(FormatError::InvalidBTreeNodeType(node.node_type));
|
||||||
}
|
}
|
||||||
|
|
||||||
if node.node_level == 0 {
|
if node.node_level == 0 {
|
||||||
// Leaf: children are SNOD addresses
|
// Leaf: children are SNOD addresses, read next (see
|
||||||
|
// `Storage::hint`).
|
||||||
|
for &snod in &node.children {
|
||||||
|
file.hint(snod, SNOD_HINT_LEN);
|
||||||
|
}
|
||||||
Ok(node.children)
|
Ok(node.children)
|
||||||
} else {
|
} else {
|
||||||
// Internal: recurse into children
|
// Internal: recurse into children. A child that fails does not
|
||||||
|
// stop the walk: the others are still descended into (reading, not
|
||||||
|
// using, what they hold), then the first error is returned. The
|
||||||
|
// result and the error are those of stopping at the first failure;
|
||||||
|
// a storage that records what it lacks (see `storage::touch`)
|
||||||
|
// learns every node the walk can reach in one attempt.
|
||||||
let mut result = Vec::new();
|
let mut result = Vec::new();
|
||||||
|
let mut failed = None;
|
||||||
for &child_addr in &node.children {
|
for &child_addr in &node.children {
|
||||||
let child_snods = collect_symbol_table_nodes_inner(
|
match collect_symbol_table_nodes_inner(
|
||||||
file_data,
|
file,
|
||||||
child_addr,
|
child_addr,
|
||||||
offset_size,
|
offset_size,
|
||||||
length_size,
|
length_size,
|
||||||
depth + 1,
|
depth + 1,
|
||||||
)?;
|
) {
|
||||||
result.extend(child_snods);
|
Ok(child_snods) if failed.is_none() => result.extend(child_snods),
|
||||||
|
Ok(_) => {}
|
||||||
|
Err(e) => {
|
||||||
|
failed.get_or_insert(e);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
match failed {
|
||||||
|
Some(e) => Err(e),
|
||||||
|
None => Ok(result),
|
||||||
}
|
}
|
||||||
Ok(result)
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -317,4 +373,48 @@ mod tests {
|
|||||||
assert_eq!(node.entries_used, 1);
|
assert_eq!(node.entries_used, 1);
|
||||||
assert_eq!(node.children, vec![0x50]);
|
assert_eq!(node.children, vec![0x50]);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Nodes and trees, cut at every length, parse identically through a
|
||||||
|
/// `read_at`-only storage.
|
||||||
|
#[test]
|
||||||
|
fn storage_parse_matches_slice_parse() {
|
||||||
|
use crate::storage::CountingStorage;
|
||||||
|
let nodes = [
|
||||||
|
build_btree_node(0, 0, &[0, 5, 10], &[0x100, 0x200], None, None, 8),
|
||||||
|
build_btree_node(0, 0, &[0, 5], &[0x100], Some(0x40), Some(0x80), 4),
|
||||||
|
build_btree_node(1, 2, &[0, 5], &[0x100], None, Some(0x80), 8),
|
||||||
|
];
|
||||||
|
for (n, node) in nodes.iter().enumerate() {
|
||||||
|
let os = if n == 1 { 4 } else { 8 };
|
||||||
|
for cut in 0..=node.len() {
|
||||||
|
let f = &node[..cut];
|
||||||
|
let storage = CountingStorage::new(f.to_vec());
|
||||||
|
let want = BTreeV1Node::parse(f, 0, os, 8);
|
||||||
|
let got = BTreeV1Node::parse_in(&storage, 0, os, 8);
|
||||||
|
assert_eq!(format!("{got:?}"), format!("{want:?}"));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let leaf1 = build_btree_node(0, 0, &[0, 5], &[0xA00], None, None, 8);
|
||||||
|
let leaf2 = build_btree_node(0, 0, &[5, 10], &[0xB00], None, None, 8);
|
||||||
|
let internal = build_btree_node(0, 1, &[0, 5, 10], &[0, 256], None, None, 8);
|
||||||
|
let mut file = vec![0u8; 512 + internal.len()];
|
||||||
|
file[..leaf1.len()].copy_from_slice(&leaf1);
|
||||||
|
file[256..256 + leaf2.len()].copy_from_slice(&leaf2);
|
||||||
|
file[512..].copy_from_slice(&internal);
|
||||||
|
for cut in [file.len(), 300, 260, 100, 10] {
|
||||||
|
let mut f = file.clone();
|
||||||
|
if cut < 512 {
|
||||||
|
// Truncate the leaves, keep the root.
|
||||||
|
f[cut..512].fill(0);
|
||||||
|
}
|
||||||
|
let storage = CountingStorage::new(f.clone());
|
||||||
|
assert_eq!(
|
||||||
|
collect_symbol_table_nodes_in(&storage, 512, 8, 8),
|
||||||
|
collect_symbol_table_nodes(&f, 512, 8, 8)
|
||||||
|
);
|
||||||
|
}
|
||||||
|
let storage = CountingStorage::new(file);
|
||||||
|
collect_symbol_table_nodes_in(&storage, 512, 8, 8).unwrap();
|
||||||
|
assert_eq!(storage.reads(), 6);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -2,11 +2,14 @@
|
|||||||
|
|
||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
use alloc::vec::Vec;
|
use alloc::vec::Vec;
|
||||||
|
use core::cmp::Ordering;
|
||||||
|
|
||||||
#[cfg(feature = "checksum")]
|
#[cfg(feature = "checksum")]
|
||||||
use byteorder::{ByteOrder, LittleEndian};
|
use byteorder::{ByteOrder, LittleEndian};
|
||||||
|
|
||||||
|
use crate::addr::to_usize;
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
|
use crate::storage::{Storage, Window, len_usize};
|
||||||
|
|
||||||
/// Parsed B-tree v2 header (signature "BTHD").
|
/// Parsed B-tree v2 header (signature "BTHD").
|
||||||
#[derive(Debug, Clone)]
|
#[derive(Debug, Clone)]
|
||||||
@@ -71,7 +74,7 @@ fn ensure_len(data: &[u8], pos: usize, needed: usize) -> Result<(), FormatError>
|
|||||||
|
|
||||||
/// Compute the number of bytes needed to represent a count, using variable-width encoding.
|
/// Compute the number of bytes needed to represent a count, using variable-width encoding.
|
||||||
/// B-tree v2 uses this for the number of records fields in internal nodes.
|
/// B-tree v2 uses this for the number of records fields in internal nodes.
|
||||||
fn bytes_for_max_records(max_nrec: u64) -> usize {
|
pub(crate) fn bytes_for_max_records(max_nrec: u64) -> usize {
|
||||||
if max_nrec == 0 {
|
if max_nrec == 0 {
|
||||||
return 1;
|
return 1;
|
||||||
}
|
}
|
||||||
@@ -97,38 +100,52 @@ impl BTreeV2Header {
|
|||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
) -> Result<BTreeV2Header, FormatError> {
|
) -> Result<BTreeV2Header, FormatError> {
|
||||||
ensure_len(file_data, offset, 4)?;
|
Self::parse_in(file_data, offset as u64, offset_size, length_size)
|
||||||
if &file_data[offset..offset + 4] != b"BTHD" {
|
}
|
||||||
|
|
||||||
|
/// [`Self::parse`] over any [`Storage`]: one bounded read of the
|
||||||
|
/// header.
|
||||||
|
pub fn parse_in<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
offset: u64,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<BTreeV2Header, FormatError> {
|
||||||
|
// Every field and the checksum; the window holds all of it or ends
|
||||||
|
// at the end of the file, so its bounds checks are the whole-file
|
||||||
|
// ones.
|
||||||
|
let full = 16 + usize::from(offset_size) + 2 + usize::from(length_size) + 4;
|
||||||
|
let w = Window::read(file, offset, full)?;
|
||||||
|
let d = &w.bytes;
|
||||||
|
w.ensure(0, 4)?;
|
||||||
|
if &d[..4] != b"BTHD" {
|
||||||
return Err(FormatError::InvalidBTreeV2Signature);
|
return Err(FormatError::InvalidBTreeV2Signature);
|
||||||
}
|
}
|
||||||
|
|
||||||
ensure_len(file_data, offset, 4 + 1 + 1 + 4 + 2 + 2 + 1 + 1)?;
|
w.ensure(0, 4 + 1 + 1 + 4 + 2 + 2 + 1 + 1)?;
|
||||||
let version = file_data[offset + 4];
|
let version = d[4];
|
||||||
if version != 0 {
|
if version != 0 {
|
||||||
return Err(FormatError::InvalidBTreeV2Version(version));
|
return Err(FormatError::InvalidBTreeV2Version(version));
|
||||||
}
|
}
|
||||||
|
|
||||||
let tree_type = file_data[offset + 5];
|
let tree_type = d[5];
|
||||||
let node_size = u32::from_le_bytes([
|
let node_size = u32::from_le_bytes([d[6], d[7], d[8], d[9]]);
|
||||||
file_data[offset + 6],
|
let record_size = u16::from_le_bytes([d[10], d[11]]);
|
||||||
file_data[offset + 7],
|
let depth = u16::from_le_bytes([d[12], d[13]]);
|
||||||
file_data[offset + 8],
|
let _split_percent = d[14];
|
||||||
file_data[offset + 9],
|
let _merge_percent = d[15];
|
||||||
]);
|
|
||||||
let record_size = u16::from_le_bytes([file_data[offset + 10], file_data[offset + 11]]);
|
|
||||||
let depth = u16::from_le_bytes([file_data[offset + 12], file_data[offset + 13]]);
|
|
||||||
let _split_percent = file_data[offset + 14];
|
|
||||||
let _merge_percent = file_data[offset + 15];
|
|
||||||
|
|
||||||
let mut pos = offset + 16;
|
let mut pos = 16;
|
||||||
let root_node_address = read_offset(file_data, pos, offset_size)?;
|
w.ensure(pos, usize::from(offset_size))?;
|
||||||
|
let root_node_address = read_offset(d, pos, offset_size)?;
|
||||||
pos += offset_size as usize;
|
pos += offset_size as usize;
|
||||||
|
|
||||||
ensure_len(file_data, pos, 2)?;
|
w.ensure(pos, 2)?;
|
||||||
let num_records_in_root = u16::from_le_bytes([file_data[pos], file_data[pos + 1]]);
|
let num_records_in_root = u16::from_le_bytes([d[pos], d[pos + 1]]);
|
||||||
pos += 2;
|
pos += 2;
|
||||||
|
|
||||||
let total_records = read_offset(file_data, pos, length_size)?;
|
w.ensure(pos, usize::from(length_size))?;
|
||||||
|
let total_records = read_offset(d, pos, length_size)?;
|
||||||
#[allow(unused_assignments)]
|
#[allow(unused_assignments)]
|
||||||
{
|
{
|
||||||
pos += length_size as usize;
|
pos += length_size as usize;
|
||||||
@@ -137,9 +154,9 @@ impl BTreeV2Header {
|
|||||||
// Validate header checksum
|
// Validate header checksum
|
||||||
#[cfg(feature = "checksum")]
|
#[cfg(feature = "checksum")]
|
||||||
{
|
{
|
||||||
ensure_len(file_data, pos, 4)?;
|
w.ensure(pos, 4)?;
|
||||||
let stored = LittleEndian::read_u32(&file_data[pos..pos + 4]);
|
let stored = LittleEndian::read_u32(&d[pos..pos + 4]);
|
||||||
let computed = crate::checksum::jenkins_lookup3(&file_data[offset..pos]);
|
let computed = crate::checksum::jenkins_lookup3(&d[..pos]);
|
||||||
if computed != stored {
|
if computed != stored {
|
||||||
return Err(FormatError::ChecksumMismatch {
|
return Err(FormatError::ChecksumMismatch {
|
||||||
expected: stored,
|
expected: stored,
|
||||||
@@ -163,7 +180,7 @@ impl BTreeV2Header {
|
|||||||
/// Compute maximum records per node for a given depth level.
|
/// Compute maximum records per node for a given depth level.
|
||||||
/// leaf: (node_size - overhead) / record_size
|
/// leaf: (node_size - overhead) / record_size
|
||||||
/// internal: depends on pointers
|
/// internal: depends on pointers
|
||||||
fn max_records_leaf(node_size: u32, record_size: u16) -> u64 {
|
pub(crate) fn max_records_leaf(node_size: u32, record_size: u16) -> u64 {
|
||||||
// Leaf overhead: signature(4) + version(1) + type(1) + checksum(4) = 10
|
// Leaf overhead: signature(4) + version(1) + type(1) + checksum(4) = 10
|
||||||
let overhead = 10u32;
|
let overhead = 10u32;
|
||||||
if node_size <= overhead || record_size == 0 {
|
if node_size <= overhead || record_size == 0 {
|
||||||
@@ -177,10 +194,17 @@ const MAX_DEPTH: u16 = 64;
|
|||||||
|
|
||||||
/// Take `n` records from the traversal's budget, or refuse the tree.
|
/// Take `n` records from the traversal's budget, or refuse the tree.
|
||||||
fn spend(budget: &mut usize, n: usize) -> Result<(), FormatError> {
|
fn spend(budget: &mut usize, n: usize) -> Result<(), FormatError> {
|
||||||
*budget = budget
|
match budget.checked_sub(n) {
|
||||||
.checked_sub(n)
|
Some(left) => {
|
||||||
.ok_or(FormatError::NestingDepthExceeded)?;
|
*budget = left;
|
||||||
Ok(())
|
Ok(())
|
||||||
|
}
|
||||||
|
None => {
|
||||||
|
// Spent: a walk that goes on after a failure stops here.
|
||||||
|
*budget = 0;
|
||||||
|
Err(FormatError::NestingDepthExceeded)
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Collect all records from a B-tree v2 by traversing from the root.
|
/// Collect all records from a B-tree v2 by traversing from the root.
|
||||||
@@ -189,6 +213,17 @@ pub fn collect_btree_v2_records(
|
|||||||
header: &BTreeV2Header,
|
header: &BTreeV2Header,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
|
) -> Result<Vec<BTreeV2Record>, FormatError> {
|
||||||
|
collect_btree_v2_records_in(file_data, header, offset_size, length_size)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`collect_btree_v2_records`] over any [`Storage`]: one bounded read per
|
||||||
|
/// node.
|
||||||
|
pub fn collect_btree_v2_records_in<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
header: &BTreeV2Header,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
) -> Result<Vec<BTreeV2Record>, FormatError> {
|
) -> Result<Vec<BTreeV2Record>, FormatError> {
|
||||||
if header.total_records == 0 || header.num_records_in_root == 0 {
|
if header.total_records == 0 || header.num_records_in_root == 0 {
|
||||||
return Ok(Vec::new());
|
return Ok(Vec::new());
|
||||||
@@ -208,24 +243,25 @@ pub fn collect_btree_v2_records(
|
|||||||
// millions of records from a few kilobytes. Counting against what the
|
// millions of records from a few kilobytes. Counting against what the
|
||||||
// file could physically contain bounds that without trusting the
|
// file could physically contain bounds that without trusting the
|
||||||
// header's own `total_records`.
|
// header's own `total_records`.
|
||||||
let mut budget = file_data.len() / usize::from(header.record_size.max(1));
|
let mut budget = len_usize(file) / usize::from(header.record_size.max(1));
|
||||||
|
|
||||||
let max_leaf_nrec = max_records_leaf(header.node_size, header.record_size);
|
let max_leaf_nrec = max_records_leaf(header.node_size, header.record_size);
|
||||||
|
|
||||||
if header.depth == 0 {
|
if header.depth == 0 {
|
||||||
// Root is a leaf
|
// Root is a leaf
|
||||||
parse_leaf_records(
|
parse_leaf_records(
|
||||||
file_data,
|
file,
|
||||||
header.root_node_address as usize,
|
to_usize(header.root_node_address)?,
|
||||||
header.num_records_in_root,
|
header.num_records_in_root,
|
||||||
header.record_size,
|
header.record_size,
|
||||||
|
header.node_size,
|
||||||
)
|
)
|
||||||
} else {
|
} else {
|
||||||
// Root is internal; traverse recursively
|
// Root is internal; traverse recursively
|
||||||
let mut records = Vec::new();
|
let mut records = Vec::new();
|
||||||
collect_internal_records(
|
collect_internal_records(
|
||||||
file_data,
|
file,
|
||||||
header.root_node_address as usize,
|
to_usize(header.root_node_address)?,
|
||||||
header.num_records_in_root,
|
header.num_records_in_root,
|
||||||
header.depth,
|
header.depth,
|
||||||
header.record_size,
|
header.record_size,
|
||||||
@@ -240,36 +276,72 @@ pub fn collect_btree_v2_records(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// A node's bytes: `want` bytes at `offset` (fewer only at the end of the
|
||||||
|
/// file), after checking its 4-byte signature. A node is read in one piece
|
||||||
|
/// when it fits in `node_size` (every valid node does); a larger claimed
|
||||||
|
/// extent — record counts from a damaged parent — is first checked against
|
||||||
|
/// the end of the file, so it costs a read only of bytes the file has.
|
||||||
|
/// Bounds errors are the whole-file ones: the signature check needs the
|
||||||
|
/// first 6 bytes, then `checks` — `(position, length)` pairs relative to
|
||||||
|
/// the node, in the order the parser checks them — must lie in the file.
|
||||||
|
fn read_node<'a, S: Storage + ?Sized>(
|
||||||
|
file: &'a S,
|
||||||
|
offset: usize,
|
||||||
|
want: usize,
|
||||||
|
node_size: u32,
|
||||||
|
signature: &[u8; 4],
|
||||||
|
checks: &[(usize, usize)],
|
||||||
|
) -> Result<Window<'a>, FormatError> {
|
||||||
|
let one_read = usize::try_from(node_size).unwrap_or(usize::MAX).max(6);
|
||||||
|
let w = Window::read(file, offset as u64, want.min(one_read))?;
|
||||||
|
w.ensure(0, 6)?;
|
||||||
|
if &w.bytes[..4] != signature {
|
||||||
|
return Err(FormatError::InvalidBTreeV2Signature);
|
||||||
|
}
|
||||||
|
if want <= one_read {
|
||||||
|
return Ok(w);
|
||||||
|
}
|
||||||
|
for &(rel, len) in checks {
|
||||||
|
Window::check_extent(file, offset as u64, rel, len)?;
|
||||||
|
}
|
||||||
|
Window::read(file, offset as u64, want)
|
||||||
|
}
|
||||||
|
|
||||||
/// Parse records from a leaf node (signature "BTLF").
|
/// Parse records from a leaf node (signature "BTLF").
|
||||||
fn parse_leaf_records(
|
fn parse_leaf_records<S: Storage + ?Sized>(
|
||||||
file_data: &[u8],
|
file: &S,
|
||||||
offset: usize,
|
offset: usize,
|
||||||
num_records: u16,
|
num_records: u16,
|
||||||
record_size: u16,
|
record_size: u16,
|
||||||
|
node_size: u32,
|
||||||
) -> Result<Vec<BTreeV2Record>, FormatError> {
|
) -> Result<Vec<BTreeV2Record>, FormatError> {
|
||||||
// signature(4) + version(1) + type(1) = 6 bytes header
|
// signature(4) + version(1) + type(1) = 6 bytes header
|
||||||
ensure_len(file_data, offset, 6)?;
|
let pos = 6;
|
||||||
if &file_data[offset..offset + 4] != b"BTLF" {
|
|
||||||
return Err(FormatError::InvalidBTreeV2Signature);
|
|
||||||
}
|
|
||||||
|
|
||||||
let pos = offset + 6;
|
|
||||||
let rs = record_size as usize;
|
let rs = record_size as usize;
|
||||||
let total = (num_records as usize)
|
let total = (num_records as usize)
|
||||||
.checked_mul(rs)
|
.checked_mul(rs)
|
||||||
.ok_or(FormatError::UnexpectedEof {
|
.ok_or(FormatError::UnexpectedEof {
|
||||||
expected: usize::MAX,
|
expected: usize::MAX,
|
||||||
available: file_data.len(),
|
available: len_usize(file),
|
||||||
})?;
|
})?;
|
||||||
ensure_len(file_data, pos, total)?;
|
let w = read_node(
|
||||||
|
file,
|
||||||
|
offset,
|
||||||
|
pos + total + 4,
|
||||||
|
node_size,
|
||||||
|
b"BTLF",
|
||||||
|
&[(pos, total)],
|
||||||
|
)?;
|
||||||
|
let d = &w.bytes;
|
||||||
|
w.ensure(pos, total)?;
|
||||||
|
|
||||||
// Validate checksum: 4 bytes after records + padding
|
// Validate checksum: 4 bytes after records + padding
|
||||||
#[cfg(feature = "checksum")]
|
#[cfg(feature = "checksum")]
|
||||||
{
|
{
|
||||||
let checksum_pos = pos + total;
|
let checksum_pos = pos + total;
|
||||||
if file_data.len() >= checksum_pos + 4 {
|
if d.len() >= checksum_pos + 4 {
|
||||||
let stored = LittleEndian::read_u32(&file_data[checksum_pos..checksum_pos + 4]);
|
let stored = LittleEndian::read_u32(&d[checksum_pos..checksum_pos + 4]);
|
||||||
let computed = crate::checksum::jenkins_lookup3(&file_data[offset..checksum_pos]);
|
let computed = crate::checksum::jenkins_lookup3(&d[..checksum_pos]);
|
||||||
if computed != stored {
|
if computed != stored {
|
||||||
return Err(FormatError::ChecksumMismatch {
|
return Err(FormatError::ChecksumMismatch {
|
||||||
expected: stored,
|
expected: stored,
|
||||||
@@ -283,16 +355,135 @@ fn parse_leaf_records(
|
|||||||
for i in 0..num_records as usize {
|
for i in 0..num_records as usize {
|
||||||
let start = pos + i * rs;
|
let start = pos + i * rs;
|
||||||
records.push(BTreeV2Record {
|
records.push(BTreeV2Record {
|
||||||
data: file_data[start..start + rs].to_vec(),
|
data: d[start..start + rs].to_vec(),
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
Ok(records)
|
Ok(records)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// An internal node read from the file: its bytes (from the signature on),
|
||||||
|
/// where its records start, and its children as `(address, record count)`.
|
||||||
|
struct InternalNode<'a> {
|
||||||
|
node: Window<'a>,
|
||||||
|
records_start: usize,
|
||||||
|
children: Vec<(u64, u16)>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl InternalNode<'_> {
|
||||||
|
/// Record `i`, `rs` bytes long.
|
||||||
|
fn record(&self, i: usize, rs: usize) -> Result<&[u8], FormatError> {
|
||||||
|
let overflow = || FormatError::UnexpectedEof {
|
||||||
|
expected: usize::MAX,
|
||||||
|
available: usize::MAX,
|
||||||
|
};
|
||||||
|
let rec_start = i
|
||||||
|
.checked_mul(rs)
|
||||||
|
.and_then(|o| self.records_start.checked_add(o))
|
||||||
|
.ok_or_else(overflow)?;
|
||||||
|
self.node.ensure(rec_start, rs)?;
|
||||||
|
Ok(&self.node.bytes[rec_start..rec_start + rs])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// An internal node's layout: where its records start, and its children as
|
||||||
|
/// `(address, record count)`.
|
||||||
|
#[allow(clippy::too_many_arguments)]
|
||||||
|
fn read_internal_node<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
offset: usize,
|
||||||
|
num_records: u16,
|
||||||
|
depth: u16,
|
||||||
|
record_size: u16,
|
||||||
|
node_size: u32,
|
||||||
|
offset_size: u8,
|
||||||
|
max_leaf_nrec: u64,
|
||||||
|
) -> Result<InternalNode<'_>, FormatError> {
|
||||||
|
let nr = num_records as usize;
|
||||||
|
let rs = record_size as usize;
|
||||||
|
|
||||||
|
// Records first
|
||||||
|
let records_total = nr.checked_mul(rs).ok_or(FormatError::UnexpectedEof {
|
||||||
|
expected: usize::MAX,
|
||||||
|
available: len_usize(file),
|
||||||
|
})?;
|
||||||
|
|
||||||
|
// Child pointer layout, as libhdf5 computes it (H5B2__hdr_init): the
|
||||||
|
// child's record count is always encoded in the width needed for a
|
||||||
|
// *leaf's* maximum, and — below the first internal level — the child
|
||||||
|
// subtree's total record count in the width needed for the most records
|
||||||
|
// a subtree of that depth can hold.
|
||||||
|
let child_depth = depth - 1;
|
||||||
|
let nrec_width = bytes_for_max_records(max_leaf_nrec);
|
||||||
|
let total_nrec_width = if depth > 1 {
|
||||||
|
bytes_for_max_records(cum_max_records(
|
||||||
|
node_size,
|
||||||
|
record_size,
|
||||||
|
offset_size,
|
||||||
|
max_leaf_nrec,
|
||||||
|
child_depth,
|
||||||
|
))
|
||||||
|
} else {
|
||||||
|
0
|
||||||
|
};
|
||||||
|
|
||||||
|
let num_children = nr + 1;
|
||||||
|
let child_ptr_size = offset_size as usize + nrec_width + total_nrec_width;
|
||||||
|
let pointers = num_children * child_ptr_size;
|
||||||
|
|
||||||
|
// signature(4) + version(1) + type(1) = 6, records, pointers, checksum.
|
||||||
|
let w = read_node(
|
||||||
|
file,
|
||||||
|
offset,
|
||||||
|
6 + records_total + pointers + 4,
|
||||||
|
node_size,
|
||||||
|
b"BTIN",
|
||||||
|
&[(6, records_total), (6 + records_total, pointers)],
|
||||||
|
)?;
|
||||||
|
let d = &w.bytes;
|
||||||
|
let mut pos = 6;
|
||||||
|
w.ensure(pos, records_total)?;
|
||||||
|
let records_start = pos;
|
||||||
|
pos += records_total;
|
||||||
|
|
||||||
|
w.ensure(pos, pointers)?;
|
||||||
|
|
||||||
|
let mut children = Vec::with_capacity(num_children);
|
||||||
|
for _ in 0..num_children {
|
||||||
|
let addr = read_offset(d, pos, offset_size)?;
|
||||||
|
pos += offset_size as usize;
|
||||||
|
let child_nrec = read_var_uint(d, pos, nrec_width)? as u16;
|
||||||
|
pos += nrec_width;
|
||||||
|
pos += total_nrec_width; // skip total records in subtree
|
||||||
|
children.push((addr, child_nrec));
|
||||||
|
}
|
||||||
|
|
||||||
|
// The checksum follows the child pointers and covers the node up to it.
|
||||||
|
// Lookups prune children by the keys in this node, so an unverified
|
||||||
|
// internal node could hide a record without any error: libhdf5 refuses
|
||||||
|
// a mismatch here, and so does this.
|
||||||
|
#[cfg(feature = "checksum")]
|
||||||
|
{
|
||||||
|
w.ensure(pos, 4)?;
|
||||||
|
let stored = LittleEndian::read_u32(&d[pos..pos + 4]);
|
||||||
|
let computed = crate::checksum::jenkins_lookup3(&d[..pos]);
|
||||||
|
if computed != stored {
|
||||||
|
return Err(FormatError::ChecksumMismatch {
|
||||||
|
expected: stored,
|
||||||
|
computed,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(InternalNode {
|
||||||
|
node: w,
|
||||||
|
records_start,
|
||||||
|
children,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
/// Recursively collect records from an internal node.
|
/// Recursively collect records from an internal node.
|
||||||
#[allow(clippy::too_many_arguments, clippy::only_used_in_recursion)]
|
#[allow(clippy::too_many_arguments, clippy::only_used_in_recursion)]
|
||||||
fn collect_internal_records(
|
fn collect_internal_records<S: Storage + ?Sized>(
|
||||||
file_data: &[u8],
|
file: &S,
|
||||||
offset: usize,
|
offset: usize,
|
||||||
num_records: u16,
|
num_records: u16,
|
||||||
depth: u16,
|
depth: u16,
|
||||||
@@ -304,145 +495,295 @@ fn collect_internal_records(
|
|||||||
budget: &mut usize,
|
budget: &mut usize,
|
||||||
out: &mut Vec<BTreeV2Record>,
|
out: &mut Vec<BTreeV2Record>,
|
||||||
) -> Result<(), FormatError> {
|
) -> Result<(), FormatError> {
|
||||||
// signature(4) + version(1) + type(1) = 6
|
|
||||||
ensure_len(file_data, offset, 6)?;
|
|
||||||
if &file_data[offset..offset + 4] != b"BTIN" {
|
|
||||||
return Err(FormatError::InvalidBTreeV2Signature);
|
|
||||||
}
|
|
||||||
|
|
||||||
let nr = num_records as usize;
|
let nr = num_records as usize;
|
||||||
let rs = record_size as usize;
|
let rs = record_size as usize;
|
||||||
let mut pos = offset + 6;
|
let node = read_internal_node(
|
||||||
|
file,
|
||||||
// Read all records first
|
offset,
|
||||||
let records_total = nr.checked_mul(rs).ok_or(FormatError::UnexpectedEof {
|
num_records,
|
||||||
expected: usize::MAX,
|
depth,
|
||||||
available: file_data.len(),
|
record_size,
|
||||||
})?;
|
node_size,
|
||||||
ensure_len(file_data, pos, records_total)?;
|
offset_size,
|
||||||
let records_start = pos;
|
max_leaf_nrec,
|
||||||
pos += records_total;
|
)?;
|
||||||
|
|
||||||
// Compute sizes for child pointers
|
|
||||||
// max_records at child depth - for variable-width nrec encoding
|
|
||||||
let child_depth = depth - 1;
|
let child_depth = depth - 1;
|
||||||
let max_nrec_child = if child_depth == 0 {
|
|
||||||
max_leaf_nrec
|
|
||||||
} else {
|
|
||||||
// For internal nodes at child_depth, the true max_nrec depends on the
|
|
||||||
// node size, record size, and the recursive width of child pointer
|
|
||||||
// entries (which themselves depend on max_nrec at deeper levels).
|
|
||||||
// Computing the exact value requires iterating from the leaf level
|
|
||||||
// upward, as described in the HDF5 spec (III.A.2 "Computing the Size
|
|
||||||
// of B-tree Nodes").
|
|
||||||
//
|
|
||||||
// We use `max_leaf_nrec * 2` as a conservative upper bound. This
|
|
||||||
// over-estimates the nrec encoding width, which means we may read
|
|
||||||
// slightly more bytes per child pointer than strictly necessary, but
|
|
||||||
// never fewer. The over-read bytes are harmless because we only
|
|
||||||
// decode `num_records` entries (the actual count from the node header).
|
|
||||||
//
|
|
||||||
// Known limitation: for very deep trees (depth > 3) with small record
|
|
||||||
// sizes, the true max could exceed this estimate, causing us to
|
|
||||||
// under-allocate the nrec encoding width and misparse child pointers.
|
|
||||||
// In practice, HDF5 B-tree v2 depths rarely exceed 2-3.
|
|
||||||
max_leaf_nrec * 2
|
|
||||||
};
|
|
||||||
let nrec_width = bytes_for_max_records(max_nrec_child);
|
|
||||||
|
|
||||||
// Total records in subtree width (only if depth > 1)
|
|
||||||
let total_nrec_width = if depth > 1 {
|
|
||||||
// Width to hold total records in a subtree
|
|
||||||
// We compute max possible total records at this subtree depth
|
|
||||||
let max_total = header_max_total_records(max_leaf_nrec, depth - 1);
|
|
||||||
bytes_for_max_records(max_total)
|
|
||||||
} else {
|
|
||||||
0
|
|
||||||
};
|
|
||||||
|
|
||||||
let num_children = nr + 1;
|
|
||||||
let child_ptr_size = offset_size as usize + nrec_width + total_nrec_width;
|
|
||||||
ensure_len(file_data, pos, num_children * child_ptr_size)?;
|
|
||||||
|
|
||||||
// Read child pointers
|
|
||||||
let mut children = Vec::with_capacity(num_children);
|
|
||||||
for _ in 0..num_children {
|
|
||||||
let addr = read_offset(file_data, pos, offset_size)?;
|
|
||||||
pos += offset_size as usize;
|
|
||||||
let child_nrec = read_var_uint(file_data, pos, nrec_width)? as u16;
|
|
||||||
pos += nrec_width;
|
|
||||||
pos += total_nrec_width; // skip total records in subtree
|
|
||||||
children.push((addr, child_nrec));
|
|
||||||
}
|
|
||||||
|
|
||||||
// Interleave: child[0], record[0], child[1], record[1], ..., child[nr]
|
// Interleave: child[0], record[0], child[1], record[1], ..., child[nr]
|
||||||
// We collect child[0] records, then record[0], then child[1], etc.
|
// We collect child[0] records, then record[0], then child[1], etc.
|
||||||
for (i, &(child_addr, child_nrec)) in children.iter().enumerate() {
|
// A child that fails does not stop the walk: the others are still
|
||||||
if child_depth == 0 {
|
// descended into (their records are dropped with the result), then the
|
||||||
// Before parsing, so a refused tree is not also a large allocation.
|
// first error is returned, as when stopping there. A storage that
|
||||||
spend(budget, usize::from(child_nrec))?;
|
// records what it lacks (see `storage::touch`) so learns every node the
|
||||||
let leaf_recs =
|
// walk can reach in one attempt. The record budget is spent as before,
|
||||||
parse_leaf_records(file_data, child_addr as usize, child_nrec, record_size)?;
|
// so the walk is no longer than a successful one.
|
||||||
out.extend(leaf_recs);
|
let mut failed = None;
|
||||||
} else {
|
for (i, &(child_addr, child_nrec)) in node.children.iter().enumerate() {
|
||||||
collect_internal_records(
|
if failed.is_some() && *budget == 0 {
|
||||||
file_data,
|
// The record budget is spent: the tree is refused, and a walk
|
||||||
child_addr as usize,
|
// over what is left could be as long as the one it bounds.
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
if let Err(e) = (|| -> Result<(), FormatError> {
|
||||||
|
if child_depth == 0 {
|
||||||
|
// Before parsing, so a refused tree is not also a large allocation.
|
||||||
|
spend(budget, usize::from(child_nrec))?;
|
||||||
|
let leaf_recs = parse_leaf_records(
|
||||||
|
file,
|
||||||
|
to_usize(child_addr)?,
|
||||||
|
child_nrec,
|
||||||
|
record_size,
|
||||||
|
node_size,
|
||||||
|
)?;
|
||||||
|
out.extend(leaf_recs);
|
||||||
|
} else {
|
||||||
|
collect_internal_records(
|
||||||
|
file,
|
||||||
|
to_usize(child_addr)?,
|
||||||
|
child_nrec,
|
||||||
|
child_depth,
|
||||||
|
record_size,
|
||||||
|
node_size,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
max_leaf_nrec,
|
||||||
|
budget,
|
||||||
|
out,
|
||||||
|
)?;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Add record[i] (except after the last child)
|
||||||
|
if i < nr {
|
||||||
|
let data = node.record(i, rs)?;
|
||||||
|
spend(budget, 1)?;
|
||||||
|
out.push(BTreeV2Record {
|
||||||
|
data: data.to_vec(),
|
||||||
|
});
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
})() {
|
||||||
|
failed.get_or_insert(e);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
match failed {
|
||||||
|
Some(e) => Err(e),
|
||||||
|
None => Ok(()),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The records of a B-tree v2 that fall in one key range, found by
|
||||||
|
/// descending the tree instead of reading all of it.
|
||||||
|
///
|
||||||
|
/// `cmp` places a record relative to the range: `Less` if the record sorts
|
||||||
|
/// before it, `Greater` if after, `Equal` if the record is in it. The tree
|
||||||
|
/// must be ordered consistently with `cmp`, as libhdf5 orders it (a link or
|
||||||
|
/// attribute name index by name hash, so all records with one hash form a
|
||||||
|
/// range whatever order their names are in). Only the nodes whose key
|
||||||
|
/// interval overlaps the range are read: O(depth) nodes plus those holding
|
||||||
|
/// the matches. Matches come in tree order.
|
||||||
|
pub fn find_btree_v2_records(
|
||||||
|
file_data: &[u8],
|
||||||
|
header: &BTreeV2Header,
|
||||||
|
offset_size: u8,
|
||||||
|
cmp: &mut dyn FnMut(&[u8]) -> Ordering,
|
||||||
|
) -> Result<Vec<BTreeV2Record>, FormatError> {
|
||||||
|
find_btree_v2_records_in(file_data, header, offset_size, cmp)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`find_btree_v2_records`] over any [`Storage`]: one bounded read per
|
||||||
|
/// node visited.
|
||||||
|
pub fn find_btree_v2_records_in<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
header: &BTreeV2Header,
|
||||||
|
offset_size: u8,
|
||||||
|
cmp: &mut dyn FnMut(&[u8]) -> Ordering,
|
||||||
|
) -> Result<Vec<BTreeV2Record>, FormatError> {
|
||||||
|
if header.total_records == 0 || header.num_records_in_root == 0 {
|
||||||
|
return Ok(Vec::new());
|
||||||
|
}
|
||||||
|
if header.depth > MAX_DEPTH {
|
||||||
|
return Err(FormatError::NestingDepthExceeded);
|
||||||
|
}
|
||||||
|
// As in `collect_btree_v2_records`: a valid tree cannot hold more
|
||||||
|
// records than the file has room for, however its children are shared.
|
||||||
|
let mut budget = len_usize(file) / usize::from(header.record_size.max(1));
|
||||||
|
let max_leaf_nrec = max_records_leaf(header.node_size, header.record_size);
|
||||||
|
let mut out = Vec::new();
|
||||||
|
find_in_node(
|
||||||
|
file,
|
||||||
|
header,
|
||||||
|
to_usize(header.root_node_address)?,
|
||||||
|
header.num_records_in_root,
|
||||||
|
header.depth,
|
||||||
|
offset_size,
|
||||||
|
max_leaf_nrec,
|
||||||
|
cmp,
|
||||||
|
&mut budget,
|
||||||
|
&mut out,
|
||||||
|
)?;
|
||||||
|
Ok(out)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[allow(clippy::too_many_arguments)]
|
||||||
|
fn find_in_node<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
header: &BTreeV2Header,
|
||||||
|
offset: usize,
|
||||||
|
num_records: u16,
|
||||||
|
depth: u16,
|
||||||
|
offset_size: u8,
|
||||||
|
max_leaf_nrec: u64,
|
||||||
|
cmp: &mut dyn FnMut(&[u8]) -> Ordering,
|
||||||
|
budget: &mut usize,
|
||||||
|
out: &mut Vec<BTreeV2Record>,
|
||||||
|
) -> Result<(), FormatError> {
|
||||||
|
spend(budget, usize::from(num_records))?;
|
||||||
|
if depth == 0 {
|
||||||
|
let records = parse_leaf_records(
|
||||||
|
file,
|
||||||
|
offset,
|
||||||
|
num_records,
|
||||||
|
header.record_size,
|
||||||
|
header.node_size,
|
||||||
|
)?;
|
||||||
|
out.extend(
|
||||||
|
records
|
||||||
|
.into_iter()
|
||||||
|
.filter(|r| cmp(&r.data) == Ordering::Equal),
|
||||||
|
);
|
||||||
|
return Ok(());
|
||||||
|
}
|
||||||
|
let rs = usize::from(header.record_size);
|
||||||
|
let node = read_internal_node(
|
||||||
|
file,
|
||||||
|
offset,
|
||||||
|
num_records,
|
||||||
|
depth,
|
||||||
|
header.record_size,
|
||||||
|
header.node_size,
|
||||||
|
offset_size,
|
||||||
|
max_leaf_nrec,
|
||||||
|
)?;
|
||||||
|
let nr = usize::from(num_records);
|
||||||
|
let mut order = Vec::with_capacity(nr);
|
||||||
|
for i in 0..nr {
|
||||||
|
order.push(cmp(node.record(i, rs)?));
|
||||||
|
}
|
||||||
|
// Child `i` holds the keys between record `i - 1` and record `i`: it can
|
||||||
|
// hold a match unless the record before it is already past the range or
|
||||||
|
// the record after it is still before it.
|
||||||
|
for (i, &(child_addr, child_nrec)) in node.children.iter().enumerate() {
|
||||||
|
let after_left = i == 0 || order[i - 1] != Ordering::Greater;
|
||||||
|
let before_right = i == nr || order[i] != Ordering::Less;
|
||||||
|
if after_left && before_right {
|
||||||
|
find_in_node(
|
||||||
|
file,
|
||||||
|
header,
|
||||||
|
to_usize(child_addr)?,
|
||||||
child_nrec,
|
child_nrec,
|
||||||
child_depth,
|
depth - 1,
|
||||||
record_size,
|
|
||||||
node_size,
|
|
||||||
offset_size,
|
offset_size,
|
||||||
length_size,
|
|
||||||
max_leaf_nrec,
|
max_leaf_nrec,
|
||||||
|
cmp,
|
||||||
budget,
|
budget,
|
||||||
out,
|
out,
|
||||||
)?;
|
)?;
|
||||||
}
|
}
|
||||||
|
if i < nr && order[i] == Ordering::Equal {
|
||||||
// Add record[i] (except after the last child)
|
|
||||||
if i < nr {
|
|
||||||
let rec_offset = i.checked_mul(rs).ok_or(FormatError::UnexpectedEof {
|
|
||||||
expected: usize::MAX,
|
|
||||||
available: file_data.len(),
|
|
||||||
})?;
|
|
||||||
let rec_start =
|
|
||||||
records_start
|
|
||||||
.checked_add(rec_offset)
|
|
||||||
.ok_or(FormatError::UnexpectedEof {
|
|
||||||
expected: usize::MAX,
|
|
||||||
available: file_data.len(),
|
|
||||||
})?;
|
|
||||||
let rec_end = rec_start
|
|
||||||
.checked_add(rs)
|
|
||||||
.ok_or(FormatError::UnexpectedEof {
|
|
||||||
expected: usize::MAX,
|
|
||||||
available: file_data.len(),
|
|
||||||
})?;
|
|
||||||
if rec_end > file_data.len() {
|
|
||||||
return Err(FormatError::UnexpectedEof {
|
|
||||||
expected: rec_end,
|
|
||||||
available: file_data.len(),
|
|
||||||
});
|
|
||||||
}
|
|
||||||
spend(budget, 1)?;
|
|
||||||
out.push(BTreeV2Record {
|
out.push(BTreeV2Record {
|
||||||
data: file_data[rec_start..rec_end].to_vec(),
|
data: node.record(i, rs)?.to_vec(),
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Estimate maximum total records at a given depth (for variable-width encoding).
|
/// Most records a subtree whose root is at `depth` can hold (libhdf5's
|
||||||
fn header_max_total_records(max_leaf_nrec: u64, depth: u16) -> u64 {
|
/// `cum_max_nrec`). See [`node_info`].
|
||||||
// Conservative: branching factor * max_leaf at each level
|
fn cum_max_records(
|
||||||
let mut total = max_leaf_nrec;
|
node_size: u32,
|
||||||
for _ in 0..depth {
|
record_size: u16,
|
||||||
total = total.saturating_mul(max_leaf_nrec.max(2));
|
offset_size: u8,
|
||||||
|
max_leaf_nrec: u64,
|
||||||
|
depth: u16,
|
||||||
|
) -> u64 {
|
||||||
|
node_info_from_leaf(node_size, record_size, offset_size, max_leaf_nrec, depth)
|
||||||
|
.last()
|
||||||
|
.map_or(max_leaf_nrec, |n| n.cum_max_nrec)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Capacity of a B-tree v2 node at one depth, as libhdf5 computes it
|
||||||
|
/// (`H5B2__hdr_init`'s `node_info`).
|
||||||
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||||
|
pub(crate) struct NodeInfo {
|
||||||
|
/// Most records one node at this depth holds.
|
||||||
|
pub(crate) max_nrec: u64,
|
||||||
|
/// Most records a subtree rooted at this depth holds.
|
||||||
|
pub(crate) cum_max_nrec: u64,
|
||||||
|
/// Bytes a subtree's total record count takes in a pointer to a node
|
||||||
|
/// at this depth (0 for a leaf, whose count is its own).
|
||||||
|
pub(crate) cum_max_nrec_size: usize,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Node capacities for depths `0..=depth` (entry `d` for depth `d`): a leaf
|
||||||
|
/// holds `max_nrec(0)` records; an internal node at depth `d` holds
|
||||||
|
/// `max_nrec(d)` records and `max_nrec(d) + 1` subtrees of depth `d - 1`,
|
||||||
|
/// where `max_nrec(d)` is what fits in a node once each record is paired
|
||||||
|
/// with a child pointer of the width depth `d` needs (address, the child's
|
||||||
|
/// record count in the width a *leaf's* maximum needs, and below the first
|
||||||
|
/// internal level the child subtree's total in the width its maximum
|
||||||
|
/// needs), with one pointer more than records.
|
||||||
|
pub(crate) fn node_info(
|
||||||
|
node_size: u32,
|
||||||
|
record_size: u16,
|
||||||
|
offset_size: u8,
|
||||||
|
depth: u16,
|
||||||
|
) -> Vec<NodeInfo> {
|
||||||
|
let max_leaf = max_records_leaf(node_size, record_size);
|
||||||
|
node_info_from_leaf(node_size, record_size, offset_size, max_leaf, depth)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn node_info_from_leaf(
|
||||||
|
node_size: u32,
|
||||||
|
record_size: u16,
|
||||||
|
offset_size: u8,
|
||||||
|
max_leaf_nrec: u64,
|
||||||
|
depth: u16,
|
||||||
|
) -> Vec<NodeInfo> {
|
||||||
|
// Internal node overhead: signature(4) + version(1) + type(1) + checksum(4).
|
||||||
|
const PREFIX: u64 = 10;
|
||||||
|
let nrec_width = bytes_for_max_records(max_leaf_nrec) as u64;
|
||||||
|
let mut info = Vec::with_capacity(usize::from(depth) + 1);
|
||||||
|
info.push(NodeInfo {
|
||||||
|
max_nrec: max_leaf_nrec,
|
||||||
|
cum_max_nrec: max_leaf_nrec,
|
||||||
|
cum_max_nrec_size: 0,
|
||||||
|
});
|
||||||
|
for d in 1..=depth {
|
||||||
|
let below = info[usize::from(d) - 1];
|
||||||
|
let ptr = u64::from(offset_size)
|
||||||
|
+ nrec_width
|
||||||
|
+ if d > 1 {
|
||||||
|
below.cum_max_nrec_size as u64
|
||||||
|
} else {
|
||||||
|
0
|
||||||
|
};
|
||||||
|
let max_nrec = u64::from(node_size)
|
||||||
|
.saturating_sub(PREFIX)
|
||||||
|
.saturating_sub(ptr)
|
||||||
|
/ (u64::from(record_size) + ptr).max(1);
|
||||||
|
let cum = max_nrec
|
||||||
|
.saturating_add(1)
|
||||||
|
.saturating_mul(below.cum_max_nrec)
|
||||||
|
.saturating_add(max_nrec);
|
||||||
|
info.push(NodeInfo {
|
||||||
|
max_nrec,
|
||||||
|
cum_max_nrec: cum,
|
||||||
|
cum_max_nrec_size: bytes_for_max_records(cum),
|
||||||
|
});
|
||||||
}
|
}
|
||||||
total
|
info
|
||||||
}
|
}
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
@@ -512,9 +853,15 @@ mod tests {
|
|||||||
child_nrec: u64,
|
child_nrec: u64,
|
||||||
) -> Vec<u8> {
|
) -> Vec<u8> {
|
||||||
let max_leaf = max_records_leaf(node_size, record_size);
|
let max_leaf = max_records_leaf(node_size, record_size);
|
||||||
let nrec_width = bytes_for_max_records(if depth == 1 { max_leaf } else { max_leaf * 2 });
|
let nrec_width = bytes_for_max_records(max_leaf);
|
||||||
let total_width = if depth > 1 {
|
let total_width = if depth > 1 {
|
||||||
bytes_for_max_records(header_max_total_records(max_leaf, depth - 1))
|
bytes_for_max_records(cum_max_records(
|
||||||
|
node_size,
|
||||||
|
record_size,
|
||||||
|
8,
|
||||||
|
max_leaf,
|
||||||
|
depth - 1,
|
||||||
|
))
|
||||||
} else {
|
} else {
|
||||||
0
|
0
|
||||||
};
|
};
|
||||||
@@ -526,6 +873,8 @@ mod tests {
|
|||||||
buf.extend_from_slice(&child_nrec.to_le_bytes()[..nrec_width]);
|
buf.extend_from_slice(&child_nrec.to_le_bytes()[..nrec_width]);
|
||||||
buf.resize(buf.len() + total_width, 0);
|
buf.resize(buf.len() + total_width, 0);
|
||||||
}
|
}
|
||||||
|
let sum = crate::checksum::jenkins_lookup3(&buf);
|
||||||
|
buf.extend_from_slice(&sum.to_le_bytes());
|
||||||
buf
|
buf
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -673,4 +1022,18 @@ mod tests {
|
|||||||
let records = collect_btree_v2_records(&header, &hdr, 8, 8).unwrap();
|
let records = collect_btree_v2_records(&header, &hdr, 8, 8).unwrap();
|
||||||
assert!(records.is_empty());
|
assert!(records.is_empty());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn subtree_capacity_matches_libhdf5() {
|
||||||
|
// A link-name index (11-byte records, 512-byte nodes, 8-byte
|
||||||
|
// addresses): libhdf5's H5B2__hdr_init gives 45 records per leaf,
|
||||||
|
// then cum_max_nrec 1 149 at depth 1 and 26 449 at depth 2 — two
|
||||||
|
// bytes of subtree count in a depth-3 root's child pointers, where
|
||||||
|
// leaf_max^3 = 91 125 would need three.
|
||||||
|
let leaf = max_records_leaf(512, 11);
|
||||||
|
assert_eq!(leaf, 45);
|
||||||
|
assert_eq!(cum_max_records(512, 11, 8, leaf, 0), 45);
|
||||||
|
assert_eq!(cum_max_records(512, 11, 8, leaf, 1), 1_149);
|
||||||
|
assert_eq!(cum_max_records(512, 11, 8, leaf, 2), 26_449);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,503 @@
|
|||||||
|
//! Writing version-2 B-trees: a header (`BTHD`) and its nodes, leaves
|
||||||
|
//! (`BTLF`) and, for more records than one leaf holds, internal nodes
|
||||||
|
//! (`BTIN`) to any depth.
|
||||||
|
//!
|
||||||
|
//! Node capacities come from [`crate::btree_v2::node_info`], the arithmetic
|
||||||
|
//! libhdf5 uses (`H5B2__hdr_init`) and the reader decodes pointers with, so
|
||||||
|
//! the pointer widths the writer encodes are the ones every reader expects.
|
||||||
|
|
||||||
|
use crate::addr::saturating_usize;
|
||||||
|
#[cfg(not(feature = "std"))]
|
||||||
|
use alloc::{format, vec, vec::Vec};
|
||||||
|
|
||||||
|
use crate::btree_v2::{NodeInfo, bytes_for_max_records, node_info};
|
||||||
|
use crate::checksum::jenkins_lookup3;
|
||||||
|
use crate::error::FormatError;
|
||||||
|
|
||||||
|
/// How a B-tree is laid out: its record type and node geometry, as the
|
||||||
|
/// header records them.
|
||||||
|
#[derive(Debug, Clone, Copy)]
|
||||||
|
pub(crate) struct BTreeV2Params {
|
||||||
|
/// Record type (5: link names, 6: link creation order, 8: attribute
|
||||||
|
/// names, 9: attribute creation order, 10/11: chunks).
|
||||||
|
pub(crate) tree_type: u8,
|
||||||
|
/// Bytes per node.
|
||||||
|
pub(crate) node_size: u32,
|
||||||
|
/// Bytes per record.
|
||||||
|
pub(crate) record_size: u16,
|
||||||
|
/// Split and merge percentages. The writer fills nodes itself; these
|
||||||
|
/// only tell libhdf5 when to split and merge as it modifies the tree.
|
||||||
|
pub(crate) split_percent: u8,
|
||||||
|
pub(crate) merge_percent: u8,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Size of a B-tree v2 header.
|
||||||
|
pub(crate) fn header_size(offset_size: u8, length_size: u8) -> usize {
|
||||||
|
4 + 1 + 1 + 4 + 2 + 2 + 1 + 1 + offset_size as usize + 2 + length_size as usize + 4
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Deepest tree the writer builds. Even at the smallest fan-out libhdf5's
|
||||||
|
/// arithmetic allows, a few levels hold more records than any file could.
|
||||||
|
const MAX_WRITE_DEPTH: u16 = 32;
|
||||||
|
|
||||||
|
/// Write a B-tree v2 holding `records` (`record_size` bytes each,
|
||||||
|
/// concatenated, already in the tree's key order) at `addr`: the header,
|
||||||
|
/// then its nodes, each `node_size` bytes. No records gives a header with
|
||||||
|
/// an undefined root.
|
||||||
|
///
|
||||||
|
/// The tree is as shallow as the node size allows: a single leaf when the
|
||||||
|
/// records fit one, otherwise internal nodes above leaves. Records are
|
||||||
|
/// spread evenly over each node's children, so every node but the root is
|
||||||
|
/// at least about half full (above libhdf5's merge threshold, which is below
|
||||||
|
/// half), and each node holds at most its depth's maximum.
|
||||||
|
pub(crate) fn build_btree_v2(
|
||||||
|
p: BTreeV2Params,
|
||||||
|
records: &[u8],
|
||||||
|
addr: u64,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<Vec<u8>, FormatError> {
|
||||||
|
let rs = usize::from(p.record_size);
|
||||||
|
if rs == 0 || !records.len().is_multiple_of(rs) {
|
||||||
|
return Err(FormatError::SerializationError(format!(
|
||||||
|
"B-tree v2 records are {} bytes, not a multiple of the record size {rs}",
|
||||||
|
records.len()
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
let n = (records.len() / rs) as u64;
|
||||||
|
let hdr_len = header_size(offset_size, length_size);
|
||||||
|
|
||||||
|
// The shallowest depth whose subtree can hold every record.
|
||||||
|
let mut info = node_info(p.node_size, p.record_size, offset_size, 0);
|
||||||
|
let max_leaf = info[0].max_nrec;
|
||||||
|
if max_leaf == 0 || max_leaf > u64::from(u16::MAX) {
|
||||||
|
return Err(FormatError::SerializationError(format!(
|
||||||
|
"a {}-byte B-tree v2 node holds {max_leaf} {}-byte records; \
|
||||||
|
a node holds 1 to 65535",
|
||||||
|
p.node_size, p.record_size
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
let mut depth = 0u16;
|
||||||
|
while info[usize::from(depth)].cum_max_nrec < n {
|
||||||
|
depth += 1;
|
||||||
|
if depth > MAX_WRITE_DEPTH {
|
||||||
|
return Err(FormatError::SerializationError(format!(
|
||||||
|
"{n} records do not fit a B-tree v2 of {}-byte nodes",
|
||||||
|
p.node_size
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
info = node_info(p.node_size, p.record_size, offset_size, depth);
|
||||||
|
let max = info[usize::from(depth)].max_nrec;
|
||||||
|
if max == 0 || max > u64::from(u16::MAX) {
|
||||||
|
return Err(FormatError::SerializationError(format!(
|
||||||
|
"a {}-byte B-tree v2 internal node holds {max} records; \
|
||||||
|
a node holds 1 to 65535",
|
||||||
|
p.node_size
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
let mut w = TreeWriter {
|
||||||
|
p,
|
||||||
|
records,
|
||||||
|
info: &info,
|
||||||
|
nrec_width: bytes_for_max_records(max_leaf),
|
||||||
|
offset_size,
|
||||||
|
first_node: addr + hdr_len as u64,
|
||||||
|
nodes: Vec::new(),
|
||||||
|
};
|
||||||
|
let root = (n > 0)
|
||||||
|
.then(|| w.node(depth, 0, saturating_usize(n)))
|
||||||
|
.transpose()?;
|
||||||
|
|
||||||
|
let mut out = Vec::with_capacity(hdr_len + w.nodes.len() * p.node_size as usize);
|
||||||
|
out.extend_from_slice(b"BTHD");
|
||||||
|
out.push(0); // version
|
||||||
|
out.push(p.tree_type);
|
||||||
|
out.extend_from_slice(&p.node_size.to_le_bytes());
|
||||||
|
out.extend_from_slice(&p.record_size.to_le_bytes());
|
||||||
|
out.extend_from_slice(&depth.to_le_bytes());
|
||||||
|
out.push(p.split_percent);
|
||||||
|
out.push(p.merge_percent);
|
||||||
|
match root {
|
||||||
|
Some(r) => push_uint(&mut out, r.addr, offset_size as usize),
|
||||||
|
None => out.extend(core::iter::repeat_n(0xFF, offset_size as usize)),
|
||||||
|
}
|
||||||
|
let root_nrec = root.map_or(0, |r| r.nrec);
|
||||||
|
out.extend_from_slice(&(root_nrec as u16).to_le_bytes());
|
||||||
|
push_uint(&mut out, n, length_size as usize);
|
||||||
|
let sum = jenkins_lookup3(&out);
|
||||||
|
out.extend_from_slice(&sum.to_le_bytes());
|
||||||
|
debug_assert_eq!(out.len(), hdr_len);
|
||||||
|
for node in &w.nodes {
|
||||||
|
out.extend_from_slice(node);
|
||||||
|
}
|
||||||
|
Ok(out)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A written node, as its parent points at it.
|
||||||
|
#[derive(Debug, Clone, Copy)]
|
||||||
|
struct NodeRef {
|
||||||
|
addr: u64,
|
||||||
|
/// Records in the node itself.
|
||||||
|
nrec: u64,
|
||||||
|
/// Records in the subtree it roots.
|
||||||
|
all_nrec: u64,
|
||||||
|
}
|
||||||
|
|
||||||
|
struct TreeWriter<'a> {
|
||||||
|
p: BTreeV2Params,
|
||||||
|
records: &'a [u8],
|
||||||
|
info: &'a [NodeInfo],
|
||||||
|
/// Width of a child's record count: what a leaf's maximum needs.
|
||||||
|
nrec_width: usize,
|
||||||
|
offset_size: u8,
|
||||||
|
/// Address of the first node (right after the header).
|
||||||
|
first_node: u64,
|
||||||
|
/// Nodes in file order (children before their parent).
|
||||||
|
nodes: Vec<Vec<u8>>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl TreeWriter<'_> {
|
||||||
|
fn record(&self, i: usize) -> &[u8] {
|
||||||
|
let rs = usize::from(self.p.record_size);
|
||||||
|
&self.records[i * rs..(i + 1) * rs]
|
||||||
|
}
|
||||||
|
|
||||||
|
fn push_node(&mut self, mut node: Vec<u8>) -> u64 {
|
||||||
|
// The checksum covers the node up to it, not the padding after.
|
||||||
|
let sum = jenkins_lookup3(&node);
|
||||||
|
node.extend_from_slice(&sum.to_le_bytes());
|
||||||
|
debug_assert!(node.len() <= self.p.node_size as usize);
|
||||||
|
node.resize(self.p.node_size as usize, 0);
|
||||||
|
let addr = self.first_node + self.nodes.len() as u64 * u64::from(self.p.node_size);
|
||||||
|
self.nodes.push(node);
|
||||||
|
addr
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write the subtree of `depth` holding records `first..first + n`.
|
||||||
|
fn node(&mut self, depth: u16, first: usize, n: usize) -> Result<NodeRef, FormatError> {
|
||||||
|
let rs = usize::from(self.p.record_size);
|
||||||
|
let mut node = Vec::with_capacity(self.p.node_size as usize);
|
||||||
|
if depth == 0 {
|
||||||
|
debug_assert!(n as u64 <= self.info[0].max_nrec);
|
||||||
|
node.extend_from_slice(b"BTLF");
|
||||||
|
node.push(0); // version
|
||||||
|
node.push(self.p.tree_type);
|
||||||
|
node.extend_from_slice(&self.records[first * rs..(first + n) * rs]);
|
||||||
|
let addr = self.push_node(node);
|
||||||
|
return Ok(NodeRef {
|
||||||
|
addr,
|
||||||
|
nrec: n as u64,
|
||||||
|
all_nrec: n as u64,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
// As few children as hold the records, at least two, with the
|
||||||
|
// records spread evenly: `k` children and `k - 1` records between
|
||||||
|
// them.
|
||||||
|
let below = self.info[usize::from(depth) - 1].cum_max_nrec;
|
||||||
|
let k = (n as u64 + 1).div_ceil(below + 1).max(2);
|
||||||
|
let max = self.info[usize::from(depth)].max_nrec;
|
||||||
|
if k - 1 > max || (n as u64) < k - 1 + k {
|
||||||
|
return Err(FormatError::SerializationError(format!(
|
||||||
|
"cannot spread {n} B-tree v2 records over {k} children at depth {depth}"
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
let k = saturating_usize(k);
|
||||||
|
let in_children = n - (k - 1);
|
||||||
|
let (base, extra) = (in_children / k, in_children % k);
|
||||||
|
|
||||||
|
let mut children = Vec::with_capacity(k);
|
||||||
|
let mut separators = Vec::with_capacity(k - 1);
|
||||||
|
let mut next = first;
|
||||||
|
for c in 0..k {
|
||||||
|
let m = base + usize::from(c < extra);
|
||||||
|
children.push(self.node(depth - 1, next, m)?);
|
||||||
|
next += m;
|
||||||
|
if c + 1 < k {
|
||||||
|
separators.push(next);
|
||||||
|
next += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
debug_assert_eq!(next, first + n);
|
||||||
|
|
||||||
|
node.extend_from_slice(b"BTIN");
|
||||||
|
node.push(0); // version
|
||||||
|
node.push(self.p.tree_type);
|
||||||
|
for &s in &separators {
|
||||||
|
node.extend_from_slice(self.record(s));
|
||||||
|
}
|
||||||
|
let total_width = if depth > 1 {
|
||||||
|
self.info[usize::from(depth) - 1].cum_max_nrec_size
|
||||||
|
} else {
|
||||||
|
0
|
||||||
|
};
|
||||||
|
for c in &children {
|
||||||
|
push_uint(&mut node, c.addr, self.offset_size as usize);
|
||||||
|
push_uint(&mut node, c.nrec, self.nrec_width);
|
||||||
|
if depth > 1 {
|
||||||
|
push_uint(&mut node, c.all_nrec, total_width);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let addr = self.push_node(node);
|
||||||
|
Ok(NodeRef {
|
||||||
|
addr,
|
||||||
|
nrec: (k - 1) as u64,
|
||||||
|
all_nrec: n as u64,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Append `v` as a `width`-byte little-endian integer.
|
||||||
|
fn push_uint(buf: &mut Vec<u8>, v: u64, width: usize) {
|
||||||
|
let bytes = v.to_le_bytes();
|
||||||
|
buf.extend_from_slice(&bytes[..width.min(8)]);
|
||||||
|
buf.extend(vec![0u8; width.saturating_sub(8)]);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
use crate::btree_v2::{BTreeV2Header, collect_btree_v2_records};
|
||||||
|
|
||||||
|
fn params(node_size: u32, record_size: u16) -> BTreeV2Params {
|
||||||
|
BTreeV2Params {
|
||||||
|
tree_type: 5,
|
||||||
|
node_size,
|
||||||
|
record_size,
|
||||||
|
split_percent: 100,
|
||||||
|
merge_percent: 40,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `n` 11-byte records: a big-endian counter, so byte order is key order.
|
||||||
|
fn records(n: usize, rs: usize) -> Vec<u8> {
|
||||||
|
let mut out = Vec::with_capacity(n * rs);
|
||||||
|
for i in 0..n {
|
||||||
|
let mut r = vec![0u8; rs];
|
||||||
|
r[..8].copy_from_slice(&(i as u64).to_be_bytes());
|
||||||
|
out.extend_from_slice(&r);
|
||||||
|
}
|
||||||
|
out
|
||||||
|
}
|
||||||
|
|
||||||
|
fn roundtrip(node_size: u32, rs: u16, n: usize, os: u8, ls: u8) -> BTreeV2Header {
|
||||||
|
let recs = records(n, usize::from(rs));
|
||||||
|
let base = 4096u64;
|
||||||
|
let tree = build_btree_v2(params(node_size, rs), &recs, base, os, ls).unwrap();
|
||||||
|
let mut file = vec![0u8; base as usize];
|
||||||
|
file.extend_from_slice(&tree);
|
||||||
|
let hdr = BTreeV2Header::parse(&file, base as usize, os, ls).unwrap();
|
||||||
|
assert_eq!(hdr.total_records, n as u64);
|
||||||
|
let got = collect_btree_v2_records(&file, &hdr, os, ls).unwrap();
|
||||||
|
assert_eq!(got.len(), n);
|
||||||
|
let flat: Vec<u8> = got.into_iter().flat_map(|r| r.data).collect();
|
||||||
|
assert_eq!(flat, recs, "node {node_size} rs {rs} n {n}");
|
||||||
|
hdr
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn one_leaf_then_deeper_trees_read_back_in_order() {
|
||||||
|
// 512-byte nodes of 11-byte records: 45 per leaf, 1149 at depth 1,
|
||||||
|
// 26 449 at depth 2.
|
||||||
|
let info = node_info(512, 11, 8, 3);
|
||||||
|
assert_eq!(
|
||||||
|
info.iter().map(|i| i.cum_max_nrec).collect::<Vec<_>>(),
|
||||||
|
[45, 1149, 26_449, 608_349]
|
||||||
|
);
|
||||||
|
for (n, depth) in [
|
||||||
|
(0, 0),
|
||||||
|
(1, 0),
|
||||||
|
(45, 0),
|
||||||
|
(46, 1),
|
||||||
|
(1149, 1),
|
||||||
|
(1150, 2),
|
||||||
|
(26_449, 2),
|
||||||
|
(26_450, 3),
|
||||||
|
(100_000, 3),
|
||||||
|
] {
|
||||||
|
let hdr = roundtrip(512, 11, n, 8, 8);
|
||||||
|
assert_eq!(hdr.depth, depth, "{n} records");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn pointer_widths_follow_the_offset_and_length_sizes() {
|
||||||
|
for (os, ls) in [(4, 4), (8, 4), (4, 8), (2, 2)] {
|
||||||
|
roundtrip(512, 11, 5000, os, ls);
|
||||||
|
}
|
||||||
|
// Wide counts: a leaf of 2048 bytes / 9-byte records (226, one byte)
|
||||||
|
// and deeper subtree totals of three bytes.
|
||||||
|
roundtrip(2048, 9, 300_000, 8, 8);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn every_node_is_within_its_capacity_and_above_the_merge_threshold() {
|
||||||
|
let rs = 17u16;
|
||||||
|
let n = 70_000usize;
|
||||||
|
let info = node_info(512, rs, 8, 3);
|
||||||
|
let recs = records(n, usize::from(rs));
|
||||||
|
let tree = build_btree_v2(params(512, rs), &recs, 0, 8, 8).unwrap();
|
||||||
|
let hdr_len = header_size(8, 8);
|
||||||
|
let nodes = (tree.len() - hdr_len) / 512;
|
||||||
|
for i in 0..nodes {
|
||||||
|
let node = &tree[hdr_len + i * 512..hdr_len + (i + 1) * 512];
|
||||||
|
let sig = &node[..4];
|
||||||
|
if sig == b"BTLF" {
|
||||||
|
continue; // counts checked through the parents below
|
||||||
|
}
|
||||||
|
assert_eq!(sig, b"BTIN");
|
||||||
|
}
|
||||||
|
// Walk from the header: each child's count within [40%, 100%].
|
||||||
|
let hdr = BTreeV2Header::parse(&tree, 0, 8, 8).unwrap();
|
||||||
|
assert_eq!(hdr.depth, 3);
|
||||||
|
assert!(u64::from(hdr.num_records_in_root) <= info[3].max_nrec);
|
||||||
|
fn walk(tree: &[u8], addr: usize, nrec: usize, depth: usize, info: &[NodeInfo], rs: usize) {
|
||||||
|
if depth == 0 {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
let nrec_w = bytes_for_max_records(info[0].max_nrec);
|
||||||
|
let tot_w = if depth > 1 {
|
||||||
|
info[depth - 1].cum_max_nrec_size
|
||||||
|
} else {
|
||||||
|
0
|
||||||
|
};
|
||||||
|
let mut pos = addr + 6 + nrec * rs;
|
||||||
|
for _ in 0..=nrec {
|
||||||
|
let a = u64::from_le_bytes(tree[pos..pos + 8].try_into().unwrap()) as usize;
|
||||||
|
pos += 8;
|
||||||
|
let mut c = 0usize;
|
||||||
|
for b in 0..nrec_w {
|
||||||
|
c |= usize::from(tree[pos + b]) << (8 * b);
|
||||||
|
}
|
||||||
|
pos += nrec_w + tot_w;
|
||||||
|
let max = info[depth - 1].max_nrec as usize;
|
||||||
|
assert!(c <= max && c * 100 > max * 40, "{c} of {max}");
|
||||||
|
walk(tree, a, c, depth - 1, info, rs);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
walk(
|
||||||
|
&tree,
|
||||||
|
hdr.root_node_address as usize,
|
||||||
|
usize::from(hdr.num_records_in_root),
|
||||||
|
3,
|
||||||
|
&info,
|
||||||
|
usize::from(rs),
|
||||||
|
);
|
||||||
|
assert!(nodes > 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Descending to a key range finds exactly the records a full read
|
||||||
|
/// holds in it — runs of equal keys that straddle node boundaries
|
||||||
|
/// included — at every depth, and nothing for keys not in the tree.
|
||||||
|
#[test]
|
||||||
|
fn a_key_range_search_matches_a_full_scan() {
|
||||||
|
use crate::btree_v2::find_btree_v2_records;
|
||||||
|
use core::cmp::Ordering;
|
||||||
|
let rs = 11usize;
|
||||||
|
// Keys 0, 0, 0, 2, 2, 2, 4, ...: runs of three, odd keys missing.
|
||||||
|
for n in [1usize, 45, 46, 1150, 30_000] {
|
||||||
|
let mut recs = Vec::with_capacity(n * rs);
|
||||||
|
for i in 0..n {
|
||||||
|
let mut r = vec![0u8; rs];
|
||||||
|
r[..8].copy_from_slice(&((i / 3 * 2) as u64).to_be_bytes());
|
||||||
|
r[8..].copy_from_slice(&[(i % 3) as u8, 0, 0]);
|
||||||
|
recs.extend_from_slice(&r);
|
||||||
|
}
|
||||||
|
let base = 4096u64;
|
||||||
|
let tree = build_btree_v2(params(512, 11), &recs, base, 8, 8).unwrap();
|
||||||
|
let mut file = vec![0u8; base as usize];
|
||||||
|
file.extend_from_slice(&tree);
|
||||||
|
let hdr = BTreeV2Header::parse(&file, base as usize, 8, 8).unwrap();
|
||||||
|
let all = collect_btree_v2_records(&file, &hdr, 8, 8).unwrap();
|
||||||
|
let key = |r: &[u8]| u64::from_be_bytes(r[..8].try_into().unwrap());
|
||||||
|
let last = key(&all[n - 1].data);
|
||||||
|
let probes = (0..=last + 1).step_by(if n > 1000 { 37 } else { 1 });
|
||||||
|
for k in probes.chain([last, last + 1, u64::MAX]) {
|
||||||
|
let found =
|
||||||
|
find_btree_v2_records(&file, &hdr, 8, &mut |r: &[u8]| key(r).cmp(&k)).unwrap();
|
||||||
|
let want: Vec<&[u8]> = all
|
||||||
|
.iter()
|
||||||
|
.map(|r| r.data.as_slice())
|
||||||
|
.filter(|r| key(r) == k)
|
||||||
|
.collect();
|
||||||
|
let got: Vec<&[u8]> = found.iter().map(|r| r.data.as_slice()).collect();
|
||||||
|
assert_eq!(got, want, "n {n} key {k}");
|
||||||
|
assert_eq!(
|
||||||
|
got.len(),
|
||||||
|
if k % 2 == 0 && k <= last {
|
||||||
|
want.len()
|
||||||
|
} else {
|
||||||
|
0
|
||||||
|
}
|
||||||
|
);
|
||||||
|
}
|
||||||
|
// Every record, or none, when the whole tree is in or out of range.
|
||||||
|
let every = find_btree_v2_records(&file, &hdr, 8, &mut |_| Ordering::Equal).unwrap();
|
||||||
|
assert_eq!(every.len(), n);
|
||||||
|
let none = find_btree_v2_records(&file, &hdr, 8, &mut |_| Ordering::Less).unwrap();
|
||||||
|
assert!(none.is_empty());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A two-level tree read through a `read_at`-only storage gives what
|
||||||
|
/// the slice gives — records, descents and errors — whole, truncated
|
||||||
|
/// at every length, and with each byte of its nodes flipped, and each
|
||||||
|
/// node costs one read.
|
||||||
|
#[test]
|
||||||
|
fn storage_reads_match_slice_reads() {
|
||||||
|
use crate::btree_v2::{
|
||||||
|
collect_btree_v2_records_in, find_btree_v2_records, find_btree_v2_records_in,
|
||||||
|
};
|
||||||
|
use crate::storage::CountingStorage;
|
||||||
|
let (rs, n, base) = (11usize, 120usize, 64usize);
|
||||||
|
let recs = records(n, rs);
|
||||||
|
let tree = build_btree_v2(params(128, 11), &recs, base as u64, 8, 8).unwrap();
|
||||||
|
let mut whole = vec![0u8; base];
|
||||||
|
whole.extend_from_slice(&tree);
|
||||||
|
let hdr = BTreeV2Header::parse(&whole, base, 8, 8).unwrap();
|
||||||
|
assert!(hdr.depth >= 1, "{hdr:?}");
|
||||||
|
let key = |r: &[u8]| u64::from_be_bytes(r[..8].try_into().unwrap());
|
||||||
|
let mut files = Vec::new();
|
||||||
|
for cut in base..=whole.len() {
|
||||||
|
files.push(whole[..cut].to_vec());
|
||||||
|
}
|
||||||
|
for at in base..whole.len() {
|
||||||
|
let mut bad = whole.clone();
|
||||||
|
bad[at] ^= 0x5a;
|
||||||
|
files.push(bad);
|
||||||
|
}
|
||||||
|
let mut ok = 0;
|
||||||
|
for f in &files {
|
||||||
|
let st = CountingStorage::new(f.clone());
|
||||||
|
let want_h = BTreeV2Header::parse(f, base, 8, 8);
|
||||||
|
let got_h = BTreeV2Header::parse_in(&st, base as u64, 8, 8);
|
||||||
|
assert_eq!(format!("{got_h:?}"), format!("{want_h:?}"));
|
||||||
|
// The nodes of the intact header, over each damaged file.
|
||||||
|
let want = collect_btree_v2_records(f, &hdr, 8, 8);
|
||||||
|
st.reset();
|
||||||
|
let got = collect_btree_v2_records_in(&st, &hdr, 8, 8);
|
||||||
|
assert_eq!(format!("{got:?}"), format!("{want:?}"));
|
||||||
|
if want.is_ok() {
|
||||||
|
ok += 1;
|
||||||
|
assert!(st.reads() <= 1 + n as u64 / 3, "{} reads", st.reads());
|
||||||
|
}
|
||||||
|
for k in [0u64, 7, 60, 119, 500] {
|
||||||
|
let want = find_btree_v2_records(f, &hdr, 8, &mut |r: &[u8]| key(r).cmp(&k));
|
||||||
|
let got = find_btree_v2_records_in(&st, &hdr, 8, &mut |r: &[u8]| key(r).cmp(&k));
|
||||||
|
assert_eq!(format!("{got:?}"), format!("{want:?}"));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
assert!(ok > 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn a_node_too_small_or_too_big_is_an_error() {
|
||||||
|
assert!(build_btree_v2(params(16, 11), &records(1, 11), 0, 8, 8).is_err());
|
||||||
|
// A leaf with room for more than 65 535 records.
|
||||||
|
assert!(build_btree_v2(params(1 << 20, 11), &records(1, 11), 0, 8, 8).is_err());
|
||||||
|
// Records that are not whole.
|
||||||
|
assert!(build_btree_v2(params(512, 11), &[0u8; 12], 0, 8, 8).is_err());
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,79 @@
|
|||||||
|
//! Large output buffers backed by transparent huge pages where the OS offers
|
||||||
|
//! them.
|
||||||
|
//!
|
||||||
|
//! A fresh multi-megabyte `Vec` is mapped lazily by the kernel: the first
|
||||||
|
//! write to each 4 KiB page takes a page fault, and the kernel zeroes the page
|
||||||
|
//! before handing it over. For a 64 MiB read that is 16384 faults, and they
|
||||||
|
//! cost far more than the copy that fills the buffer — single-threaded
|
||||||
|
//! contiguous reads ran at about a quarter of h5py's speed because of them.
|
||||||
|
//! numpy (so h5py) avoids this by asking for transparent huge pages
|
||||||
|
//! (`madvise(MADV_HUGEPAGE)`) on every allocation of 4 MiB or more, which
|
||||||
|
//! turns 512 faults into one; this module does the same.
|
||||||
|
//!
|
||||||
|
//! The advice only changes how the pages are backed, never their contents, so
|
||||||
|
//! it is harmless when it cannot be honoured (THP disabled, not Linux, a
|
||||||
|
//! region that is part of the heap): the buffer is then exactly what it would
|
||||||
|
//! have been without it.
|
||||||
|
|
||||||
|
#[cfg(not(feature = "std"))]
|
||||||
|
use alloc::vec::Vec;
|
||||||
|
|
||||||
|
/// Buffers smaller than this are left alone (numpy uses the same threshold).
|
||||||
|
#[cfg(any(target_os = "linux", test))]
|
||||||
|
pub(crate) const HUGE_PAGE_THRESHOLD: usize = 4 << 20;
|
||||||
|
|
||||||
|
/// Advise the kernel to back `[ptr, ptr + len)` with transparent huge pages,
|
||||||
|
/// when `len` is large enough to benefit. Call it before the first write so
|
||||||
|
/// the faults happen at huge-page granularity.
|
||||||
|
#[inline]
|
||||||
|
pub(crate) fn advise_huge_pages(ptr: *const u8, len: usize) {
|
||||||
|
#[cfg(target_os = "linux")]
|
||||||
|
if len >= HUGE_PAGE_THRESHOLD {
|
||||||
|
const PAGE: usize = 4096;
|
||||||
|
let start = (ptr as usize).next_multiple_of(PAGE);
|
||||||
|
let end = (ptr as usize + len) & !(PAGE - 1);
|
||||||
|
if end > start {
|
||||||
|
// SAFETY: `[start, end)` lies inside an allocation of `len` bytes
|
||||||
|
// at `ptr` that the caller owns, and is page aligned as madvise
|
||||||
|
// requires. MADV_HUGEPAGE does not change the memory's contents or
|
||||||
|
// validity; on failure (EINVAL when THP is compiled out, etc.) the
|
||||||
|
// region is simply left as it was, so the result is ignored.
|
||||||
|
unsafe {
|
||||||
|
libc::madvise(start as *mut libc::c_void, end - start, libc::MADV_HUGEPAGE);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
#[cfg(not(target_os = "linux"))]
|
||||||
|
let _ = (ptr, len);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `Vec::with_capacity(count)` for a buffer about to be filled in bulk, with
|
||||||
|
/// huge-page advice when it is large (see the module docs).
|
||||||
|
#[inline]
|
||||||
|
pub(crate) fn vec_for_bulk<T>(count: usize) -> Vec<T> {
|
||||||
|
let v: Vec<T> = Vec::with_capacity(count);
|
||||||
|
advise_huge_pages(
|
||||||
|
v.as_ptr().cast::<u8>(),
|
||||||
|
v.capacity().saturating_mul(core::mem::size_of::<T>()),
|
||||||
|
);
|
||||||
|
v
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn bulk_vec_is_an_ordinary_vec() {
|
||||||
|
for count in [0usize, 1, 1000, HUGE_PAGE_THRESHOLD / 4 + 3] {
|
||||||
|
let mut v: Vec<u32> = vec_for_bulk(count);
|
||||||
|
assert!(v.capacity() >= count);
|
||||||
|
v.extend((0..count as u32).map(|i| i.wrapping_mul(2654435761)));
|
||||||
|
assert!(
|
||||||
|
v.iter()
|
||||||
|
.enumerate()
|
||||||
|
.all(|(i, &x)| x == (i as u32).wrapping_mul(2654435761))
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -14,6 +14,45 @@ pub fn jenkins_lookup3(data: &[u8]) -> u32 {
|
|||||||
hashlittle(data, 0)
|
hashlittle(data, 0)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// HDF5's Fletcher-32 checksum, as the Fletcher-32 I/O filter (filter id 3)
|
||||||
|
/// stores it after each chunk.
|
||||||
|
///
|
||||||
|
/// A line-for-line port of `H5_checksum_fletcher32` (H5checksum.c, libhdf5
|
||||||
|
/// 1.8 through 1.14): big-endian 16-bit words summed in blocks of 360, each
|
||||||
|
/// sum reduced after a block by the ones'-complement fold
|
||||||
|
/// `(s & 0xffff) + (s >> 16)` rather than `% 65535`, an odd trailing byte
|
||||||
|
/// taken as the high byte of a last word, and a final fold of both sums.
|
||||||
|
/// The fold and `% 65535` differ whenever a sum is a non-zero multiple of
|
||||||
|
/// 65535: the fold leaves 0xffff where the modulo gives 0, so the two
|
||||||
|
/// disagree on about one chunk in 32768 and libhdf5 rejects the other's
|
||||||
|
/// checksum. This must stay the only implementation.
|
||||||
|
pub fn fletcher32(data: &[u8]) -> u32 {
|
||||||
|
let mut sum1: u32 = 0;
|
||||||
|
let mut sum2: u32 = 0;
|
||||||
|
// 360 words keep both sums inside 32 bits between folds (the bound
|
||||||
|
// libhdf5 uses: after a fold sum1 < 0x10200, so sum2 stays below
|
||||||
|
// 360 * 361 / 2 * 0xffff + 360 * 0x10200 + 0x1fffe < 2^32). The adds wrap
|
||||||
|
// like the C unsigned arithmetic all the same.
|
||||||
|
let (words, odd) = data.as_chunks::<2>();
|
||||||
|
for block in words.chunks(360) {
|
||||||
|
for w in block {
|
||||||
|
sum1 = sum1.wrapping_add((u32::from(w[0]) << 8) | u32::from(w[1]));
|
||||||
|
sum2 = sum2.wrapping_add(sum1);
|
||||||
|
}
|
||||||
|
sum1 = (sum1 & 0xffff) + (sum1 >> 16);
|
||||||
|
sum2 = (sum2 & 0xffff) + (sum2 >> 16);
|
||||||
|
}
|
||||||
|
if let [last] = odd {
|
||||||
|
sum1 = sum1.wrapping_add(u32::from(*last) << 8);
|
||||||
|
sum2 = sum2.wrapping_add(sum1);
|
||||||
|
sum1 = (sum1 & 0xffff) + (sum1 >> 16);
|
||||||
|
sum2 = (sum2 & 0xffff) + (sum2 >> 16);
|
||||||
|
}
|
||||||
|
sum1 = (sum1 & 0xffff) + (sum1 >> 16);
|
||||||
|
sum2 = (sum2 & 0xffff) + (sum2 >> 16);
|
||||||
|
(sum2 << 16) | sum1
|
||||||
|
}
|
||||||
|
|
||||||
/// Compute CRC32 (IEEE / ISO 3309) over data.
|
/// Compute CRC32 (IEEE / ISO 3309) over data.
|
||||||
///
|
///
|
||||||
/// When the `fast-checksum` feature is enabled, this uses hardware CRC32
|
/// When the `fast-checksum` feature is enabled, this uses hardware CRC32
|
||||||
@@ -207,6 +246,19 @@ fn hashlittle(data: &[u8], initval: u32) -> u32 {
|
|||||||
mod tests {
|
mod tests {
|
||||||
use super::*;
|
use super::*;
|
||||||
|
|
||||||
|
/// Values of libhdf5's `H5_checksum_fletcher32` (h5py 3.x's bundled
|
||||||
|
/// libhdf5, called through ctypes). The first three are sums that are
|
||||||
|
/// multiples of 65535, where `% 65535` gave 0 instead of 0xffff.
|
||||||
|
#[test]
|
||||||
|
fn fletcher32_matches_libhdf5() {
|
||||||
|
assert_eq!(fletcher32(&[0x00, 0x01, 0xff, 0xfe]), 0x0001_ffff);
|
||||||
|
assert_eq!(fletcher32(&[0xff; 720]), 0xffff_ffff);
|
||||||
|
assert_eq!(fletcher32(&[0xff; 721]), 0xff00_ff00);
|
||||||
|
assert_eq!(fletcher32(&[0xff; 1441]), 0xff00_ff00);
|
||||||
|
assert_eq!(fletcher32(&[]), 0);
|
||||||
|
assert_eq!(fletcher32(&[7]), 0x0700_0700);
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn empty_input() {
|
fn empty_input() {
|
||||||
// Empty input should return the initial state after no mixing
|
// Empty input should return the initial state after no mixing
|
||||||
|
|||||||
@@ -223,13 +223,32 @@ pub const DEFAULT_CACHE_BYTES: usize = 16 * 1024 * 1024; // 16 MiB
|
|||||||
/// coordinate map and reduces collision chains compared to power-of-two sizes.
|
/// coordinate map and reduces collision chains compared to power-of-two sizes.
|
||||||
pub const DEFAULT_MAX_SLOTS: usize = 521;
|
pub const DEFAULT_MAX_SLOTS: usize = 521;
|
||||||
|
|
||||||
|
/// Most datasets whose chunk index a [`ChunkCache`] keeps at once.
|
||||||
|
pub const MAX_INDEXED_DATASETS: usize = 64;
|
||||||
|
|
||||||
|
/// Most chunk-index entries, summed over all datasets, a [`ChunkCache`] keeps.
|
||||||
|
/// Least-recently-used datasets' indexes are dropped past this (the dataset
|
||||||
|
/// being read is always kept), so a file with many or huge chunked datasets
|
||||||
|
/// cannot grow the cache without bound.
|
||||||
|
pub const MAX_INDEXED_CHUNKS: usize = 1 << 20;
|
||||||
|
|
||||||
|
/// The dataset key the address-less (legacy) methods use when
|
||||||
|
/// [`ChunkCache::ensure_dataset`] has not been called.
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
const UNBOUND_DATASET: u64 = u64::MAX;
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// LRU entry
|
// LRU entry
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// Decompressed chunks are keyed by dataset *and* coordinate: every chunked
|
||||||
|
/// dataset has a chunk at (0, 0, ...), so the coordinate alone is ambiguous.
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
type SlotKey = (u64, ChunkCoord);
|
||||||
|
|
||||||
#[cfg(feature = "std")]
|
#[cfg(feature = "std")]
|
||||||
struct CachedChunk {
|
struct CachedChunk {
|
||||||
coord: ChunkCoord,
|
key: SlotKey,
|
||||||
/// Shared so a cache hit is a refcount bump, not a copy of the whole
|
/// Shared so a cache hit is a refcount bump, not a copy of the whole
|
||||||
/// (potentially large) decompressed chunk.
|
/// (potentially large) decompressed chunk.
|
||||||
data: Arc<CacheAlignedBuffer>,
|
data: Arc<CacheAlignedBuffer>,
|
||||||
@@ -237,21 +256,54 @@ struct CachedChunk {
|
|||||||
last_access: u64,
|
last_access: u64,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Per-dataset index state.
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
#[derive(Default)]
|
||||||
|
struct DatasetEntry {
|
||||||
|
/// Chunk coordinate -> ChunkInfo (offset + size in file).
|
||||||
|
index: Option<Arc<HashMap<ChunkCoord, ChunkInfo>>>,
|
||||||
|
/// The same chunks in the order the chunk index lists them: what
|
||||||
|
/// [`ChunkCache::chunks_for`] returns, so a cached read walks (and, on a
|
||||||
|
/// damaged file, fails at) the chunks in the same order as an uncached
|
||||||
|
/// one, rather than in hash-map order.
|
||||||
|
ordered: Option<Arc<Vec<ChunkInfo>>>,
|
||||||
|
/// Pre-built chunk index for O(1) coordinate lookups.
|
||||||
|
chunk_index: Option<Arc<ChunkIndex>>,
|
||||||
|
/// Pre-computed chunk layout for fast assembly.
|
||||||
|
chunk_layout: Option<Arc<ChunkLayout>>,
|
||||||
|
/// Tick of the last use, for dropping the least recently used dataset.
|
||||||
|
last_used: u64,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
impl DatasetEntry {
|
||||||
|
fn weight(&self) -> usize {
|
||||||
|
self.index.as_ref().map_or(0, |m| m.len())
|
||||||
|
+ self.ordered.as_ref().map_or(0, |o| o.len())
|
||||||
|
+ self.chunk_index.as_ref().map_or(0, |c| c.num_chunks())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// ChunkCache
|
// ChunkCache
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
/// A per-dataset chunk cache with hash-based index and LRU eviction.
|
/// A per-file chunk cache: chunk indexes per dataset, plus an LRU of
|
||||||
|
/// decompressed chunks, all keyed by dataset.
|
||||||
///
|
///
|
||||||
/// # Usage
|
/// A dataset is identified by the address of its chunk index (B-tree, fixed
|
||||||
|
/// or extensible array, ...), which is unique within a file. Every method
|
||||||
|
/// that takes an `addr` works on that dataset only, so threads reading
|
||||||
|
/// different datasets through one shared cache never see each other's
|
||||||
|
/// chunks. The address-less methods (`has_index`, `populate_index`,
|
||||||
|
/// `get_decompressed`, ...) act on the dataset last bound with
|
||||||
|
/// [`Self::ensure_dataset`]; that binding is shared state, so concurrent
|
||||||
|
/// readers must use the `*_in` / `*_for` methods instead (the chunked
|
||||||
|
/// readers in [`crate::chunked_read`] do).
|
||||||
///
|
///
|
||||||
/// ```ignore
|
/// Memory is bounded: decompressed data by `max_bytes`/`max_slots` across
|
||||||
/// let cache = ChunkCache::new();
|
/// all datasets, indexes by [`MAX_INDEXED_DATASETS`] and
|
||||||
/// // Pass &cache to read_chunked_data — it will populate the index lazily.
|
/// [`MAX_INDEXED_CHUNKS`].
|
||||||
/// ```
|
|
||||||
///
|
|
||||||
/// The cache is wrapped in `Mutex` internally so it can be mutated through
|
|
||||||
/// shared references (thread-safe).
|
|
||||||
///
|
///
|
||||||
/// Only available with the `std` feature because it requires `std::sync::Mutex`.
|
/// Only available with the `std` feature because it requires `std::sync::Mutex`.
|
||||||
#[cfg(feature = "std")]
|
#[cfg(feature = "std")]
|
||||||
@@ -261,26 +313,20 @@ pub struct ChunkCache {
|
|||||||
|
|
||||||
#[cfg(feature = "std")]
|
#[cfg(feature = "std")]
|
||||||
struct CacheInner {
|
struct CacheInner {
|
||||||
/// Hash index: chunk coordinate -> ChunkInfo (offset + size in file).
|
/// Per-dataset chunk indexes, keyed by chunk-index address.
|
||||||
/// Populated once per dataset on first access.
|
datasets: HashMap<u64, DatasetEntry>,
|
||||||
index: Option<HashMap<ChunkCoord, ChunkInfo>>,
|
|
||||||
|
|
||||||
/// Address of the dataset (its chunk-index base address) that the cached
|
/// Dataset the address-less methods act on (see `ensure_dataset`).
|
||||||
/// index, chunk index, layout, and decompressed slots currently belong to.
|
current: Option<u64>,
|
||||||
/// The cache is shared per file across datasets, so every cached-read entry
|
|
||||||
/// checks this and resets the per-dataset state when the dataset changes —
|
|
||||||
/// otherwise one dataset's chunk index (with its own rank) would be reused
|
|
||||||
/// for another, corrupting reads.
|
|
||||||
index_addr: Option<u64>,
|
|
||||||
|
|
||||||
/// LRU cache of decompressed chunk data.
|
/// LRU cache of decompressed chunk data.
|
||||||
slots: Vec<CachedChunk>,
|
slots: Vec<CachedChunk>,
|
||||||
|
|
||||||
/// Coordinate -> index into `slots`, for O(1) lookup instead of a linear
|
/// Key -> index into `slots`, for O(1) lookup instead of a linear
|
||||||
/// scan. Kept in sync with `slots` on every insert/evict/clear — in
|
/// scan. Kept in sync with `slots` on every insert/evict/clear — in
|
||||||
/// particular, `slots.swap_remove(i)` moves the last element into slot
|
/// particular, `slots.swap_remove(i)` moves the last element into slot
|
||||||
/// `i`, so the moved element's index entry must be updated too.
|
/// `i`, so the moved element's index entry must be updated too.
|
||||||
slot_index: HashMap<ChunkCoord, usize>,
|
slot_index: HashMap<SlotKey, usize>,
|
||||||
|
|
||||||
/// Current total bytes of cached decompressed data.
|
/// Current total bytes of cached decompressed data.
|
||||||
current_bytes: usize,
|
current_bytes: usize,
|
||||||
@@ -294,17 +340,145 @@ struct CacheInner {
|
|||||||
/// Monotonic counter for LRU ordering.
|
/// Monotonic counter for LRU ordering.
|
||||||
tick: u64,
|
tick: u64,
|
||||||
|
|
||||||
/// Last accessed chunk coordinate (for sequential detection).
|
/// Last accessed chunk (for sequential detection).
|
||||||
last_coord: Option<ChunkCoord>,
|
last_coord: Option<SlotKey>,
|
||||||
|
|
||||||
/// Access pattern statistics.
|
/// Access pattern statistics.
|
||||||
stats: AccessStats,
|
stats: AccessStats,
|
||||||
|
}
|
||||||
|
|
||||||
/// Pre-built chunk index for O(1) coordinate lookups.
|
#[cfg(feature = "std")]
|
||||||
chunk_index: Option<ChunkIndex>,
|
impl CacheInner {
|
||||||
|
fn current(&self) -> u64 {
|
||||||
|
self.current.unwrap_or(UNBOUND_DATASET)
|
||||||
|
}
|
||||||
|
|
||||||
/// Pre-computed chunk layout for fast assembly.
|
fn touch(&mut self, addr: u64) -> &mut DatasetEntry {
|
||||||
chunk_layout: Option<ChunkLayout>,
|
self.tick += 1;
|
||||||
|
let tick = self.tick;
|
||||||
|
let entry = self.datasets.entry(addr).or_default();
|
||||||
|
entry.last_used = tick;
|
||||||
|
entry
|
||||||
|
}
|
||||||
|
|
||||||
|
fn entry(&self, addr: u64) -> Option<&DatasetEntry> {
|
||||||
|
self.datasets.get(&addr)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Drop least-recently-used datasets' indexes (never `keep`'s) until the
|
||||||
|
/// dataset and chunk-entry budgets hold.
|
||||||
|
fn trim_datasets(&mut self, keep: u64) {
|
||||||
|
loop {
|
||||||
|
let total: usize = self.datasets.values().map(DatasetEntry::weight).sum();
|
||||||
|
if self.datasets.len() <= MAX_INDEXED_DATASETS && total <= MAX_INDEXED_CHUNKS {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
let victim = self
|
||||||
|
.datasets
|
||||||
|
.iter()
|
||||||
|
.filter(|(a, _)| **a != keep)
|
||||||
|
.min_by_key(|(_, e)| e.last_used)
|
||||||
|
.map(|(a, _)| *a);
|
||||||
|
match victim {
|
||||||
|
Some(a) => {
|
||||||
|
self.datasets.remove(&a);
|
||||||
|
}
|
||||||
|
None => return,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn get_decompressed(&mut self, addr: u64, coord: &[u64]) -> Option<Arc<CacheAlignedBuffer>> {
|
||||||
|
self.tick += 1;
|
||||||
|
let tick = self.tick;
|
||||||
|
|
||||||
|
// Track sequential vs random access
|
||||||
|
let is_sequential = self.last_coord.as_ref().is_some_and(|(prev_addr, prev)| {
|
||||||
|
// Sequential if exactly one dimension changed
|
||||||
|
let changes: usize = prev
|
||||||
|
.iter()
|
||||||
|
.zip(coord.iter())
|
||||||
|
.filter(|(a, b)| a != b)
|
||||||
|
.count();
|
||||||
|
*prev_addr == addr && changes <= 1
|
||||||
|
});
|
||||||
|
if is_sequential {
|
||||||
|
self.stats.sequential_count += 1;
|
||||||
|
} else if self.last_coord.is_some() {
|
||||||
|
self.stats.random_count += 1;
|
||||||
|
}
|
||||||
|
let key: SlotKey = (addr, coord.to_vec());
|
||||||
|
let found = if let Some(&idx) = self.slot_index.get(&key) {
|
||||||
|
self.slots[idx].last_access = tick;
|
||||||
|
Some(Arc::clone(&self.slots[idx].data))
|
||||||
|
} else {
|
||||||
|
None
|
||||||
|
};
|
||||||
|
self.last_coord = Some(key);
|
||||||
|
if let Some(ref data) = found {
|
||||||
|
self.stats.hits += 1;
|
||||||
|
self.stats.bytes_read += data.len() as u64;
|
||||||
|
} else {
|
||||||
|
self.stats.misses += 1;
|
||||||
|
}
|
||||||
|
found
|
||||||
|
}
|
||||||
|
|
||||||
|
fn put_decompressed(
|
||||||
|
&mut self,
|
||||||
|
key: SlotKey,
|
||||||
|
data: Arc<CacheAlignedBuffer>,
|
||||||
|
) -> Arc<CacheAlignedBuffer> {
|
||||||
|
let data_len = data.len();
|
||||||
|
|
||||||
|
// Don't cache if single chunk exceeds budget — still return the data
|
||||||
|
// to the caller, just don't retain it.
|
||||||
|
if data_len > self.max_bytes {
|
||||||
|
return data;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Check if already present
|
||||||
|
self.tick += 1;
|
||||||
|
let tick = self.tick;
|
||||||
|
if let Some(&idx) = self.slot_index.get(&key) {
|
||||||
|
self.slots[idx].last_access = tick;
|
||||||
|
return Arc::clone(&self.slots[idx].data); // already cached
|
||||||
|
}
|
||||||
|
|
||||||
|
// Evict until we have room
|
||||||
|
while self.slots.len() >= self.max_slots
|
||||||
|
|| (self.current_bytes + data_len > self.max_bytes && !self.slots.is_empty())
|
||||||
|
{
|
||||||
|
// Find LRU slot
|
||||||
|
let lru_idx = self
|
||||||
|
.slots
|
||||||
|
.iter()
|
||||||
|
.enumerate()
|
||||||
|
.min_by_key(|(_, s)| s.last_access)
|
||||||
|
.map(|(i, _)| i)
|
||||||
|
.unwrap();
|
||||||
|
let removed = self.slots.swap_remove(lru_idx);
|
||||||
|
self.slot_index.remove(&removed.key);
|
||||||
|
// swap_remove moved the former last element into `lru_idx` (unless
|
||||||
|
// it *was* the last element) — fix up that element's index entry.
|
||||||
|
if lru_idx < self.slots.len() {
|
||||||
|
let moved_key = self.slots[lru_idx].key.clone();
|
||||||
|
self.slot_index.insert(moved_key, lru_idx);
|
||||||
|
}
|
||||||
|
self.current_bytes -= removed.data.len();
|
||||||
|
self.stats.evictions += 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
self.current_bytes += data_len;
|
||||||
|
let new_idx = self.slots.len();
|
||||||
|
self.slot_index.insert(key.clone(), new_idx);
|
||||||
|
self.slots.push(CachedChunk {
|
||||||
|
key,
|
||||||
|
data: Arc::clone(&data),
|
||||||
|
last_access: tick,
|
||||||
|
});
|
||||||
|
data
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Access pattern statistics tracked by the chunk cache.
|
/// Access pattern statistics tracked by the chunk cache.
|
||||||
@@ -356,8 +530,8 @@ impl ChunkCache {
|
|||||||
pub fn with_capacity(max_bytes: usize, max_slots: usize) -> Self {
|
pub fn with_capacity(max_bytes: usize, max_slots: usize) -> Self {
|
||||||
Self {
|
Self {
|
||||||
inner: std::sync::Mutex::new(CacheInner {
|
inner: std::sync::Mutex::new(CacheInner {
|
||||||
index: None,
|
datasets: HashMap::new(),
|
||||||
index_addr: None,
|
current: None,
|
||||||
slots: Vec::with_capacity(max_slots.min(64)),
|
slots: Vec::with_capacity(max_slots.min(64)),
|
||||||
slot_index: HashMap::with_capacity(max_slots.min(64)),
|
slot_index: HashMap::with_capacity(max_slots.min(64)),
|
||||||
current_bytes: 0,
|
current_bytes: 0,
|
||||||
@@ -366,340 +540,345 @@ impl ChunkCache {
|
|||||||
tick: 0,
|
tick: 0,
|
||||||
last_coord: None,
|
last_coord: None,
|
||||||
stats: AccessStats::default(),
|
stats: AccessStats::default(),
|
||||||
chunk_index: None,
|
|
||||||
chunk_layout: None,
|
|
||||||
}),
|
}),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// ----- Index operations -----
|
fn lock(&self) -> std::sync::MutexGuard<'_, CacheInner> {
|
||||||
|
self.inner.lock().unwrap_or_else(|e| e.into_inner())
|
||||||
|
}
|
||||||
|
|
||||||
/// The most decompressed bytes this cache will hold.
|
/// The most decompressed bytes this cache will hold.
|
||||||
pub fn max_bytes(&self) -> usize {
|
pub fn max_bytes(&self) -> usize {
|
||||||
self.inner.lock().map(|g| g.max_bytes).unwrap_or(0)
|
self.lock().max_bytes
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Bind the cache to the dataset at chunk-index address `addr`.
|
// ----- Dataset-keyed operations (safe to use concurrently) -----
|
||||||
|
|
||||||
|
/// The chunk list of the dataset whose chunk index is at `addr`.
|
||||||
///
|
///
|
||||||
/// The cache is shared per file across all of its datasets. If the cache
|
/// On the first call for a dataset, `build` scans its chunk index; the
|
||||||
/// currently holds state for a different dataset, all per-dataset state
|
/// result is kept (offsets truncated to `rank` for the lookup key), so
|
||||||
/// (chunk index, chunk-index map, layout, and decompressed slots) is
|
/// later calls skip the scan. `build` runs without the cache lock held;
|
||||||
/// dropped so the next access rebuilds it for this dataset. Reading the
|
/// if two threads race to build the same dataset's index, the first
|
||||||
/// same dataset again is a no-op, preserving the cache's benefit for
|
/// stored one wins and both return equivalent lists.
|
||||||
/// repeated/sequential access. Returns `true` if a reset occurred.
|
pub fn chunks_for<E>(
|
||||||
pub fn ensure_dataset(&self, addr: u64) -> bool {
|
&self,
|
||||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
addr: u64,
|
||||||
if inner.index_addr == Some(addr) {
|
rank: usize,
|
||||||
return false;
|
build: impl FnOnce() -> Result<Vec<ChunkInfo>, E>,
|
||||||
|
) -> Result<Vec<ChunkInfo>, E> {
|
||||||
|
if let Some(ordered) = self.lock().touch(addr).ordered.clone() {
|
||||||
|
return Ok(ordered.as_ref().clone());
|
||||||
}
|
}
|
||||||
inner.index = None;
|
let chunks = build()?;
|
||||||
inner.chunk_index = None;
|
let map: HashMap<ChunkCoord, ChunkInfo> = chunks
|
||||||
inner.chunk_layout = None;
|
.iter()
|
||||||
inner.slots.clear();
|
.map(|ci| (ci.offsets.iter().take(rank).copied().collect(), ci.clone()))
|
||||||
inner.slot_index.clear();
|
.collect();
|
||||||
inner.current_bytes = 0;
|
let mut inner = self.lock();
|
||||||
inner.last_coord = None;
|
let entry = inner.touch(addr);
|
||||||
inner.index_addr = Some(addr);
|
// Another thread may have built this dataset's index meanwhile: keep
|
||||||
true
|
// the first one, so every reader sees the same order.
|
||||||
|
let ordered = Arc::clone(entry.ordered.get_or_insert_with(|| Arc::new(chunks)));
|
||||||
|
entry.index.get_or_insert_with(|| Arc::new(map));
|
||||||
|
inner.trim_datasets(addr);
|
||||||
|
Ok(ordered.as_ref().clone())
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Returns `true` if the chunk index has been built.
|
fn index_for<E>(
|
||||||
|
&self,
|
||||||
|
addr: u64,
|
||||||
|
rank: usize,
|
||||||
|
build: impl FnOnce() -> Result<Vec<ChunkInfo>, E>,
|
||||||
|
) -> Result<Arc<HashMap<ChunkCoord, ChunkInfo>>, E> {
|
||||||
|
if let Some(index) = self.lock().touch(addr).index.clone() {
|
||||||
|
return Ok(index);
|
||||||
|
}
|
||||||
|
let chunks = build()?;
|
||||||
|
let map: HashMap<ChunkCoord, ChunkInfo> = chunks
|
||||||
|
.iter()
|
||||||
|
.map(|ci| (ci.offsets.iter().take(rank).copied().collect(), ci.clone()))
|
||||||
|
.collect();
|
||||||
|
let mut inner = self.lock();
|
||||||
|
let entry = inner.touch(addr);
|
||||||
|
if entry.index.is_none() {
|
||||||
|
entry.ordered = Some(Arc::new(chunks));
|
||||||
|
}
|
||||||
|
let index = Arc::clone(entry.index.get_or_insert_with(|| Arc::new(map)));
|
||||||
|
inner.trim_datasets(addr);
|
||||||
|
Ok(index)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The pre-computed assembly layout of the dataset at `addr`, building
|
||||||
|
/// its chunk index (via `build`, as in [`Self::chunks_for`]) and layout on
|
||||||
|
/// first use.
|
||||||
|
pub fn chunk_layout_for<E>(
|
||||||
|
&self,
|
||||||
|
addr: u64,
|
||||||
|
rank: usize,
|
||||||
|
build: impl FnOnce() -> Result<Vec<ChunkInfo>, E>,
|
||||||
|
ds_dims: &[usize],
|
||||||
|
chunk_dims: &[usize],
|
||||||
|
elem_size: usize,
|
||||||
|
) -> Result<Arc<ChunkLayout>, E> {
|
||||||
|
let (layout, chunk_index) = {
|
||||||
|
let mut inner = self.lock();
|
||||||
|
let entry = inner.touch(addr);
|
||||||
|
(entry.chunk_layout.clone(), entry.chunk_index.clone())
|
||||||
|
};
|
||||||
|
if let Some(layout) = layout {
|
||||||
|
return Ok(layout);
|
||||||
|
}
|
||||||
|
let chunk_index = match chunk_index {
|
||||||
|
Some(ci) => ci,
|
||||||
|
None => {
|
||||||
|
let index = self.index_for(addr, rank, build)?;
|
||||||
|
let chunks: Vec<ChunkInfo> = index.values().cloned().collect();
|
||||||
|
Arc::new(ChunkIndex::build(&chunks, rank))
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let layout = ChunkLayout::build(&chunk_index, ds_dims, chunk_dims, elem_size);
|
||||||
|
let mut inner = self.lock();
|
||||||
|
let entry = inner.touch(addr);
|
||||||
|
entry.chunk_index.get_or_insert(chunk_index);
|
||||||
|
let layout = Arc::clone(entry.chunk_layout.get_or_insert_with(|| Arc::new(layout)));
|
||||||
|
inner.trim_datasets(addr);
|
||||||
|
Ok(layout)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Cached decompressed chunk at `coord` of the dataset at `addr`.
|
||||||
|
///
|
||||||
|
/// O(1) lookup; the clone is an `Arc` refcount bump, not a copy of the
|
||||||
|
/// underlying decompressed data.
|
||||||
|
pub fn get_decompressed_in(&self, addr: u64, coord: &[u64]) -> Option<Arc<CacheAlignedBuffer>> {
|
||||||
|
self.lock().get_decompressed(addr, coord)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Cache decompressed chunk data for `coord` of the dataset at `addr`.
|
||||||
|
/// Returns the `Arc`-shared buffer now cached (or already cached).
|
||||||
|
pub fn put_decompressed_in(
|
||||||
|
&self,
|
||||||
|
addr: u64,
|
||||||
|
coord: ChunkCoord,
|
||||||
|
data: Vec<u8>,
|
||||||
|
) -> Arc<CacheAlignedBuffer> {
|
||||||
|
self.put_decompressed_aligned_in(addr, coord, CacheAlignedBuffer::from_vec(data))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`Self::put_decompressed_in`] for an already-aligned buffer.
|
||||||
|
pub fn put_decompressed_aligned_in(
|
||||||
|
&self,
|
||||||
|
addr: u64,
|
||||||
|
coord: ChunkCoord,
|
||||||
|
data: CacheAlignedBuffer,
|
||||||
|
) -> Arc<CacheAlignedBuffer> {
|
||||||
|
let data = Arc::new(data);
|
||||||
|
self.lock().put_decompressed((addr, coord), data)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Record that the given chunk coordinates of the dataset at `addr` are
|
||||||
|
/// predicted to be accessed soon (bookkeeping only).
|
||||||
|
///
|
||||||
|
/// This does **not** prefetch or pre-decompress anything — it only
|
||||||
|
/// checks whether each coordinate is already in the chunk index and
|
||||||
|
/// updates access-pattern stats accordingly.
|
||||||
|
pub fn prefetch_hint_in(&self, addr: u64, next_coords: &[ChunkCoord]) {
|
||||||
|
let mut inner = self.lock();
|
||||||
|
let Some(index) = inner.entry(addr).and_then(|e| e.index.clone()) else {
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
let known = next_coords
|
||||||
|
.iter()
|
||||||
|
.filter(|c| index.contains_key(*c))
|
||||||
|
.count();
|
||||||
|
inner.stats.sequential_count += known as u64;
|
||||||
|
}
|
||||||
|
|
||||||
|
// ----- Address-less operations on the bound dataset -----
|
||||||
|
|
||||||
|
/// Bind the address-less methods to the dataset at chunk-index address
|
||||||
|
/// `addr`. Returns `true` if this changed the bound dataset.
|
||||||
|
///
|
||||||
|
/// Each dataset's state is kept separately, so switching loses nothing
|
||||||
|
/// and never exposes one dataset's index or chunks to another. The
|
||||||
|
/// binding itself is shared, though: concurrent readers should use the
|
||||||
|
/// `addr`-taking methods rather than bind and then call these.
|
||||||
|
pub fn ensure_dataset(&self, addr: u64) -> bool {
|
||||||
|
let mut inner = self.lock();
|
||||||
|
let changed = inner.current != Some(addr);
|
||||||
|
inner.current = Some(addr);
|
||||||
|
changed
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Returns `true` if the bound dataset's chunk index has been built.
|
||||||
pub fn has_index(&self) -> bool {
|
pub fn has_index(&self) -> bool {
|
||||||
self.inner
|
let inner = self.lock();
|
||||||
.lock()
|
inner
|
||||||
.unwrap_or_else(|e| e.into_inner())
|
.entry(inner.current())
|
||||||
.index
|
.is_some_and(|e| e.index.is_some())
|
||||||
.is_some()
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Build the chunk index from a pre-collected list of `ChunkInfo`.
|
/// Build the bound dataset's chunk index from a pre-collected list of
|
||||||
|
/// `ChunkInfo`.
|
||||||
///
|
///
|
||||||
/// The `rank` parameter is used to truncate offsets to spatial dims only
|
/// The `rank` parameter is used to truncate offsets to spatial dims only
|
||||||
/// (B-tree v1 stores rank+1 offsets).
|
/// (B-tree v1 stores rank+1 offsets).
|
||||||
pub fn populate_index(&self, chunks: &[ChunkInfo], rank: usize) {
|
pub fn populate_index(&self, chunks: &[ChunkInfo], rank: usize) {
|
||||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let addr = self.lock().current();
|
||||||
if inner.index.is_some() {
|
let _ = self.index_for::<core::convert::Infallible>(addr, rank, || Ok(chunks.to_vec()));
|
||||||
return; // already populated
|
|
||||||
}
|
|
||||||
let mut map = HashMap::with_capacity(chunks.len());
|
|
||||||
|
|
||||||
for ci in chunks {
|
|
||||||
let coord: ChunkCoord = ci.offsets.iter().take(rank).copied().collect();
|
|
||||||
map.insert(coord, ci.clone());
|
|
||||||
}
|
|
||||||
inner.index = Some(map);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Look up a chunk by its spatial coordinate in the index.
|
/// Look up a chunk by its spatial coordinate in the bound dataset's index.
|
||||||
pub fn lookup_index(&self, coord: &[u64]) -> Option<ChunkInfo> {
|
pub fn lookup_index(&self, coord: &[u64]) -> Option<ChunkInfo> {
|
||||||
let inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let inner = self.lock();
|
||||||
inner.index.as_ref()?.get(coord).cloned()
|
inner
|
||||||
|
.entry(inner.current())?
|
||||||
|
.index
|
||||||
|
.as_ref()?
|
||||||
|
.get(coord)
|
||||||
|
.cloned()
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Return all indexed chunks as a `Vec<ChunkInfo>` (order unspecified).
|
/// Return all of the bound dataset's indexed chunks (order unspecified).
|
||||||
pub fn all_indexed_chunks(&self) -> Option<Vec<ChunkInfo>> {
|
pub fn all_indexed_chunks(&self) -> Option<Vec<ChunkInfo>> {
|
||||||
let inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let inner = self.lock();
|
||||||
inner.index.as_ref().map(|m| m.values().cloned().collect())
|
let index = inner.entry(inner.current())?.index.as_ref()?;
|
||||||
|
Some(index.values().cloned().collect())
|
||||||
}
|
}
|
||||||
|
|
||||||
// ----- Chunk index (pre-built coordinate → ChunkInfo map) -----
|
/// Returns `true` if the bound dataset's `ChunkIndex` has been built.
|
||||||
|
|
||||||
/// Returns `true` if the chunk B-tree index has been built.
|
|
||||||
pub fn has_chunk_index(&self) -> bool {
|
pub fn has_chunk_index(&self) -> bool {
|
||||||
self.inner
|
let inner = self.lock();
|
||||||
.lock()
|
inner
|
||||||
.unwrap_or_else(|e| e.into_inner())
|
.entry(inner.current())
|
||||||
.chunk_index
|
.is_some_and(|e| e.chunk_index.is_some())
|
||||||
.is_some()
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Build and store the chunk B-tree index from a pre-collected list of `ChunkInfo`.
|
/// Build and store the bound dataset's `ChunkIndex`.
|
||||||
pub fn populate_chunk_index(&self, chunks: &[ChunkInfo], rank: usize) {
|
pub fn populate_chunk_index(&self, chunks: &[ChunkInfo], rank: usize) {
|
||||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let built = Arc::new(ChunkIndex::build(chunks, rank));
|
||||||
if inner.chunk_index.is_some() {
|
let mut inner = self.lock();
|
||||||
return;
|
let addr = inner.current();
|
||||||
}
|
inner.touch(addr).chunk_index.get_or_insert(built);
|
||||||
inner.chunk_index = Some(ChunkIndex::build(chunks, rank));
|
inner.trim_datasets(addr);
|
||||||
}
|
}
|
||||||
|
|
||||||
// ----- Chunk layout (pre-computed assembly plan) -----
|
/// Returns `true` if the bound dataset's chunk layout has been computed.
|
||||||
|
|
||||||
/// Returns `true` if the chunk layout has been computed.
|
|
||||||
pub fn has_chunk_layout(&self) -> bool {
|
pub fn has_chunk_layout(&self) -> bool {
|
||||||
self.inner
|
let inner = self.lock();
|
||||||
.lock()
|
inner
|
||||||
.unwrap_or_else(|e| e.into_inner())
|
.entry(inner.current())
|
||||||
.chunk_layout
|
.is_some_and(|e| e.chunk_layout.is_some())
|
||||||
.is_some()
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Build and store the pre-computed chunk layout for fast assembly.
|
/// Build and store the bound dataset's chunk layout (needs its
|
||||||
|
/// `ChunkIndex`; does nothing without one).
|
||||||
pub fn populate_chunk_layout(&self, ds_dims: &[usize], chunk_dims: &[usize], elem_size: usize) {
|
pub fn populate_chunk_layout(&self, ds_dims: &[usize], chunk_dims: &[usize], elem_size: usize) {
|
||||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let mut inner = self.lock();
|
||||||
if inner.chunk_layout.is_some() {
|
let addr = inner.current();
|
||||||
|
let entry = inner.touch(addr);
|
||||||
|
if entry.chunk_layout.is_some() {
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
if let Some(ref idx) = inner.chunk_index {
|
if let Some(idx) = entry.chunk_index.clone() {
|
||||||
inner.chunk_layout = Some(ChunkLayout::build(idx, ds_dims, chunk_dims, elem_size));
|
entry.chunk_layout = Some(Arc::new(ChunkLayout::build(
|
||||||
|
&idx, ds_dims, chunk_dims, elem_size,
|
||||||
|
)));
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Execute a function with a reference to the chunk layout.
|
/// Execute a function with a reference to the bound dataset's chunk
|
||||||
///
|
/// layout. Returns `None` if the layout hasn't been computed yet.
|
||||||
/// Returns `None` if the layout hasn't been computed yet.
|
|
||||||
pub fn with_chunk_layout<F, R>(&self, f: F) -> Option<R>
|
pub fn with_chunk_layout<F, R>(&self, f: F) -> Option<R>
|
||||||
where
|
where
|
||||||
F: FnOnce(&ChunkLayout) -> R,
|
F: FnOnce(&ChunkLayout) -> R,
|
||||||
{
|
{
|
||||||
let inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let layout = {
|
||||||
inner.chunk_layout.as_ref().map(f)
|
let inner = self.lock();
|
||||||
|
inner.entry(inner.current())?.chunk_layout.clone()?
|
||||||
|
};
|
||||||
|
Some(f(&layout))
|
||||||
}
|
}
|
||||||
|
|
||||||
// ----- Decompressed data cache (LRU) -----
|
/// Try to get cached decompressed data for a chunk of the bound dataset.
|
||||||
|
|
||||||
/// Try to get cached decompressed data for a chunk coordinate.
|
|
||||||
///
|
///
|
||||||
/// O(1) lookup. Returns an owned copy for API compatibility with callers
|
/// Returns an owned copy; prefer [`Self::get_decompressed_aligned`] when
|
||||||
/// that need a `Vec<u8>`; prefer [`Self::get_decompressed_aligned`] when
|
/// an `Arc`-shared buffer works for the caller.
|
||||||
/// an `Arc`-shared buffer works for the caller, since that avoids the
|
|
||||||
/// copy entirely.
|
|
||||||
pub fn get_decompressed(&self, coord: &[u64]) -> Option<Vec<u8>> {
|
pub fn get_decompressed(&self, coord: &[u64]) -> Option<Vec<u8>> {
|
||||||
self.get_decompressed_aligned(coord)
|
self.get_decompressed_aligned(coord)
|
||||||
.map(|arc| arc.as_slice().to_vec())
|
.map(|arc| arc.as_slice().to_vec())
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Try to get a reference-counted clone of the aligned buffer for a chunk.
|
/// Reference-counted cached buffer for a chunk of the bound dataset.
|
||||||
///
|
|
||||||
/// O(1) index lookup; the clone is an `Arc` refcount bump, not a copy of
|
|
||||||
/// the underlying decompressed data.
|
|
||||||
pub fn get_decompressed_aligned(&self, coord: &[u64]) -> Option<Arc<CacheAlignedBuffer>> {
|
pub fn get_decompressed_aligned(&self, coord: &[u64]) -> Option<Arc<CacheAlignedBuffer>> {
|
||||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let mut inner = self.lock();
|
||||||
inner.tick += 1;
|
let addr = inner.current();
|
||||||
let tick = inner.tick;
|
inner.get_decompressed(addr, coord)
|
||||||
|
|
||||||
// Track sequential vs random access
|
|
||||||
let is_sequential = inner.last_coord.as_ref().is_some_and(|prev| {
|
|
||||||
// Sequential if exactly one dimension changed
|
|
||||||
let changes: usize = prev
|
|
||||||
.iter()
|
|
||||||
.zip(coord.iter())
|
|
||||||
.filter(|(a, b)| a != b)
|
|
||||||
.count();
|
|
||||||
changes <= 1
|
|
||||||
});
|
|
||||||
if is_sequential {
|
|
||||||
inner.stats.sequential_count += 1;
|
|
||||||
} else if inner.last_coord.is_some() {
|
|
||||||
inner.stats.random_count += 1;
|
|
||||||
}
|
|
||||||
inner.last_coord = Some(coord.to_vec());
|
|
||||||
|
|
||||||
let found = if let Some(&idx) = inner.slot_index.get(coord) {
|
|
||||||
inner.slots[idx].last_access = tick;
|
|
||||||
Some(Arc::clone(&inner.slots[idx].data))
|
|
||||||
} else {
|
|
||||||
None
|
|
||||||
};
|
|
||||||
if let Some(ref data) = found {
|
|
||||||
inner.stats.hits += 1;
|
|
||||||
inner.stats.bytes_read += data.len() as u64;
|
|
||||||
} else {
|
|
||||||
inner.stats.misses += 1;
|
|
||||||
}
|
|
||||||
found
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Insert decompressed chunk data into the LRU cache.
|
/// Insert decompressed chunk data for the bound dataset into the LRU
|
||||||
///
|
/// cache, returning the `Arc`-shared buffer now cached.
|
||||||
/// The data is stored in a [`CacheAlignedBuffer`] so subsequent reads
|
|
||||||
/// return cache-line-aligned memory. Returns the `Arc`-shared buffer that
|
|
||||||
/// is now cached (or already was), so the caller can reuse it directly
|
|
||||||
/// instead of holding a separate copy of the same data.
|
|
||||||
pub fn put_decompressed(&self, coord: ChunkCoord, data: Vec<u8>) -> Arc<CacheAlignedBuffer> {
|
pub fn put_decompressed(&self, coord: ChunkCoord, data: Vec<u8>) -> Arc<CacheAlignedBuffer> {
|
||||||
let aligned = CacheAlignedBuffer::from_vec(data);
|
self.put_decompressed_aligned(coord, CacheAlignedBuffer::from_vec(data))
|
||||||
self.put_decompressed_aligned(coord, aligned)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Insert an already-aligned buffer into the LRU cache.
|
/// Insert an already-aligned buffer for the bound dataset.
|
||||||
///
|
|
||||||
/// Returns the `Arc`-shared buffer now held by the cache (the one just
|
|
||||||
/// inserted, or the existing cached copy if `coord` was already present).
|
|
||||||
pub fn put_decompressed_aligned(
|
pub fn put_decompressed_aligned(
|
||||||
&self,
|
&self,
|
||||||
coord: ChunkCoord,
|
coord: ChunkCoord,
|
||||||
data: CacheAlignedBuffer,
|
data: CacheAlignedBuffer,
|
||||||
) -> Arc<CacheAlignedBuffer> {
|
) -> Arc<CacheAlignedBuffer> {
|
||||||
let data = Arc::new(data);
|
let data = Arc::new(data);
|
||||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let mut inner = self.lock();
|
||||||
let data_len = data.len();
|
let addr = inner.current();
|
||||||
|
inner.put_decompressed((addr, coord), data)
|
||||||
// Don't cache if single chunk exceeds budget — still return the data
|
|
||||||
// to the caller, just don't retain it.
|
|
||||||
if data_len > inner.max_bytes {
|
|
||||||
return data;
|
|
||||||
}
|
|
||||||
|
|
||||||
// Check if already present
|
|
||||||
inner.tick += 1;
|
|
||||||
let tick = inner.tick;
|
|
||||||
if let Some(&idx) = inner.slot_index.get(&coord) {
|
|
||||||
inner.slots[idx].last_access = tick;
|
|
||||||
return Arc::clone(&inner.slots[idx].data); // already cached
|
|
||||||
}
|
|
||||||
|
|
||||||
// Evict until we have room
|
|
||||||
while inner.slots.len() >= inner.max_slots
|
|
||||||
|| (inner.current_bytes + data_len > inner.max_bytes && !inner.slots.is_empty())
|
|
||||||
{
|
|
||||||
// Find LRU slot
|
|
||||||
let lru_idx = inner
|
|
||||||
.slots
|
|
||||||
.iter()
|
|
||||||
.enumerate()
|
|
||||||
.min_by_key(|(_, s)| s.last_access)
|
|
||||||
.map(|(i, _)| i)
|
|
||||||
.unwrap();
|
|
||||||
let removed = inner.slots.swap_remove(lru_idx);
|
|
||||||
inner.slot_index.remove(&removed.coord);
|
|
||||||
// swap_remove moved the former last element into `lru_idx` (unless
|
|
||||||
// it *was* the last element) — fix up that element's index entry.
|
|
||||||
if lru_idx < inner.slots.len() {
|
|
||||||
let moved_coord = inner.slots[lru_idx].coord.clone();
|
|
||||||
inner.slot_index.insert(moved_coord, lru_idx);
|
|
||||||
}
|
|
||||||
inner.current_bytes -= removed.data.len();
|
|
||||||
inner.stats.evictions += 1;
|
|
||||||
}
|
|
||||||
|
|
||||||
inner.current_bytes += data_len;
|
|
||||||
let new_idx = inner.slots.len();
|
|
||||||
inner.slot_index.insert(coord.clone(), new_idx);
|
|
||||||
inner.slots.push(CachedChunk {
|
|
||||||
coord,
|
|
||||||
data: Arc::clone(&data),
|
|
||||||
last_access: tick,
|
|
||||||
});
|
|
||||||
data
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Clear the entire cache (index + decompressed data).
|
/// [`Self::prefetch_hint_in`] for the bound dataset.
|
||||||
|
pub fn prefetch_hint(&self, next_coords: &[ChunkCoord]) {
|
||||||
|
let addr = self.lock().current();
|
||||||
|
self.prefetch_hint_in(addr, next_coords);
|
||||||
|
}
|
||||||
|
|
||||||
|
// ----- Whole-cache operations -----
|
||||||
|
|
||||||
|
/// Clear the entire cache (indexes + decompressed data + stats).
|
||||||
pub fn clear(&self) {
|
pub fn clear(&self) {
|
||||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let mut inner = self.lock();
|
||||||
inner.index = None;
|
inner.datasets.clear();
|
||||||
inner.index_addr = None;
|
inner.current = None;
|
||||||
inner.slots.clear();
|
inner.slots.clear();
|
||||||
inner.slot_index.clear();
|
inner.slot_index.clear();
|
||||||
inner.current_bytes = 0;
|
inner.current_bytes = 0;
|
||||||
inner.tick = 0;
|
inner.tick = 0;
|
||||||
inner.last_coord = None;
|
inner.last_coord = None;
|
||||||
inner.stats = AccessStats::default();
|
inner.stats = AccessStats::default();
|
||||||
inner.chunk_index = None;
|
|
||||||
inner.chunk_layout = None;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Record that the given chunk coordinates are predicted to be accessed
|
|
||||||
/// soon (bookkeeping only).
|
|
||||||
///
|
|
||||||
/// This does **not** prefetch or pre-decompress anything — it only
|
|
||||||
/// checks whether each coordinate is already in the chunk index and
|
|
||||||
/// updates access-pattern stats accordingly. Real prefetching (e.g.
|
|
||||||
/// background pre-decompression) is not implemented.
|
|
||||||
pub fn prefetch_hint(&self, next_coords: &[ChunkCoord]) {
|
|
||||||
let inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
|
||||||
if inner.index.is_none() {
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
drop(inner);
|
|
||||||
// For each predicted coordinate, verify it exists in the index.
|
|
||||||
// The index is already populated, so this is a no-op for known chunks.
|
|
||||||
// The purpose is to signal intent — callers can pre-decompress if needed.
|
|
||||||
// We touch the stats to record that prefetch hints were issued.
|
|
||||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
|
||||||
for coord in next_coords {
|
|
||||||
let exists = inner
|
|
||||||
.index
|
|
||||||
.as_ref()
|
|
||||||
.map(|idx| idx.contains_key(coord))
|
|
||||||
.unwrap_or(false);
|
|
||||||
if exists {
|
|
||||||
inner.stats.sequential_count += 1;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Return the current access pattern statistics.
|
/// Return the current access pattern statistics.
|
||||||
pub fn access_stats(&self) -> AccessStats {
|
pub fn access_stats(&self) -> AccessStats {
|
||||||
self.inner
|
self.lock().stats.clone()
|
||||||
.lock()
|
|
||||||
.unwrap_or_else(|e| e.into_inner())
|
|
||||||
.stats
|
|
||||||
.clone()
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Update the sweep direction label in the access stats.
|
/// Update the sweep direction label in the access stats.
|
||||||
pub fn set_sweep_direction(&self, direction: &'static str) {
|
pub fn set_sweep_direction(&self, direction: &'static str) {
|
||||||
self.inner
|
self.lock().stats.sweep_direction = Some(direction);
|
||||||
.lock()
|
|
||||||
.unwrap_or_else(|e| e.into_inner())
|
|
||||||
.stats
|
|
||||||
.sweep_direction = Some(direction);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Number of decompressed chunks currently cached.
|
/// Number of decompressed chunks currently cached (all datasets).
|
||||||
pub fn cached_chunk_count(&self) -> usize {
|
pub fn cached_chunk_count(&self) -> usize {
|
||||||
self.inner
|
self.lock().slots.len()
|
||||||
.lock()
|
|
||||||
.unwrap_or_else(|e| e.into_inner())
|
|
||||||
.slots
|
|
||||||
.len()
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Total bytes of decompressed data currently cached.
|
/// Total bytes of decompressed data currently cached (all datasets).
|
||||||
pub fn cached_bytes(&self) -> usize {
|
pub fn cached_bytes(&self) -> usize {
|
||||||
self.inner
|
self.lock().current_bytes
|
||||||
.lock()
|
}
|
||||||
.unwrap_or_else(|e| e.into_inner())
|
|
||||||
.current_bytes
|
/// Number of datasets whose chunk index is currently kept.
|
||||||
|
pub fn indexed_dataset_count(&self) -> usize {
|
||||||
|
self.lock().datasets.len()
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -808,6 +987,92 @@ mod tests {
|
|||||||
assert_eq!(cache.cached_bytes(), 0);
|
assert_eq!(cache.cached_bytes(), 0);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn datasets_sharing_coordinates_stay_separate() {
|
||||||
|
let cache = ChunkCache::new();
|
||||||
|
let a = vec![make_chunk(vec![0, 0], 0x100, 8)];
|
||||||
|
let b = vec![make_chunk(vec![0, 0], 0x900, 8)];
|
||||||
|
let got_a = cache.chunks_for::<()>(1, 1, || Ok(a.clone())).unwrap();
|
||||||
|
let got_b = cache.chunks_for::<()>(2, 1, || Ok(b.clone())).unwrap();
|
||||||
|
assert_eq!(got_a[0].address, 0x100);
|
||||||
|
assert_eq!(got_b[0].address, 0x900);
|
||||||
|
// Built once per dataset: a second lookup doesn't call the builder.
|
||||||
|
let again = cache
|
||||||
|
.chunks_for::<()>(1, 1, || panic!("index rebuilt"))
|
||||||
|
.unwrap();
|
||||||
|
assert_eq!(again[0].address, 0x100);
|
||||||
|
|
||||||
|
cache.put_decompressed_in(1, vec![0], vec![1; 4]);
|
||||||
|
cache.put_decompressed_in(2, vec![0], vec![2; 4]);
|
||||||
|
assert_eq!(
|
||||||
|
cache.get_decompressed_in(1, &[0]).unwrap().as_slice(),
|
||||||
|
&[1; 4]
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
cache.get_decompressed_in(2, &[0]).unwrap().as_slice(),
|
||||||
|
&[2; 4]
|
||||||
|
);
|
||||||
|
assert!(cache.get_decompressed_in(3, &[0]).is_none());
|
||||||
|
assert_eq!(cache.cached_chunk_count(), 2);
|
||||||
|
|
||||||
|
// The bound-dataset methods see only the bound dataset.
|
||||||
|
cache.ensure_dataset(2);
|
||||||
|
assert_eq!(cache.lookup_index(&[0]).unwrap().address, 0x900);
|
||||||
|
assert_eq!(cache.get_decompressed(&[0]).unwrap(), vec![2; 4]);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn dataset_indexes_are_bounded() {
|
||||||
|
let cache = ChunkCache::new();
|
||||||
|
for addr in 0..(MAX_INDEXED_DATASETS as u64 + 10) {
|
||||||
|
cache
|
||||||
|
.chunks_for::<()>(addr, 1, || Ok(vec![make_chunk(vec![0], addr, 8)]))
|
||||||
|
.unwrap();
|
||||||
|
}
|
||||||
|
assert_eq!(cache.indexed_dataset_count(), MAX_INDEXED_DATASETS);
|
||||||
|
|
||||||
|
// One huge index evicts the others but is itself kept.
|
||||||
|
let huge: Vec<ChunkInfo> = (0..MAX_INDEXED_CHUNKS as u64)
|
||||||
|
.map(|i| make_chunk(vec![i], i, 8))
|
||||||
|
.collect();
|
||||||
|
let got = cache.chunks_for::<()>(9999, 1, || Ok(huge)).unwrap();
|
||||||
|
assert_eq!(got.len(), MAX_INDEXED_CHUNKS);
|
||||||
|
assert_eq!(cache.indexed_dataset_count(), 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn concurrent_readers_of_different_datasets_see_their_own_chunks() {
|
||||||
|
let cache = std::sync::Arc::new(ChunkCache::with_capacity(1 << 20, 64));
|
||||||
|
let handles: Vec<_> = (0..8u64)
|
||||||
|
.map(|t| {
|
||||||
|
let cache = std::sync::Arc::clone(&cache);
|
||||||
|
std::thread::spawn(move || {
|
||||||
|
for round in 0..500u64 {
|
||||||
|
let addr = (t + round) % 16;
|
||||||
|
let coord = vec![round % 4];
|
||||||
|
let chunks = cache
|
||||||
|
.chunks_for::<()>(addr, 1, || {
|
||||||
|
Ok((0..4).map(|c| make_chunk(vec![c], addr, 8)).collect())
|
||||||
|
})
|
||||||
|
.unwrap();
|
||||||
|
assert!(chunks.iter().all(|c| c.address == addr));
|
||||||
|
let want = vec![addr as u8; 8];
|
||||||
|
let got = match cache.get_decompressed_in(addr, &coord) {
|
||||||
|
Some(hit) => hit.to_vec(),
|
||||||
|
None => cache
|
||||||
|
.put_decompressed_in(addr, coord, want.clone())
|
||||||
|
.to_vec(),
|
||||||
|
};
|
||||||
|
assert_eq!(got, want);
|
||||||
|
}
|
||||||
|
})
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
for h in handles {
|
||||||
|
h.join().unwrap();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn duplicate_insert_is_noop() {
|
fn duplicate_insert_is_noop() {
|
||||||
let cache = ChunkCache::new();
|
let cache = ChunkCache::new();
|
||||||
|
|||||||
@@ -0,0 +1,228 @@
|
|||||||
|
//! Chunk-index linearisation shared by the Fixed Array and Extensible Array
|
||||||
|
//! chunk indexes (reader and writer).
|
||||||
|
//!
|
||||||
|
//! Both indexes store one element per chunk at a *linear* index, and the
|
||||||
|
//! library derives that index from the chunk's scaled coordinates
|
||||||
|
//! (`offset / chunk_dim`) using the dataset's **maximum** dimensions, not its
|
||||||
|
//! current ones (`H5D__farray_idx_get_addr` / `H5D__earray_idx_get_addr`,
|
||||||
|
//! via `layout->max_down_chunks`). A dataset whose current shape is smaller
|
||||||
|
//! than its maxshape therefore has gaps in the index, and laying it out by the
|
||||||
|
//! current shape puts every chunk after the first row in the wrong place.
|
||||||
|
//!
|
||||||
|
//! The Extensible Array adds one more step: its one unlimited dimension has no
|
||||||
|
//! finite chunk count, so the library *swizzles* the coordinates to make that
|
||||||
|
//! dimension the slowest-varying one (`H5VM_swizzle_coords`, which moves
|
||||||
|
//! `coords[unlim_dim]` to the front and shifts the dimensions before it right
|
||||||
|
//! by one) before linearising with `swizzled_max_down_chunks`. When the
|
||||||
|
//! unlimited dimension is already dimension 0 no swizzle happens.
|
||||||
|
|
||||||
|
#[cfg(not(feature = "std"))]
|
||||||
|
extern crate alloc;
|
||||||
|
|
||||||
|
#[cfg(not(feature = "std"))]
|
||||||
|
use alloc::{vec, vec::Vec};
|
||||||
|
|
||||||
|
use crate::error::FormatError;
|
||||||
|
|
||||||
|
/// How a chunk index maps linear element indexes to chunk coordinates.
|
||||||
|
#[derive(Debug, Clone)]
|
||||||
|
pub(crate) struct ChunkGrid {
|
||||||
|
/// Spatial chunk dimensions, in dataset order.
|
||||||
|
chunk_dims: Vec<u64>,
|
||||||
|
/// Chunks per dimension covering the *current* extent, in dataset order.
|
||||||
|
cur_chunks: Vec<u64>,
|
||||||
|
/// Dataset dimension stored at each linearisation position (slowest
|
||||||
|
/// first). The identity except for a swizzled Extensible Array.
|
||||||
|
order: Vec<usize>,
|
||||||
|
/// Linear stride of each linearisation position.
|
||||||
|
down: Vec<u64>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl ChunkGrid {
|
||||||
|
/// Grid for a Fixed Array index: row-major over the chunk counts of the
|
||||||
|
/// maximum dimensions (`max_dims`, falling back to the current dimensions
|
||||||
|
/// when the dataspace records none).
|
||||||
|
pub(crate) fn fixed_array(
|
||||||
|
cur_dims: &[u64],
|
||||||
|
max_dims: Option<&[u64]>,
|
||||||
|
chunk_dims: &[u64],
|
||||||
|
) -> Result<Self, FormatError> {
|
||||||
|
Self::build(cur_dims, max_dims, chunk_dims, None)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Grid for an Extensible Array index: like the Fixed Array, but the
|
||||||
|
/// unlimited dimension (the one whose maximum is `H5S_UNLIMITED`) is moved
|
||||||
|
/// to the slowest-varying position first.
|
||||||
|
pub(crate) fn extensible_array(
|
||||||
|
cur_dims: &[u64],
|
||||||
|
max_dims: Option<&[u64]>,
|
||||||
|
chunk_dims: &[u64],
|
||||||
|
) -> Result<Self, FormatError> {
|
||||||
|
let unlim = max_dims.and_then(|m| m.iter().position(|&d| d == u64::MAX));
|
||||||
|
Self::build(cur_dims, max_dims, chunk_dims, unlim)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn build(
|
||||||
|
cur_dims: &[u64],
|
||||||
|
max_dims: Option<&[u64]>,
|
||||||
|
chunk_dims: &[u64],
|
||||||
|
unlim: Option<usize>,
|
||||||
|
) -> Result<Self, FormatError> {
|
||||||
|
let rank = chunk_dims.len();
|
||||||
|
if cur_dims.len() != rank || max_dims.is_some_and(|m| m.len() != rank) {
|
||||||
|
return Err(FormatError::ChunkedReadError(
|
||||||
|
"chunk index rank does not match the dataspace".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
if chunk_dims.contains(&0) {
|
||||||
|
return Err(FormatError::ChunkedReadError(
|
||||||
|
"chunk dimension is zero".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
let cur_chunks: Vec<u64> = cur_dims
|
||||||
|
.iter()
|
||||||
|
.zip(chunk_dims)
|
||||||
|
.map(|(&d, &c)| d.div_ceil(c))
|
||||||
|
.collect();
|
||||||
|
// Chunk counts of the maximum extent. An unlimited dimension has no
|
||||||
|
// finite count; it only ever sits in the slowest position, where its
|
||||||
|
// count never enters a stride. A (corrupt) maximum smaller than the
|
||||||
|
// current extent is widened so no allocated chunk becomes unreachable.
|
||||||
|
let max_chunks: Vec<u64> = (0..rank)
|
||||||
|
.map(|d| {
|
||||||
|
let max = max_dims.map_or(cur_dims[d], |m| m[d]);
|
||||||
|
if max == u64::MAX {
|
||||||
|
u64::MAX
|
||||||
|
} else {
|
||||||
|
max.div_ceil(chunk_dims[d]).max(cur_chunks[d])
|
||||||
|
}
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
|
||||||
|
let mut order: Vec<usize> = (0..rank).collect();
|
||||||
|
if let Some(u) = unlim {
|
||||||
|
order.remove(u);
|
||||||
|
order.insert(0, u);
|
||||||
|
}
|
||||||
|
let mut down = vec![1u64; rank];
|
||||||
|
for p in (0..rank.saturating_sub(1)).rev() {
|
||||||
|
let next = max_chunks[order[p + 1]];
|
||||||
|
if next == u64::MAX {
|
||||||
|
// Only reachable with more than one unlimited dimension, which
|
||||||
|
// neither index type can describe.
|
||||||
|
return Err(FormatError::ChunkedReadError(
|
||||||
|
"array chunk index with more than one unlimited dimension".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
down[p] = down[p + 1].checked_mul(next).ok_or_else(|| {
|
||||||
|
FormatError::Overflow("chunk index linear stride overflows u64".into())
|
||||||
|
})?;
|
||||||
|
}
|
||||||
|
Ok(Self {
|
||||||
|
chunk_dims: chunk_dims.to_vec(),
|
||||||
|
cur_chunks,
|
||||||
|
order,
|
||||||
|
down,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Dataset-space offsets of the chunk stored at linear `index`, or `None`
|
||||||
|
/// when that chunk lies outside the current extent (the index still has a
|
||||||
|
/// slot for it; the library ignores such chunks on read).
|
||||||
|
pub(crate) fn offsets(&self, index: u64) -> Option<Vec<u64>> {
|
||||||
|
let rank = self.chunk_dims.len();
|
||||||
|
let mut offsets = vec![0u64; rank];
|
||||||
|
let mut rem = index;
|
||||||
|
for p in 0..rank {
|
||||||
|
let d = self.order[p];
|
||||||
|
// A zero stride: a later dimension has no chunks (its maximum,
|
||||||
|
// or with none recorded its current extent, is 0), so no slot of
|
||||||
|
// the index is a chunk of the dataset.
|
||||||
|
if self.down[p] == 0 {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
let scaled = rem / self.down[p];
|
||||||
|
rem %= self.down[p];
|
||||||
|
if scaled >= self.cur_chunks[d] {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
offsets[d] = scaled * self.chunk_dims[d];
|
||||||
|
}
|
||||||
|
Some(offsets)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Linear index of the chunk with scaled coordinates `scaled`
|
||||||
|
/// (`offset / chunk_dim` per dimension, in dataset order).
|
||||||
|
pub(crate) fn linear_index(&self, scaled: &[u64]) -> u64 {
|
||||||
|
self.order
|
||||||
|
.iter()
|
||||||
|
.zip(&self.down)
|
||||||
|
.map(|(&d, &stride)| scaled[d] * stride)
|
||||||
|
.sum()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn fixed_array_uses_max_dims() {
|
||||||
|
// shape (4, 6), chunks (2, 3), maxshape (20, 10): 10 x 4 chunk grid.
|
||||||
|
let g = ChunkGrid::fixed_array(&[4, 6], Some(&[20, 10]), &[2, 3]).unwrap();
|
||||||
|
assert_eq!(g.offsets(0), Some(vec![0, 0]));
|
||||||
|
assert_eq!(g.offsets(1), Some(vec![0, 3]));
|
||||||
|
assert_eq!(g.offsets(2), None); // column chunk 2 is beyond the extent
|
||||||
|
assert_eq!(g.offsets(4), Some(vec![2, 0]));
|
||||||
|
assert_eq!(g.offsets(5), Some(vec![2, 3]));
|
||||||
|
assert_eq!(g.offsets(8), None); // row chunk 2 is beyond the extent
|
||||||
|
assert_eq!(g.linear_index(&[1, 1]), 5);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn extensible_array_swizzles_unlimited_dim() {
|
||||||
|
// maxshape (10, None): dim 1 is unlimited and becomes slowest.
|
||||||
|
let g = ChunkGrid::extensible_array(&[4, 6], Some(&[10, u64::MAX]), &[2, 3]).unwrap();
|
||||||
|
// max chunks of dim 0 = 5, so index = c1 * 5 + c0.
|
||||||
|
assert_eq!(g.linear_index(&[1, 0]), 1);
|
||||||
|
assert_eq!(g.linear_index(&[0, 1]), 5);
|
||||||
|
assert_eq!(g.offsets(5), Some(vec![0, 3]));
|
||||||
|
assert_eq!(g.offsets(6), Some(vec![2, 3]));
|
||||||
|
assert_eq!(g.offsets(2), None);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn extensible_array_unlimited_first_is_row_major() {
|
||||||
|
let g = ChunkGrid::extensible_array(&[4, 6], Some(&[u64::MAX, 30]), &[2, 3]).unwrap();
|
||||||
|
// max chunks of dim 1 = 10.
|
||||||
|
assert_eq!(g.linear_index(&[1, 1]), 11);
|
||||||
|
assert_eq!(g.offsets(11), Some(vec![2, 3]));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn zero_extent_has_no_chunks() {
|
||||||
|
// No maximum recorded and a zero current dimension: every stride
|
||||||
|
// before it is 0 (this divided by zero).
|
||||||
|
let g = ChunkGrid::fixed_array(&[1, 0], None, &[6, 6]).unwrap();
|
||||||
|
for i in 0..16 {
|
||||||
|
assert_eq!(g.offsets(i), None);
|
||||||
|
}
|
||||||
|
let g = ChunkGrid::fixed_array(&[0, 0, 3], Some(&[4, 0, 3]), &[2, 2, 3]).unwrap();
|
||||||
|
for i in 0..16 {
|
||||||
|
assert_eq!(g.offsets(i), None);
|
||||||
|
}
|
||||||
|
let g = ChunkGrid::extensible_array(&[0, 5], Some(&[u64::MAX, 0]), &[2, 2]).unwrap();
|
||||||
|
for i in 0..16 {
|
||||||
|
assert_eq!(g.offsets(i), None);
|
||||||
|
}
|
||||||
|
// A zero last dimension leaves the other strides alone.
|
||||||
|
let g = ChunkGrid::fixed_array(&[4, 0], Some(&[4, 6]), &[2, 3]).unwrap();
|
||||||
|
assert_eq!(g.offsets(0), None);
|
||||||
|
assert_eq!(g.linear_index(&[1, 1]), 3);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn rejects_two_unlimited_dims_after_the_first() {
|
||||||
|
assert!(ChunkGrid::fixed_array(&[4, 6], Some(&[u64::MAX, u64::MAX]), &[2, 3]).is_err());
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -18,6 +18,7 @@ use alloc::collections::BTreeMap;
|
|||||||
#[cfg(feature = "std")]
|
#[cfg(feature = "std")]
|
||||||
use std::collections::HashMap;
|
use std::collections::HashMap;
|
||||||
|
|
||||||
|
use crate::addr::to_usize;
|
||||||
use crate::chunk_cache::ChunkCoord;
|
use crate::chunk_cache::ChunkCoord;
|
||||||
use crate::chunked_read::ChunkInfo;
|
use crate::chunked_read::ChunkInfo;
|
||||||
|
|
||||||
@@ -167,7 +168,15 @@ impl ChunkLayout {
|
|||||||
|
|
||||||
for (_coord, ci) in index.iter() {
|
for (_coord, ci) in index.iter() {
|
||||||
let coord: ChunkCoord = ci.offsets.iter().take(rank).copied().collect();
|
let coord: ChunkCoord = ci.offsets.iter().take(rank).copied().collect();
|
||||||
let chunk_offsets: Vec<usize> = coord.iter().map(|&o| o as usize).collect();
|
// `ds_dims` are `usize`: a chunk at an offset past `usize::MAX`
|
||||||
|
// (only on a 32-bit target) lies outside the dataset.
|
||||||
|
let Ok(chunk_offsets) = coord
|
||||||
|
.iter()
|
||||||
|
.map(|&o| to_usize(o))
|
||||||
|
.collect::<Result<Vec<usize>, _>>()
|
||||||
|
else {
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
|
||||||
let copies = if rank == 0 {
|
let copies = if rank == 0 {
|
||||||
// Scalar dataset — single copy
|
// Scalar dataset — single copy
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -1,12 +1,14 @@
|
|||||||
//! HDF5 Data Layout message parsing (message type 0x0008).
|
//! HDF5 Data Layout message parsing (message type 0x0008).
|
||||||
|
|
||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
use alloc::{string::String, vec::Vec};
|
use alloc::{format, string::String, vec::Vec};
|
||||||
|
|
||||||
#[cfg(feature = "std")]
|
#[cfg(feature = "std")]
|
||||||
use std::string::String;
|
use std::string::String;
|
||||||
|
|
||||||
|
use crate::addr::to_usize;
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
|
use crate::storage::Storage;
|
||||||
|
|
||||||
/// A single VDS (Virtual Dataset) source mapping.
|
/// A single VDS (Virtual Dataset) source mapping.
|
||||||
///
|
///
|
||||||
@@ -24,6 +26,34 @@ pub struct VdsMapping {
|
|||||||
pub virtual_selection: Vec<u8>,
|
pub virtual_selection: Vec<u8>,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Most dimensions a layout message can list (libhdf5 `H5O_LAYOUT_NDIMS`):
|
||||||
|
/// 32 dataspace dimensions plus the element size.
|
||||||
|
const MAX_LAYOUT_NDIMS: usize = 33;
|
||||||
|
|
||||||
|
/// libhdf5's checks on a chunked layout message's dimensions
|
||||||
|
/// (`H5O__layout_decode`): at most [`MAX_LAYOUT_NDIMS`], no dimension 0, and
|
||||||
|
/// before version 4 at least one dataspace dimension plus the element size.
|
||||||
|
/// A zero chunk dimension used to read the dataset as all fill values.
|
||||||
|
fn check_chunk_dims(dims: Vec<u32>, layout_version: u8) -> Result<Vec<u32>, FormatError> {
|
||||||
|
if dims.len() > MAX_LAYOUT_NDIMS {
|
||||||
|
return Err(FormatError::InvalidChunkDimensions(
|
||||||
|
"dimensionality is too large".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
if layout_version < 4 && dims.len() < 2 {
|
||||||
|
return Err(FormatError::InvalidChunkDimensions(
|
||||||
|
"bad dimensions for chunked storage".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
if let Some(u) = dims.iter().position(|&d| d == 0) {
|
||||||
|
return Err(FormatError::InvalidChunkDimensions(format!(
|
||||||
|
"bad chunk dimension value when parsing layout message - chunk dimension must be \
|
||||||
|
positive: mesg->u.chunk.dim[{u}] = 0"
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
Ok(dims)
|
||||||
|
}
|
||||||
|
|
||||||
/// Parsed HDF5 data layout message.
|
/// Parsed HDF5 data layout message.
|
||||||
#[derive(Debug, Clone, PartialEq)]
|
#[derive(Debug, Clone, PartialEq)]
|
||||||
pub enum DataLayout {
|
pub enum DataLayout {
|
||||||
@@ -45,7 +75,9 @@ pub enum DataLayout {
|
|||||||
chunk_dimensions: Vec<u32>,
|
chunk_dimensions: Vec<u32>,
|
||||||
/// B-tree address, or `None` if undefined.
|
/// B-tree address, or `None` if undefined.
|
||||||
btree_address: Option<u64>,
|
btree_address: Option<u64>,
|
||||||
/// Layout version (3 or 4).
|
/// Layout version (3 or 4). Version 1/2 messages (HDF5 1.4/1.6-era)
|
||||||
|
/// use the same version-1 B-tree chunk index as version 3 and are
|
||||||
|
/// reported as 3.
|
||||||
version: u8,
|
version: u8,
|
||||||
/// Chunk index type (v4 only).
|
/// Chunk index type (v4 only).
|
||||||
chunk_index_type: Option<u8>,
|
chunk_index_type: Option<u8>,
|
||||||
@@ -53,6 +85,11 @@ pub enum DataLayout {
|
|||||||
single_chunk_filtered_size: Option<u64>,
|
single_chunk_filtered_size: Option<u64>,
|
||||||
/// Filter mask for v4 single chunk with filters.
|
/// Filter mask for v4 single chunk with filters.
|
||||||
single_chunk_filter_mask: Option<u32>,
|
single_chunk_filter_mask: Option<u32>,
|
||||||
|
/// Layout v4 flag bit 0 (`H5D_CHUNK_DONT_FILTER_PARTIAL_CHUNKS`):
|
||||||
|
/// partial edge chunks — those extending past the dataset's current
|
||||||
|
/// extent in some dimension — are stored without the filter pipeline,
|
||||||
|
/// even though their filter mask is 0. Always `false` for v3.
|
||||||
|
dont_filter_partial_edge_chunks: bool,
|
||||||
},
|
},
|
||||||
/// Virtual dataset layout (v4 only).
|
/// Virtual dataset layout (v4 only).
|
||||||
Virtual {
|
Virtual {
|
||||||
@@ -67,21 +104,33 @@ pub enum DataLayout {
|
|||||||
},
|
},
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Version-1 VDS mapping flag: the source file name is stored by an earlier
|
||||||
|
/// entry, whose index follows in place of the name.
|
||||||
|
const VDS_SOURCE_FILE_SHARED: u8 = 0x01;
|
||||||
|
/// Version-1 VDS mapping flag: likewise for the source dataset name.
|
||||||
|
const VDS_SOURCE_DSET_SHARED: u8 = 0x02;
|
||||||
|
/// Version-1 VDS mapping flag: the source is in the virtual file itself
|
||||||
|
/// (`"."`); no file name is stored.
|
||||||
|
const VDS_SOURCE_SAME_FILE: u8 = 0x04;
|
||||||
|
const VDS_ALL_FLAGS: u8 = VDS_SOURCE_FILE_SHARED | VDS_SOURCE_DSET_SHARED | VDS_SOURCE_SAME_FILE;
|
||||||
|
|
||||||
/// Parse VDS mappings from global-heap object data.
|
/// Parse VDS mappings from global-heap object data.
|
||||||
///
|
///
|
||||||
/// The global-heap block holding a VDS mapping list is laid out as
|
/// The global-heap block holding a VDS mapping list is laid out as
|
||||||
/// (reverse-engineered and validated against HDF5 2.0):
|
/// (`H5D__virtual_store_layout` / `H5D__virtual_load_layout` in libhdf5):
|
||||||
///
|
///
|
||||||
/// ```text
|
/// ```text
|
||||||
/// version(1) · nused(length_size, LE) · entry[nused] · checksum(4)
|
/// version(1) · nused(length_size, LE) · entry[nused] · checksum(4)
|
||||||
/// ```
|
/// ```
|
||||||
///
|
///
|
||||||
/// Each entry is:
|
/// Each entry is:
|
||||||
/// - source file name — a null-terminated string in **block version 0**; in
|
/// - **block version 1 only:** a flags byte. `0x04`: the source is in the
|
||||||
/// **block version 1** a same-file reference is encoded as a single `0x04`
|
/// virtual file itself and no file name is stored; `0x01`/`0x02`: the
|
||||||
/// marker byte (the source file is the virtual file itself) in place of the
|
/// source file/dataset name is that of an earlier entry, whose index
|
||||||
/// name;
|
/// (`length_size` bytes) is stored instead of the name. libhdf5 2.0 writes
|
||||||
/// - source dataset name (null-terminated string);
|
/// version 1 when the file's low version bound is 2.0 and it saves space;
|
||||||
|
/// - source file name (null-terminated string, unless flagged above);
|
||||||
|
/// - source dataset name (null-terminated string, unless flagged above);
|
||||||
/// - source selection (serialized `H5S` dataspace selection — self-describing
|
/// - source selection (serialized `H5S` dataspace selection — self-describing
|
||||||
/// in length);
|
/// in length);
|
||||||
/// - virtual selection (serialized `H5S` dataspace selection).
|
/// - virtual selection (serialized `H5S` dataspace selection).
|
||||||
@@ -107,7 +156,7 @@ pub fn parse_vds_mappings(
|
|||||||
// `nused` is untrusted; don't pre-allocate from it. Each entry consumes at
|
// `nused` is untrusted; don't pre-allocate from it. Each entry consumes at
|
||||||
// least a few bytes, so the loop is naturally bounded by the heap data and
|
// least a few bytes, so the loop is naturally bounded by the heap data and
|
||||||
// a bogus `nused` simply errors out on the first short read.
|
// a bogus `nused` simply errors out on the first short read.
|
||||||
let mut mappings = Vec::new();
|
let mut mappings: Vec<VdsMapping> = Vec::new();
|
||||||
// Reads one self-describing selection at `pos`, returning its raw bytes and
|
// Reads one self-describing selection at `pos`, returning its raw bytes and
|
||||||
// advancing past it — bounds-checked so a corrupt selection can't overrun.
|
// advancing past it — bounds-checked so a corrupt selection can't overrun.
|
||||||
let read_selection = |heap_data: &[u8], pos: &mut usize| -> Result<Vec<u8>, FormatError> {
|
let read_selection = |heap_data: &[u8], pos: &mut usize| -> Result<Vec<u8>, FormatError> {
|
||||||
@@ -127,17 +176,57 @@ pub fn parse_vds_mappings(
|
|||||||
Ok(bytes)
|
Ok(bytes)
|
||||||
};
|
};
|
||||||
|
|
||||||
for _ in 0..nused {
|
if version > 1 {
|
||||||
// Source file name (with the version-1 same-file marker handled).
|
return Err(FormatError::ChunkedReadError(
|
||||||
let source_file = if version >= 1 && heap_data.get(pos) == Some(&0x04) {
|
"unsupported VDS mapping block version".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
for i in 0..nused {
|
||||||
|
// Version 1 prefixes each entry with a flags byte; a name may then be
|
||||||
|
// omitted (same file) or replaced by the index of an earlier entry
|
||||||
|
// holding the same name (`H5D__virtual_load_layout`).
|
||||||
|
let flags = if version >= 1 {
|
||||||
|
let f = *heap_data.get(pos).ok_or(FormatError::UnexpectedEof {
|
||||||
|
expected: pos + 1,
|
||||||
|
available: heap_data.len(),
|
||||||
|
})?;
|
||||||
pos += 1;
|
pos += 1;
|
||||||
|
if f & !VDS_ALL_FLAGS != 0 {
|
||||||
|
return Err(FormatError::ChunkedReadError(
|
||||||
|
"unknown VDS mapping flags".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
f
|
||||||
|
} else {
|
||||||
|
0
|
||||||
|
};
|
||||||
|
// Index of an earlier entry, for a shared name.
|
||||||
|
let earlier = |pos: &mut usize| -> Result<usize, FormatError> {
|
||||||
|
let idx = read_length(heap_data, *pos, length_size)?;
|
||||||
|
*pos += ls;
|
||||||
|
if idx >= i {
|
||||||
|
return Err(FormatError::ChunkedReadError(
|
||||||
|
"VDS mapping shares a name with a later entry".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
to_usize(idx)
|
||||||
|
};
|
||||||
|
|
||||||
|
let source_file = if flags & VDS_SOURCE_SAME_FILE != 0 {
|
||||||
String::from(".")
|
String::from(".")
|
||||||
|
} else if flags & VDS_SOURCE_FILE_SHARED != 0 {
|
||||||
|
let idx = earlier(&mut pos)?;
|
||||||
|
mappings[idx].source_file.clone()
|
||||||
} else {
|
} else {
|
||||||
read_null_terminated_string(heap_data, &mut pos)?
|
read_null_terminated_string(heap_data, &mut pos)?
|
||||||
};
|
};
|
||||||
|
|
||||||
// Source dataset name.
|
let source_dataset = if flags & VDS_SOURCE_DSET_SHARED != 0 {
|
||||||
let source_dataset = read_null_terminated_string(heap_data, &mut pos)?;
|
let idx = earlier(&mut pos)?;
|
||||||
|
mappings[idx].source_dataset.clone()
|
||||||
|
} else {
|
||||||
|
read_null_terminated_string(heap_data, &mut pos)?
|
||||||
|
};
|
||||||
|
|
||||||
// Source selection, then virtual selection (both self-describing length).
|
// Source selection, then virtual selection (both self-describing length).
|
||||||
let source_selection = read_selection(heap_data, &mut pos)?;
|
let source_selection = read_selection(heap_data, &mut pos)?;
|
||||||
@@ -222,6 +311,16 @@ impl DataLayout {
|
|||||||
&mut self,
|
&mut self,
|
||||||
file_data: &[u8],
|
file_data: &[u8],
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
|
) -> Result<(), FormatError> {
|
||||||
|
self.resolve_vds_mappings_in(file_data, length_size)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`Self::resolve_vds_mappings`] over any [`Storage`]: one read of the
|
||||||
|
/// global heap collection holding the mappings.
|
||||||
|
pub fn resolve_vds_mappings_in<S: Storage + ?Sized>(
|
||||||
|
&mut self,
|
||||||
|
file_data: &S,
|
||||||
|
length_size: u8,
|
||||||
) -> Result<(), FormatError> {
|
) -> Result<(), FormatError> {
|
||||||
if let DataLayout::Virtual {
|
if let DataLayout::Virtual {
|
||||||
global_heap_address,
|
global_heap_address,
|
||||||
@@ -231,11 +330,8 @@ impl DataLayout {
|
|||||||
} = self
|
} = self
|
||||||
&& let Some(addr) = *global_heap_address
|
&& let Some(addr) = *global_heap_address
|
||||||
{
|
{
|
||||||
let coll = crate::global_heap::GlobalHeapCollection::parse(
|
let coll =
|
||||||
file_data,
|
crate::global_heap::GlobalHeapCollection::parse_in(file_data, addr, length_size)?;
|
||||||
addr as usize,
|
|
||||||
length_size,
|
|
||||||
)?;
|
|
||||||
let obj = coll.get_object(*global_heap_index as u16).ok_or(
|
let obj = coll.get_object(*global_heap_index as u16).ok_or(
|
||||||
FormatError::GlobalHeapObjectNotFound {
|
FormatError::GlobalHeapObjectNotFound {
|
||||||
collection_address: addr,
|
collection_address: addr,
|
||||||
@@ -256,6 +352,7 @@ impl DataLayout {
|
|||||||
let layout_class = data[1];
|
let layout_class = data[1];
|
||||||
|
|
||||||
match version {
|
match version {
|
||||||
|
1 | 2 => Self::parse_v1_v2(data, offset_size),
|
||||||
3 => Self::parse_v3(data, layout_class, offset_size, length_size),
|
3 => Self::parse_v3(data, layout_class, offset_size, length_size),
|
||||||
// v5 (emitted by HDF5 1.14+/2.0 with `libver=latest`) uses the same
|
// v5 (emitted by HDF5 1.14+/2.0 with `libver=latest`) uses the same
|
||||||
// message structure as v4 — only the version number was bumped.
|
// message structure as v4 — only the version number was bumped.
|
||||||
@@ -264,6 +361,87 @@ impl DataLayout {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Layout message versions 1 and 2 (HDF5 before 1.6.3):
|
||||||
|
///
|
||||||
|
/// ```text
|
||||||
|
/// version(1) · dimensionality(1) · layout class(1) · reserved(5)
|
||||||
|
/// · address(offset_size) — contiguous and chunked only
|
||||||
|
/// · dimension sizes(4 × dimensionality)
|
||||||
|
/// · compact data size(4) · compact raw data — compact only
|
||||||
|
/// ```
|
||||||
|
///
|
||||||
|
/// The dimension sizes are the dataset's (contiguous/compact) or the
|
||||||
|
/// chunk's (chunked) extent plus a trailing element-size dimension, as in
|
||||||
|
/// version 3's chunked form. libhdf5 ignores them for contiguous storage
|
||||||
|
/// and sizes the data from the dataspace; the product of the stored
|
||||||
|
/// dimensions is that same size, and a disagreement (a dimension that was
|
||||||
|
/// truncated to 32 bits) is caught by the reader's size check rather than
|
||||||
|
/// returning wrong data.
|
||||||
|
fn parse_v1_v2(data: &[u8], offset_size: u8) -> Result<DataLayout, FormatError> {
|
||||||
|
ensure_len(data, 0, 8)?;
|
||||||
|
let dimensionality = data[1] as usize;
|
||||||
|
let layout_class = data[2];
|
||||||
|
// H5O_LAYOUT_NDIMS: 32 dataspace dimensions + the element-size one.
|
||||||
|
if dimensionality > 33 {
|
||||||
|
return Err(FormatError::Overflow(format!(
|
||||||
|
"data layout dimensionality {dimensionality} exceeds 33"
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
let mut p = 8;
|
||||||
|
let os = offset_size as usize;
|
||||||
|
let address = match layout_class {
|
||||||
|
1 | 2 => {
|
||||||
|
ensure_len(data, p, os)?;
|
||||||
|
let a = if is_undefined(data, p, offset_size) {
|
||||||
|
None
|
||||||
|
} else {
|
||||||
|
Some(read_offset(data, p, offset_size)?)
|
||||||
|
};
|
||||||
|
p += os;
|
||||||
|
a
|
||||||
|
}
|
||||||
|
0 => None,
|
||||||
|
_ => return Err(FormatError::InvalidLayoutClass(layout_class)),
|
||||||
|
};
|
||||||
|
ensure_len(data, p, dimensionality * 4)?;
|
||||||
|
let dims: Vec<u32> = data[p..p + dimensionality * 4]
|
||||||
|
.as_chunks::<4>()
|
||||||
|
.0
|
||||||
|
.iter()
|
||||||
|
.map(|c| u32::from_le_bytes(*c))
|
||||||
|
.collect();
|
||||||
|
p += dimensionality * 4;
|
||||||
|
match layout_class {
|
||||||
|
0 => {
|
||||||
|
ensure_len(data, p, 4)?;
|
||||||
|
let size =
|
||||||
|
u32::from_le_bytes([data[p], data[p + 1], data[p + 2], data[p + 3]]) as usize;
|
||||||
|
ensure_len(data, p + 4, size)?;
|
||||||
|
Ok(DataLayout::Compact {
|
||||||
|
data: data[p + 4..p + 4 + size].to_vec(),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
1 => {
|
||||||
|
let size = dims
|
||||||
|
.iter()
|
||||||
|
.try_fold(1u64, |acc, &d| acc.checked_mul(d as u64))
|
||||||
|
.ok_or_else(|| {
|
||||||
|
FormatError::Overflow(format!("contiguous layout size {dims:?}"))
|
||||||
|
})?;
|
||||||
|
Ok(DataLayout::Contiguous { address, size })
|
||||||
|
}
|
||||||
|
_ => Ok(DataLayout::Chunked {
|
||||||
|
chunk_dimensions: check_chunk_dims(dims, 2)?,
|
||||||
|
btree_address: address,
|
||||||
|
version: 3,
|
||||||
|
chunk_index_type: None,
|
||||||
|
single_chunk_filtered_size: None,
|
||||||
|
single_chunk_filter_mask: None,
|
||||||
|
dont_filter_partial_edge_chunks: false,
|
||||||
|
}),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
fn parse_v3(
|
fn parse_v3(
|
||||||
data: &[u8],
|
data: &[u8],
|
||||||
layout_class: u8,
|
layout_class: u8,
|
||||||
@@ -316,12 +494,13 @@ impl DataLayout {
|
|||||||
p += 4;
|
p += 4;
|
||||||
}
|
}
|
||||||
Ok(DataLayout::Chunked {
|
Ok(DataLayout::Chunked {
|
||||||
chunk_dimensions,
|
chunk_dimensions: check_chunk_dims(chunk_dimensions, 3)?,
|
||||||
btree_address,
|
btree_address,
|
||||||
version: 3,
|
version: 3,
|
||||||
chunk_index_type: None,
|
chunk_index_type: None,
|
||||||
single_chunk_filtered_size: None,
|
single_chunk_filtered_size: None,
|
||||||
single_chunk_filter_mask: None,
|
single_chunk_filter_mask: None,
|
||||||
|
dont_filter_partial_edge_chunks: false,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
_ => Err(FormatError::InvalidLayoutClass(layout_class)),
|
_ => Err(FormatError::InvalidLayoutClass(layout_class)),
|
||||||
@@ -364,47 +543,40 @@ impl DataLayout {
|
|||||||
let dimensionality = data[pos + 1] as usize;
|
let dimensionality = data[pos + 1] as usize;
|
||||||
let dim_size_encoded_length = data[pos + 2] as usize;
|
let dim_size_encoded_length = data[pos + 2] as usize;
|
||||||
let mut p = pos + 3;
|
let mut p = pos + 3;
|
||||||
|
if dimensionality > MAX_LAYOUT_NDIMS {
|
||||||
|
return Err(FormatError::InvalidChunkDimensions(
|
||||||
|
"dimensionality is too large".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
// dimension sizes
|
// Each dimension takes 1 to 8 bytes (libhdf5 writes the
|
||||||
|
// fewest that hold the largest one, so 3, 5, 6 and 7 occur:
|
||||||
|
// a chunk dimension of 70 000 takes 3). libhdf5 refuses 0
|
||||||
|
// and more than 8.
|
||||||
|
if dim_size_encoded_length == 0 || dim_size_encoded_length > 8 {
|
||||||
|
return Err(FormatError::InvalidChunkDimensions(
|
||||||
|
"encoded chunk dimension size is too large".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
ensure_len(data, p, dimensionality * dim_size_encoded_length)?;
|
ensure_len(data, p, dimensionality * dim_size_encoded_length)?;
|
||||||
let mut chunk_dimensions = Vec::with_capacity(dimensionality);
|
let mut chunk_dimensions = Vec::with_capacity(dimensionality);
|
||||||
for _ in 0..dimensionality {
|
for _ in 0..dimensionality {
|
||||||
let val = match dim_size_encoded_length {
|
let val = data[p..p + dim_size_encoded_length]
|
||||||
1 => data[p] as u32,
|
.iter()
|
||||||
2 => u16::from_le_bytes([data[p], data[p + 1]]) as u32,
|
.rev()
|
||||||
4 => u32::from_le_bytes([data[p], data[p + 1], data[p + 2], data[p + 3]]),
|
.fold(0u64, |acc, &b| (acc << 8) | u64::from(b));
|
||||||
8 => {
|
// Chunk dimensions are held as u32; HDF5 2.0 can write
|
||||||
// V4 chunked encodes dimension sizes as 8 bytes, but
|
// larger ones (layout version 5), which are refused
|
||||||
// our ChunkedStorageV4 stores them as u32. We read only
|
// rather than truncated.
|
||||||
// the low 4 bytes (little-endian). This silently
|
let val = u32::try_from(val).map_err(|_| {
|
||||||
// truncates dimensions > 4 GiB, which are not expected
|
FormatError::InvalidChunkDimensions(format!(
|
||||||
// in practice (HDF5 chunk dimensions are always small).
|
"chunk dimension {val} is larger than 2^32 - 1, which is not supported"
|
||||||
// If the high bytes are non-zero, the file is malformed
|
))
|
||||||
// or uses dimensions we cannot represent.
|
})?;
|
||||||
let high = u32::from_le_bytes([
|
|
||||||
data[p + 4],
|
|
||||||
data[p + 5],
|
|
||||||
data[p + 6],
|
|
||||||
data[p + 7],
|
|
||||||
]);
|
|
||||||
if high != 0 {
|
|
||||||
return Err(FormatError::UnexpectedEof {
|
|
||||||
expected: p + 8,
|
|
||||||
available: data.len(),
|
|
||||||
});
|
|
||||||
}
|
|
||||||
u32::from_le_bytes([data[p], data[p + 1], data[p + 2], data[p + 3]])
|
|
||||||
}
|
|
||||||
_ => {
|
|
||||||
return Err(FormatError::UnexpectedEof {
|
|
||||||
expected: p + dim_size_encoded_length,
|
|
||||||
available: data.len(),
|
|
||||||
});
|
|
||||||
}
|
|
||||||
};
|
|
||||||
chunk_dimensions.push(val);
|
chunk_dimensions.push(val);
|
||||||
p += dim_size_encoded_length;
|
p += dim_size_encoded_length;
|
||||||
}
|
}
|
||||||
|
let chunk_dimensions = check_chunk_dims(chunk_dimensions, 4)?;
|
||||||
|
|
||||||
// chunk index type
|
// chunk index type
|
||||||
ensure_len(data, p, 1)?;
|
ensure_len(data, p, 1)?;
|
||||||
@@ -505,6 +677,7 @@ impl DataLayout {
|
|||||||
chunk_index_type: Some(chunk_index_type),
|
chunk_index_type: Some(chunk_index_type),
|
||||||
single_chunk_filtered_size,
|
single_chunk_filtered_size,
|
||||||
single_chunk_filter_mask,
|
single_chunk_filter_mask,
|
||||||
|
dont_filter_partial_edge_chunks: flags & 0x01 != 0,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
3 => {
|
3 => {
|
||||||
@@ -539,6 +712,202 @@ impl DataLayout {
|
|||||||
mod tests {
|
mod tests {
|
||||||
use super::*;
|
use super::*;
|
||||||
|
|
||||||
|
/// Version 1/2 header: version, dimensionality, class, reserved(5).
|
||||||
|
fn v1v2_header(version: u8, ndims: u8, class: u8) -> Vec<u8> {
|
||||||
|
vec![version, ndims, class, 0, 0, 0, 0, 0]
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn v2_compact() {
|
||||||
|
let mut buf = v1v2_header(2, 2, 0);
|
||||||
|
// dims (3 elements of 2 bytes) — no address for compact
|
||||||
|
buf.extend_from_slice(&3u32.to_le_bytes());
|
||||||
|
buf.extend_from_slice(&2u32.to_le_bytes());
|
||||||
|
buf.extend_from_slice(&6u32.to_le_bytes()); // compact size (u32 in v1/v2)
|
||||||
|
buf.extend_from_slice(&[1, 0, 2, 0, 3, 0]);
|
||||||
|
assert_eq!(
|
||||||
|
DataLayout::parse(&buf, 8, 8).unwrap(),
|
||||||
|
DataLayout::Compact {
|
||||||
|
data: vec![1, 0, 2, 0, 3, 0]
|
||||||
|
}
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn v1_contiguous_size_from_dimensions() {
|
||||||
|
let mut buf = v1v2_header(1, 3, 1);
|
||||||
|
buf.extend_from_slice(&0x800u32.to_le_bytes()); // 4-byte address
|
||||||
|
for d in [10u32, 20, 4] {
|
||||||
|
buf.extend_from_slice(&d.to_le_bytes());
|
||||||
|
}
|
||||||
|
assert_eq!(
|
||||||
|
DataLayout::parse(&buf, 4, 4).unwrap(),
|
||||||
|
DataLayout::Contiguous {
|
||||||
|
address: Some(0x800),
|
||||||
|
size: 800,
|
||||||
|
}
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn v1_contiguous_undefined_address() {
|
||||||
|
let mut buf = v1v2_header(1, 2, 1);
|
||||||
|
buf.extend_from_slice(&[0xFF; 8]);
|
||||||
|
buf.extend_from_slice(&5u32.to_le_bytes());
|
||||||
|
buf.extend_from_slice(&8u32.to_le_bytes());
|
||||||
|
assert_eq!(
|
||||||
|
DataLayout::parse(&buf, 8, 8).unwrap(),
|
||||||
|
DataLayout::Contiguous {
|
||||||
|
address: None,
|
||||||
|
size: 40,
|
||||||
|
}
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn v1_chunked_maps_to_btree_v1_index() {
|
||||||
|
let mut buf = v1v2_header(1, 3, 2);
|
||||||
|
buf.extend_from_slice(&0x1234u64.to_le_bytes());
|
||||||
|
for d in [50u32, 50, 4] {
|
||||||
|
buf.extend_from_slice(&d.to_le_bytes());
|
||||||
|
}
|
||||||
|
assert_eq!(
|
||||||
|
DataLayout::parse(&buf, 8, 8).unwrap(),
|
||||||
|
DataLayout::Chunked {
|
||||||
|
chunk_dimensions: vec![50, 50, 4],
|
||||||
|
btree_address: Some(0x1234),
|
||||||
|
version: 3,
|
||||||
|
chunk_index_type: None,
|
||||||
|
single_chunk_filtered_size: None,
|
||||||
|
single_chunk_filter_mask: None,
|
||||||
|
dont_filter_partial_edge_chunks: false,
|
||||||
|
}
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A v3 chunked layout message with these dims (element size last).
|
||||||
|
fn v3_chunked_msg(dims: &[u32]) -> Vec<u8> {
|
||||||
|
let mut buf = vec![3u8, 2, dims.len() as u8];
|
||||||
|
buf.extend_from_slice(&0x1000u64.to_le_bytes());
|
||||||
|
for d in dims {
|
||||||
|
buf.extend_from_slice(&d.to_le_bytes());
|
||||||
|
}
|
||||||
|
buf
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn chunk_dimensions_are_checked_when_the_layout_is_parsed() {
|
||||||
|
assert!(DataLayout::parse(&v3_chunked_msg(&[4, 4, 8]), 8, 8).is_ok());
|
||||||
|
// A zero chunk dimension used to read as all fill values.
|
||||||
|
let err = DataLayout::parse(&v3_chunked_msg(&[4, 0, 8]), 8, 8).unwrap_err();
|
||||||
|
assert!(
|
||||||
|
matches!(&err, FormatError::InvalidChunkDimensions(m) if m.contains("dim[1] = 0")),
|
||||||
|
"{err:?}"
|
||||||
|
);
|
||||||
|
// Only the element-size dimension: libhdf5 "bad dimensions".
|
||||||
|
assert_eq!(
|
||||||
|
DataLayout::parse(&v3_chunked_msg(&[8]), 8, 8).unwrap_err(),
|
||||||
|
FormatError::InvalidChunkDimensions("bad dimensions for chunked storage".into())
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
DataLayout::parse(&v3_chunked_msg(&[1; 34]), 8, 8).unwrap_err(),
|
||||||
|
FormatError::InvalidChunkDimensions("dimensionality is too large".into())
|
||||||
|
);
|
||||||
|
// v1/v2 and v4 messages get the zero check too.
|
||||||
|
let mut v1 = v1v2_header(1, 2, 2);
|
||||||
|
v1.extend_from_slice(&0x1000u64.to_le_bytes());
|
||||||
|
v1.extend_from_slice(&0u32.to_le_bytes());
|
||||||
|
v1.extend_from_slice(&8u32.to_le_bytes());
|
||||||
|
assert!(matches!(
|
||||||
|
DataLayout::parse(&v1, 8, 8),
|
||||||
|
Err(FormatError::InvalidChunkDimensions(_))
|
||||||
|
));
|
||||||
|
let mut v4 = vec![4u8, 2, 0, 2, 4];
|
||||||
|
v4.extend_from_slice(&0u32.to_le_bytes());
|
||||||
|
v4.extend_from_slice(&8u32.to_le_bytes());
|
||||||
|
v4.push(3); // fixed array index
|
||||||
|
v4.push(0); // page bits
|
||||||
|
v4.extend_from_slice(&0x1000u64.to_le_bytes());
|
||||||
|
assert!(matches!(
|
||||||
|
DataLayout::parse(&v4, 8, 8),
|
||||||
|
Err(FormatError::InvalidChunkDimensions(_))
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A v4 chunked layout (fixed array index) whose `dims` are each
|
||||||
|
/// encoded in `width` bytes.
|
||||||
|
fn v4_chunked_msg(width: u8, dims: &[u64]) -> Vec<u8> {
|
||||||
|
let mut m = vec![4u8, 2, 0, dims.len() as u8, width];
|
||||||
|
for &d in dims {
|
||||||
|
m.extend_from_slice(&d.to_le_bytes()[..width.min(8) as usize]);
|
||||||
|
}
|
||||||
|
m.push(3); // fixed array index
|
||||||
|
m.push(0); // page bits
|
||||||
|
m.extend_from_slice(&0x1000u64.to_le_bytes());
|
||||||
|
m
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn v4_chunk_dimensions_take_1_to_8_bytes() {
|
||||||
|
// libhdf5 encodes each dimension in the fewest bytes that hold the
|
||||||
|
// largest: a chunk dimension of 70 000 takes 3, and 3, 5, 6 and 7
|
||||||
|
// were refused ("UnexpectedEof").
|
||||||
|
for width in 1..=8u8 {
|
||||||
|
let dims = [if width >= 3 { 70_000 } else { 200 }, 8];
|
||||||
|
let layout = DataLayout::parse(&v4_chunked_msg(width, &dims), 8, 8)
|
||||||
|
.unwrap_or_else(|e| panic!("width {width}: {e:?}"));
|
||||||
|
assert!(
|
||||||
|
matches!(&layout, DataLayout::Chunked { chunk_dimensions, .. }
|
||||||
|
if chunk_dimensions.iter().map(|&d| u64::from(d)).eq(dims)),
|
||||||
|
"width {width}: {layout:?}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
// libhdf5 refuses 0 and more than 8 bytes.
|
||||||
|
for width in [0u8, 9] {
|
||||||
|
assert_eq!(
|
||||||
|
DataLayout::parse(&v4_chunked_msg(width, &[4, 8]), 8, 8).unwrap_err(),
|
||||||
|
FormatError::InvalidChunkDimensions(
|
||||||
|
"encoded chunk dimension size is too large".into()
|
||||||
|
)
|
||||||
|
);
|
||||||
|
}
|
||||||
|
// A dimension past u32 cannot be represented and is refused, not
|
||||||
|
// truncated.
|
||||||
|
assert!(matches!(
|
||||||
|
DataLayout::parse(&v4_chunked_msg(5, &[1 << 32, 8]), 8, 8),
|
||||||
|
Err(FormatError::InvalidChunkDimensions(m)) if m.contains("2^32")
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn v1v2_rejects_bad_class_dimensionality_and_truncation() {
|
||||||
|
assert_eq!(
|
||||||
|
DataLayout::parse(&v1v2_header(1, 1, 3), 8, 8).unwrap_err(),
|
||||||
|
FormatError::InvalidLayoutClass(3)
|
||||||
|
);
|
||||||
|
assert!(matches!(
|
||||||
|
DataLayout::parse(&v1v2_header(2, 34, 1), 8, 8).unwrap_err(),
|
||||||
|
FormatError::Overflow(_)
|
||||||
|
));
|
||||||
|
// Chunked, dims cut short.
|
||||||
|
let mut buf = v1v2_header(1, 2, 2);
|
||||||
|
buf.extend_from_slice(&0x10u64.to_le_bytes());
|
||||||
|
buf.extend_from_slice(&7u32.to_le_bytes());
|
||||||
|
assert!(matches!(
|
||||||
|
DataLayout::parse(&buf, 8, 8).unwrap_err(),
|
||||||
|
FormatError::UnexpectedEof { .. }
|
||||||
|
));
|
||||||
|
// Compact, raw data shorter than its declared size.
|
||||||
|
let mut buf = v1v2_header(2, 1, 0);
|
||||||
|
buf.extend_from_slice(&4u32.to_le_bytes());
|
||||||
|
buf.extend_from_slice(&100u32.to_le_bytes());
|
||||||
|
buf.extend_from_slice(&[0; 4]);
|
||||||
|
assert!(matches!(
|
||||||
|
DataLayout::parse(&buf, 8, 8).unwrap_err(),
|
||||||
|
FormatError::UnexpectedEof { .. }
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn v3_compact() {
|
fn v3_compact() {
|
||||||
let mut buf = vec![3u8, 0]; // version=3, class=0 (compact)
|
let mut buf = vec![3u8, 0]; // version=3, class=0 (compact)
|
||||||
@@ -602,6 +971,7 @@ mod tests {
|
|||||||
chunk_index_type: None,
|
chunk_index_type: None,
|
||||||
single_chunk_filtered_size: None,
|
single_chunk_filtered_size: None,
|
||||||
single_chunk_filter_mask: None,
|
single_chunk_filter_mask: None,
|
||||||
|
dont_filter_partial_edge_chunks: false,
|
||||||
}
|
}
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
@@ -679,10 +1049,35 @@ mod tests {
|
|||||||
chunk_index_type: Some(1),
|
chunk_index_type: Some(1),
|
||||||
single_chunk_filtered_size: None,
|
single_chunk_filtered_size: None,
|
||||||
single_chunk_filter_mask: None,
|
single_chunk_filter_mask: None,
|
||||||
|
dont_filter_partial_edge_chunks: false,
|
||||||
}
|
}
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn v4_chunked_dont_filter_partial_edge_chunks_flag() {
|
||||||
|
let mut buf = vec![4u8, 2]; // version=4, class=2
|
||||||
|
buf.push(0x01); // flags bit 0 = don't filter partial edge chunks
|
||||||
|
buf.push(2); // dimensionality=2
|
||||||
|
buf.push(4); // dim_size_encoded_length=4
|
||||||
|
buf.extend_from_slice(&5u32.to_le_bytes());
|
||||||
|
buf.extend_from_slice(&4u32.to_le_bytes());
|
||||||
|
buf.push(3); // Fixed Array
|
||||||
|
buf.push(10); // max_dblk_page_nelmts_bits
|
||||||
|
buf.extend_from_slice(&0x3000u64.to_le_bytes());
|
||||||
|
match DataLayout::parse(&buf, 8, 8).unwrap() {
|
||||||
|
DataLayout::Chunked {
|
||||||
|
dont_filter_partial_edge_chunks,
|
||||||
|
btree_address,
|
||||||
|
..
|
||||||
|
} => {
|
||||||
|
assert!(dont_filter_partial_edge_chunks);
|
||||||
|
assert_eq!(btree_address, Some(0x3000));
|
||||||
|
}
|
||||||
|
other => panic!("expected Chunked, got {other:?}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn v4_chunked_single_chunk_with_filters() {
|
fn v4_chunked_single_chunk_with_filters() {
|
||||||
let mut buf = vec![4u8, 2]; // version=4, class=2
|
let mut buf = vec![4u8, 2]; // version=4, class=2
|
||||||
@@ -705,6 +1100,7 @@ mod tests {
|
|||||||
chunk_index_type: Some(1),
|
chunk_index_type: Some(1),
|
||||||
single_chunk_filtered_size: Some(1024),
|
single_chunk_filtered_size: Some(1024),
|
||||||
single_chunk_filter_mask: Some(0),
|
single_chunk_filter_mask: Some(0),
|
||||||
|
dont_filter_partial_edge_chunks: false,
|
||||||
}
|
}
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
@@ -815,6 +1211,62 @@ mod tests {
|
|||||||
assert_eq!(v1.iter_linear_1d(8).unwrap(), vec![4, 5, 6, 7]);
|
assert_eq!(v1.iter_linear_1d(8).unwrap(), vec![4, 5, 6, 7]);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn parse_vds_mappings_v1_shared_names() {
|
||||||
|
// Written by HDF5 2.0 (h5py, libver=("v200", "v200")) for three
|
||||||
|
// mappings from `a_rather_long_source_file.h5:a_rather_long_dataset_name`
|
||||||
|
// and one from the same file: the entries carry flags 0x00, 0x03, 0x03
|
||||||
|
// and 0x06, so names after the first are stored as entry indices.
|
||||||
|
let blob: &[u8] = &[
|
||||||
|
0x01, 0x04, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x61, 0x5f, 0x72, 0x61,
|
||||||
|
0x74, 0x68, 0x65, 0x72, 0x5f, 0x6c, 0x6f, 0x6e, 0x67, 0x5f, 0x73, 0x6f, 0x75, 0x72,
|
||||||
|
0x63, 0x65, 0x5f, 0x66, 0x69, 0x6c, 0x65, 0x2e, 0x68, 0x35, 0x00, 0x61, 0x5f, 0x72,
|
||||||
|
0x61, 0x74, 0x68, 0x65, 0x72, 0x5f, 0x6c, 0x6f, 0x6e, 0x67, 0x5f, 0x64, 0x61, 0x74,
|
||||||
|
0x61, 0x73, 0x65, 0x74, 0x5f, 0x6e, 0x61, 0x6d, 0x65, 0x00, 0x02, 0x00, 0x00, 0x00,
|
||||||
|
0x03, 0x00, 0x00, 0x00, 0x01, 0x02, 0x01, 0x00, 0x00, 0x00, 0x00, 0x00, 0x01, 0x00,
|
||||||
|
0x01, 0x00, 0x04, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00, 0x00, 0x00, 0x01, 0x02,
|
||||||
|
0x02, 0x00, 0x00, 0x00, 0x00, 0x00, 0x01, 0x00, 0x01, 0x00, 0x01, 0x00, 0x00, 0x00,
|
||||||
|
0x01, 0x00, 0x01, 0x00, 0x04, 0x00, 0x03, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
|
||||||
|
0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03,
|
||||||
|
0x00, 0x00, 0x00, 0x01, 0x02, 0x01, 0x00, 0x00, 0x00, 0x04, 0x00, 0x01, 0x00, 0x01,
|
||||||
|
0x00, 0x04, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00, 0x00, 0x00, 0x01, 0x02, 0x02,
|
||||||
|
0x00, 0x00, 0x00, 0x01, 0x00, 0x01, 0x00, 0x01, 0x00, 0x01, 0x00, 0x00, 0x00, 0x01,
|
||||||
|
0x00, 0x01, 0x00, 0x04, 0x00, 0x03, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
|
||||||
|
0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00,
|
||||||
|
0x00, 0x00, 0x01, 0x02, 0x01, 0x00, 0x00, 0x00, 0x08, 0x00, 0x01, 0x00, 0x01, 0x00,
|
||||||
|
0x04, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00, 0x00, 0x00, 0x01, 0x02, 0x02, 0x00,
|
||||||
|
0x00, 0x00, 0x02, 0x00, 0x01, 0x00, 0x01, 0x00, 0x01, 0x00, 0x00, 0x00, 0x01, 0x00,
|
||||||
|
0x01, 0x00, 0x04, 0x00, 0x06, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x02,
|
||||||
|
0x00, 0x00, 0x00, 0x03, 0x00, 0x00, 0x00, 0x01, 0x02, 0x01, 0x00, 0x00, 0x00, 0x00,
|
||||||
|
0x00, 0x01, 0x00, 0x01, 0x00, 0x04, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00, 0x00,
|
||||||
|
0x00, 0x01, 0x02, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00, 0x01, 0x00, 0x01, 0x00, 0x01,
|
||||||
|
0x00, 0x00, 0x00, 0x01, 0x00, 0x01, 0x00, 0x04, 0x00, 0x8e, 0xa7, 0xea, 0x7a,
|
||||||
|
];
|
||||||
|
let mappings = parse_vds_mappings(blob, 8).unwrap();
|
||||||
|
let names: Vec<(&str, &str)> = mappings
|
||||||
|
.iter()
|
||||||
|
.map(|m| (m.source_file.as_str(), m.source_dataset.as_str()))
|
||||||
|
.collect();
|
||||||
|
let (file, dset) = ("a_rather_long_source_file.h5", "a_rather_long_dataset_name");
|
||||||
|
assert_eq!(
|
||||||
|
names,
|
||||||
|
vec![(file, dset), (file, dset), (file, dset), (".", dset)]
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn parse_vds_mappings_v1_forward_reference_is_error() {
|
||||||
|
// Entry 0 claiming to share entry 0's file name must not index past
|
||||||
|
// the entries decoded so far.
|
||||||
|
let mut blob = vec![0x01u8, 1, 0, 0, 0, 0, 0, 0, 0, 0x01];
|
||||||
|
blob.extend_from_slice(&[0u8; 8]);
|
||||||
|
blob.extend_from_slice(b"d\0");
|
||||||
|
assert!(parse_vds_mappings(&blob, 8).is_err());
|
||||||
|
// Unknown flag bits are refused.
|
||||||
|
let blob = [0x01u8, 1, 0, 0, 0, 0, 0, 0, 0, 0x08, b'd', 0];
|
||||||
|
assert!(parse_vds_mappings(&blob, 8).is_err());
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn parse_vds_mappings_external_v0() {
|
fn parse_vds_mappings_external_v0() {
|
||||||
// Block version 0 with an explicit (external) source file name.
|
// Block version 0 with an explicit (external) source file name.
|
||||||
@@ -862,4 +1314,44 @@ mod tests {
|
|||||||
let blob = [0x01u8, 0, 0, 0, 0, 0, 0, 0, 0];
|
let blob = [0x01u8, 0, 0, 0, 0, 0, 0, 0, 0];
|
||||||
assert!(parse_vds_mappings(&blob, 8).unwrap().is_empty());
|
assert!(parse_vds_mappings(&blob, 8).unwrap().is_empty());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// A virtual dataset's mappings resolve identically through a
|
||||||
|
/// read_at-only CountingStorage, in two reads of the global heap.
|
||||||
|
#[test]
|
||||||
|
fn vds_mappings_through_storage_match_slice() {
|
||||||
|
use crate::message_type::MessageType;
|
||||||
|
use crate::object_header::ObjectHeader;
|
||||||
|
use crate::storage::CountingStorage;
|
||||||
|
let file: &[u8] = include_bytes!("../tests/fixtures/vds_same_file.h5");
|
||||||
|
let sb = crate::superblock::Superblock::parse(file, 0).unwrap();
|
||||||
|
let (os, ls) = (sb.offset_size, sb.length_size);
|
||||||
|
let storage = CountingStorage::new(file.to_vec());
|
||||||
|
let mut virtuals = 0;
|
||||||
|
for child in
|
||||||
|
crate::group_v2::resolve_group_children(file, &sb, sb.root_group_address).unwrap()
|
||||||
|
{
|
||||||
|
let h =
|
||||||
|
ObjectHeader::parse(file, child.object_header_address as usize, os, ls).unwrap();
|
||||||
|
let Some(msg) = h
|
||||||
|
.messages
|
||||||
|
.iter()
|
||||||
|
.find(|m| m.msg_type == MessageType::DataLayout)
|
||||||
|
else {
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
let mut want = DataLayout::parse(&msg.data, os, ls).unwrap();
|
||||||
|
if !matches!(want, DataLayout::Virtual { .. }) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let mut got = want.clone();
|
||||||
|
want.resolve_vds_mappings(file, ls).unwrap();
|
||||||
|
storage.reset();
|
||||||
|
got.resolve_vds_mappings_in(&storage, ls).unwrap();
|
||||||
|
assert_eq!(format!("{got:?}"), format!("{want:?}"));
|
||||||
|
assert!(matches!(&got, DataLayout::Virtual { mappings, .. } if !mappings.is_empty()));
|
||||||
|
assert_eq!(storage.reads(), 2);
|
||||||
|
virtuals += 1;
|
||||||
|
}
|
||||||
|
assert!(virtuals >= 1);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
+1155
-492
File diff suppressed because it is too large
Load Diff
@@ -7,6 +7,9 @@ use alloc::vec::Vec;
|
|||||||
|
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
|
|
||||||
|
/// Most dimensions a dataspace can have (`H5S_MAX_RANK`).
|
||||||
|
pub const MAX_RANK: u8 = 32;
|
||||||
|
|
||||||
/// Type of dataspace.
|
/// Type of dataspace.
|
||||||
#[derive(Debug, Clone, PartialEq)]
|
#[derive(Debug, Clone, PartialEq)]
|
||||||
pub enum DataspaceType {
|
pub enum DataspaceType {
|
||||||
@@ -67,6 +70,12 @@ impl Dataspace {
|
|||||||
let version = data[0];
|
let version = data[0];
|
||||||
let rank = data[1];
|
let rank = data[1];
|
||||||
let flags = data[2];
|
let flags = data[2];
|
||||||
|
// H5O__sdspace_decode's checks.
|
||||||
|
if rank > MAX_RANK {
|
||||||
|
return Err(FormatError::InvalidDataspace(
|
||||||
|
"simple dataspace dimensionality is too large",
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
let (space_type, header_size) = match version {
|
let (space_type, header_size) = match version {
|
||||||
1 => {
|
1 => {
|
||||||
@@ -88,6 +97,11 @@ impl Dataspace {
|
|||||||
2 => DataspaceType::Null,
|
2 => DataspaceType::Null,
|
||||||
_ => return Err(FormatError::InvalidDataspaceType(type_byte)),
|
_ => return Err(FormatError::InvalidDataspaceType(type_byte)),
|
||||||
};
|
};
|
||||||
|
if st != DataspaceType::Simple && rank > 0 {
|
||||||
|
return Err(FormatError::InvalidDataspace(
|
||||||
|
"invalid rank for scalar or NULL dataspace",
|
||||||
|
));
|
||||||
|
}
|
||||||
(st, 4usize)
|
(st, 4usize)
|
||||||
}
|
}
|
||||||
_ => return Err(FormatError::InvalidDataspaceVersion(version)),
|
_ => return Err(FormatError::InvalidDataspaceVersion(version)),
|
||||||
@@ -107,8 +121,13 @@ impl Dataspace {
|
|||||||
// Read max dimensions if flags bit 0 is set
|
// Read max dimensions if flags bit 0 is set
|
||||||
let max_dimensions = if flags & 0x01 != 0 {
|
let max_dimensions = if flags & 0x01 != 0 {
|
||||||
let mut max_dims = Vec::with_capacity(rank as usize);
|
let mut max_dims = Vec::with_capacity(rank as usize);
|
||||||
for _ in 0..rank {
|
for &dim in &dimensions {
|
||||||
let val = read_length(data, pos, length_size)?;
|
let val = read_length(data, pos, length_size)?;
|
||||||
|
if dim > val {
|
||||||
|
return Err(FormatError::InvalidDataspace(
|
||||||
|
"dataspace dimension size is greater than its maximum size",
|
||||||
|
));
|
||||||
|
}
|
||||||
max_dims.push(val);
|
max_dims.push(val);
|
||||||
pos += ls;
|
pos += ls;
|
||||||
}
|
}
|
||||||
@@ -176,7 +195,6 @@ impl Dataspace {
|
|||||||
match self.space_type {
|
match self.space_type {
|
||||||
DataspaceType::Null => Ok(0),
|
DataspaceType::Null => Ok(0),
|
||||||
DataspaceType::Scalar => Ok(1),
|
DataspaceType::Scalar => Ok(1),
|
||||||
DataspaceType::Simple if self.dimensions.is_empty() => Ok(0),
|
|
||||||
DataspaceType::Simple => self
|
DataspaceType::Simple => self
|
||||||
.dimensions
|
.dimensions
|
||||||
.iter()
|
.iter()
|
||||||
@@ -195,18 +213,14 @@ impl Dataspace {
|
|||||||
match self.space_type {
|
match self.space_type {
|
||||||
DataspaceType::Null => 0,
|
DataspaceType::Null => 0,
|
||||||
DataspaceType::Scalar => 1,
|
DataspaceType::Scalar => 1,
|
||||||
DataspaceType::Simple => {
|
// A simple dataspace of rank 0 holds one element, as in libhdf5
|
||||||
if self.dimensions.is_empty() {
|
// (the product of no dimensions). Saturate rather than wrap: a
|
||||||
0
|
// wrapped product could under-size a buffer. Size-critical
|
||||||
} else {
|
// callers use `checked_num_elements`.
|
||||||
// Saturate rather than wrap: a wrapped product could
|
DataspaceType::Simple => self
|
||||||
// under-size a buffer. Size-critical callers use
|
.dimensions
|
||||||
// `checked_num_elements`.
|
.iter()
|
||||||
self.dimensions
|
.fold(1u64, |acc, &d| acc.saturating_mul(d)),
|
||||||
.iter()
|
|
||||||
.fold(1u64, |acc, &d| acc.saturating_mul(d))
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -352,4 +366,44 @@ mod tests {
|
|||||||
let ds = Dataspace::parse(&data, 8).unwrap();
|
let ds = Dataspace::parse(&data, 8).unwrap();
|
||||||
assert_eq!(ds.max_dimensions, Some(vec![10]));
|
assert_eq!(ds.max_dimensions, Some(vec![10]));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// A simple dataspace of rank 0 (cve-2020-18494's `/dset1`) holds one
|
||||||
|
/// element in libhdf5, which h5py reads as shape `()`. It was 0.
|
||||||
|
#[test]
|
||||||
|
fn simple_rank_zero_holds_one_element() {
|
||||||
|
let data = build_v2_dataspace(0, 0, 1, &[], None);
|
||||||
|
let ds = Dataspace::parse(&data, 8).unwrap();
|
||||||
|
assert_eq!(ds.space_type, DataspaceType::Simple);
|
||||||
|
assert_eq!(ds.num_elements(), 1);
|
||||||
|
assert_eq!(ds.checked_num_elements().unwrap(), 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `H5O__sdspace_decode`'s checks.
|
||||||
|
#[test]
|
||||||
|
fn refuses_what_libhdf5_refuses() {
|
||||||
|
let too_many = build_v2_dataspace(33, 0, 1, &[1; 33], None);
|
||||||
|
assert!(matches!(
|
||||||
|
Dataspace::parse(&too_many, 8),
|
||||||
|
Err(FormatError::InvalidDataspace(_))
|
||||||
|
));
|
||||||
|
let scalar_with_rank = build_v2_dataspace(1, 0, 0, &[4], None);
|
||||||
|
assert!(matches!(
|
||||||
|
Dataspace::parse(&scalar_with_rank, 8),
|
||||||
|
Err(FormatError::InvalidDataspace(_))
|
||||||
|
));
|
||||||
|
let null_with_rank = build_v2_dataspace(1, 0, 2, &[4], None);
|
||||||
|
assert!(matches!(
|
||||||
|
Dataspace::parse(&null_with_rank, 8),
|
||||||
|
Err(FormatError::InvalidDataspace(_))
|
||||||
|
));
|
||||||
|
let over_max = build_v1_dataspace(2, 0x01, &[5, 20], Some(&[10, 10]));
|
||||||
|
assert!(matches!(
|
||||||
|
Dataspace::parse(&over_max, 8),
|
||||||
|
Err(FormatError::InvalidDataspace(_))
|
||||||
|
));
|
||||||
|
// 32 dimensions, and a size equal to the maximum or unlimited, are fine.
|
||||||
|
assert!(Dataspace::parse(&build_v2_dataspace(32, 0, 1, &[1; 32], None), 8).is_ok());
|
||||||
|
let at_max = build_v1_dataspace(2, 0x01, &[10, 20], Some(&[10, u64::MAX]));
|
||||||
|
assert!(Dataspace::parse(&at_max, 8).is_ok());
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -3,11 +3,14 @@
|
|||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
extern crate alloc;
|
extern crate alloc;
|
||||||
|
|
||||||
|
use crate::addr::saturating_usize;
|
||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
use alloc::{vec, vec::Vec};
|
use alloc::{vec, vec::Vec};
|
||||||
|
|
||||||
use crate::checksum::jenkins_lookup3;
|
use crate::checksum::jenkins_lookup3;
|
||||||
use crate::chunked_write::WrittenChunk;
|
use crate::chunked_write::{
|
||||||
|
WrittenChunk, filtered_chunk_size_len, push_addr, push_index_element, push_v4_chunk_dims,
|
||||||
|
};
|
||||||
|
|
||||||
/// Serialize a v4 Extensible Array layout message.
|
/// Serialize a v4 Extensible Array layout message.
|
||||||
pub(crate) fn serialize_v4_extensible_array(
|
pub(crate) fn serialize_v4_extensible_array(
|
||||||
@@ -24,45 +27,17 @@ pub(crate) fn serialize_v4_extensible_array(
|
|||||||
let ndims = chunk_dims.len() as u8 + 1;
|
let ndims = chunk_dims.len() as u8 + 1;
|
||||||
buf.push(ndims);
|
buf.push(ndims);
|
||||||
|
|
||||||
let max_dim = chunk_dims
|
push_v4_chunk_dims(&mut buf, chunk_dims, element_size);
|
||||||
.iter()
|
|
||||||
.map(|&d| d as u64)
|
|
||||||
.chain(core::iter::once(element_size as u64))
|
|
||||||
.max()
|
|
||||||
.unwrap_or(1);
|
|
||||||
let dim_encoded_len: u8 = if max_dim <= 0xFF {
|
|
||||||
1
|
|
||||||
} else if max_dim <= 0xFFFF {
|
|
||||||
2
|
|
||||||
} else {
|
|
||||||
4
|
|
||||||
};
|
|
||||||
buf.push(dim_encoded_len);
|
|
||||||
|
|
||||||
for &d in chunk_dims {
|
|
||||||
match dim_encoded_len {
|
|
||||||
1 => buf.push(d as u8),
|
|
||||||
2 => buf.extend_from_slice(&(d as u16).to_le_bytes()),
|
|
||||||
4 => buf.extend_from_slice(&d.to_le_bytes()),
|
|
||||||
_ => unreachable!("unexpected dim_encoded_len: {dim_encoded_len}"),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
match dim_encoded_len {
|
|
||||||
1 => buf.push(element_size as u8),
|
|
||||||
2 => buf.extend_from_slice(&(element_size as u16).to_le_bytes()),
|
|
||||||
4 => buf.extend_from_slice(&element_size.to_le_bytes()),
|
|
||||||
_ => unreachable!("unexpected dim_encoded_len: {dim_encoded_len}"),
|
|
||||||
}
|
|
||||||
|
|
||||||
// chunk index type = 4 (Extensible Array)
|
// chunk index type = 4 (Extensible Array)
|
||||||
buf.push(4);
|
buf.push(4);
|
||||||
|
|
||||||
// EA creation parameters (must match AEHD and HDF5 C library defaults)
|
// EA creation parameters (must match AEHD and HDF5 C library defaults)
|
||||||
buf.push(32); // max_nelmts_bits
|
buf.push(MAX_NELMTS_BITS);
|
||||||
buf.push(4); // idx_blk_elmts
|
buf.push(IDX_BLK_ELMTS);
|
||||||
buf.push(4); // super_blk_min_data_ptrs
|
buf.push(SUP_BLK_MIN_DATA_PTRS);
|
||||||
buf.push(16); // data_blk_min_elmts
|
buf.push(DATA_BLK_MIN_ELMTS);
|
||||||
buf.push(10); // max_dblk_page_nelmts_bits
|
buf.push(MAX_DBLK_PAGE_NELMTS_BITS);
|
||||||
|
|
||||||
// EA header address
|
// EA header address
|
||||||
match offset_size {
|
match offset_size {
|
||||||
@@ -74,304 +49,281 @@ pub(crate) fn serialize_v4_extensible_array(
|
|||||||
buf
|
buf
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// EA creation parameters — the HDF5 library's defaults for chunk indexes
|
||||||
|
// (`H5D_EARRAY_*`); the layout message above and the header must agree.
|
||||||
|
const MAX_NELMTS_BITS: u8 = 32;
|
||||||
|
const IDX_BLK_ELMTS: u8 = 4;
|
||||||
|
const SUP_BLK_MIN_DATA_PTRS: u8 = 4;
|
||||||
|
const DATA_BLK_MIN_ELMTS: u8 = 16;
|
||||||
|
const MAX_DBLK_PAGE_NELMTS_BITS: u8 = 10;
|
||||||
|
|
||||||
|
/// One data block of the array: its first element (relative to the end of
|
||||||
|
/// the index block's own elements), element count, and address when it is
|
||||||
|
/// allocated.
|
||||||
|
struct DataBlock {
|
||||||
|
start: usize,
|
||||||
|
nelmts: usize,
|
||||||
|
addr: Option<u64>,
|
||||||
|
}
|
||||||
|
|
||||||
/// Build a complete Extensible Array at a known absolute address.
|
/// Build a complete Extensible Array at a known absolute address.
|
||||||
///
|
///
|
||||||
/// For simplicity, we put all elements inline in the index block when the
|
/// `slots[i]` is the element at linear index `i` (see `chunk_grid`); `None`
|
||||||
/// number of chunks is small (up to idx_blk_elmts), otherwise use inline +
|
/// marks an unallocated chunk. The first `IDX_BLK_ELMTS` elements live in
|
||||||
/// direct data blocks.
|
/// the index block, the rest in data blocks grouped by super block level
|
||||||
|
/// exactly as `H5EA__hdr_init` sizes them: level `u` has `2^(u/2)` data
|
||||||
|
/// blocks of `DATA_BLK_MIN_ELMTS * 2^ceil(u/2)` elements. The data blocks of
|
||||||
|
/// the first levels are addressed straight from the index block; later
|
||||||
|
/// levels go through a super block (EASB). Data blocks larger than a page
|
||||||
|
/// (`2^MAX_DBLK_PAGE_NELMTS_BITS` elements) are paged, with their page-init
|
||||||
|
/// bits kept in the owning super block. Only blocks holding a defined element
|
||||||
|
/// are allocated; the rest keep the undefined address, as in a file the
|
||||||
|
/// library wrote.
|
||||||
pub fn build_extensible_array_at(
|
pub fn build_extensible_array_at(
|
||||||
chunks: &[WrittenChunk],
|
slots: &[Option<WrittenChunk>],
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
has_filters: bool,
|
has_filters: bool,
|
||||||
ea_base_address: u64,
|
ea_base_address: u64,
|
||||||
) -> Vec<u8> {
|
) -> Vec<u8> {
|
||||||
let os = offset_size as usize;
|
let os = offset_size as usize;
|
||||||
let num_elements = chunks.len();
|
let chunk_size_bytes = has_filters.then(|| filtered_chunk_size_len(slots));
|
||||||
|
let elem_size = os + chunk_size_bytes.map_or(0, |n| n + 4);
|
||||||
// Compute element encoding size (same logic as Fixed Array)
|
|
||||||
let chunk_size_bytes: usize = if has_filters {
|
|
||||||
let max_raw = chunks.iter().map(|c| c.raw_size).max().unwrap_or(1);
|
|
||||||
let log2_val = if max_raw <= 1 {
|
|
||||||
0
|
|
||||||
} else {
|
|
||||||
63 - max_raw.leading_zeros()
|
|
||||||
};
|
|
||||||
let len = 1 + ((log2_val + 8) / 8) as usize;
|
|
||||||
len.min(8)
|
|
||||||
} else {
|
|
||||||
0
|
|
||||||
};
|
|
||||||
|
|
||||||
let elem_size = if has_filters {
|
|
||||||
os + chunk_size_bytes + 4
|
|
||||||
} else {
|
|
||||||
os
|
|
||||||
};
|
|
||||||
|
|
||||||
let client_id: u8 = if has_filters { 1 } else { 0 };
|
let client_id: u8 = if has_filters { 1 } else { 0 };
|
||||||
|
let arr_off_size = (MAX_NELMTS_BITS as usize).div_ceil(8);
|
||||||
|
let page_nelmts = 1usize << MAX_DBLK_PAGE_NELMTS_BITS;
|
||||||
|
let idx_blk = IDX_BLK_ELMTS as usize;
|
||||||
|
|
||||||
// EA creation parameters — must match HDF5 C library defaults exactly
|
// Elements past the last defined one are never realised
|
||||||
let max_nelmts_bits: u8 = 32;
|
// (`max_idx_set` is one past the highest index ever set).
|
||||||
let idx_blk_elmts: u8 = 4;
|
let max_idx_set = slots.iter().rposition(Option::is_some).map_or(0, |i| i + 1);
|
||||||
let min_dblk_nelmts: u8 = 16;
|
let slots = &slots[..max_idx_set];
|
||||||
let super_blk_min_nelmts: u8 = 4;
|
let defined_in = |start: usize, n: usize| -> bool {
|
||||||
let max_dblk_nelmts_bits: u8 = 10;
|
let lo = idx_blk.saturating_add(start).min(slots.len());
|
||||||
|
let hi = idx_blk
|
||||||
|
.saturating_add(start)
|
||||||
|
.saturating_add(n)
|
||||||
|
.min(slots.len());
|
||||||
|
slots[lo..hi].iter().any(Option::is_some)
|
||||||
|
};
|
||||||
|
|
||||||
// EAHD size: fixed(12) + 6 stats(6*length_size) + addr(offset_size) + checksum(4)
|
// Super block levels: (ndblks, dblk_nelmts, first element).
|
||||||
|
let log2_dmin = (DATA_BLK_MIN_ELMTS as u32).trailing_zeros() as usize;
|
||||||
|
let nsblks = 1 + MAX_NELMTS_BITS as usize - log2_dmin;
|
||||||
|
let ndblk_addrs = 2 * (SUP_BLK_MIN_DATA_PTRS as usize - 1);
|
||||||
|
let mut levels: Vec<(usize, usize, usize)> = Vec::with_capacity(nsblks);
|
||||||
|
let mut start = 0usize;
|
||||||
|
for u in 0..nsblks {
|
||||||
|
let ndblks = 1usize << (u / 2);
|
||||||
|
let nelmts = (DATA_BLK_MIN_ELMTS as usize) << u.div_ceil(2);
|
||||||
|
levels.push((ndblks, nelmts, start));
|
||||||
|
// Saturate: on 32-bit targets the last levels only need to compare
|
||||||
|
// as "beyond the end".
|
||||||
|
start = start.saturating_add(ndblks.saturating_mul(nelmts));
|
||||||
|
}
|
||||||
|
// Levels whose data blocks the index block addresses directly.
|
||||||
|
let mut direct_levels = 0;
|
||||||
|
let mut n = 0;
|
||||||
|
while n < ndblk_addrs {
|
||||||
|
n += levels[direct_levels].0;
|
||||||
|
direct_levels += 1;
|
||||||
|
}
|
||||||
|
let nsblk_addrs = nsblks - direct_levels;
|
||||||
|
|
||||||
|
let dblk_size = |nelmts: usize| -> usize {
|
||||||
|
let prefix = 4 + 1 + 1 + os + arr_off_size + 4;
|
||||||
|
if nelmts > page_nelmts {
|
||||||
|
prefix + (nelmts / page_nelmts) * (page_nelmts * elem_size + 4)
|
||||||
|
} else {
|
||||||
|
prefix + nelmts * elem_size
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let sblk_bitmap_len = |ndblks: usize, nelmts: usize| -> usize {
|
||||||
|
if nelmts > page_nelmts {
|
||||||
|
ndblks * (nelmts / page_nelmts).div_ceil(8)
|
||||||
|
} else {
|
||||||
|
0
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
// Plan addresses: header, index block, the direct data blocks, then each
|
||||||
|
// allocated super block followed by its allocated data blocks.
|
||||||
let aehd_size = 4 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 6 * length_size as usize + os + 4;
|
let aehd_size = 4 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 6 * length_size as usize + os + 4;
|
||||||
let aeib_address = ea_base_address + aehd_size as u64;
|
let aeib_address = ea_base_address + aehd_size as u64;
|
||||||
|
let aeib_size = 4 + 1 + 1 + os + idx_blk * elem_size + ndblk_addrs * os + nsblk_addrs * os + 4;
|
||||||
|
let mut cursor = aeib_address + aeib_size as u64;
|
||||||
|
|
||||||
// Determine how many elements go inline vs data blocks
|
let mut ndata_blks = 0u64;
|
||||||
let n_inline = (idx_blk_elmts as usize).min(num_elements);
|
let mut data_blk_size = 0u64;
|
||||||
let remaining_after_inline = num_elements.saturating_sub(n_inline);
|
let mut nsuper_blks = 0u64;
|
||||||
|
let mut super_blk_size = 0u64;
|
||||||
|
let mut realized = idx_blk as u64;
|
||||||
|
|
||||||
// Compute super block layout per HDF5 spec
|
let mut plan_dblk = |cursor: &mut u64, start: usize, nelmts: usize| -> DataBlock {
|
||||||
let sblk_min = super_blk_min_nelmts as usize;
|
let addr = defined_in(start, nelmts).then(|| {
|
||||||
let log2_dblk_min = if min_dblk_nelmts <= 1 {
|
let a = *cursor;
|
||||||
0
|
let size = dblk_size(nelmts) as u64;
|
||||||
} else {
|
*cursor += size;
|
||||||
(min_dblk_nelmts as u32).trailing_zeros() as usize
|
ndata_blks += 1;
|
||||||
|
data_blk_size += size;
|
||||||
|
realized += nelmts as u64;
|
||||||
|
a
|
||||||
|
});
|
||||||
|
DataBlock {
|
||||||
|
start,
|
||||||
|
nelmts,
|
||||||
|
addr,
|
||||||
|
}
|
||||||
};
|
};
|
||||||
let nsblks = (max_nelmts_bits as usize).saturating_sub(log2_dblk_min) + 1;
|
|
||||||
|
|
||||||
// Direct data block addresses (from super blocks 0..sblk_min-1)
|
let mut direct: Vec<DataBlock> = Vec::with_capacity(ndblk_addrs);
|
||||||
let mut dblk_sizes: Vec<usize> = Vec::new();
|
for &(ndblks, nelmts, first) in &levels[..direct_levels] {
|
||||||
for sblk_idx in 0..sblk_min.min(nsblks) {
|
for k in 0..ndblks {
|
||||||
let ndblks = 1usize << (sblk_idx / 2);
|
direct.push(plan_dblk(&mut cursor, first + k * nelmts, nelmts));
|
||||||
let dblk_nelmts = (min_dblk_nelmts as usize) * (1 << sblk_idx.div_ceil(2));
|
|
||||||
for _ in 0..ndblks {
|
|
||||||
dblk_sizes.push(dblk_nelmts);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
let n_direct_dblks = dblk_sizes.len();
|
// (super block address, level, its data blocks)
|
||||||
|
let mut supers: Vec<(Option<u64>, usize, Vec<DataBlock>)> = Vec::with_capacity(nsblk_addrs);
|
||||||
// Super block addresses (for super blocks sblk_min..nsblks-1)
|
for (u, &(ndblks, nelmts, first)) in levels.iter().enumerate().skip(direct_levels) {
|
||||||
let n_sblk_addrs = nsblks.saturating_sub(sblk_min);
|
if !defined_in(first, ndblks.saturating_mul(nelmts)) {
|
||||||
|
supers.push((None, u, Vec::new()));
|
||||||
// EAIB size
|
continue;
|
||||||
let aeib_size = 4
|
|
||||||
+ 1
|
|
||||||
+ 1
|
|
||||||
+ os
|
|
||||||
+ idx_blk_elmts as usize * elem_size
|
|
||||||
+ n_direct_dblks * os
|
|
||||||
+ n_sblk_addrs * os
|
|
||||||
+ 4;
|
|
||||||
|
|
||||||
// Build AEHD
|
|
||||||
let mut aehd = Vec::with_capacity(aehd_size);
|
|
||||||
aehd.extend_from_slice(b"EAHD");
|
|
||||||
aehd.push(0); // version
|
|
||||||
aehd.push(client_id);
|
|
||||||
aehd.push(elem_size as u8);
|
|
||||||
aehd.push(max_nelmts_bits);
|
|
||||||
aehd.push(idx_blk_elmts);
|
|
||||||
aehd.push(min_dblk_nelmts);
|
|
||||||
aehd.push(super_blk_min_nelmts);
|
|
||||||
aehd.push(max_dblk_nelmts_bits);
|
|
||||||
|
|
||||||
// Count data blocks that will have chunks
|
|
||||||
let n_active_dblks: u64 = if remaining_after_inline > 0 {
|
|
||||||
let mut count = 0u64;
|
|
||||||
let mut ci = n_inline;
|
|
||||||
for &sz in &dblk_sizes {
|
|
||||||
if ci < num_elements {
|
|
||||||
count += 1;
|
|
||||||
ci += sz;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
count
|
let sb_size =
|
||||||
} else {
|
4 + 1 + 1 + os + arr_off_size + sblk_bitmap_len(ndblks, nelmts) + ndblks * os + 4;
|
||||||
0
|
let sb_addr = cursor;
|
||||||
};
|
cursor += sb_size as u64;
|
||||||
let blk_off_size = (max_nelmts_bits as usize).div_ceil(8);
|
nsuper_blks += 1;
|
||||||
let aedb_header_overhead = 4 + 1 + 1 + os + blk_off_size + 4;
|
super_blk_size += sb_size as u64;
|
||||||
let data_blk_total_size: u64 = if remaining_after_inline > 0 {
|
let dblks = (0..ndblks)
|
||||||
let mut total = 0u64;
|
.map(|k| plan_dblk(&mut cursor, first + k * nelmts, nelmts))
|
||||||
let mut ci = n_inline;
|
.collect();
|
||||||
for &sz in &dblk_sizes {
|
supers.push((Some(sb_addr), u, dblks));
|
||||||
if ci < num_elements {
|
}
|
||||||
total += (aedb_header_overhead + sz * elem_size) as u64;
|
|
||||||
ci += sz;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
total
|
|
||||||
} else {
|
|
||||||
0
|
|
||||||
};
|
|
||||||
let max_idx_set: u64 = if remaining_after_inline > 0 {
|
|
||||||
let mut max_set = idx_blk_elmts as u64;
|
|
||||||
let mut ci = n_inline;
|
|
||||||
for &sz in &dblk_sizes {
|
|
||||||
if ci < num_elements {
|
|
||||||
max_set += sz as u64;
|
|
||||||
ci += sz;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
max_set
|
|
||||||
} else {
|
|
||||||
idx_blk_elmts as u64
|
|
||||||
};
|
|
||||||
|
|
||||||
|
let slot = |i: usize| slots.get(i).and_then(Option::as_ref);
|
||||||
let write_length = |buf: &mut Vec<u8>, val: u64| match length_size {
|
let write_length = |buf: &mut Vec<u8>, val: u64| match length_size {
|
||||||
4 => buf.extend_from_slice(&(val as u32).to_le_bytes()),
|
4 => buf.extend_from_slice(&(val as u32).to_le_bytes()),
|
||||||
_ => buf.extend_from_slice(&val.to_le_bytes()),
|
_ => buf.extend_from_slice(&val.to_le_bytes()),
|
||||||
};
|
};
|
||||||
let write_addr = |buf: &mut Vec<u8>, val: u64| match offset_size {
|
let write_addr_opt = |buf: &mut Vec<u8>, addr: Option<u64>| match addr {
|
||||||
4 => buf.extend_from_slice(&(val as u32).to_le_bytes()),
|
Some(a) => push_addr(buf, a, offset_size),
|
||||||
_ => buf.extend_from_slice(&val.to_le_bytes()),
|
None => buf.extend(core::iter::repeat_n(0xFF, os)),
|
||||||
|
};
|
||||||
|
let block_prefix = |buf: &mut Vec<u8>, sig: &[u8; 4], block_off: usize| {
|
||||||
|
buf.extend_from_slice(sig);
|
||||||
|
buf.push(0); // version
|
||||||
|
buf.push(client_id);
|
||||||
|
push_addr(buf, ea_base_address, offset_size);
|
||||||
|
buf.extend_from_slice(&(block_off as u64).to_le_bytes()[..arr_off_size]);
|
||||||
|
};
|
||||||
|
// Serialise one data block (paged or not) onto `out`.
|
||||||
|
let write_dblk = |out: &mut Vec<u8>, db: &DataBlock| {
|
||||||
|
let at = out.len();
|
||||||
|
block_prefix(out, b"EADB", db.start);
|
||||||
|
let first = idx_blk + db.start;
|
||||||
|
if db.nelmts > page_nelmts {
|
||||||
|
// Paged: the prefix carries only its own checksum; each page
|
||||||
|
// follows with one of its own.
|
||||||
|
let sum = jenkins_lookup3(&out[at..]);
|
||||||
|
out.extend_from_slice(&sum.to_le_bytes());
|
||||||
|
for p in 0..db.nelmts / page_nelmts {
|
||||||
|
let page_at = out.len();
|
||||||
|
for e in 0..page_nelmts {
|
||||||
|
let i = first + p * page_nelmts + e;
|
||||||
|
push_index_element(out, slot(i), offset_size, chunk_size_bytes);
|
||||||
|
}
|
||||||
|
let sum = jenkins_lookup3(&out[page_at..]);
|
||||||
|
out.extend_from_slice(&sum.to_le_bytes());
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
for i in first..first + db.nelmts {
|
||||||
|
push_index_element(out, slot(i), offset_size, chunk_size_bytes);
|
||||||
|
}
|
||||||
|
let sum = jenkins_lookup3(&out[at..]);
|
||||||
|
out.extend_from_slice(&sum.to_le_bytes());
|
||||||
|
}
|
||||||
|
debug_assert_eq!(out.len() - at, dblk_size(db.nelmts));
|
||||||
};
|
};
|
||||||
|
|
||||||
write_length(&mut aehd, 0);
|
// Header (EAHD). The six statistics are, in order: super blocks, their
|
||||||
write_length(&mut aehd, 0);
|
// bytes, data blocks, their bytes, max index set, elements realised.
|
||||||
write_length(&mut aehd, n_active_dblks);
|
let mut out = Vec::with_capacity(saturating_usize(cursor - ea_base_address));
|
||||||
write_length(&mut aehd, data_blk_total_size);
|
out.extend_from_slice(b"EAHD");
|
||||||
write_length(&mut aehd, num_elements as u64);
|
out.push(0); // version
|
||||||
write_length(&mut aehd, max_idx_set);
|
out.push(client_id);
|
||||||
|
out.push(elem_size as u8);
|
||||||
|
out.push(MAX_NELMTS_BITS);
|
||||||
|
out.push(IDX_BLK_ELMTS);
|
||||||
|
out.push(DATA_BLK_MIN_ELMTS);
|
||||||
|
out.push(SUP_BLK_MIN_DATA_PTRS);
|
||||||
|
out.push(MAX_DBLK_PAGE_NELMTS_BITS);
|
||||||
|
write_length(&mut out, nsuper_blks);
|
||||||
|
write_length(&mut out, super_blk_size);
|
||||||
|
write_length(&mut out, ndata_blks);
|
||||||
|
write_length(&mut out, data_blk_size);
|
||||||
|
write_length(&mut out, max_idx_set as u64);
|
||||||
|
write_length(&mut out, realized);
|
||||||
|
push_addr(&mut out, aeib_address, offset_size);
|
||||||
|
let sum = jenkins_lookup3(&out);
|
||||||
|
out.extend_from_slice(&sum.to_le_bytes());
|
||||||
|
debug_assert_eq!(out.len(), aehd_size);
|
||||||
|
|
||||||
write_addr(&mut aehd, aeib_address);
|
// Index block (EAIB): inline elements, data block and super block
|
||||||
|
// addresses.
|
||||||
let aehd_checksum = jenkins_lookup3(&aehd);
|
let ib_start = out.len();
|
||||||
aehd.extend_from_slice(&aehd_checksum.to_le_bytes());
|
out.extend_from_slice(b"EAIB");
|
||||||
debug_assert_eq!(aehd.len(), aehd_size);
|
out.push(0);
|
||||||
|
out.push(client_id);
|
||||||
// Build AEIB
|
push_addr(&mut out, ea_base_address, offset_size);
|
||||||
let mut aeib = Vec::with_capacity(aeib_size);
|
for i in 0..idx_blk {
|
||||||
aeib.extend_from_slice(b"EAIB");
|
push_index_element(&mut out, slot(i), offset_size, chunk_size_bytes);
|
||||||
aeib.push(0);
|
|
||||||
aeib.push(client_id);
|
|
||||||
|
|
||||||
match offset_size {
|
|
||||||
4 => aeib.extend_from_slice(&(ea_base_address as u32).to_le_bytes()),
|
|
||||||
8 => aeib.extend_from_slice(&ea_base_address.to_le_bytes()),
|
|
||||||
_ => aeib.extend_from_slice(&ea_base_address.to_le_bytes()),
|
|
||||||
}
|
}
|
||||||
|
for db in &direct {
|
||||||
// Inline elements
|
write_addr_opt(&mut out, db.addr);
|
||||||
#[allow(clippy::needless_range_loop)]
|
|
||||||
for i in 0..idx_blk_elmts as usize {
|
|
||||||
if i < n_inline {
|
|
||||||
write_chunk_element(
|
|
||||||
&mut aeib,
|
|
||||||
&chunks[i],
|
|
||||||
offset_size,
|
|
||||||
has_filters,
|
|
||||||
chunk_size_bytes,
|
|
||||||
);
|
|
||||||
} else {
|
|
||||||
write_undefined_element(&mut aeib, offset_size, has_filters, chunk_size_bytes);
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
for (sb_addr, _, _) in &supers {
|
||||||
|
write_addr_opt(&mut out, *sb_addr);
|
||||||
|
}
|
||||||
|
let sum = jenkins_lookup3(&out[ib_start..]);
|
||||||
|
out.extend_from_slice(&sum.to_le_bytes());
|
||||||
|
debug_assert_eq!(out.len() - ib_start, aeib_size);
|
||||||
|
|
||||||
// Data block addresses + build data blocks
|
for db in direct.iter().filter(|d| d.addr.is_some()) {
|
||||||
let mut data_blocks_buf = Vec::new();
|
write_dblk(&mut out, db);
|
||||||
let dblks_base = aeib_address + aeib_size as u64;
|
}
|
||||||
let mut dblk_cursor = dblks_base;
|
for (sb_addr, u, dblks) in &supers {
|
||||||
let mut chunk_idx = n_inline;
|
if sb_addr.is_none() {
|
||||||
|
|
||||||
for &nelmts in &dblk_sizes {
|
|
||||||
if chunk_idx >= num_elements {
|
|
||||||
match offset_size {
|
|
||||||
4 => aeib.extend_from_slice(&u32::MAX.to_le_bytes()),
|
|
||||||
8 => aeib.extend_from_slice(&u64::MAX.to_le_bytes()),
|
|
||||||
_ => aeib.extend_from_slice(&u64::MAX.to_le_bytes()),
|
|
||||||
}
|
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
let (ndblks, nelmts, first) = levels[*u];
|
||||||
match offset_size {
|
let sb_start = out.len();
|
||||||
4 => aeib.extend_from_slice(&(dblk_cursor as u32).to_le_bytes()),
|
block_prefix(&mut out, b"EASB", first);
|
||||||
8 => aeib.extend_from_slice(&dblk_cursor.to_le_bytes()),
|
if nelmts > page_nelmts {
|
||||||
_ => aeib.extend_from_slice(&dblk_cursor.to_le_bytes()),
|
// Page-init bits, `npages` per data block, packed MSB-first
|
||||||
}
|
// (`H5VM_bit_set`): every page of an allocated data block is
|
||||||
|
// written.
|
||||||
// Build EADB
|
let npages = nelmts / page_nelmts;
|
||||||
let mut aedb = Vec::new();
|
let mut bitmap = vec![0u8; sblk_bitmap_len(ndblks, nelmts)];
|
||||||
aedb.extend_from_slice(b"EADB");
|
for (k, db) in dblks.iter().enumerate() {
|
||||||
aedb.push(0);
|
if db.addr.is_some() {
|
||||||
aedb.push(client_id);
|
for p in 0..npages {
|
||||||
match offset_size {
|
let bit = k * npages + p;
|
||||||
4 => aedb.extend_from_slice(&(ea_base_address as u32).to_le_bytes()),
|
bitmap[bit / 8] |= 0x80 >> (bit % 8);
|
||||||
8 => aedb.extend_from_slice(&ea_base_address.to_le_bytes()),
|
}
|
||||||
_ => aedb.extend_from_slice(&ea_base_address.to_le_bytes()),
|
}
|
||||||
}
|
|
||||||
|
|
||||||
let blk_off_size = (max_nelmts_bits as usize).div_ceil(8);
|
|
||||||
let blk_off_val = (chunk_idx - n_inline) as u64;
|
|
||||||
aedb.extend_from_slice(&blk_off_val.to_le_bytes()[..blk_off_size]);
|
|
||||||
|
|
||||||
for slot in 0..nelmts {
|
|
||||||
if chunk_idx + slot < num_elements {
|
|
||||||
write_chunk_element(
|
|
||||||
&mut aedb,
|
|
||||||
&chunks[chunk_idx + slot],
|
|
||||||
offset_size,
|
|
||||||
has_filters,
|
|
||||||
chunk_size_bytes,
|
|
||||||
);
|
|
||||||
} else {
|
|
||||||
write_undefined_element(&mut aedb, offset_size, has_filters, chunk_size_bytes);
|
|
||||||
}
|
}
|
||||||
|
out.extend_from_slice(&bitmap);
|
||||||
}
|
}
|
||||||
|
for db in dblks {
|
||||||
let aedb_checksum = jenkins_lookup3(&aedb);
|
write_addr_opt(&mut out, db.addr);
|
||||||
aedb.extend_from_slice(&aedb_checksum.to_le_bytes());
|
}
|
||||||
|
let sum = jenkins_lookup3(&out[sb_start..]);
|
||||||
dblk_cursor += aedb.len() as u64;
|
out.extend_from_slice(&sum.to_le_bytes());
|
||||||
data_blocks_buf.extend_from_slice(&aedb);
|
for db in dblks.iter().filter(|d| d.addr.is_some()) {
|
||||||
chunk_idx += nelmts;
|
write_dblk(&mut out, db);
|
||||||
}
|
|
||||||
|
|
||||||
// Super block addresses (all undefined)
|
|
||||||
for _ in 0..n_sblk_addrs {
|
|
||||||
match offset_size {
|
|
||||||
4 => aeib.extend_from_slice(&u32::MAX.to_le_bytes()),
|
|
||||||
8 => aeib.extend_from_slice(&u64::MAX.to_le_bytes()),
|
|
||||||
_ => aeib.extend_from_slice(&u64::MAX.to_le_bytes()),
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
debug_assert_eq!(out.len() as u64, cursor - ea_base_address);
|
||||||
let aeib_checksum = jenkins_lookup3(&aeib);
|
out
|
||||||
aeib.extend_from_slice(&aeib_checksum.to_le_bytes());
|
|
||||||
debug_assert_eq!(aeib.len(), aeib_size);
|
|
||||||
|
|
||||||
let mut combined = aehd;
|
|
||||||
combined.extend_from_slice(&aeib);
|
|
||||||
combined.extend_from_slice(&data_blocks_buf);
|
|
||||||
combined
|
|
||||||
}
|
|
||||||
|
|
||||||
fn write_chunk_element(
|
|
||||||
buf: &mut Vec<u8>,
|
|
||||||
chunk: &WrittenChunk,
|
|
||||||
offset_size: u8,
|
|
||||||
has_filters: bool,
|
|
||||||
chunk_size_bytes: usize,
|
|
||||||
) {
|
|
||||||
match offset_size {
|
|
||||||
4 => buf.extend_from_slice(&(chunk.address as u32).to_le_bytes()),
|
|
||||||
8 => buf.extend_from_slice(&chunk.address.to_le_bytes()),
|
|
||||||
_ => buf.extend_from_slice(&chunk.address.to_le_bytes()),
|
|
||||||
}
|
|
||||||
if has_filters {
|
|
||||||
let cs_bytes = chunk.compressed_size.to_le_bytes();
|
|
||||||
buf.extend_from_slice(&cs_bytes[..chunk_size_bytes]);
|
|
||||||
buf.extend_from_slice(&chunk.filter_mask.to_le_bytes());
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn write_undefined_element(
|
|
||||||
buf: &mut Vec<u8>,
|
|
||||||
offset_size: u8,
|
|
||||||
has_filters: bool,
|
|
||||||
chunk_size_bytes: usize,
|
|
||||||
) {
|
|
||||||
let os = offset_size as usize;
|
|
||||||
// Use extend with repeat to avoid heap-allocating a temporary Vec on each call.
|
|
||||||
buf.extend(core::iter::repeat_n(0xFF, os));
|
|
||||||
if has_filters {
|
|
||||||
buf.extend(core::iter::repeat_n(0x00, chunk_size_bytes));
|
|
||||||
buf.extend_from_slice(&0u32.to_le_bytes());
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -12,7 +12,11 @@ use std::string::String;
|
|||||||
use core::fmt;
|
use core::fmt;
|
||||||
|
|
||||||
/// Errors that can occur when parsing HDF5 binary format structures.
|
/// Errors that can occur when parsing HDF5 binary format structures.
|
||||||
|
///
|
||||||
|
/// Non-exhaustive: new failure modes (new storage backends, new file
|
||||||
|
/// features) add variants, so a `match` needs a wildcard arm.
|
||||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||||
|
#[non_exhaustive]
|
||||||
pub enum FormatError {
|
pub enum FormatError {
|
||||||
/// The HDF5 magic signature was not found at any valid offset.
|
/// The HDF5 magic signature was not found at any valid offset.
|
||||||
SignatureNotFound,
|
SignatureNotFound,
|
||||||
@@ -80,6 +84,9 @@ pub enum FormatError {
|
|||||||
InvalidLocalHeapSignature,
|
InvalidLocalHeapSignature,
|
||||||
/// Invalid local heap version.
|
/// Invalid local heap version.
|
||||||
InvalidLocalHeapVersion(u8),
|
InvalidLocalHeapVersion(u8),
|
||||||
|
/// A local heap's free list points outside its data segment (libhdf5:
|
||||||
|
/// "bad heap free list").
|
||||||
|
InvalidLocalHeapFreeList,
|
||||||
/// Invalid B-tree v1 signature.
|
/// Invalid B-tree v1 signature.
|
||||||
InvalidBTreeSignature,
|
InvalidBTreeSignature,
|
||||||
/// Invalid B-tree node type.
|
/// Invalid B-tree node type.
|
||||||
@@ -117,6 +124,14 @@ pub enum FormatError {
|
|||||||
/// A message is marked shared but was parsed without access to the file,
|
/// A message is marked shared but was parsed without access to the file,
|
||||||
/// so the reference to the real message could not be followed.
|
/// so the reference to the real message could not be followed.
|
||||||
UnresolvedSharedMessage,
|
UnresolvedSharedMessage,
|
||||||
|
/// A shared-message reference points at an object header that holds no
|
||||||
|
/// (unshared) message of the referenced type (raw message type id).
|
||||||
|
SharedMessageTargetMissing(u16),
|
||||||
|
/// A superblock was parsed at a non-zero offset of the buffer (the file
|
||||||
|
/// has a user block of this many bytes). HDF5 addresses are relative to
|
||||||
|
/// the superblock, so the buffer must start there: see
|
||||||
|
/// `signature::split_user_block`.
|
||||||
|
UserBlockNotStripped(u64),
|
||||||
/// A selection does not fit the dataset it was applied to (wrong rank, or
|
/// A selection does not fit the dataset it was applied to (wrong rank, or
|
||||||
/// it reaches past a dimension's extent).
|
/// it reaches past a dimension's extent).
|
||||||
SelectionOutOfBounds(String),
|
SelectionOutOfBounds(String),
|
||||||
@@ -190,6 +205,55 @@ pub enum FormatError {
|
|||||||
DuplicateDatasetName(String),
|
DuplicateDatasetName(String),
|
||||||
/// Integer overflow in size computation (malformed data protection).
|
/// Integer overflow in size computation (malformed data protection).
|
||||||
Overflow(String),
|
Overflow(String),
|
||||||
|
/// An object header that libhdf5 refuses to load (the reason is
|
||||||
|
/// libhdf5's own error text): a misaligned or overrunning message, a
|
||||||
|
/// wrong message count, contradictory message flags, a message of a
|
||||||
|
/// class that cannot be shared flagged shareable, …
|
||||||
|
InvalidObjectHeader(&'static str),
|
||||||
|
/// A datatype message libhdf5 refuses to decode (the reason is
|
||||||
|
/// libhdf5's own error text): size 0, bit fields outside the type,
|
||||||
|
/// an empty enum name, a compound member outside its compound, …
|
||||||
|
InvalidDatatype(String),
|
||||||
|
/// A chunked layout whose chunk dimensions libhdf5 refuses: a zero
|
||||||
|
/// dimension, a rank that does not match the dataspace, an element size
|
||||||
|
/// that is not the datatype's, or a chunk of 4 GiB or more indexed by a
|
||||||
|
/// version-1 B-tree.
|
||||||
|
InvalidChunkDimensions(String),
|
||||||
|
/// The superblock's end-of-file address lies past the end of the file:
|
||||||
|
/// the file was truncated (libhdf5 refuses to open it).
|
||||||
|
TruncatedFile {
|
||||||
|
/// End of file recorded in the superblock (relative to byte 0).
|
||||||
|
stored_eof: u64,
|
||||||
|
/// The file's actual length in bytes.
|
||||||
|
actual_len: u64,
|
||||||
|
},
|
||||||
|
/// A link libhdf5 refuses to list: a symbol-table entry with an empty
|
||||||
|
/// name ("invalid link name"). Listing the group fails, as in libhdf5.
|
||||||
|
InvalidLinkName,
|
||||||
|
/// A dataspace message libhdf5 refuses to decode (the reason is
|
||||||
|
/// libhdf5's own error text): more than 32 dimensions, a rank on a
|
||||||
|
/// scalar or null dataspace, a dimension larger than its maximum.
|
||||||
|
InvalidDataspace(&'static str),
|
||||||
|
/// A dataset whose storage libhdf5 refuses when it opens the dataset
|
||||||
|
/// (the reason is libhdf5's own error text): an element count times
|
||||||
|
/// element size that overflows, contiguous storage past the end of the
|
||||||
|
/// file, compact data of the wrong size.
|
||||||
|
InvalidDatasetStorage(&'static str),
|
||||||
|
/// A superblock extension message libhdf5 refuses to decode when it
|
||||||
|
/// opens the file (the reason is libhdf5's own error text): a File Space
|
||||||
|
/// Info message that runs off its end or has a bad page size, a metadata
|
||||||
|
/// cache image outside the file, …
|
||||||
|
InvalidSuperblockExtension(&'static str),
|
||||||
|
/// A metadata cache image block libhdf5 refuses to load (the reason is
|
||||||
|
/// libhdf5's own error text).
|
||||||
|
InvalidCacheImage(&'static str),
|
||||||
|
/// The [`Storage`](crate::storage::Storage) backend failed to serve a
|
||||||
|
/// read (an I/O or network error, or a short read inside the file).
|
||||||
|
Storage(String),
|
||||||
|
/// The operation still needs the whole file as one slice and the
|
||||||
|
/// [`Storage`](crate::storage::Storage) backend has no contiguous view
|
||||||
|
/// (`as_contiguous()` is `None`); the text names the operation.
|
||||||
|
ContiguousStorageRequired(&'static str),
|
||||||
}
|
}
|
||||||
|
|
||||||
impl fmt::Display for FormatError {
|
impl fmt::Display for FormatError {
|
||||||
@@ -270,6 +334,9 @@ impl fmt::Display for FormatError {
|
|||||||
FormatError::InvalidLocalHeapSignature => {
|
FormatError::InvalidLocalHeapSignature => {
|
||||||
write!(f, "invalid local heap signature")
|
write!(f, "invalid local heap signature")
|
||||||
}
|
}
|
||||||
|
FormatError::InvalidLocalHeapFreeList => {
|
||||||
|
write!(f, "bad local heap free list")
|
||||||
|
}
|
||||||
FormatError::InvalidLocalHeapVersion(v) => {
|
FormatError::InvalidLocalHeapVersion(v) => {
|
||||||
write!(f, "invalid local heap version: {v}")
|
write!(f, "invalid local heap version: {v}")
|
||||||
}
|
}
|
||||||
@@ -339,6 +406,16 @@ impl fmt::Display for FormatError {
|
|||||||
FormatError::SelectionOutOfBounds(msg) => {
|
FormatError::SelectionOutOfBounds(msg) => {
|
||||||
write!(f, "selection out of bounds: {msg}")
|
write!(f, "selection out of bounds: {msg}")
|
||||||
}
|
}
|
||||||
|
FormatError::UserBlockNotStripped(n) => write!(
|
||||||
|
f,
|
||||||
|
"file has a {n}-byte user block: parse the bytes from the superblock on \
|
||||||
|
(signature::split_user_block)"
|
||||||
|
),
|
||||||
|
FormatError::SharedMessageTargetMissing(t) => write!(
|
||||||
|
f,
|
||||||
|
"shared message reference points at an object header with no message of type \
|
||||||
|
{t:#06x}"
|
||||||
|
),
|
||||||
FormatError::UnresolvedSharedMessage => write!(
|
FormatError::UnresolvedSharedMessage => write!(
|
||||||
f,
|
f,
|
||||||
"message is shared but no file data was available to resolve it"
|
"message is shared but no file data was available to resolve it"
|
||||||
@@ -382,9 +459,17 @@ impl fmt::Display for FormatError {
|
|||||||
FormatError::InvalidFilterPipelineVersion(v) => {
|
FormatError::InvalidFilterPipelineVersion(v) => {
|
||||||
write!(f, "invalid filter pipeline version: {v}")
|
write!(f, "invalid filter pipeline version: {v}")
|
||||||
}
|
}
|
||||||
FormatError::UnsupportedFilter(id) => {
|
FormatError::UnsupportedFilter(id) => match crate::filter_registry::known_filter(*id) {
|
||||||
write!(f, "unsupported filter: {id}")
|
Some((name, Some(feature))) => write!(
|
||||||
}
|
f,
|
||||||
|
"unsupported filter: {id} ({name}; this build lacks the `{feature}` feature)"
|
||||||
|
),
|
||||||
|
Some((name, None)) => write!(
|
||||||
|
f,
|
||||||
|
"unsupported filter: {id} ({name}, not implemented by clawhdf5)"
|
||||||
|
),
|
||||||
|
None => write!(f, "unsupported filter: {id}"),
|
||||||
|
},
|
||||||
FormatError::FilterError(msg) => {
|
FormatError::FilterError(msg) => {
|
||||||
write!(f, "filter error: {msg}")
|
write!(f, "filter error: {msg}")
|
||||||
}
|
}
|
||||||
@@ -421,6 +506,50 @@ impl fmt::Display for FormatError {
|
|||||||
FormatError::Overflow(msg) => {
|
FormatError::Overflow(msg) => {
|
||||||
write!(f, "integer overflow: {msg}")
|
write!(f, "integer overflow: {msg}")
|
||||||
}
|
}
|
||||||
|
FormatError::InvalidObjectHeader(why) => {
|
||||||
|
write!(f, "corrupt object header: {why}")
|
||||||
|
}
|
||||||
|
FormatError::InvalidDatatype(why) => {
|
||||||
|
write!(f, "invalid datatype: {why}")
|
||||||
|
}
|
||||||
|
FormatError::InvalidChunkDimensions(why) => {
|
||||||
|
write!(f, "invalid chunk dimensions: {why}")
|
||||||
|
}
|
||||||
|
FormatError::TruncatedFile {
|
||||||
|
stored_eof,
|
||||||
|
actual_len,
|
||||||
|
} => {
|
||||||
|
write!(
|
||||||
|
f,
|
||||||
|
"truncated file: the superblock records end of file {stored_eof}, \
|
||||||
|
but the file is {actual_len} bytes"
|
||||||
|
)
|
||||||
|
}
|
||||||
|
FormatError::InvalidLinkName => {
|
||||||
|
write!(f, "invalid link name: a group entry has an empty name")
|
||||||
|
}
|
||||||
|
FormatError::InvalidDataspace(why) => {
|
||||||
|
write!(f, "invalid dataspace: {why}")
|
||||||
|
}
|
||||||
|
FormatError::InvalidDatasetStorage(why) => {
|
||||||
|
write!(f, "invalid dataset storage: {why}")
|
||||||
|
}
|
||||||
|
FormatError::InvalidSuperblockExtension(why) => {
|
||||||
|
write!(f, "invalid superblock extension: {why}")
|
||||||
|
}
|
||||||
|
FormatError::InvalidCacheImage(why) => {
|
||||||
|
write!(f, "invalid metadata cache image: {why}")
|
||||||
|
}
|
||||||
|
FormatError::Storage(why) => {
|
||||||
|
write!(f, "storage read failed: {why}")
|
||||||
|
}
|
||||||
|
FormatError::ContiguousStorageRequired(what) => {
|
||||||
|
write!(
|
||||||
|
f,
|
||||||
|
"{what} needs the whole file in memory, which this storage backend does \
|
||||||
|
not provide"
|
||||||
|
)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -9,18 +9,23 @@ extern crate alloc;
|
|||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
use alloc::{format, vec, vec::Vec};
|
use alloc::{format, vec, vec::Vec};
|
||||||
|
|
||||||
|
use crate::addr::to_usize;
|
||||||
|
use crate::chunk_grid::ChunkGrid;
|
||||||
use crate::chunked_read::ChunkInfo;
|
use crate::chunked_read::ChunkInfo;
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
|
use crate::storage::{PAGED_BLOCK_ONE_READ_MAX, Storage, Window, read_exact_at};
|
||||||
|
|
||||||
/// Verify the Jenkins lookup3 checksum stored immediately after
|
/// Verify the Jenkins lookup3 checksum stored immediately after
|
||||||
/// `data[start..end]`, as every Extensible Array structure carries one.
|
/// `data[start..end]`, as every Extensible Array structure carries one. `w`
|
||||||
|
/// is a window of the file and `start`/`end` are relative to it.
|
||||||
///
|
///
|
||||||
/// A corrupt chunk index yields addresses pointing at the wrong bytes, so a
|
/// A corrupt chunk index yields addresses pointing at the wrong bytes, so a
|
||||||
/// mismatch is an error: otherwise the damage surfaces as plausible data read
|
/// mismatch is an error: otherwise the damage surfaces as plausible data read
|
||||||
/// from the wrong chunk.
|
/// from the wrong chunk.
|
||||||
#[cfg(feature = "checksum")]
|
#[cfg(feature = "checksum")]
|
||||||
fn verify_checksum(data: &[u8], start: usize, end: usize) -> Result<(), FormatError> {
|
fn verify_checksum(w: &Window<'_>, start: usize, end: usize) -> Result<(), FormatError> {
|
||||||
ensure_len(data, end, 4)?;
|
w.ensure(end, 4)?;
|
||||||
|
let data: &[u8] = &w.bytes;
|
||||||
let stored = u32::from_le_bytes([data[end], data[end + 1], data[end + 2], data[end + 3]]);
|
let stored = u32::from_le_bytes([data[end], data[end + 1], data[end + 2], data[end + 3]]);
|
||||||
let computed = crate::checksum::jenkins_lookup3(&data[start..end]);
|
let computed = crate::checksum::jenkins_lookup3(&data[start..end]);
|
||||||
if computed != stored {
|
if computed != stored {
|
||||||
@@ -33,7 +38,7 @@ fn verify_checksum(data: &[u8], start: usize, end: usize) -> Result<(), FormatEr
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[cfg(not(feature = "checksum"))]
|
#[cfg(not(feature = "checksum"))]
|
||||||
fn verify_checksum(_data: &[u8], _start: usize, _end: usize) -> Result<(), FormatError> {
|
fn verify_checksum(_w: &Window<'_>, _start: usize, _end: usize) -> Result<(), FormatError> {
|
||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -79,19 +84,6 @@ fn read_offset(data: &[u8], pos: usize, size: u8) -> Result<u64, FormatError> {
|
|||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
fn ensure_len(data: &[u8], offset: usize, needed: usize) -> Result<(), FormatError> {
|
|
||||||
if offset
|
|
||||||
.checked_add(needed)
|
|
||||||
.is_none_or(|end| end > data.len())
|
|
||||||
{
|
|
||||||
return Err(FormatError::UnexpectedEof {
|
|
||||||
expected: offset.saturating_add(needed),
|
|
||||||
available: data.len(),
|
|
||||||
});
|
|
||||||
}
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
fn is_undefined_addr(addr: u64, offset_size: u8) -> bool {
|
fn is_undefined_addr(addr: u64, offset_size: u8) -> bool {
|
||||||
match offset_size {
|
match offset_size {
|
||||||
2 => addr == 0xFFFF,
|
2 => addr == 0xFFFF,
|
||||||
@@ -129,6 +121,16 @@ impl ExtensibleArrayHeader {
|
|||||||
offset: usize,
|
offset: usize,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
|
) -> Result<Self, FormatError> {
|
||||||
|
Self::parse_in(file_data, offset as u64, offset_size, length_size)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`Self::parse`] over any [`Storage`]: one read of the header.
|
||||||
|
pub fn parse_in<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
offset: u64,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
) -> Result<Self, FormatError> {
|
) -> Result<Self, FormatError> {
|
||||||
// EAHD: signature(4) + version(1) + client_id(1) + element_size(1) +
|
// EAHD: signature(4) + version(1) + client_id(1) + element_size(1) +
|
||||||
// max_nelmts_bits(1) + idx_blk_elmts(1) + min_dblk_nelmts(1) +
|
// max_nelmts_bits(1) + idx_blk_elmts(1) + min_dblk_nelmts(1) +
|
||||||
@@ -136,9 +138,10 @@ impl ExtensibleArrayHeader {
|
|||||||
// 6 stats fields (each length_size) + index_block_address(offset_size) + checksum(4)
|
// 6 stats fields (each length_size) + index_block_address(offset_size) + checksum(4)
|
||||||
let min_size =
|
let min_size =
|
||||||
4 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 6 * length_size as usize + offset_size as usize + 4;
|
4 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 6 * length_size as usize + offset_size as usize + 4;
|
||||||
ensure_len(file_data, offset, min_size)?;
|
let w = Window::read(file, offset, min_size)?;
|
||||||
|
w.ensure(0, min_size)?;
|
||||||
|
|
||||||
let d = &file_data[offset..];
|
let d: &[u8] = &w.bytes;
|
||||||
if &d[0..4] != b"EAHD" {
|
if &d[0..4] != b"EAHD" {
|
||||||
return Err(FormatError::ChunkedReadError(
|
return Err(FormatError::ChunkedReadError(
|
||||||
"invalid Extensible Array header signature".into(),
|
"invalid Extensible Array header signature".into(),
|
||||||
@@ -171,7 +174,7 @@ impl ExtensibleArrayHeader {
|
|||||||
pos += ls; // skip max_idx_set (6th stats field)
|
pos += ls; // skip max_idx_set (6th stats field)
|
||||||
let index_block_address = read_offset(d, pos, offset_size)?;
|
let index_block_address = read_offset(d, pos, offset_size)?;
|
||||||
pos += offset_size as usize;
|
pos += offset_size as usize;
|
||||||
verify_checksum(file_data, offset, offset + pos)?;
|
verify_checksum(&w, 0, pos)?;
|
||||||
|
|
||||||
Ok(ExtensibleArrayHeader {
|
Ok(ExtensibleArrayHeader {
|
||||||
client_id,
|
client_id,
|
||||||
@@ -192,35 +195,33 @@ impl ExtensibleArrayHeader {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Read a single element from the extensible array element data.
|
/// Read a single element at offset `pos` of the window `w`.
|
||||||
/// Returns (chunk_info, bytes_consumed) or None if unallocated.
|
/// Returns (chunk_info, bytes_consumed) or None if unallocated.
|
||||||
#[allow(clippy::too_many_arguments)]
|
#[allow(clippy::too_many_arguments)]
|
||||||
fn read_element(
|
fn read_element(
|
||||||
data: &[u8],
|
w: &Window<'_>,
|
||||||
pos: usize,
|
pos: usize,
|
||||||
client_id: u8,
|
client_id: u8,
|
||||||
element_size: u8,
|
element_size: u8,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
chunk_byte_size: u64,
|
chunk_byte_size: u64,
|
||||||
linear_index: usize,
|
linear_index: usize,
|
||||||
num_chunks_per_dim: &[u64],
|
grid: &ChunkGrid,
|
||||||
chunk_dimensions: &[u32],
|
|
||||||
) -> Result<(Option<ChunkInfo>, usize), FormatError> {
|
) -> Result<(Option<ChunkInfo>, usize), FormatError> {
|
||||||
let os = offset_size as usize;
|
let os = offset_size as usize;
|
||||||
|
let data: &[u8] = &w.bytes;
|
||||||
|
|
||||||
if client_id == 0 {
|
if client_id == 0 {
|
||||||
// Non-filtered: just address
|
// Non-filtered: just address
|
||||||
if pos + os > data.len() {
|
w.ensure(pos, os)?;
|
||||||
return Err(FormatError::UnexpectedEof {
|
|
||||||
expected: pos + os,
|
|
||||||
available: data.len(),
|
|
||||||
});
|
|
||||||
}
|
|
||||||
if is_undefined(data, pos, offset_size) {
|
if is_undefined(data, pos, offset_size) {
|
||||||
return Ok((None, os));
|
return Ok((None, os));
|
||||||
}
|
}
|
||||||
let address = read_offset(data, pos, offset_size)?;
|
let address = read_offset(data, pos, offset_size)?;
|
||||||
let offsets = index_to_chunk_offsets(linear_index, num_chunks_per_dim, chunk_dimensions);
|
// A slot beyond the current extent is ignored, as the library does.
|
||||||
|
let Some(offsets) = grid.offsets(linear_index as u64) else {
|
||||||
|
return Ok((None, os));
|
||||||
|
};
|
||||||
Ok((
|
Ok((
|
||||||
Some(ChunkInfo {
|
Some(ChunkInfo {
|
||||||
chunk_size: chunk_byte_size as u32,
|
chunk_size: chunk_byte_size as u32,
|
||||||
@@ -240,15 +241,7 @@ fn read_element(
|
|||||||
}
|
}
|
||||||
let chunk_size_bytes = es - os - 4;
|
let chunk_size_bytes = es - os - 4;
|
||||||
let elem_total = os + chunk_size_bytes + 4;
|
let elem_total = os + chunk_size_bytes + 4;
|
||||||
if pos
|
w.ensure(pos, elem_total)?;
|
||||||
.checked_add(elem_total)
|
|
||||||
.is_none_or(|end| end > data.len())
|
|
||||||
{
|
|
||||||
return Err(FormatError::UnexpectedEof {
|
|
||||||
expected: pos.saturating_add(elem_total),
|
|
||||||
available: data.len(),
|
|
||||||
});
|
|
||||||
}
|
|
||||||
if is_undefined(data, pos, offset_size) {
|
if is_undefined(data, pos, offset_size) {
|
||||||
return Ok((None, elem_total));
|
return Ok((None, elem_total));
|
||||||
}
|
}
|
||||||
@@ -261,7 +254,9 @@ fn read_element(
|
|||||||
data[fm_off + 2],
|
data[fm_off + 2],
|
||||||
data[fm_off + 3],
|
data[fm_off + 3],
|
||||||
]);
|
]);
|
||||||
let offsets = index_to_chunk_offsets(linear_index, num_chunks_per_dim, chunk_dimensions);
|
let Some(offsets) = grid.offsets(linear_index as u64) else {
|
||||||
|
return Ok((None, elem_total));
|
||||||
|
};
|
||||||
Ok((
|
Ok((
|
||||||
Some(ChunkInfo {
|
Some(ChunkInfo {
|
||||||
chunk_size: chunk_size as u32,
|
chunk_size: chunk_size as u32,
|
||||||
@@ -274,27 +269,6 @@ fn read_element(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Convert a linear chunk index to N-dimensional chunk offsets in dataset space.
|
|
||||||
fn index_to_chunk_offsets(
|
|
||||||
index: usize,
|
|
||||||
num_chunks_per_dim: &[u64],
|
|
||||||
chunk_dimensions: &[u32],
|
|
||||||
) -> Vec<u64> {
|
|
||||||
let rank = num_chunks_per_dim.len();
|
|
||||||
let mut offsets = vec![0u64; rank];
|
|
||||||
let mut remaining = index as u64;
|
|
||||||
for d in (0..rank).rev() {
|
|
||||||
let nchunks = num_chunks_per_dim[d];
|
|
||||||
if nchunks == 0 {
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
let chunk_idx = remaining % nchunks;
|
|
||||||
remaining /= nchunks;
|
|
||||||
offsets[d] = chunk_idx * chunk_dimensions[d] as u64;
|
|
||||||
}
|
|
||||||
offsets
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Collect elements from a data block at the given offset.
|
/// Collect elements from a data block at the given offset.
|
||||||
#[allow(clippy::too_many_arguments)]
|
#[allow(clippy::too_many_arguments)]
|
||||||
/// Layout of super block `u`, per the HDF5 spec: the number of data blocks it
|
/// Layout of super block `u`, per the HDF5 spec: the number of data blocks it
|
||||||
@@ -331,37 +305,43 @@ fn page_nelmts(header: &ExtensibleArrayHeader) -> Option<usize> {
|
|||||||
/// paged. The bitmap lives in the super block, not here — a paged data block
|
/// paged. The bitmap lives in the super block, not here — a paged data block
|
||||||
/// stores only its prefix, then one slot per page.
|
/// stores only its prefix, then one slot per page.
|
||||||
#[allow(clippy::too_many_arguments)]
|
#[allow(clippy::too_many_arguments)]
|
||||||
fn read_data_block_elements(
|
fn read_data_block_elements<S: Storage + ?Sized>(
|
||||||
file_data: &[u8],
|
file: &S,
|
||||||
db_offset: usize,
|
db_offset: u64,
|
||||||
nelmts: usize,
|
nelmts: usize,
|
||||||
header: &ExtensibleArrayHeader,
|
header: &ExtensibleArrayHeader,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
chunk_byte_size: u64,
|
chunk_byte_size: u64,
|
||||||
start_index: usize,
|
start_index: usize,
|
||||||
num_chunks_per_dim: &[u64],
|
grid: &ChunkGrid,
|
||||||
chunk_dimensions: &[u32],
|
|
||||||
page_init: &[u8],
|
page_init: &[u8],
|
||||||
first_page: usize,
|
first_page: usize,
|
||||||
) -> Result<Vec<ChunkInfo>, FormatError> {
|
) -> Result<Vec<ChunkInfo>, FormatError> {
|
||||||
// EADB: signature(4) + version(1) + client_id(1) + header_address(offset_size)
|
// EADB: signature(4) + version(1) + client_id(1) + header_address(offset_size)
|
||||||
// + block offset(arr_off_size)
|
// + block offset(arr_off_size)
|
||||||
let db_header_size = 4 + 1 + 1 + offset_size as usize + arr_off_size(header);
|
let db_header_size = 4 + 1 + 1 + offset_size as usize + arr_off_size(header);
|
||||||
ensure_len(file_data, db_offset, db_header_size)?;
|
let prefix = read_exact_at(file, db_offset, db_header_size)?;
|
||||||
|
|
||||||
if &file_data[db_offset..db_offset + 4] != b"EADB" {
|
if &prefix[0..4] != b"EADB" {
|
||||||
return Err(FormatError::ChunkedReadError(
|
return Err(FormatError::ChunkedReadError(
|
||||||
"invalid Extensible Array data block signature".into(),
|
"invalid Extensible Array data block signature".into(),
|
||||||
));
|
));
|
||||||
}
|
}
|
||||||
|
|
||||||
let mut pos = db_offset + db_header_size;
|
// Positions below are relative to the data block.
|
||||||
|
let mut pos = db_header_size;
|
||||||
let page = page_nelmts(header).ok_or_else(|| {
|
let page = page_nelmts(header).ok_or_else(|| {
|
||||||
FormatError::Overflow("Extensible Array page element count overflows usize".into())
|
FormatError::Overflow("Extensible Array page element count overflows usize".into())
|
||||||
})?;
|
})?;
|
||||||
|
let elem_bytes = if header.client_id == 0 {
|
||||||
|
offset_size as usize
|
||||||
|
} else {
|
||||||
|
header.element_size as usize
|
||||||
|
};
|
||||||
|
|
||||||
let mut chunks = Vec::new();
|
let mut chunks = Vec::new();
|
||||||
let read_run = |from: usize,
|
let read_run = |w: &Window<'_>,
|
||||||
|
from: usize,
|
||||||
count: usize,
|
count: usize,
|
||||||
first_index: usize,
|
first_index: usize,
|
||||||
chunks: &mut Vec<ChunkInfo>|
|
chunks: &mut Vec<ChunkInfo>|
|
||||||
@@ -369,15 +349,14 @@ fn read_data_block_elements(
|
|||||||
let mut p = from;
|
let mut p = from;
|
||||||
for i in 0..count {
|
for i in 0..count {
|
||||||
let (info, consumed) = read_element(
|
let (info, consumed) = read_element(
|
||||||
file_data,
|
w,
|
||||||
p,
|
p,
|
||||||
header.client_id,
|
header.client_id,
|
||||||
header.element_size,
|
header.element_size,
|
||||||
offset_size,
|
offset_size,
|
||||||
chunk_byte_size,
|
chunk_byte_size,
|
||||||
first_index + i,
|
first_index + i,
|
||||||
num_chunks_per_dim,
|
grid,
|
||||||
chunk_dimensions,
|
|
||||||
)?;
|
)?;
|
||||||
if let Some(ci) = info {
|
if let Some(ci) = info {
|
||||||
chunks.push(ci);
|
chunks.push(ci);
|
||||||
@@ -388,18 +367,19 @@ fn read_data_block_elements(
|
|||||||
};
|
};
|
||||||
|
|
||||||
if nelmts <= page {
|
if nelmts <= page {
|
||||||
// Prefix and elements are covered by one checksum.
|
// Prefix and elements are covered by one checksum. One window holds
|
||||||
let elem_bytes = if header.client_id == 0 {
|
// all of it (or ends at the end of the file), so its bounds checks
|
||||||
offset_size as usize
|
// are the whole-file ones.
|
||||||
} else {
|
|
||||||
header.element_size as usize
|
|
||||||
};
|
|
||||||
let end = nelmts
|
let end = nelmts
|
||||||
.checked_mul(elem_bytes)
|
.checked_mul(elem_bytes)
|
||||||
.and_then(|b| pos.checked_add(b))
|
.and_then(|b| pos.checked_add(b))
|
||||||
.ok_or_else(|| FormatError::Overflow("Extensible Array data block span".into()))?;
|
.ok_or_else(|| FormatError::Overflow("Extensible Array data block span".into()))?;
|
||||||
verify_checksum(file_data, db_offset, end)?;
|
// The checksum's bounds check comes first: make it before reading.
|
||||||
read_run(pos, nelmts, start_index, &mut chunks)?;
|
#[cfg(feature = "checksum")]
|
||||||
|
Window::check_extent(file, db_offset, end, 4)?;
|
||||||
|
let w = Window::read(file, db_offset, end.saturating_add(4))?;
|
||||||
|
verify_checksum(&w, 0, end)?;
|
||||||
|
read_run(&w, pos, nelmts, start_index, &mut chunks)?;
|
||||||
return Ok(chunks);
|
return Ok(chunks);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -407,18 +387,32 @@ fn read_data_block_elements(
|
|||||||
// each holding `page` elements followed by a checksum. Pages whose bit is
|
// each holding `page` elements followed by a checksum. Pages whose bit is
|
||||||
// clear were never written; their slot still occupies the file, so stride
|
// clear were never written; their slot still occupies the file, so stride
|
||||||
// over it rather than reading zeros as addresses.
|
// over it rather than reading zeros as addresses.
|
||||||
verify_checksum(file_data, db_offset, pos)?;
|
let npages = nelmts.div_ceil(page);
|
||||||
pos += 4;
|
// The whole data block in one window when it is small: every position
|
||||||
let elem_bytes = if header.client_id == 0 {
|
// checked below lies inside it (or past the end of the file). A larger
|
||||||
offset_size as usize
|
// block is read as its prefix, then each page in use on its own.
|
||||||
|
let block_len = pos
|
||||||
|
.saturating_add(4)
|
||||||
|
.saturating_add(npages.saturating_mul(page.saturating_mul(elem_bytes).saturating_add(4)));
|
||||||
|
let whole = if block_len <= PAGED_BLOCK_ONE_READ_MAX {
|
||||||
|
Some(Window::read(file, db_offset, block_len)?)
|
||||||
} else {
|
} else {
|
||||||
header.element_size as usize
|
None
|
||||||
};
|
};
|
||||||
|
let head_w;
|
||||||
|
let head = match &whole {
|
||||||
|
Some(w) => w,
|
||||||
|
None => {
|
||||||
|
head_w = Window::read(file, db_offset, pos + 4)?;
|
||||||
|
&head_w
|
||||||
|
}
|
||||||
|
};
|
||||||
|
verify_checksum(head, 0, pos)?;
|
||||||
|
pos += 4;
|
||||||
let page_stride = page
|
let page_stride = page
|
||||||
.checked_mul(elem_bytes)
|
.checked_mul(elem_bytes)
|
||||||
.and_then(|b| b.checked_add(4))
|
.and_then(|b| b.checked_add(4))
|
||||||
.ok_or_else(|| FormatError::Overflow("Extensible Array page stride".into()))?;
|
.ok_or_else(|| FormatError::Overflow("Extensible Array page stride".into()))?;
|
||||||
let npages = nelmts.div_ceil(page);
|
|
||||||
for p in 0..npages {
|
for p in 0..npages {
|
||||||
// One bit per page across the whole super block, packed contiguously
|
// One bit per page across the whole super block, packed contiguously
|
||||||
// and MSB-first within each byte, as H5VM_bit_get reads it.
|
// and MSB-first within each byte, as H5VM_bit_get reads it.
|
||||||
@@ -428,10 +422,20 @@ fn read_data_block_elements(
|
|||||||
.is_some_and(|byte| byte & (0x80 >> (bit % 8)) != 0);
|
.is_some_and(|byte| byte & (0x80 >> (bit % 8)) != 0);
|
||||||
if initialised {
|
if initialised {
|
||||||
let count = core::cmp::min(page, nelmts - p * page);
|
let count = core::cmp::min(page, nelmts - p * page);
|
||||||
|
// `w` holds the page from `base` on (positions below are
|
||||||
|
// relative to it, and `pos` to the data block).
|
||||||
|
let page_w;
|
||||||
|
let (w, base) = match &whole {
|
||||||
|
Some(w) => (w, 0),
|
||||||
|
None => {
|
||||||
|
page_w = Window::read(file, db_offset.saturating_add(pos as u64), page_stride)?;
|
||||||
|
(&page_w, pos)
|
||||||
|
}
|
||||||
|
};
|
||||||
// Each page carries its own checksum, over a full page's worth of
|
// Each page carries its own checksum, over a full page's worth of
|
||||||
// slots even when the last one holds fewer live elements.
|
// slots even when the last one holds fewer live elements.
|
||||||
verify_checksum(file_data, pos, pos + page * elem_bytes)?;
|
verify_checksum(w, pos - base, pos - base + page * elem_bytes)?;
|
||||||
read_run(pos, count, start_index + p * page, &mut chunks)?;
|
read_run(w, pos - base, count, start_index + p * page, &mut chunks)?;
|
||||||
}
|
}
|
||||||
pos = pos
|
pos = pos
|
||||||
.checked_add(page_stride)
|
.checked_add(page_stride)
|
||||||
@@ -449,25 +453,45 @@ pub fn read_extensible_array_chunks(
|
|||||||
file_data: &[u8],
|
file_data: &[u8],
|
||||||
header: &ExtensibleArrayHeader,
|
header: &ExtensibleArrayHeader,
|
||||||
dataset_dims: &[u64],
|
dataset_dims: &[u64],
|
||||||
|
max_dims: Option<&[u64]>,
|
||||||
|
chunk_dimensions: &[u32],
|
||||||
|
element_size: u32,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<Vec<ChunkInfo>, FormatError> {
|
||||||
|
read_extensible_array_chunks_in(
|
||||||
|
&file_data,
|
||||||
|
header,
|
||||||
|
dataset_dims,
|
||||||
|
max_dims,
|
||||||
|
chunk_dimensions,
|
||||||
|
element_size,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`read_extensible_array_chunks`] over any [`Storage`]: one read of the
|
||||||
|
/// index block's prefix, one of the whole index block, and the same for
|
||||||
|
/// every super block and data block it references.
|
||||||
|
#[allow(clippy::too_many_arguments)]
|
||||||
|
pub fn read_extensible_array_chunks_in<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
header: &ExtensibleArrayHeader,
|
||||||
|
dataset_dims: &[u64],
|
||||||
|
max_dims: Option<&[u64]>,
|
||||||
chunk_dimensions: &[u32],
|
chunk_dimensions: &[u32],
|
||||||
element_size: u32,
|
element_size: u32,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
_length_size: u8,
|
_length_size: u8,
|
||||||
) -> Result<Vec<ChunkInfo>, FormatError> {
|
) -> Result<Vec<ChunkInfo>, FormatError> {
|
||||||
let rank = chunk_dimensions.len();
|
|
||||||
let os = offset_size as usize;
|
let os = offset_size as usize;
|
||||||
|
|
||||||
let mut num_chunks_per_dim = Vec::with_capacity(rank);
|
// Linear indexes follow the maximum dimensions, with the unlimited
|
||||||
for d in 0..rank {
|
// dimension swizzled to the slowest position (see `chunk_grid`).
|
||||||
let ch_dim = chunk_dimensions[d] as u64;
|
let dims_u64: Vec<u64> = chunk_dimensions.iter().map(|&d| d as u64).collect();
|
||||||
if ch_dim == 0 {
|
let grid = ChunkGrid::extensible_array(dataset_dims, max_dims, &dims_u64)?;
|
||||||
return Err(FormatError::ChunkedReadError(
|
let grid = &grid;
|
||||||
"chunk dimension is zero".into(),
|
|
||||||
));
|
|
||||||
}
|
|
||||||
let ds_dim = dataset_dims[d];
|
|
||||||
num_chunks_per_dim.push(ds_dim.div_ceil(ch_dim));
|
|
||||||
}
|
|
||||||
|
|
||||||
let chunk_byte_size: u64 =
|
let chunk_byte_size: u64 =
|
||||||
chunk_dimensions.iter().map(|&d| d as u64).product::<u64>() * element_size as u64;
|
chunk_dimensions.iter().map(|&d| d as u64).product::<u64>() * element_size as u64;
|
||||||
@@ -475,19 +499,20 @@ pub fn read_extensible_array_chunks(
|
|||||||
// Parse index block (EAIB): signature(4) + version(1) + client_id(1)
|
// Parse index block (EAIB): signature(4) + version(1) + client_id(1)
|
||||||
// + header address(offset_size), then the inline elements, then the
|
// + header address(offset_size), then the inline elements, then the
|
||||||
// direct data block addresses, then the super block addresses.
|
// direct data block addresses, then the super block addresses.
|
||||||
let ib_offset = header.index_block_address as usize;
|
// Positions below are relative to the index block.
|
||||||
|
let ib_offset = header.index_block_address;
|
||||||
let ib_header_size = 4 + 1 + 1 + os;
|
let ib_header_size = 4 + 1 + 1 + os;
|
||||||
ensure_len(file_data, ib_offset, ib_header_size)?;
|
let prefix = read_exact_at(file, ib_offset, ib_header_size)?;
|
||||||
|
|
||||||
if &file_data[ib_offset..ib_offset + 4] != b"EAIB" {
|
if &prefix[0..4] != b"EAIB" {
|
||||||
return Err(FormatError::ChunkedReadError(
|
return Err(FormatError::ChunkedReadError(
|
||||||
"invalid Extensible Array index block signature".into(),
|
"invalid Extensible Array index block signature".into(),
|
||||||
));
|
));
|
||||||
}
|
}
|
||||||
let mut pos = ib_offset + ib_header_size;
|
let mut pos = ib_header_size;
|
||||||
|
|
||||||
let mut chunks = Vec::new();
|
let mut chunks = Vec::new();
|
||||||
let total_elements = header.num_elements as usize;
|
let total_elements = to_usize(header.num_elements)?;
|
||||||
|
|
||||||
let dmin = header.min_dblk_nelmts as usize;
|
let dmin = header.min_dblk_nelmts as usize;
|
||||||
if dmin == 0 || !dmin.is_power_of_two() {
|
if dmin == 0 || !dmin.is_power_of_two() {
|
||||||
@@ -544,21 +569,26 @@ pub fn read_extensible_array_chunks(
|
|||||||
.and_then(|n| n.checked_mul(os).and_then(|b| p.checked_add(b)))
|
.and_then(|n| n.checked_mul(os).and_then(|b| p.checked_add(b)))
|
||||||
})
|
})
|
||||||
.ok_or_else(|| FormatError::Overflow("Extensible Array index block span".into()))?;
|
.ok_or_else(|| FormatError::Overflow("Extensible Array index block span".into()))?;
|
||||||
verify_checksum(file_data, ib_offset, ib_end)?;
|
// The whole index block in one window: every position read below is
|
||||||
|
// before `ib_end`.
|
||||||
|
// The checksum's bounds check comes first: make it before reading.
|
||||||
|
#[cfg(feature = "checksum")]
|
||||||
|
Window::check_extent(file, ib_offset, ib_end, 4)?;
|
||||||
|
let w = Window::read(file, ib_offset, ib_end.saturating_add(4))?;
|
||||||
|
verify_checksum(&w, 0, ib_end)?;
|
||||||
|
|
||||||
// 1. Elements stored inline in the index block.
|
// 1. Elements stored inline in the index block.
|
||||||
let n_inline = (header.idx_blk_elmts as usize).min(total_elements);
|
let n_inline = (header.idx_blk_elmts as usize).min(total_elements);
|
||||||
for i in 0..n_inline {
|
for i in 0..n_inline {
|
||||||
let (info, consumed) = read_element(
|
let (info, consumed) = read_element(
|
||||||
file_data,
|
&w,
|
||||||
pos,
|
pos,
|
||||||
header.client_id,
|
header.client_id,
|
||||||
header.element_size,
|
header.element_size,
|
||||||
offset_size,
|
offset_size,
|
||||||
chunk_byte_size,
|
chunk_byte_size,
|
||||||
i,
|
i,
|
||||||
&num_chunks_per_dim,
|
grid,
|
||||||
chunk_dimensions,
|
|
||||||
)?;
|
)?;
|
||||||
if let Some(ci) = info {
|
if let Some(ci) = info {
|
||||||
chunks.push(ci);
|
chunks.push(ci);
|
||||||
@@ -575,8 +605,8 @@ pub fn read_extensible_array_chunks(
|
|||||||
if global_index >= total_elements {
|
if global_index >= total_elements {
|
||||||
return Ok(chunks);
|
return Ok(chunks);
|
||||||
}
|
}
|
||||||
ensure_len(file_data, pos, os)?;
|
w.ensure(pos, os)?;
|
||||||
let addr = read_offset(file_data, pos, offset_size)?;
|
let addr = read_offset(&w.bytes, pos, offset_size)?;
|
||||||
pos += os;
|
pos += os;
|
||||||
if !is_undefined_addr(addr, offset_size) {
|
if !is_undefined_addr(addr, offset_size) {
|
||||||
if dblk_nelmts > page_nelmts(header).unwrap_or(usize::MAX) {
|
if dblk_nelmts > page_nelmts(header).unwrap_or(usize::MAX) {
|
||||||
@@ -587,15 +617,14 @@ pub fn read_extensible_array_chunks(
|
|||||||
));
|
));
|
||||||
}
|
}
|
||||||
chunks.extend(read_data_block_elements(
|
chunks.extend(read_data_block_elements(
|
||||||
file_data,
|
file,
|
||||||
addr as usize,
|
addr,
|
||||||
dblk_nelmts,
|
dblk_nelmts,
|
||||||
header,
|
header,
|
||||||
offset_size,
|
offset_size,
|
||||||
chunk_byte_size,
|
chunk_byte_size,
|
||||||
global_index,
|
global_index,
|
||||||
&num_chunks_per_dim,
|
grid,
|
||||||
chunk_dimensions,
|
|
||||||
&[],
|
&[],
|
||||||
0,
|
0,
|
||||||
)?);
|
)?);
|
||||||
@@ -609,24 +638,23 @@ pub fn read_extensible_array_chunks(
|
|||||||
if global_index >= total_elements {
|
if global_index >= total_elements {
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
ensure_len(file_data, pos, os)?;
|
w.ensure(pos, os)?;
|
||||||
let sb_addr = read_offset(file_data, pos, offset_size)?;
|
let sb_addr = read_offset(&w.bytes, pos, offset_size)?;
|
||||||
pos += os;
|
pos += os;
|
||||||
let (ndblks, dblk_nelmts) = sblk_info(u, dmin).ok_or_else(|| {
|
let (ndblks, dblk_nelmts) = sblk_info(u, dmin).ok_or_else(|| {
|
||||||
FormatError::Overflow("Extensible Array super block layout overflows usize".into())
|
FormatError::Overflow("Extensible Array super block layout overflows usize".into())
|
||||||
})?;
|
})?;
|
||||||
if !is_undefined_addr(sb_addr, offset_size) {
|
if !is_undefined_addr(sb_addr, offset_size) {
|
||||||
chunks.extend(read_super_block(
|
chunks.extend(read_super_block(
|
||||||
file_data,
|
file,
|
||||||
sb_addr as usize,
|
sb_addr,
|
||||||
ndblks,
|
ndblks,
|
||||||
dblk_nelmts,
|
dblk_nelmts,
|
||||||
header,
|
header,
|
||||||
offset_size,
|
offset_size,
|
||||||
chunk_byte_size,
|
chunk_byte_size,
|
||||||
global_index,
|
global_index,
|
||||||
&num_chunks_per_dim,
|
grid,
|
||||||
chunk_dimensions,
|
|
||||||
)?);
|
)?);
|
||||||
}
|
}
|
||||||
global_index =
|
global_index =
|
||||||
@@ -644,23 +672,22 @@ pub fn read_extensible_array_chunks(
|
|||||||
/// + block offset + the page-init bitmap for every data block it owns
|
/// + block offset + the page-init bitmap for every data block it owns
|
||||||
/// + one address per data block + checksum.
|
/// + one address per data block + checksum.
|
||||||
#[allow(clippy::too_many_arguments)]
|
#[allow(clippy::too_many_arguments)]
|
||||||
fn read_super_block(
|
fn read_super_block<S: Storage + ?Sized>(
|
||||||
file_data: &[u8],
|
file: &S,
|
||||||
sb_offset: usize,
|
sb_offset: u64,
|
||||||
ndblks: usize,
|
ndblks: usize,
|
||||||
dblk_nelmts: usize,
|
dblk_nelmts: usize,
|
||||||
header: &ExtensibleArrayHeader,
|
header: &ExtensibleArrayHeader,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
chunk_byte_size: u64,
|
chunk_byte_size: u64,
|
||||||
start_index: usize,
|
start_index: usize,
|
||||||
num_chunks_per_dim: &[u64],
|
grid: &ChunkGrid,
|
||||||
chunk_dimensions: &[u32],
|
|
||||||
) -> Result<Vec<ChunkInfo>, FormatError> {
|
) -> Result<Vec<ChunkInfo>, FormatError> {
|
||||||
let os = offset_size as usize;
|
let os = offset_size as usize;
|
||||||
let sb_header_size = 4 + 1 + 1 + os + arr_off_size(header);
|
let sb_header_size = 4 + 1 + 1 + os + arr_off_size(header);
|
||||||
ensure_len(file_data, sb_offset, sb_header_size)?;
|
let prefix = read_exact_at(file, sb_offset, sb_header_size)?;
|
||||||
|
|
||||||
if &file_data[sb_offset..sb_offset + 4] != b"EASB" {
|
if &prefix[0..4] != b"EASB" {
|
||||||
return Err(FormatError::ChunkedReadError(
|
return Err(FormatError::ChunkedReadError(
|
||||||
"invalid Extensible Array super block signature".into(),
|
"invalid Extensible Array super block signature".into(),
|
||||||
));
|
));
|
||||||
@@ -682,36 +709,44 @@ fn read_super_block(
|
|||||||
let bitmap_bytes = per_dblk_bitmap
|
let bitmap_bytes = per_dblk_bitmap
|
||||||
.checked_mul(ndblks)
|
.checked_mul(ndblks)
|
||||||
.ok_or_else(|| FormatError::Overflow("Extensible Array page bitmap size".into()))?;
|
.ok_or_else(|| FormatError::Overflow("Extensible Array page bitmap size".into()))?;
|
||||||
let bitmap_start = sb_offset + sb_header_size;
|
// Positions below are relative to the super block, whose bytes (up to
|
||||||
ensure_len(file_data, bitmap_start, bitmap_bytes)?;
|
// its checksum) are all in one window.
|
||||||
let bitmap = &file_data[bitmap_start..bitmap_start + bitmap_bytes];
|
let bitmap_start = sb_header_size;
|
||||||
|
// The bitmap's bounds check, then (with checksums) the checksum's, come
|
||||||
|
// before anything else is read from the block: make them before reading
|
||||||
|
// it, so size fields stretching it past the end of the file cost no read.
|
||||||
|
Window::check_extent(file, sb_offset, bitmap_start, bitmap_bytes)?;
|
||||||
let mut pos = bitmap_start + bitmap_bytes;
|
let mut pos = bitmap_start + bitmap_bytes;
|
||||||
let mut chunks = Vec::new();
|
|
||||||
let mut global_idx = start_index;
|
|
||||||
|
|
||||||
// One checksum covers the prefix, the bitmap and every data block address.
|
// One checksum covers the prefix, the bitmap and every data block address.
|
||||||
let sb_end = ndblks
|
let sb_end = ndblks
|
||||||
.checked_mul(os)
|
.checked_mul(os)
|
||||||
.and_then(|b| pos.checked_add(b))
|
.and_then(|b| pos.checked_add(b))
|
||||||
.ok_or_else(|| FormatError::Overflow("Extensible Array super block span".into()))?;
|
.ok_or_else(|| FormatError::Overflow("Extensible Array super block span".into()))?;
|
||||||
verify_checksum(file_data, sb_offset, sb_end)?;
|
#[cfg(feature = "checksum")]
|
||||||
|
Window::check_extent(file, sb_offset, sb_end, 4)?;
|
||||||
|
let w = Window::read(file, sb_offset, sb_end.saturating_add(4))?;
|
||||||
|
w.ensure(bitmap_start, bitmap_bytes)?;
|
||||||
|
let bitmap = &w.bytes[bitmap_start..bitmap_start + bitmap_bytes];
|
||||||
|
|
||||||
|
let mut chunks = Vec::new();
|
||||||
|
let mut global_idx = start_index;
|
||||||
|
verify_checksum(&w, 0, sb_end)?;
|
||||||
|
|
||||||
for i in 0..ndblks {
|
for i in 0..ndblks {
|
||||||
ensure_len(file_data, pos, os)?;
|
w.ensure(pos, os)?;
|
||||||
let addr = read_offset(file_data, pos, offset_size)?;
|
let addr = read_offset(&w.bytes, pos, offset_size)?;
|
||||||
pos += os;
|
pos += os;
|
||||||
if !is_undefined_addr(addr, offset_size) {
|
if !is_undefined_addr(addr, offset_size) {
|
||||||
chunks.extend(read_data_block_elements(
|
chunks.extend(read_data_block_elements(
|
||||||
file_data,
|
file,
|
||||||
addr as usize,
|
addr,
|
||||||
dblk_nelmts,
|
dblk_nelmts,
|
||||||
header,
|
header,
|
||||||
offset_size,
|
offset_size,
|
||||||
chunk_byte_size,
|
chunk_byte_size,
|
||||||
global_idx,
|
global_idx,
|
||||||
num_chunks_per_dim,
|
grid,
|
||||||
chunk_dimensions,
|
|
||||||
bitmap,
|
bitmap,
|
||||||
i * npages,
|
i * npages,
|
||||||
)?);
|
)?);
|
||||||
@@ -735,35 +770,18 @@ mod tests {
|
|||||||
}
|
}
|
||||||
#[test]
|
#[test]
|
||||||
fn index_to_offsets_1d() {
|
fn index_to_offsets_1d() {
|
||||||
let num_chunks = vec![5u64];
|
let g = ChunkGrid::fixed_array(&[100], None, &[20]).unwrap();
|
||||||
let chunk_dims = vec![20u32];
|
assert_eq!(g.offsets(0).unwrap(), vec![0]);
|
||||||
assert_eq!(index_to_chunk_offsets(0, &num_chunks, &chunk_dims), vec![0]);
|
assert_eq!(g.offsets(1).unwrap(), vec![20]);
|
||||||
assert_eq!(
|
assert_eq!(g.offsets(4).unwrap(), vec![80]);
|
||||||
index_to_chunk_offsets(1, &num_chunks, &chunk_dims),
|
|
||||||
vec![20]
|
|
||||||
);
|
|
||||||
assert_eq!(
|
|
||||||
index_to_chunk_offsets(4, &num_chunks, &chunk_dims),
|
|
||||||
vec![80]
|
|
||||||
);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn index_to_offsets_2d() {
|
fn index_to_offsets_2d() {
|
||||||
let num_chunks = vec![3u64, 2];
|
let g = ChunkGrid::fixed_array(&[10, 6], None, &[4, 3]).unwrap();
|
||||||
let chunk_dims = vec![4u32, 3];
|
assert_eq!(g.offsets(0).unwrap(), vec![0, 0]);
|
||||||
assert_eq!(
|
assert_eq!(g.offsets(1).unwrap(), vec![0, 3]);
|
||||||
index_to_chunk_offsets(0, &num_chunks, &chunk_dims),
|
assert_eq!(g.offsets(2).unwrap(), vec![4, 0]);
|
||||||
vec![0, 0]
|
|
||||||
);
|
|
||||||
assert_eq!(
|
|
||||||
index_to_chunk_offsets(1, &num_chunks, &chunk_dims),
|
|
||||||
vec![0, 3]
|
|
||||||
);
|
|
||||||
assert_eq!(
|
|
||||||
index_to_chunk_offsets(2, &num_chunks, &chunk_dims),
|
|
||||||
vec![4, 0]
|
|
||||||
);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
@@ -830,7 +848,7 @@ mod tests {
|
|||||||
index_block_address: (usize::MAX - 4) as u64,
|
index_block_address: (usize::MAX - 4) as u64,
|
||||||
};
|
};
|
||||||
let buf = vec![0u8; 64];
|
let buf = vec![0u8; 64];
|
||||||
let r = read_extensible_array_chunks(&buf, &header, &[100], &[20], 8, 8, 8);
|
let r = read_extensible_array_chunks(&buf, &header, &[100], None, &[20], 8, 8, 8);
|
||||||
assert!(r.is_err());
|
assert!(r.is_err());
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -913,9 +931,17 @@ mod tests {
|
|||||||
let header = ExtensibleArrayHeader::parse(&file_data, aehd_offset, os, ls).unwrap();
|
let header = ExtensibleArrayHeader::parse(&file_data, aehd_offset, os, ls).unwrap();
|
||||||
let ds_dims = vec![40u64]; // 2 chunks × 20 elements
|
let ds_dims = vec![40u64]; // 2 chunks × 20 elements
|
||||||
let chunk_dims = vec![20u32];
|
let chunk_dims = vec![20u32];
|
||||||
let chunks =
|
let chunks = read_extensible_array_chunks(
|
||||||
read_extensible_array_chunks(&file_data, &header, &ds_dims, &chunk_dims, 8, os, ls)
|
&file_data,
|
||||||
.unwrap();
|
&header,
|
||||||
|
&ds_dims,
|
||||||
|
None,
|
||||||
|
&chunk_dims,
|
||||||
|
8,
|
||||||
|
os,
|
||||||
|
ls,
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
assert_eq!(chunks.len(), 2);
|
assert_eq!(chunks.len(), 2);
|
||||||
assert_eq!(chunks[0].address, base_addr);
|
assert_eq!(chunks[0].address, base_addr);
|
||||||
@@ -925,11 +951,11 @@ mod tests {
|
|||||||
assert_eq!(chunks[1].offsets, vec![20]);
|
assert_eq!(chunks[1].offsets, vec![20]);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Build a synthetic EA with inline elements + one direct data block.
|
/// A synthetic EA with inline elements + one direct data block: the
|
||||||
#[test]
|
/// file, with the header at 0x100 (8-byte offsets and lengths, 4 chunks
|
||||||
fn read_inline_plus_data_blocks() {
|
/// of 10 elements from 0x1000 on).
|
||||||
|
fn build_inline_plus_data_blocks() -> Vec<u8> {
|
||||||
let os: u8 = 8;
|
let os: u8 = 8;
|
||||||
let ls: u8 = 8;
|
|
||||||
let osv = os as usize;
|
let osv = os as usize;
|
||||||
let chunk_byte_size = 10u64 * 8; // 10 elements × 8 bytes
|
let chunk_byte_size = 10u64 * 8; // 10 elements × 8 bytes
|
||||||
let idx_blk_elmts = 2u8;
|
let idx_blk_elmts = 2u8;
|
||||||
@@ -1019,13 +1045,30 @@ mod tests {
|
|||||||
dbpos += osv;
|
dbpos += osv;
|
||||||
}
|
}
|
||||||
stamp_checksum(&mut file_data, aedb_offset, dbpos);
|
stamp_checksum(&mut file_data, aedb_offset, dbpos);
|
||||||
|
file_data
|
||||||
|
}
|
||||||
|
|
||||||
let header = ExtensibleArrayHeader::parse(&file_data, aehd_offset, os, ls).unwrap();
|
/// Build a synthetic EA with inline elements + one direct data block.
|
||||||
|
#[test]
|
||||||
|
fn read_inline_plus_data_blocks() {
|
||||||
|
let (os, ls) = (8u8, 8u8);
|
||||||
|
let chunk_byte_size = 10u64 * 8;
|
||||||
|
let base_addr = 0x1000u64;
|
||||||
|
let file_data = build_inline_plus_data_blocks();
|
||||||
|
let header = ExtensibleArrayHeader::parse(&file_data, 0x100, os, ls).unwrap();
|
||||||
let ds_dims = vec![40u64];
|
let ds_dims = vec![40u64];
|
||||||
let chunk_dims = vec![10u32];
|
let chunk_dims = vec![10u32];
|
||||||
let chunks =
|
let chunks = read_extensible_array_chunks(
|
||||||
read_extensible_array_chunks(&file_data, &header, &ds_dims, &chunk_dims, 8, os, ls)
|
&file_data,
|
||||||
.unwrap();
|
&header,
|
||||||
|
&ds_dims,
|
||||||
|
None,
|
||||||
|
&chunk_dims,
|
||||||
|
8,
|
||||||
|
os,
|
||||||
|
ls,
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
assert_eq!(chunks.len(), 4);
|
assert_eq!(chunks.len(), 4);
|
||||||
for (i, c) in chunks.iter().enumerate() {
|
for (i, c) in chunks.iter().enumerate() {
|
||||||
@@ -1047,10 +1090,9 @@ mod tests {
|
|||||||
#[test]
|
#[test]
|
||||||
fn read_element_unallocated() {
|
fn read_element_unallocated() {
|
||||||
let data = vec![0xFFu8; 16];
|
let data = vec![0xFFu8; 16];
|
||||||
let num_chunks = vec![5u64];
|
let grid = ChunkGrid::fixed_array(&[50], None, &[10]).unwrap();
|
||||||
let chunk_dims = vec![10u32];
|
|
||||||
let (info, consumed) =
|
let (info, consumed) =
|
||||||
read_element(&data, 0, 0, 8, 8, 80, 0, &num_chunks, &chunk_dims).unwrap();
|
read_element(&Window::whole(&data), 0, 0, 8, 8, 80, 0, &grid).unwrap();
|
||||||
assert!(info.is_none());
|
assert!(info.is_none());
|
||||||
assert_eq!(consumed, 8);
|
assert_eq!(consumed, 8);
|
||||||
}
|
}
|
||||||
@@ -1069,18 +1111,16 @@ mod tests {
|
|||||||
// Filter mask
|
// Filter mask
|
||||||
data[12..16].copy_from_slice(&0u32.to_le_bytes());
|
data[12..16].copy_from_slice(&0u32.to_le_bytes());
|
||||||
|
|
||||||
let num_chunks = vec![5u64];
|
let grid = ChunkGrid::fixed_array(&[50], None, &[10]).unwrap();
|
||||||
let chunk_dims = vec![10u32];
|
|
||||||
let (info, consumed) = read_element(
|
let (info, consumed) = read_element(
|
||||||
&data,
|
&Window::whole(&data),
|
||||||
0,
|
0,
|
||||||
1,
|
1,
|
||||||
elem_size as u8,
|
elem_size as u8,
|
||||||
os,
|
os,
|
||||||
80,
|
80,
|
||||||
2,
|
2,
|
||||||
&num_chunks,
|
&grid,
|
||||||
&chunk_dims,
|
|
||||||
)
|
)
|
||||||
.unwrap();
|
.unwrap();
|
||||||
let ci = info.unwrap();
|
let ci = info.unwrap();
|
||||||
@@ -1090,4 +1130,38 @@ mod tests {
|
|||||||
assert_eq!(ci.offsets, vec![20]);
|
assert_eq!(ci.offsets, vec![20]);
|
||||||
assert_eq!(consumed, elem_size);
|
assert_eq!(consumed, elem_size);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The Storage path reads exactly what the slice path reads: the array
|
||||||
|
/// whole, cut at every length through its structures, and with a byte
|
||||||
|
/// damaged in each of them, through a read_at-only CountingStorage.
|
||||||
|
#[test]
|
||||||
|
fn storage_reads_match_slice_reads() {
|
||||||
|
use crate::storage::CountingStorage;
|
||||||
|
let full = build_inline_plus_data_blocks();
|
||||||
|
let mut files = Vec::new();
|
||||||
|
for cut in 0x100..0x340 {
|
||||||
|
files.push(full[..cut].to_vec());
|
||||||
|
}
|
||||||
|
for at in [0x104, 0x150, 0x204, 0x216, 0x230, 0x304, 0x318] {
|
||||||
|
let mut damaged = full.clone();
|
||||||
|
damaged[at] ^= 1;
|
||||||
|
files.push(damaged);
|
||||||
|
}
|
||||||
|
files.push(full);
|
||||||
|
let mut compared = 0;
|
||||||
|
for f in files {
|
||||||
|
let storage = CountingStorage::new(f.clone());
|
||||||
|
let want = ExtensibleArrayHeader::parse(&f, 0x100, 8, 8);
|
||||||
|
let got = ExtensibleArrayHeader::parse_in(&storage, 0x100, 8, 8);
|
||||||
|
assert_eq!(format!("{got:?}"), format!("{want:?}"));
|
||||||
|
let Ok(h) = want else { continue };
|
||||||
|
for dims in [&[40u64][..], &[25]] {
|
||||||
|
let want = read_extensible_array_chunks(&f, &h, dims, None, &[10], 8, 8, 8);
|
||||||
|
let got = read_extensible_array_chunks_in(&storage, &h, dims, None, &[10], 8, 8, 8);
|
||||||
|
assert_eq!(format!("{got:?}"), format!("{want:?}"), "{} bytes", f.len());
|
||||||
|
compared += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
assert!(compared > 100);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -12,7 +12,8 @@
|
|||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
use alloc::{format, vec, vec::Vec};
|
use alloc::{format, vec, vec::Vec};
|
||||||
|
|
||||||
use crate::chunked_read::{alloc_output, checked_byte_len, list_chunks};
|
use crate::addr::to_usize;
|
||||||
|
use crate::chunked_read::{alloc_output, checked_byte_len, list_chunks_in};
|
||||||
use crate::data_layout::DataLayout;
|
use crate::data_layout::DataLayout;
|
||||||
use crate::dataspace::Dataspace;
|
use crate::dataspace::Dataspace;
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
@@ -98,15 +99,63 @@ pub fn parse_fill_value(msg: &HeaderMessage) -> Result<Option<Vec<u8>>, FormatEr
|
|||||||
|
|
||||||
/// The fill value that applies to a dataset given its header messages. The new
|
/// The fill value that applies to a dataset given its header messages. The new
|
||||||
/// message wins over the old one when both are present.
|
/// message wins over the old one when both are present.
|
||||||
|
///
|
||||||
|
/// A *shared* fill value message holds only a reference to the real message,
|
||||||
|
/// which cannot be followed without the file: this returns
|
||||||
|
/// [`FormatError::UnresolvedSharedMessage`] for one (it used to answer "zeros").
|
||||||
|
/// Use [`dataset_fill_value_in`] when the file bytes are at hand.
|
||||||
pub fn dataset_fill_value(messages: &[HeaderMessage]) -> Result<Option<Vec<u8>>, FormatError> {
|
pub fn dataset_fill_value(messages: &[HeaderMessage]) -> Result<Option<Vec<u8>>, FormatError> {
|
||||||
|
fill_value_from(messages, |_| Err(FormatError::UnresolvedSharedMessage))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`dataset_fill_value`] for a dataset in `file_data`, following a shared
|
||||||
|
/// fill value message to where it lives: another object header, or the
|
||||||
|
/// file's shared-message (SOHM) heap, as libhdf5 writes it when the file has
|
||||||
|
/// a SOHM index for fill values.
|
||||||
|
pub fn dataset_fill_value_in(
|
||||||
|
file_data: &[u8],
|
||||||
|
messages: &[HeaderMessage],
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<Option<Vec<u8>>, FormatError> {
|
||||||
|
dataset_fill_value_from_storage(&file_data, messages, offset_size, length_size)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`dataset_fill_value_in`] with the file behind any
|
||||||
|
/// [`Storage`](crate::storage::Storage) (a `&dyn Storage` too). (The trait
|
||||||
|
/// is not imported here: its `len` would shadow the slice method in this
|
||||||
|
/// module.)
|
||||||
|
pub fn dataset_fill_value_from_storage<S: crate::storage::Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
messages: &[HeaderMessage],
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<Option<Vec<u8>>, FormatError> {
|
||||||
|
fill_value_from(messages, |msg| {
|
||||||
|
crate::shared_message::message_data_with_sohm_in(file, msg, offset_size, length_size)
|
||||||
|
.map(|data| data.into_owned())
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
fn fill_value_from(
|
||||||
|
messages: &[HeaderMessage],
|
||||||
|
resolve_shared: impl Fn(&HeaderMessage) -> Result<Vec<u8>, FormatError>,
|
||||||
|
) -> Result<Option<Vec<u8>>, FormatError> {
|
||||||
for wanted in [MessageType::FillValue, MessageType::FillValueOld] {
|
for wanted in [MessageType::FillValue, MessageType::FillValueOld] {
|
||||||
if let Some(msg) = messages.iter().find(|m| m.msg_type == wanted) {
|
if let Some(msg) = messages.iter().find(|m| m.msg_type == wanted) {
|
||||||
if crate::shared_message::is_shared(msg.flags) {
|
let value = if crate::shared_message::is_shared(msg.flags) {
|
||||||
// A shared fill value is legal but vanishingly rare; treat it
|
let data = resolve_shared(msg)?;
|
||||||
// as the default rather than misparsing the reference.
|
parse_fill_value(&HeaderMessage {
|
||||||
return Ok(None);
|
msg_type: msg.msg_type,
|
||||||
}
|
size: data.len(),
|
||||||
if let Some(value) = parse_fill_value(msg)? {
|
flags: msg.flags & !0x02,
|
||||||
|
creation_order: msg.creation_order,
|
||||||
|
data,
|
||||||
|
})?
|
||||||
|
} else {
|
||||||
|
parse_fill_value(msg)?
|
||||||
|
};
|
||||||
|
if let Some(value) = value {
|
||||||
return Ok(Some(value));
|
return Ok(Some(value));
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -164,6 +213,30 @@ pub fn read_full_with_fill<E: From<FormatError>>(
|
|||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
read: impl FnOnce() -> Result<Vec<u8>, E>,
|
read: impl FnOnce() -> Result<Vec<u8>, E>,
|
||||||
|
) -> Result<Vec<u8>, E> {
|
||||||
|
read_full_with_fill_in(
|
||||||
|
messages,
|
||||||
|
file_data,
|
||||||
|
layout,
|
||||||
|
dataspace,
|
||||||
|
elem_size,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
read,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`read_full_with_fill`] over any [`Storage`](crate::storage::Storage).
|
||||||
|
#[allow(clippy::too_many_arguments)]
|
||||||
|
pub fn read_full_with_fill_in<E: From<FormatError>, S: crate::storage::Storage + ?Sized>(
|
||||||
|
messages: &[HeaderMessage],
|
||||||
|
file_data: &S,
|
||||||
|
layout: &DataLayout,
|
||||||
|
dataspace: &Dataspace,
|
||||||
|
elem_size: usize,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
read: impl FnOnce() -> Result<Vec<u8>, E>,
|
||||||
) -> Result<Vec<u8>, E> {
|
) -> Result<Vec<u8>, E> {
|
||||||
// A dataset with external raw data also has no data address in this
|
// A dataset with external raw data also has no data address in this
|
||||||
// file. It is NOT unallocated — its values live elsewhere — so it must
|
// file. It is NOT unallocated — its values live elsewhere — so it must
|
||||||
@@ -174,12 +247,12 @@ pub fn read_full_with_fill<E: From<FormatError>>(
|
|||||||
{
|
{
|
||||||
return Err(FormatError::ExternalDataFilesUnsupported.into());
|
return Err(FormatError::ExternalDataFilesUnsupported.into());
|
||||||
}
|
}
|
||||||
let fill = dataset_fill_value(messages)?;
|
let fill = dataset_fill_value_from_storage(file_data, messages, offset_size, length_size)?;
|
||||||
if !has_storage(layout) {
|
if !has_storage(layout) {
|
||||||
return Ok(filled_dataset(dataspace, elem_size, fill.as_deref())?);
|
return Ok(filled_dataset(dataspace, elem_size, fill.as_deref())?);
|
||||||
}
|
}
|
||||||
let mut output = read()?;
|
let mut output = read()?;
|
||||||
apply_to_unallocated_chunks(
|
apply_to_unallocated_chunks_in(
|
||||||
&mut output,
|
&mut output,
|
||||||
file_data,
|
file_data,
|
||||||
layout,
|
layout,
|
||||||
@@ -205,6 +278,30 @@ pub fn apply_to_unallocated_chunks(
|
|||||||
fill: Option<&[u8]>,
|
fill: Option<&[u8]>,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
|
) -> Result<(), FormatError> {
|
||||||
|
apply_to_unallocated_chunks_in(
|
||||||
|
output,
|
||||||
|
file_data,
|
||||||
|
layout,
|
||||||
|
dataspace,
|
||||||
|
elem_size,
|
||||||
|
fill,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`apply_to_unallocated_chunks`] over any [`Storage`](crate::storage::Storage).
|
||||||
|
#[allow(clippy::too_many_arguments)]
|
||||||
|
pub fn apply_to_unallocated_chunks_in<S: crate::storage::Storage + ?Sized>(
|
||||||
|
output: &mut [u8],
|
||||||
|
file_data: &S,
|
||||||
|
layout: &DataLayout,
|
||||||
|
dataspace: &Dataspace,
|
||||||
|
elem_size: usize,
|
||||||
|
fill: Option<&[u8]>,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
) -> Result<(), FormatError> {
|
) -> Result<(), FormatError> {
|
||||||
let Some(fill) = fill.filter(|f| f.len() == elem_size && !is_default(Some(f))) else {
|
let Some(fill) = fill.filter(|f| f.len() == elem_size && !is_default(Some(f))) else {
|
||||||
return Ok(());
|
return Ok(());
|
||||||
@@ -212,7 +309,7 @@ pub fn apply_to_unallocated_chunks(
|
|||||||
if !matches!(layout, DataLayout::Chunked { .. }) || elem_size == 0 {
|
if !matches!(layout, DataLayout::Chunked { .. }) || elem_size == 0 {
|
||||||
return Ok(());
|
return Ok(());
|
||||||
}
|
}
|
||||||
let (chunks, chunk_dims) = list_chunks(
|
let (chunks, chunk_dims) = list_chunks_in(
|
||||||
file_data,
|
file_data,
|
||||||
layout,
|
layout,
|
||||||
dataspace,
|
dataspace,
|
||||||
@@ -221,7 +318,11 @@ pub fn apply_to_unallocated_chunks(
|
|||||||
length_size,
|
length_size,
|
||||||
)?;
|
)?;
|
||||||
let rank = chunk_dims.len();
|
let rank = chunk_dims.len();
|
||||||
let ds_dims: Vec<usize> = dataspace.dimensions.iter().map(|&d| d as usize).collect();
|
let ds_dims: Vec<usize> = dataspace
|
||||||
|
.dimensions
|
||||||
|
.iter()
|
||||||
|
.map(|&d| to_usize(d))
|
||||||
|
.collect::<Result<_, _>>()?;
|
||||||
if rank == 0 || ds_dims.len() != rank || chunk_dims.contains(&0) {
|
if rank == 0 || ds_dims.len() != rank || chunk_dims.contains(&0) {
|
||||||
return Ok(());
|
return Ok(());
|
||||||
}
|
}
|
||||||
@@ -253,7 +354,7 @@ pub fn apply_to_unallocated_chunks(
|
|||||||
let mut cell = 0usize;
|
let mut cell = 0usize;
|
||||||
let mut in_range = true;
|
let mut in_range = true;
|
||||||
for d in 0..rank {
|
for d in 0..rank {
|
||||||
let coord = chunk.offsets[d] as usize / chunk_dims[d];
|
let coord = to_usize(chunk.offsets[d])? / chunk_dims[d];
|
||||||
if coord >= grid[d] {
|
if coord >= grid[d] {
|
||||||
in_range = false;
|
in_range = false;
|
||||||
break;
|
break;
|
||||||
@@ -404,4 +505,37 @@ mod tests {
|
|||||||
.collect();
|
.collect();
|
||||||
assert_eq!(filled, [2, 3, 7, 8]);
|
assert_eq!(filled, [2, 3, 7, 8]);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Fill values, shared ones in the SOHM heap included, resolve
|
||||||
|
/// identically through a read_at-only CountingStorage.
|
||||||
|
#[test]
|
||||||
|
fn storage_reads_match_slice_reads() {
|
||||||
|
use crate::object_header::ObjectHeader;
|
||||||
|
use crate::storage::CountingStorage;
|
||||||
|
let file: &[u8] = include_bytes!("../tests/fixtures/shared_fill_value.h5");
|
||||||
|
let sb = crate::superblock::Superblock::parse(file, 0).unwrap();
|
||||||
|
let (os, ls) = (sb.offset_size, sb.length_size);
|
||||||
|
let storage = CountingStorage::new(file.to_vec());
|
||||||
|
let mut shared = 0;
|
||||||
|
let children =
|
||||||
|
crate::group_v2::resolve_group_children(file, &sb, sb.root_group_address).unwrap();
|
||||||
|
assert!(children.len() >= 3);
|
||||||
|
for child in children {
|
||||||
|
let h =
|
||||||
|
ObjectHeader::parse(file, child.object_header_address as usize, os, ls).unwrap();
|
||||||
|
shared += h
|
||||||
|
.messages
|
||||||
|
.iter()
|
||||||
|
.filter(|m| {
|
||||||
|
m.msg_type == MessageType::FillValue
|
||||||
|
&& crate::shared_message::is_shared(m.flags)
|
||||||
|
})
|
||||||
|
.count();
|
||||||
|
let want = dataset_fill_value_in(file, &h.messages, os, ls);
|
||||||
|
assert_eq!(want, Ok(Some((-7i32).to_le_bytes().to_vec())));
|
||||||
|
let got = dataset_fill_value_from_storage(&storage, &h.messages, os, ls);
|
||||||
|
assert_eq!(got, want, "{}", child.name);
|
||||||
|
}
|
||||||
|
assert!(shared >= 2);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -19,8 +19,36 @@ pub const FILTER_SCALEOFFSET: u16 = 6;
|
|||||||
pub const FILTER_LZ4: u16 = 32004;
|
pub const FILTER_LZ4: u16 = 32004;
|
||||||
/// Zstandard compression.
|
/// Zstandard compression.
|
||||||
pub const FILTER_ZSTD: u16 = 32015;
|
pub const FILTER_ZSTD: u16 = 32015;
|
||||||
/// Pcodec lossless numerical codec (clawhdf5 internal; not yet HDF5-registered).
|
/// bzip2 (registered by PyTables; hdf5plugin's `BZip2`).
|
||||||
pub const FILTER_PCODEC: u16 = 32023;
|
pub const FILTER_BZIP2: u16 = 307;
|
||||||
|
/// LZF — h5py's built-in `compression="lzf"`.
|
||||||
|
pub const FILTER_LZF: u16 = 32000;
|
||||||
|
/// Blosc 1 (hdf5-blosc; hdf5plugin's `Blosc`).
|
||||||
|
pub const FILTER_BLOSC: u16 = 32001;
|
||||||
|
/// Bitshuffle, optionally with LZ4 or Zstandard (hdf5plugin's `Bitshuffle`).
|
||||||
|
pub const FILTER_BITSHUFFLE: u16 = 32008;
|
||||||
|
/// ZFP lossy (and lossless) compression of numeric arrays (H5Z-ZFP;
|
||||||
|
/// hdf5plugin's `Zfp`). Read-only, with the `zfp` feature.
|
||||||
|
pub const FILTER_ZFP: u16 = 32013;
|
||||||
|
/// Blosc 2 (hdf5plugin's `Blosc2`).
|
||||||
|
pub const FILTER_BLOSC2: u16 = 32026;
|
||||||
|
/// Pcodec lossless numerical codec — a **private, unregistered** clawhdf5
|
||||||
|
/// filter. Pcodec has no ID in the HDF Group's filter registry (checked
|
||||||
|
/// 2026-09-25, `hdf5_plugins/docs/RegisteredFilterPlugins.md`), so it uses an
|
||||||
|
/// ID from the registry's testing/private range (256–511). No libhdf5 plugin
|
||||||
|
/// decodes it: h5py/libhdf5 report the filter as unavailable. Only clawhdf5
|
||||||
|
/// (with the `pcodec` feature) reads these datasets.
|
||||||
|
pub const FILTER_PCODEC: u16 = 480;
|
||||||
|
/// Filter name written with [`FILTER_PCODEC`].
|
||||||
|
pub const FILTER_PCODEC_NAME: &str = "pcodec (clawhdf5 private)";
|
||||||
|
/// The ID clawhdf5 up to 2.7.0 wrote pcodec under. It is registered to
|
||||||
|
/// Granular BitRound (GBR), whose decode is a pass-through, so libhdf5 with
|
||||||
|
/// that plugin would have returned the compressed bytes as data. Read as
|
||||||
|
/// pcodec only when the filter is named exactly [`FILTER_PCODEC_LEGACY_NAME`],
|
||||||
|
/// the name those versions wrote; never written.
|
||||||
|
pub const FILTER_PCODEC_LEGACY: u16 = 32023;
|
||||||
|
/// The filter name clawhdf5 up to 2.7.0 wrote with [`FILTER_PCODEC_LEGACY`].
|
||||||
|
pub const FILTER_PCODEC_LEGACY_NAME: &str = "pcodec";
|
||||||
|
|
||||||
/// Description of a single filter in a pipeline.
|
/// Description of a single filter in a pipeline.
|
||||||
#[derive(Debug, Clone, PartialEq)]
|
#[derive(Debug, Clone, PartialEq)]
|
||||||
|
|||||||
@@ -0,0 +1,496 @@
|
|||||||
|
//! Filter registry: every filter is looked up here by its HDF5 filter ID.
|
||||||
|
//!
|
||||||
|
//! Two tiers:
|
||||||
|
//!
|
||||||
|
//! * **Built-in filters** — a static table of the filters compiled into this
|
||||||
|
//! build: the HDF5 standard filters (deflate, shuffle, Fletcher32, szip,
|
||||||
|
//! N-Bit, scale-offset) and the plugin filters whose cargo features are
|
||||||
|
//! enabled (LZ4, Zstandard, pcodec, LZF, bitshuffle, bzip2, blosc,
|
||||||
|
//! blosc2, zfp).
|
||||||
|
//! [`builtin_filters`] lists them.
|
||||||
|
//! * **Registered filters** (`std` only) — codecs the application supplies
|
||||||
|
//! for any other ID with [`register_filter`] (a [`FilterCodec`], or just a
|
||||||
|
//! decoding closure). A registered codec cannot shadow a built-in one,
|
||||||
|
//! except under 32023: that ID belongs to Granular BitRound, and the
|
||||||
|
//! built-in entry there only reads the pcodec chunks clawhdf5 <= 2.7.0
|
||||||
|
//! wrote (filter name `"pcodec"`), so a codec registered for 32023 handles
|
||||||
|
//! every other chunk with that ID, and writes.
|
||||||
|
//!
|
||||||
|
//! An ID in neither tier fails with [`FormatError::UnsupportedFilter`], as it
|
||||||
|
//! always has.
|
||||||
|
//!
|
||||||
|
//! ```
|
||||||
|
//! # #[cfg(feature = "std")] {
|
||||||
|
//! use clawhdf5_format::filter_registry::{self, FilterContext};
|
||||||
|
//! use clawhdf5_format::error::FormatError;
|
||||||
|
//!
|
||||||
|
//! // A toy filter in the private-use range: every byte XORed with 0x5A.
|
||||||
|
//! filter_registry::register_filter(300, |input: &[u8], _ctx: &FilterContext<'_>| {
|
||||||
|
//! Ok::<_, FormatError>(input.iter().map(|b| b ^ 0x5A).collect())
|
||||||
|
//! })
|
||||||
|
//! .unwrap();
|
||||||
|
//! assert!(filter_registry::is_filter_available(300));
|
||||||
|
//! filter_registry::unregister_filter(300);
|
||||||
|
//! # }
|
||||||
|
//! ```
|
||||||
|
|
||||||
|
#[cfg(not(feature = "std"))]
|
||||||
|
extern crate alloc;
|
||||||
|
|
||||||
|
#[cfg(not(feature = "std"))]
|
||||||
|
use alloc::vec::Vec;
|
||||||
|
|
||||||
|
use crate::error::FormatError;
|
||||||
|
use crate::filter_pipeline::FilterDescription;
|
||||||
|
|
||||||
|
/// What a codec is told about the filter it is applying.
|
||||||
|
#[derive(Debug, Clone, Copy)]
|
||||||
|
pub struct FilterContext<'a> {
|
||||||
|
/// The filter as recorded in the dataset's filter pipeline: its ID, name,
|
||||||
|
/// flags and client data (`cd_values`).
|
||||||
|
pub filter: &'a FilterDescription,
|
||||||
|
/// Size in bytes of one dataset element (the datatype's size).
|
||||||
|
pub element_size: usize,
|
||||||
|
/// Decoding only: the most bytes this stage may produce — what entered
|
||||||
|
/// the filter when the chunk was written. 0 means unknown; a decoder then
|
||||||
|
/// falls back to a fixed ceiling. Always 0 when encoding.
|
||||||
|
pub max_output: usize,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl FilterContext<'_> {
|
||||||
|
/// The filter's client data (`cd_values`).
|
||||||
|
pub fn client_data(&self) -> &[u32] {
|
||||||
|
&self.filter.client_data
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The largest output a decoder should allow: [`Self::max_output`], or
|
||||||
|
/// 256 MiB when that is unknown.
|
||||||
|
pub fn output_limit(&self) -> usize {
|
||||||
|
if self.max_output != 0 {
|
||||||
|
self.max_output
|
||||||
|
} else {
|
||||||
|
crate::filters::MAX_DECOMPRESS_SIZE
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A filter implementation.
|
||||||
|
///
|
||||||
|
/// `decode` undoes the filter (the read direction). `encode` applies it (the
|
||||||
|
/// write direction); the default refuses with
|
||||||
|
/// [`FormatError::UnsupportedFilter`], which is right for a read-only codec.
|
||||||
|
pub trait FilterCodec: Send + Sync {
|
||||||
|
/// Undo the filter on one chunk. The output must not exceed
|
||||||
|
/// [`FilterContext::output_limit`]; the pipeline rejects a larger one.
|
||||||
|
fn decode(&self, input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError>;
|
||||||
|
|
||||||
|
/// Apply the filter to one chunk.
|
||||||
|
fn encode(&self, input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||||
|
let _ = input;
|
||||||
|
Err(FormatError::UnsupportedFilter(ctx.filter.filter_id))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Any `Fn(&[u8], &FilterContext) -> Result<Vec<u8>, FormatError>` is a
|
||||||
|
/// decode-only codec.
|
||||||
|
impl<F> FilterCodec for F
|
||||||
|
where
|
||||||
|
F: Fn(&[u8], &FilterContext<'_>) -> Result<Vec<u8>, FormatError> + Send + Sync,
|
||||||
|
{
|
||||||
|
fn decode(&self, input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||||
|
self(input, ctx)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Signature of a built-in filter's decoder or encoder.
|
||||||
|
pub type BuiltinFn = fn(&[u8], &FilterContext<'_>) -> Result<Vec<u8>, FormatError>;
|
||||||
|
|
||||||
|
/// A filter compiled into this build.
|
||||||
|
#[derive(Debug, Clone, Copy)]
|
||||||
|
pub struct BuiltinFilter {
|
||||||
|
/// HDF5 filter ID.
|
||||||
|
pub id: u16,
|
||||||
|
/// Human-readable name.
|
||||||
|
pub name: &'static str,
|
||||||
|
/// Decoder.
|
||||||
|
pub(crate) decode: BuiltinFn,
|
||||||
|
/// Encoder, if this build can write the filter.
|
||||||
|
pub(crate) encode: Option<BuiltinFn>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl BuiltinFilter {
|
||||||
|
/// Whether this build can write the filter as well as read it.
|
||||||
|
pub fn can_encode(&self) -> bool {
|
||||||
|
self.encode.is_some()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether the built-in entry only borrows its ID for some chunks, so a
|
||||||
|
/// registered codec may take the rest: the legacy pcodec entry under
|
||||||
|
/// Granular BitRound's 32023, which claims only chunks named `"pcodec"`.
|
||||||
|
fn is_shared(&self) -> bool {
|
||||||
|
self.id == crate::filter_pipeline::FILTER_PCODEC_LEGACY
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether this entry decodes chunks written with `filter`.
|
||||||
|
fn claims(&self, filter: &crate::filter_pipeline::FilterDescription) -> bool {
|
||||||
|
!self.is_shared()
|
||||||
|
|| filter.name.as_deref() == Some(crate::filter_pipeline::FILTER_PCODEC_LEGACY_NAME)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The filters compiled into this build, in ID order.
|
||||||
|
pub fn builtin_filters() -> &'static [BuiltinFilter] {
|
||||||
|
crate::filters::BUILTIN_FILTERS
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The built-in filter with this ID, if it is compiled in.
|
||||||
|
pub fn builtin_filter(id: u16) -> Option<&'static BuiltinFilter> {
|
||||||
|
builtin_filters().iter().find(|f| f.id == id)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Why a filter ID may be missing from this build: the filter's name, and
|
||||||
|
/// the cargo feature that provides it (`None`: clawhdf5 does not implement
|
||||||
|
/// it — register a codec for it with [`register_filter`]). `None` for an ID
|
||||||
|
/// clawhdf5 knows nothing about.
|
||||||
|
pub fn known_filter(id: u16) -> Option<(&'static str, Option<&'static str>)> {
|
||||||
|
Some(match id {
|
||||||
|
1 => ("deflate", Some("deflate")),
|
||||||
|
4 => ("SZIP", Some("szip")),
|
||||||
|
307 => ("bzip2", Some("bzip2")),
|
||||||
|
480 => ("pcodec", Some("pcodec")),
|
||||||
|
32000 => ("LZF", Some("lzf")),
|
||||||
|
32001 => ("Blosc", Some("blosc")),
|
||||||
|
32004 => ("LZ4", Some("lz4")),
|
||||||
|
32008 => ("bitshuffle", Some("bitshuffle")),
|
||||||
|
32013 => ("ZFP", Some("zfp")),
|
||||||
|
32015 => ("Zstandard", Some("zstd")),
|
||||||
|
32019 => ("JPEG", None),
|
||||||
|
32022 => ("BitGroom", None),
|
||||||
|
32023 => ("Granular BitRound", None),
|
||||||
|
32026 => ("Blosc2", Some("blosc2")),
|
||||||
|
_ => return None,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether a chunk filtered with `id` can be decoded: a built-in filter or a
|
||||||
|
/// registered one.
|
||||||
|
pub fn is_filter_available(id: u16) -> bool {
|
||||||
|
if builtin_filter(id).is_some() {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
{
|
||||||
|
registered(id).is_some()
|
||||||
|
}
|
||||||
|
#[cfg(not(feature = "std"))]
|
||||||
|
{
|
||||||
|
false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether chunks filtered with `id` may be decoded by a codec the
|
||||||
|
/// application registered (whose stored sizes this crate cannot bound).
|
||||||
|
pub(crate) fn may_be_registered(id: u16) -> bool {
|
||||||
|
if builtin_filter(id).is_some_and(|b| !b.is_shared()) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
{
|
||||||
|
registered(id).is_some()
|
||||||
|
}
|
||||||
|
#[cfg(not(feature = "std"))]
|
||||||
|
{
|
||||||
|
false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
mod custom {
|
||||||
|
use super::FilterCodec;
|
||||||
|
use std::collections::BTreeMap;
|
||||||
|
use std::sync::{Arc, PoisonError, RwLock};
|
||||||
|
|
||||||
|
pub(super) type Registry = BTreeMap<u16, Arc<dyn FilterCodec>>;
|
||||||
|
|
||||||
|
static REGISTRY: RwLock<Registry> = RwLock::new(BTreeMap::new());
|
||||||
|
|
||||||
|
pub(super) fn with_read<R>(f: impl FnOnce(&Registry) -> R) -> R {
|
||||||
|
// A panic while holding the lock cannot leave the map half-updated
|
||||||
|
// (every update is a single insert/remove), so poisoning is ignored.
|
||||||
|
f(®ISTRY.read().unwrap_or_else(PoisonError::into_inner))
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(super) fn with_write<R>(f: impl FnOnce(&mut Registry) -> R) -> R {
|
||||||
|
f(&mut REGISTRY.write().unwrap_or_else(PoisonError::into_inner))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Register a codec for filter `id`, process-wide. It is used for every
|
||||||
|
/// chunk read (and, if it implements [`FilterCodec::encode`], written) with
|
||||||
|
/// that filter ID, by every file.
|
||||||
|
///
|
||||||
|
/// A plain closure `Fn(&[u8], &FilterContext) -> Result<Vec<u8>, FormatError>`
|
||||||
|
/// registers a decoder. Replaces (and returns) an earlier registration for
|
||||||
|
/// the same ID. Fails with [`FormatError::FilterError`] if `id` is a built-in
|
||||||
|
/// filter of this build: those cannot be overridden. The exception is 32023
|
||||||
|
/// (Granular BitRound): with the `pcodec` feature the built-in entry there
|
||||||
|
/// reads only chunks whose filter is named `"pcodec"` (clawhdf5 <= 2.7.0's
|
||||||
|
/// files); a codec registered for 32023 decodes every other chunk with that
|
||||||
|
/// ID and does all the writing.
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
pub fn register_filter<C>(
|
||||||
|
id: u16,
|
||||||
|
codec: C,
|
||||||
|
) -> Result<Option<std::sync::Arc<dyn FilterCodec>>, FormatError>
|
||||||
|
where
|
||||||
|
C: FilterCodec + 'static,
|
||||||
|
{
|
||||||
|
if let Some(builtin) = builtin_filter(id).filter(|b| !b.is_shared()) {
|
||||||
|
return Err(FormatError::FilterError(format!(
|
||||||
|
"filter {id} ({}) is built in and cannot be re-registered",
|
||||||
|
builtin.name
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
let codec: std::sync::Arc<dyn FilterCodec> = std::sync::Arc::new(codec);
|
||||||
|
Ok(custom::with_write(|r| r.insert(id, codec)))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Remove the codec registered for `id`. Returns whether one was registered.
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
pub fn unregister_filter(id: u16) -> bool {
|
||||||
|
custom::with_write(|r| r.remove(&id).is_some())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The codec registered for `id`, if any.
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
pub fn registered(id: u16) -> Option<std::sync::Arc<dyn FilterCodec>> {
|
||||||
|
custom::with_read(|r| r.get(&id).cloned())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Undo filter `ctx.filter` on `input`: the built-in decoder if there is one
|
||||||
|
/// that claims the chunk, else a registered one, else the built-in decoder's
|
||||||
|
/// own refusal or [`FormatError::UnsupportedFilter`].
|
||||||
|
pub(crate) fn decode(input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||||
|
let id = ctx.filter.filter_id;
|
||||||
|
let builtin = builtin_filter(id);
|
||||||
|
if let Some(builtin) = builtin.filter(|b| b.claims(ctx.filter)) {
|
||||||
|
return (builtin.decode)(input, ctx);
|
||||||
|
}
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
if let Some(codec) = registered(id) {
|
||||||
|
let out = codec.decode(input, ctx)?;
|
||||||
|
// A registered codec is outside our control: hold it to the same
|
||||||
|
// bound the built-in decoders enforce.
|
||||||
|
if out.len() > ctx.output_limit() {
|
||||||
|
return Err(FormatError::DecompressionError(format!(
|
||||||
|
"filter {id}: decoded {} bytes, more than the {} the chunk can hold",
|
||||||
|
out.len(),
|
||||||
|
ctx.output_limit()
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
return Ok(out);
|
||||||
|
}
|
||||||
|
match builtin {
|
||||||
|
Some(builtin) => (builtin.decode)(input, ctx),
|
||||||
|
None => Err(FormatError::UnsupportedFilter(id)),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Apply filter `ctx.filter` to `input`.
|
||||||
|
pub(crate) fn encode(input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||||
|
let id = ctx.filter.filter_id;
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
if builtin_filter(id).is_some_and(|b| b.is_shared())
|
||||||
|
&& let Some(codec) = registered(id)
|
||||||
|
{
|
||||||
|
return codec.encode(input, ctx);
|
||||||
|
}
|
||||||
|
if let Some(builtin) = builtin_filter(id) {
|
||||||
|
return match builtin.encode {
|
||||||
|
Some(encode) => encode(input, ctx),
|
||||||
|
None => Err(FormatError::UnsupportedFilter(id)),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
if let Some(codec) = registered(id) {
|
||||||
|
return codec.encode(input, ctx);
|
||||||
|
}
|
||||||
|
Err(FormatError::UnsupportedFilter(id))
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(all(test, feature = "std"))]
|
||||||
|
pub(crate) mod tests {
|
||||||
|
use super::*;
|
||||||
|
use crate::filter_pipeline::{FILTER_FLETCHER32, FILTER_SHUFFLE, FilterPipeline};
|
||||||
|
use crate::filters::{compress_chunk, decompress_chunk};
|
||||||
|
|
||||||
|
fn pipeline(id: u16) -> FilterPipeline {
|
||||||
|
FilterPipeline {
|
||||||
|
version: 2,
|
||||||
|
filters: vec![FilterDescription {
|
||||||
|
filter_id: id,
|
||||||
|
name: Some("test".into()),
|
||||||
|
flags: 0,
|
||||||
|
client_data: vec![7],
|
||||||
|
}],
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
struct Xor;
|
||||||
|
impl FilterCodec for Xor {
|
||||||
|
fn decode(&self, input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||||
|
let k = ctx.client_data()[0] as u8;
|
||||||
|
Ok(input.iter().map(|b| b ^ k).collect())
|
||||||
|
}
|
||||||
|
fn encode(&self, input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||||
|
self.decode(input, ctx)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Each test uses its own ID: the registry is process-wide and tests run
|
||||||
|
// in parallel.
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn unknown_filter_keeps_its_error() {
|
||||||
|
let err = decompress_chunk(b"abc", &pipeline(311), 3, 1).unwrap_err();
|
||||||
|
assert_eq!(err, FormatError::UnsupportedFilter(311));
|
||||||
|
let err = compress_chunk(b"abc", &pipeline(311), 1).unwrap_err();
|
||||||
|
assert_eq!(err, FormatError::UnsupportedFilter(311));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn registered_codec_round_trips_through_the_pipeline() {
|
||||||
|
assert!(!is_filter_available(312));
|
||||||
|
assert!(register_filter(312, Xor).unwrap().is_none());
|
||||||
|
assert!(is_filter_available(312));
|
||||||
|
let data = b"hello, registry".to_vec();
|
||||||
|
let enc = compress_chunk(&data, &pipeline(312), 1).unwrap();
|
||||||
|
assert_ne!(enc, data);
|
||||||
|
assert_eq!(
|
||||||
|
decompress_chunk(&enc, &pipeline(312), data.len(), 1).unwrap(),
|
||||||
|
data
|
||||||
|
);
|
||||||
|
assert!(unregister_filter(312));
|
||||||
|
assert!(!unregister_filter(312));
|
||||||
|
assert_eq!(
|
||||||
|
decompress_chunk(&enc, &pipeline(312), data.len(), 1).unwrap_err(),
|
||||||
|
FormatError::UnsupportedFilter(312)
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn closure_registers_a_decoder_only() {
|
||||||
|
register_filter(313, |input: &[u8], _ctx: &FilterContext<'_>| {
|
||||||
|
Ok(input.iter().rev().copied().collect())
|
||||||
|
})
|
||||||
|
.unwrap();
|
||||||
|
assert_eq!(
|
||||||
|
decompress_chunk(b"abc", &pipeline(313), 3, 1).unwrap(),
|
||||||
|
b"cba"
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
compress_chunk(b"abc", &pipeline(313), 1).unwrap_err(),
|
||||||
|
FormatError::UnsupportedFilter(313)
|
||||||
|
);
|
||||||
|
unregister_filter(313);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn registered_decoder_output_is_bounded() {
|
||||||
|
register_filter(314, |_input: &[u8], _ctx: &FilterContext<'_>| {
|
||||||
|
Ok(vec![0u8; 1000])
|
||||||
|
})
|
||||||
|
.unwrap();
|
||||||
|
let err = decompress_chunk(b"abc", &pipeline(314), 10, 1).unwrap_err();
|
||||||
|
assert!(matches!(err, FormatError::DecompressionError(_)), "{err:?}");
|
||||||
|
unregister_filter(314);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn builtins_cannot_be_overridden() {
|
||||||
|
for id in [FILTER_SHUFFLE, FILTER_FLETCHER32] {
|
||||||
|
let Err(err) = register_filter(id, Xor) else {
|
||||||
|
panic!("built-in filter {id} was re-registered");
|
||||||
|
};
|
||||||
|
assert!(matches!(err, FormatError::FilterError(_)), "{err:?}");
|
||||||
|
}
|
||||||
|
assert!(builtin_filter(FILTER_SHUFFLE).is_some());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Serialises the tests that register or read filter 32023 (the
|
||||||
|
/// registry is process-wide).
|
||||||
|
pub(crate) static ID_32023: std::sync::Mutex<()> = std::sync::Mutex::new(());
|
||||||
|
|
||||||
|
/// 32023 is Granular BitRound's ID; the `pcodec` build's built-in entry
|
||||||
|
/// there reads only clawhdf5 <= 2.7.0's pcodec chunks (named "pcodec"),
|
||||||
|
/// so a codec can be registered for the rest, and writes with it.
|
||||||
|
#[test]
|
||||||
|
fn a_codec_can_be_registered_for_granular_bitround() {
|
||||||
|
let _guard = ID_32023
|
||||||
|
.lock()
|
||||||
|
.unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||||
|
let named = |name: Option<&str>| FilterPipeline {
|
||||||
|
version: 2,
|
||||||
|
filters: vec![FilterDescription {
|
||||||
|
filter_id: 32023,
|
||||||
|
name: name.map(Into::into),
|
||||||
|
flags: 0,
|
||||||
|
client_data: vec![7],
|
||||||
|
}],
|
||||||
|
};
|
||||||
|
let prev = register_filter(32023, Xor).expect("32023 must be registrable");
|
||||||
|
assert!(prev.is_none());
|
||||||
|
let data = b"granular bitround".to_vec();
|
||||||
|
for name in [None, Some("granular_bitround"), Some("test")] {
|
||||||
|
let pl = named(name);
|
||||||
|
let enc = compress_chunk(&data, &pl, 1).unwrap();
|
||||||
|
assert_ne!(enc, data);
|
||||||
|
assert_eq!(decompress_chunk(&enc, &pl, data.len(), 1).unwrap(), data);
|
||||||
|
}
|
||||||
|
// clawhdf5 <= 2.7.0's pcodec chunks still go to the built-in reader.
|
||||||
|
#[cfg(feature = "pcodec")]
|
||||||
|
{
|
||||||
|
let raw: Vec<u8> = (0..64)
|
||||||
|
.flat_map(|i| (f64::from(i) * 0.5).to_le_bytes())
|
||||||
|
.collect();
|
||||||
|
let comp = crate::filters::pcodec_compress(&raw, 8).unwrap();
|
||||||
|
let mut pl = named(Some("pcodec"));
|
||||||
|
pl.filters[0].client_data = vec![8];
|
||||||
|
assert_eq!(decompress_chunk(&comp, &pl, raw.len(), 8).unwrap(), raw);
|
||||||
|
}
|
||||||
|
assert!(unregister_filter(32023));
|
||||||
|
let pl = named(None);
|
||||||
|
assert!(matches!(
|
||||||
|
decompress_chunk(&data, &pl, data.len(), 1),
|
||||||
|
Err(FormatError::UnsupportedFilter(32023))
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn unsupported_filter_error_names_the_filter() {
|
||||||
|
let msg = FormatError::UnsupportedFilter(32026).to_string();
|
||||||
|
assert!(msg.contains("Blosc2") && msg.contains("`blosc2`"), "{msg}");
|
||||||
|
let msg = FormatError::UnsupportedFilter(32013).to_string();
|
||||||
|
assert!(msg.contains("ZFP") && msg.contains("`zfp`"), "{msg}");
|
||||||
|
let msg = FormatError::UnsupportedFilter(32019).to_string();
|
||||||
|
assert!(
|
||||||
|
msg.contains("JPEG") && msg.contains("not implemented"),
|
||||||
|
"{msg}"
|
||||||
|
);
|
||||||
|
let msg = FormatError::UnsupportedFilter(32000).to_string();
|
||||||
|
assert!(msg.contains("LZF") && msg.contains("`lzf`"), "{msg}");
|
||||||
|
assert_eq!(
|
||||||
|
FormatError::UnsupportedFilter(399).to_string(),
|
||||||
|
"unsupported filter: 399"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn builtin_table_is_sorted_and_unique() {
|
||||||
|
let ids: Vec<u16> = builtin_filters().iter().map(|f| f.id).collect();
|
||||||
|
let mut sorted = ids.clone();
|
||||||
|
sorted.sort_unstable();
|
||||||
|
sorted.dedup();
|
||||||
|
assert_eq!(ids, sorted);
|
||||||
|
}
|
||||||
|
}
|
||||||
+1709
-332
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,446 @@
|
|||||||
|
//! Bitshuffle (HDF5 filter 32008) and the bit transpose it shares with blosc.
|
||||||
|
//!
|
||||||
|
//! **The transform.** A block of `n` elements (`n` a multiple of 8) of
|
||||||
|
//! `es` bytes each is viewed as an `n × 8·es` bit matrix — row *i* is
|
||||||
|
//! element *i*, column `8·j + k` is bit *k* (LSB first) of its byte *j* — and
|
||||||
|
//! transposed: the output is `8·es` rows of `n` bits, row `8·j + k` holding
|
||||||
|
//! bit *k* of byte *j* of every element in order, packed LSB first. That is
|
||||||
|
//! what `bshuf_trans_bit_elem` produces (checked against hdf5plugin's
|
||||||
|
//! library bit for bit).
|
||||||
|
//!
|
||||||
|
//! **The filter** (`bshuf_h5filter.c`). `cd_values`: `[0..2]` bitshuffle
|
||||||
|
//! version, `[2]` element size, `[3]` block size in elements (0 = default:
|
||||||
|
//! 8192 bytes' worth, rounded down to a multiple of 8, at least 128),
|
||||||
|
//! `[4]` compression (0 none, 2 LZ4, 3 Zstandard), `[5]` Zstandard level.
|
||||||
|
//! The chunk is cut into blocks of `block size` elements; the tail shorter
|
||||||
|
//! than a block is transposed as one block rounded down to a multiple of 8
|
||||||
|
//! elements, and the last `n mod 8` elements are stored as they are.
|
||||||
|
//! Uncompressed, that is the whole chunk. Compressed, the chunk starts with a
|
||||||
|
//! 12-byte header — the decoded size (u64 big-endian) and the block size in
|
||||||
|
//! bytes (u32 big-endian) — and each transposed block is stored as a u32
|
||||||
|
//! big-endian length and an LZ4 block / Zstandard frame; the untransposed
|
||||||
|
//! tail follows the last block.
|
||||||
|
|
||||||
|
#[cfg(not(feature = "std"))]
|
||||||
|
extern crate alloc;
|
||||||
|
#[cfg(not(feature = "std"))]
|
||||||
|
use alloc::{format, vec, vec::Vec};
|
||||||
|
|
||||||
|
use crate::error::FormatError;
|
||||||
|
#[cfg(feature = "bitshuffle")]
|
||||||
|
use crate::filter_registry::FilterContext;
|
||||||
|
|
||||||
|
/// Transpose an 8×8 bit matrix packed in a u64 (byte *r* = row *r*, bit *c*
|
||||||
|
/// of that byte = column *c*). An involution.
|
||||||
|
#[inline]
|
||||||
|
fn transpose8(mut x: u64) -> u64 {
|
||||||
|
let t = (x ^ (x >> 7)) & 0x00AA_00AA_00AA_00AA;
|
||||||
|
x = x ^ t ^ (t << 7);
|
||||||
|
let t = (x ^ (x >> 14)) & 0x0000_CCCC_0000_CCCC;
|
||||||
|
x = x ^ t ^ (t << 14);
|
||||||
|
let t = (x ^ (x >> 28)) & 0x0000_0000_F0F0_F0F0;
|
||||||
|
x ^ t ^ (t << 28)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Bit-transpose one block: `input` and `out` are `n * es` bytes, `n` a
|
||||||
|
/// multiple of 8.
|
||||||
|
pub(crate) fn bitshuffle_block(input: &[u8], out: &mut [u8], n: usize, es: usize) {
|
||||||
|
debug_assert!(n.is_multiple_of(8) && input.len() == n * es && out.len() == n * es);
|
||||||
|
let row = n / 8;
|
||||||
|
for j in 0..es {
|
||||||
|
for g in 0..row {
|
||||||
|
let mut x = 0u64;
|
||||||
|
for t in 0..8 {
|
||||||
|
x |= u64::from(input[(8 * g + t) * es + j]) << (8 * t);
|
||||||
|
}
|
||||||
|
let y = transpose8(x);
|
||||||
|
for k in 0..8 {
|
||||||
|
out[(8 * j + k) * row + g] = (y >> (8 * k)) as u8;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Undo [`bitshuffle_block`].
|
||||||
|
pub(crate) fn bitunshuffle_block(input: &[u8], out: &mut [u8], n: usize, es: usize) {
|
||||||
|
debug_assert!(n.is_multiple_of(8) && input.len() == n * es && out.len() == n * es);
|
||||||
|
let row = n / 8;
|
||||||
|
for j in 0..es {
|
||||||
|
for g in 0..row {
|
||||||
|
let mut y = 0u64;
|
||||||
|
for k in 0..8 {
|
||||||
|
y |= u64::from(input[(8 * j + k) * row + g]) << (8 * k);
|
||||||
|
}
|
||||||
|
let x = transpose8(y);
|
||||||
|
for t in 0..8 {
|
||||||
|
out[(8 * g + t) * es + j] = (x >> (8 * t)) as u8;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `bshuf_default_block_size`: 8 KiB of elements, a multiple of 8, >= 128.
|
||||||
|
#[cfg(feature = "bitshuffle")]
|
||||||
|
fn default_block_size(es: usize) -> usize {
|
||||||
|
((8192 / es) / 8 * 8).max(128)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(feature = "bitshuffle")]
|
||||||
|
fn err(msg: &str) -> FormatError {
|
||||||
|
FormatError::DecompressionError(format!("bitshuffle: {msg}"))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `cd_values[4]`: the compression bitshuffle applies after the transpose.
|
||||||
|
#[cfg(feature = "bitshuffle")]
|
||||||
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||||
|
enum Codec {
|
||||||
|
None,
|
||||||
|
Lz4,
|
||||||
|
Zstd,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(feature = "bitshuffle")]
|
||||||
|
fn codec(cd: &[u32]) -> Result<Codec, FormatError> {
|
||||||
|
match cd.get(4).copied().unwrap_or(0) {
|
||||||
|
0 => Ok(Codec::None),
|
||||||
|
2 => Ok(Codec::Lz4),
|
||||||
|
3 => Ok(Codec::Zstd),
|
||||||
|
other => Err(FormatError::FilterError(format!(
|
||||||
|
"bitshuffle: unknown compression {other}"
|
||||||
|
))),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The element counts of the transposed blocks for `size` elements.
|
||||||
|
#[cfg(feature = "bitshuffle")]
|
||||||
|
fn blocks(size: usize, block: usize) -> impl Iterator<Item = usize> {
|
||||||
|
let full = size / block;
|
||||||
|
let last = (size % block) / 8 * 8;
|
||||||
|
core::iter::repeat_n(block, full).chain((last > 0).then_some(last))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Decode a bitshuffle-filtered chunk.
|
||||||
|
#[cfg(feature = "bitshuffle")]
|
||||||
|
pub(crate) fn bitshuffle_decode(
|
||||||
|
input: &[u8],
|
||||||
|
ctx: &FilterContext<'_>,
|
||||||
|
) -> Result<Vec<u8>, FormatError> {
|
||||||
|
let cd = ctx.client_data();
|
||||||
|
let es = match cd.get(2) {
|
||||||
|
Some(&e) if e != 0 => e as usize,
|
||||||
|
_ => return Err(err("missing element size")),
|
||||||
|
};
|
||||||
|
let codec = codec(cd)?;
|
||||||
|
let limit = ctx.output_limit();
|
||||||
|
if codec == Codec::None {
|
||||||
|
if input.len() > limit {
|
||||||
|
return Err(err("output exceeds the chunk size"));
|
||||||
|
}
|
||||||
|
let block = match cd.get(3) {
|
||||||
|
Some(&b) if b != 0 => b as usize,
|
||||||
|
_ => default_block_size(es),
|
||||||
|
};
|
||||||
|
if !block.is_multiple_of(8) {
|
||||||
|
return Err(err("block size is not a multiple of 8"));
|
||||||
|
}
|
||||||
|
if !input.len().is_multiple_of(es) {
|
||||||
|
return Err(err("chunk is not a whole number of elements"));
|
||||||
|
}
|
||||||
|
let size = input.len() / es;
|
||||||
|
let mut out = vec![0u8; input.len()];
|
||||||
|
let mut pos = 0;
|
||||||
|
for n in blocks(size, block) {
|
||||||
|
let bytes = n * es;
|
||||||
|
bitunshuffle_block(&input[pos..pos + bytes], &mut out[pos..pos + bytes], n, es);
|
||||||
|
pos += bytes;
|
||||||
|
}
|
||||||
|
out[pos..].copy_from_slice(&input[pos..]);
|
||||||
|
return Ok(out);
|
||||||
|
}
|
||||||
|
|
||||||
|
let header = input.get(..12).ok_or_else(|| err("truncated header"))?;
|
||||||
|
let total = u64::from_be_bytes(header[..8].try_into().unwrap());
|
||||||
|
let block_bytes = u32::from_be_bytes(header[8..12].try_into().unwrap()) as usize;
|
||||||
|
let total = usize::try_from(total)
|
||||||
|
.ok()
|
||||||
|
.filter(|&t| t <= limit)
|
||||||
|
.ok_or_else(|| err("decoded size exceeds the chunk size"))?;
|
||||||
|
if !total.is_multiple_of(es) {
|
||||||
|
return Err(err("chunk is not a whole number of elements"));
|
||||||
|
}
|
||||||
|
if block_bytes == 0 || !block_bytes.is_multiple_of(es) {
|
||||||
|
return Err(err("bad block size"));
|
||||||
|
}
|
||||||
|
let block = block_bytes / es;
|
||||||
|
if !block.is_multiple_of(8) {
|
||||||
|
return Err(err("block size is not a multiple of 8"));
|
||||||
|
}
|
||||||
|
let size = total / es;
|
||||||
|
let mut out = vec![0u8; total];
|
||||||
|
let mut tmp = vec![0u8; block_bytes.min(total)];
|
||||||
|
let mut ip = 12usize;
|
||||||
|
let mut op = 0usize;
|
||||||
|
let mut zstd = None;
|
||||||
|
for n in blocks(size, block) {
|
||||||
|
let bytes = n * es;
|
||||||
|
let len = input
|
||||||
|
.get(ip..ip + 4)
|
||||||
|
.map(|b| u32::from_be_bytes(b.try_into().unwrap()) as usize)
|
||||||
|
.ok_or_else(|| err("truncated block header"))?;
|
||||||
|
ip += 4;
|
||||||
|
let comp = input
|
||||||
|
.get(ip..ip.saturating_add(len))
|
||||||
|
.ok_or_else(|| err("truncated block"))?;
|
||||||
|
ip += len;
|
||||||
|
let dst = &mut tmp[..bytes];
|
||||||
|
let got = match codec {
|
||||||
|
Codec::Lz4 => lz4_flex::block::decompress_into(comp, dst)
|
||||||
|
.map_err(|e| err(&format!("lz4: {e}")))?,
|
||||||
|
Codec::Zstd => zstd_decode_into(
|
||||||
|
zstd.get_or_insert_with(ruzstd::decoding::FrameDecoder::new),
|
||||||
|
comp,
|
||||||
|
dst,
|
||||||
|
)?,
|
||||||
|
Codec::None => unreachable!(),
|
||||||
|
};
|
||||||
|
if got != bytes {
|
||||||
|
return Err(err("block decoded to the wrong size"));
|
||||||
|
}
|
||||||
|
bitunshuffle_block(dst, &mut out[op..op + bytes], n, es);
|
||||||
|
op += bytes;
|
||||||
|
}
|
||||||
|
let tail = total - op;
|
||||||
|
let rest = input
|
||||||
|
.get(ip..ip + tail)
|
||||||
|
.ok_or_else(|| err("truncated trailing elements"))?;
|
||||||
|
out[op..].copy_from_slice(rest);
|
||||||
|
Ok(out)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Decode Zstandard frames into exactly `dst`, failing if they hold more.
|
||||||
|
///
|
||||||
|
/// ruzstd reserves a frame's declared window (by default up to 100 MiB)
|
||||||
|
/// before decoding it, so the window is capped at what the output could
|
||||||
|
/// need: twice `dst` (window sizes are rounded up), and at least 128 KiB.
|
||||||
|
/// The encoders behind these filters (c-blosc, c-blosc2, bitshuffle)
|
||||||
|
/// compress each block in one call with its size known, so libzstd's
|
||||||
|
/// window never exceeds the block.
|
||||||
|
#[cfg(any(feature = "bitshuffle", feature = "blosc"))]
|
||||||
|
pub(crate) fn zstd_decode_into(
|
||||||
|
decoder: &mut ruzstd::decoding::FrameDecoder,
|
||||||
|
frames: &[u8],
|
||||||
|
dst: &mut [u8],
|
||||||
|
) -> Result<usize, FormatError> {
|
||||||
|
decoder.set_max_window_size((2 * dst.len()).max(1 << 17) as u64);
|
||||||
|
decoder
|
||||||
|
.decode_all(frames, dst)
|
||||||
|
.map_err(|e| FormatError::DecompressionError(format!("zstd: {e}")))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Compress with ruzstd. It implements one level (roughly zstd's level 1),
|
||||||
|
/// so the requested level only matters to other encoders.
|
||||||
|
#[cfg(any(feature = "bitshuffle", feature = "blosc"))]
|
||||||
|
pub(crate) fn zstd_encode(data: &[u8]) -> Vec<u8> {
|
||||||
|
ruzstd::encoding::compress_to_vec(data, ruzstd::encoding::CompressionLevel::Fastest)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Encode a chunk with the bitshuffle filter.
|
||||||
|
#[cfg(feature = "bitshuffle")]
|
||||||
|
pub(crate) fn bitshuffle_encode(
|
||||||
|
input: &[u8],
|
||||||
|
ctx: &FilterContext<'_>,
|
||||||
|
) -> Result<Vec<u8>, FormatError> {
|
||||||
|
let cd = ctx.client_data();
|
||||||
|
let es = match cd.get(2) {
|
||||||
|
Some(&e) if e != 0 => e as usize,
|
||||||
|
_ => ctx.element_size.max(1),
|
||||||
|
};
|
||||||
|
let codec = codec(cd)?;
|
||||||
|
let block = match cd.get(3) {
|
||||||
|
Some(&b) if b != 0 => b as usize,
|
||||||
|
_ => default_block_size(es),
|
||||||
|
};
|
||||||
|
let cerr = |m: &str| FormatError::CompressionError(format!("bitshuffle: {m}"));
|
||||||
|
if !block.is_multiple_of(8) {
|
||||||
|
return Err(cerr("block size is not a multiple of 8"));
|
||||||
|
}
|
||||||
|
if !input.len().is_multiple_of(es) {
|
||||||
|
return Err(cerr("chunk is not a whole number of elements"));
|
||||||
|
}
|
||||||
|
let size = input.len() / es;
|
||||||
|
let mut out = Vec::with_capacity(input.len() + 12 + input.len() / 64);
|
||||||
|
if codec != Codec::None {
|
||||||
|
out.extend_from_slice(&(input.len() as u64).to_be_bytes());
|
||||||
|
let block_bytes =
|
||||||
|
u32::try_from(block * es).map_err(|_| cerr("block size does not fit in 32 bits"))?;
|
||||||
|
out.extend_from_slice(&block_bytes.to_be_bytes());
|
||||||
|
}
|
||||||
|
let mut tmp = vec![0u8; (block * es).min(input.len())];
|
||||||
|
let mut pos = 0;
|
||||||
|
for n in blocks(size, block) {
|
||||||
|
let bytes = n * es;
|
||||||
|
let dst = &mut tmp[..bytes];
|
||||||
|
bitshuffle_block(&input[pos..pos + bytes], dst, n, es);
|
||||||
|
match codec {
|
||||||
|
Codec::None => out.extend_from_slice(dst),
|
||||||
|
Codec::Lz4 | Codec::Zstd => {
|
||||||
|
let comp = if codec == Codec::Lz4 {
|
||||||
|
lz4_flex::block::compress(dst)
|
||||||
|
} else {
|
||||||
|
zstd_encode(dst)
|
||||||
|
};
|
||||||
|
out.extend_from_slice(&(comp.len() as u32).to_be_bytes());
|
||||||
|
out.extend_from_slice(&comp);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
pos += bytes;
|
||||||
|
}
|
||||||
|
out.extend_from_slice(&input[pos..]);
|
||||||
|
Ok(out)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
/// The definition, one bit at a time.
|
||||||
|
fn naive(input: &[u8], n: usize, es: usize) -> Vec<u8> {
|
||||||
|
let mut out = vec![0u8; n * es];
|
||||||
|
for i in 0..n {
|
||||||
|
for j in 0..es {
|
||||||
|
for k in 0..8 {
|
||||||
|
if input[i * es + j] >> k & 1 == 1 {
|
||||||
|
let p = (8 * j + k) * n + i;
|
||||||
|
out[p / 8] |= 1 << (p % 8);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
out
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn transpose_matches_the_definition_and_inverts() {
|
||||||
|
for (n, es) in [(8, 1), (16, 2), (24, 4), (128, 8), (64, 3), (8, 16)] {
|
||||||
|
let input: Vec<u8> = (0..n * es)
|
||||||
|
.map(|i| (i as u32).wrapping_mul(2_654_435_761).rotate_left(7) as u8)
|
||||||
|
.collect();
|
||||||
|
let mut out = vec![0u8; n * es];
|
||||||
|
bitshuffle_block(&input, &mut out, n, es);
|
||||||
|
assert_eq!(out, naive(&input, n, es), "n={n} es={es}");
|
||||||
|
let mut back = vec![0u8; n * es];
|
||||||
|
bitunshuffle_block(&out, &mut back, n, es);
|
||||||
|
assert_eq!(back, input);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(feature = "bitshuffle")]
|
||||||
|
fn ctx_for(cd: Vec<u32>) -> crate::filter_pipeline::FilterDescription {
|
||||||
|
crate::filter_pipeline::FilterDescription {
|
||||||
|
filter_id: crate::filter_pipeline::FILTER_BITSHUFFLE,
|
||||||
|
name: None,
|
||||||
|
flags: 0,
|
||||||
|
client_data: cd,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(feature = "bitshuffle")]
|
||||||
|
#[test]
|
||||||
|
fn filter_round_trips_every_mode() {
|
||||||
|
for es in [1usize, 2, 4, 8] {
|
||||||
|
for n in [0usize, 1, 7, 8, 100, 1000, 5003] {
|
||||||
|
let data: Vec<u8> = (0..n * es)
|
||||||
|
.map(|i| (i % 97) as u8 ^ (i / 300) as u8)
|
||||||
|
.collect();
|
||||||
|
for (comp, block) in [(0, 0), (0, 16), (2, 0), (2, 64), (3, 0), (3, 1024)] {
|
||||||
|
let f = ctx_for(vec![0, 4, es as u32, block, comp]);
|
||||||
|
let ctx = FilterContext {
|
||||||
|
filter: &f,
|
||||||
|
element_size: es,
|
||||||
|
max_output: data.len(),
|
||||||
|
};
|
||||||
|
let enc = bitshuffle_encode(&data, &ctx).unwrap();
|
||||||
|
let dec = bitshuffle_decode(&enc, &ctx).unwrap();
|
||||||
|
assert_eq!(dec, data, "es={es} n={n} comp={comp} block={block}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(feature = "bitshuffle")]
|
||||||
|
#[test]
|
||||||
|
fn rejects_oversized_and_truncated_chunks() {
|
||||||
|
let data = vec![5u8; 4096];
|
||||||
|
let f = ctx_for(vec![0, 4, 4, 0, 2]);
|
||||||
|
let mut ctx = FilterContext {
|
||||||
|
filter: &f,
|
||||||
|
element_size: 4,
|
||||||
|
max_output: data.len(),
|
||||||
|
};
|
||||||
|
let enc = bitshuffle_encode(&data, &ctx).unwrap();
|
||||||
|
assert!(bitshuffle_decode(&enc[..enc.len() - 1], &ctx).is_err());
|
||||||
|
ctx.max_output = 100;
|
||||||
|
assert!(bitshuffle_decode(&enc, &ctx).is_err());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Random and mutated chunks, in every mode, and hostile `cd_values`:
|
||||||
|
/// errors are fine, panics are not.
|
||||||
|
#[cfg(feature = "bitshuffle")]
|
||||||
|
#[test]
|
||||||
|
fn fuzzed_chunks_never_panic() {
|
||||||
|
use crate::test_fuzz::{Rng, fuzz_decoder};
|
||||||
|
let data: Vec<u8> = (0..3001u32)
|
||||||
|
.flat_map(|i| ((i / 7) as u16).to_le_bytes())
|
||||||
|
.collect();
|
||||||
|
for (comp, block) in [(0, 0), (0, 16), (2, 0), (2, 64), (3, 0), (3, 1024)] {
|
||||||
|
let f = ctx_for(vec![0, 4, 2, block, comp]);
|
||||||
|
let ctx = FilterContext {
|
||||||
|
filter: &f,
|
||||||
|
element_size: 2,
|
||||||
|
max_output: data.len(),
|
||||||
|
};
|
||||||
|
let seeds = vec![
|
||||||
|
bitshuffle_encode(&data, &ctx).unwrap(),
|
||||||
|
bitshuffle_encode(&data[..34], &ctx).unwrap(),
|
||||||
|
bitshuffle_encode(&data[..512], &ctx).unwrap(),
|
||||||
|
];
|
||||||
|
fuzz_decoder(
|
||||||
|
0xb5 + comp as u64 * 7 + block as u64,
|
||||||
|
&seeds,
|
||||||
|
4_000,
|
||||||
|
data.len(),
|
||||||
|
|s| bitshuffle_decode(s, &ctx),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
// Hostile filter parameters on a valid chunk.
|
||||||
|
let mut rng = Rng::new(0xcd);
|
||||||
|
let good = ctx_for(vec![0, 4, 2, 0, 2]);
|
||||||
|
let enc = bitshuffle_encode(
|
||||||
|
&data,
|
||||||
|
&FilterContext {
|
||||||
|
filter: &good,
|
||||||
|
element_size: 2,
|
||||||
|
max_output: data.len(),
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
for _ in 0..3_000 {
|
||||||
|
let cd: Vec<u32> = (0..rng.below(7))
|
||||||
|
.map(|_| match rng.below(4) {
|
||||||
|
0 => rng.below(5) as u32,
|
||||||
|
1 => u32::MAX - rng.below(4) as u32,
|
||||||
|
2 => 1 << rng.below(32),
|
||||||
|
_ => rng.next_u64() as u32,
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
let f = ctx_for(cd);
|
||||||
|
let ctx = FilterContext {
|
||||||
|
filter: &f,
|
||||||
|
element_size: 2,
|
||||||
|
max_output: data.len(),
|
||||||
|
};
|
||||||
|
let _ = bitshuffle_decode(&enc, &ctx);
|
||||||
|
let _ = bitshuffle_decode(&data, &ctx);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,711 @@
|
|||||||
|
//! Blosc 1 (HDF5 filter 32001, `hdf5-blosc`, hdf5plugin's `Blosc`), in pure
|
||||||
|
//! Rust: the Blosc 1 frame, its byte shuffle and bit shuffle, and the
|
||||||
|
//! BloscLZ, LZ4/LZ4HC, Snappy, Zlib and Zstandard codecs inside it.
|
||||||
|
//!
|
||||||
|
//! **Frame** (c-blosc 1.x, format version 2). A 16-byte header — version
|
||||||
|
//! (2), codec format version (1), flags, type size, then little-endian `u32`
|
||||||
|
//! decoded size, block size and frame size. Flags: bit 0 byte shuffle, bit
|
||||||
|
//! 1 stored raw ("memcpyed": the data follows the header), bit 2 bit
|
||||||
|
//! shuffle, bit 4 "do not split", bits 5-7 the codec (0 BloscLZ, 1 LZ4 and
|
||||||
|
//! LZ4HC, 2 Snappy, 3 Zlib, 4 Zstandard). Unless stored raw, a table of
|
||||||
|
//! `u32` block offsets follows, one per block of `block size` bytes (the
|
||||||
|
//! last one may be shorter). A block is one stream, or — when the "do not
|
||||||
|
//! split" flag is clear, the type size is at most 16, the block holds at
|
||||||
|
//! least 128 elements, and it is not the short last block — `type size`
|
||||||
|
//! streams, one per byte plane. Each stream is a `u32` length and the
|
||||||
|
//! codec's output; a length equal to the stream's decoded size means the
|
||||||
|
//! bytes are stored raw. The decoded block is then unshuffled (byte shuffle
|
||||||
|
//! for type size > 1; bit shuffle when the block holds a multiple of 8
|
||||||
|
//! elements, the trailing partial element copied as is).
|
||||||
|
//!
|
||||||
|
//! **Filter** (`blosc_filter.c`) `cd_values`: `[0]` filter revision, `[1]`
|
||||||
|
//! Blosc format version, `[2]` type size, `[3]` chunk size in bytes, `[4]`
|
||||||
|
//! compression level, `[5]` shuffle (0 none, 1 byte, 2 bit), `[6]`
|
||||||
|
//! compressor (0 blosclz, 1 lz4, 2 lz4hc, 3 snappy, 4 zlib, 5 zstd). The
|
||||||
|
//! decoder needs only the frame.
|
||||||
|
|
||||||
|
use crate::error::FormatError;
|
||||||
|
use crate::filter_registry::FilterContext;
|
||||||
|
use crate::filters_bitshuffle::{bitshuffle_block, bitunshuffle_block};
|
||||||
|
|
||||||
|
const HEADER: usize = 16;
|
||||||
|
const FLAG_SHUFFLE: u8 = 0x01;
|
||||||
|
const FLAG_MEMCPYED: u8 = 0x02;
|
||||||
|
const FLAG_BITSHUFFLE: u8 = 0x04;
|
||||||
|
const FLAG_FUTURE: u8 = 0x08;
|
||||||
|
const FLAG_DONT_SPLIT: u8 = 0x10;
|
||||||
|
const MAX_SPLITS: usize = 16;
|
||||||
|
const MIN_BUFFERSIZE: usize = 128;
|
||||||
|
|
||||||
|
fn err(msg: &str) -> FormatError {
|
||||||
|
FormatError::DecompressionError(format!("blosc: {msg}"))
|
||||||
|
}
|
||||||
|
|
||||||
|
fn le32(b: &[u8], at: usize) -> Result<usize, FormatError> {
|
||||||
|
b.get(at..at + 4)
|
||||||
|
.map(|s| u32::from_le_bytes(s.try_into().unwrap()) as usize)
|
||||||
|
.ok_or_else(|| err("truncated frame"))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The codec inside a Blosc frame (flags bits 5-7).
|
||||||
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||||
|
pub(crate) enum Codec {
|
||||||
|
BloscLz,
|
||||||
|
Lz4,
|
||||||
|
Snappy,
|
||||||
|
Zlib,
|
||||||
|
Zstd,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Codec {
|
||||||
|
pub(crate) fn from_flags(flags: u8) -> Result<Codec, FormatError> {
|
||||||
|
match flags >> 5 {
|
||||||
|
0 => Ok(Codec::BloscLz),
|
||||||
|
1 => Ok(Codec::Lz4),
|
||||||
|
2 => Ok(Codec::Snappy),
|
||||||
|
3 => Ok(Codec::Zlib),
|
||||||
|
4 => Ok(Codec::Zstd),
|
||||||
|
other => Err(err(&format!("unknown codec {other}"))),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Decode one codec stream into exactly `dst`.
|
||||||
|
pub(crate) fn decode_stream(
|
||||||
|
codec: Codec,
|
||||||
|
src: &[u8],
|
||||||
|
dst: &mut [u8],
|
||||||
|
zstd: &mut Option<ruzstd::decoding::FrameDecoder>,
|
||||||
|
) -> Result<(), FormatError> {
|
||||||
|
let n = match codec {
|
||||||
|
Codec::BloscLz => blosclz_decompress(src, dst),
|
||||||
|
Codec::Lz4 => {
|
||||||
|
lz4_flex::block::decompress_into(src, dst).map_err(|e| err(&format!("lz4: {e}")))?
|
||||||
|
}
|
||||||
|
Codec::Snappy => {
|
||||||
|
let len = snap::raw::decompress_len(src).map_err(|e| err(&format!("snappy: {e}")))?;
|
||||||
|
if len != dst.len() {
|
||||||
|
return Err(err("snappy stream has the wrong size"));
|
||||||
|
}
|
||||||
|
snap::raw::Decoder::new()
|
||||||
|
.decompress(src, dst)
|
||||||
|
.map_err(|e| err(&format!("snappy: {e}")))?
|
||||||
|
}
|
||||||
|
Codec::Zlib => {
|
||||||
|
let out = crate::filters::inflate_bounded(src, dst.len(), dst.len())
|
||||||
|
.map_err(|e| err(&format!("zlib: {e}")))?;
|
||||||
|
let n = out.len();
|
||||||
|
if n == dst.len() {
|
||||||
|
dst.copy_from_slice(&out);
|
||||||
|
}
|
||||||
|
n
|
||||||
|
}
|
||||||
|
Codec::Zstd => crate::filters_bitshuffle::zstd_decode_into(
|
||||||
|
zstd.get_or_insert_with(ruzstd::decoding::FrameDecoder::new),
|
||||||
|
src,
|
||||||
|
dst,
|
||||||
|
)?,
|
||||||
|
};
|
||||||
|
if n != dst.len() {
|
||||||
|
return Err(err("stream decoded to the wrong size"));
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Decode a Blosc-filtered chunk: one Blosc 1 frame.
|
||||||
|
///
|
||||||
|
/// An HDF5 chunk is never empty, so a frame that decodes to nothing where
|
||||||
|
/// the chunk size is known is corrupt (libhdf5's filter fails it too).
|
||||||
|
pub(crate) fn blosc_decode(input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||||
|
let out = blosc_decompress(input, ctx.output_limit())?;
|
||||||
|
if out.is_empty() && ctx.max_output != 0 {
|
||||||
|
return Err(err("empty frame for a non-empty chunk"));
|
||||||
|
}
|
||||||
|
Ok(out)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Decompress a Blosc 1 frame, refusing more than `limit` bytes of output.
|
||||||
|
pub fn blosc_decompress(input: &[u8], limit: usize) -> Result<Vec<u8>, FormatError> {
|
||||||
|
if input.len() < HEADER {
|
||||||
|
return Err(err("truncated header"));
|
||||||
|
}
|
||||||
|
let version = input[0];
|
||||||
|
let codec_version = input[1];
|
||||||
|
let flags = input[2];
|
||||||
|
let typesize = input[3] as usize;
|
||||||
|
let nbytes = le32(input, 4)?;
|
||||||
|
let blocksize = le32(input, 8)?;
|
||||||
|
let cbytes = le32(input, 12)?;
|
||||||
|
if version != 1 && version != 2 {
|
||||||
|
return Err(err(&format!(
|
||||||
|
"frame format version {version} is not Blosc 1 (a Blosc 2 chunk?)"
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
if flags & FLAG_FUTURE != 0 {
|
||||||
|
return Err(err("unknown header flags"));
|
||||||
|
}
|
||||||
|
if nbytes > limit {
|
||||||
|
return Err(err("decoded size exceeds the chunk size"));
|
||||||
|
}
|
||||||
|
if cbytes > input.len() {
|
||||||
|
return Err(err("frame is longer than the chunk"));
|
||||||
|
}
|
||||||
|
if cbytes < HEADER {
|
||||||
|
return Err(err("truncated frame"));
|
||||||
|
}
|
||||||
|
let src = &input[..cbytes];
|
||||||
|
if nbytes == 0 {
|
||||||
|
return Ok(Vec::new());
|
||||||
|
}
|
||||||
|
if blocksize == 0 || typesize == 0 {
|
||||||
|
return Err(err("bad block or type size"));
|
||||||
|
}
|
||||||
|
let mut out = vec![0u8; nbytes];
|
||||||
|
if flags & FLAG_MEMCPYED != 0 {
|
||||||
|
if cbytes != nbytes + HEADER {
|
||||||
|
return Err(err("stored frame has the wrong size"));
|
||||||
|
}
|
||||||
|
out.copy_from_slice(&src[HEADER..]);
|
||||||
|
return Ok(out);
|
||||||
|
}
|
||||||
|
let codec = Codec::from_flags(flags)?;
|
||||||
|
if codec_version != 1 {
|
||||||
|
return Err(err(&format!(
|
||||||
|
"unsupported {codec:?} format version {codec_version}"
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
let nblocks = nbytes.div_ceil(blocksize);
|
||||||
|
let leftover = nbytes % blocksize;
|
||||||
|
if nblocks > (cbytes - HEADER) / 4 {
|
||||||
|
return Err(err("block table is truncated"));
|
||||||
|
}
|
||||||
|
let block_len = blocksize.min(nbytes);
|
||||||
|
let mut tmp = vec![0u8; block_len];
|
||||||
|
let mut zstd = None;
|
||||||
|
let dont_split = flags & FLAG_DONT_SPLIT != 0;
|
||||||
|
for j in 0..nblocks {
|
||||||
|
let is_leftover = j == nblocks - 1 && leftover > 0;
|
||||||
|
let bsize = if is_leftover { leftover } else { blocksize };
|
||||||
|
let nsplits = if !dont_split
|
||||||
|
&& typesize <= MAX_SPLITS
|
||||||
|
&& bsize / typesize >= MIN_BUFFERSIZE
|
||||||
|
&& !is_leftover
|
||||||
|
{
|
||||||
|
typesize
|
||||||
|
} else {
|
||||||
|
1
|
||||||
|
};
|
||||||
|
let neblock = bsize / nsplits;
|
||||||
|
let mut pos = le32(src, HEADER + 4 * j)?;
|
||||||
|
let tmp = &mut tmp[..bsize];
|
||||||
|
for s in 0..nsplits {
|
||||||
|
let clen = src
|
||||||
|
.get(pos..)
|
||||||
|
.and_then(|rest| rest.get(..4))
|
||||||
|
.map(|b| u32::from_le_bytes(b.try_into().unwrap()) as usize)
|
||||||
|
.ok_or_else(|| err("block offset out of range"))?;
|
||||||
|
pos += 4;
|
||||||
|
let stream = src
|
||||||
|
.get(pos..pos.saturating_add(clen))
|
||||||
|
.ok_or_else(|| err("stream runs past the frame"))?;
|
||||||
|
let dst = &mut tmp[s * neblock..(s + 1) * neblock];
|
||||||
|
if clen == neblock {
|
||||||
|
dst.copy_from_slice(stream);
|
||||||
|
} else {
|
||||||
|
decode_stream(codec, stream, dst, &mut zstd)?;
|
||||||
|
}
|
||||||
|
pos += clen;
|
||||||
|
}
|
||||||
|
// `bsize` is a whole number of splits by construction (`nsplits` > 1
|
||||||
|
// only for full blocks, and c-blosc sizes those in whole elements).
|
||||||
|
if nsplits * neblock != bsize {
|
||||||
|
return Err(err("block is not a whole number of streams"));
|
||||||
|
}
|
||||||
|
let dest = &mut out[j * blocksize..j * blocksize + bsize];
|
||||||
|
unshuffle_block(flags, typesize, tmp, dest);
|
||||||
|
}
|
||||||
|
Ok(out)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Undo the frame's shuffle on one decoded block.
|
||||||
|
fn unshuffle_block(flags: u8, typesize: usize, src: &[u8], dest: &mut [u8]) {
|
||||||
|
let bsize = src.len();
|
||||||
|
if flags & FLAG_SHUFFLE != 0 && typesize > 1 {
|
||||||
|
let n = bsize / typesize;
|
||||||
|
for i in 0..n {
|
||||||
|
for b in 0..typesize {
|
||||||
|
dest[i * typesize + b] = src[b * n + i];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
dest[n * typesize..].copy_from_slice(&src[n * typesize..]);
|
||||||
|
} else if flags & FLAG_BITSHUFFLE != 0 && bsize >= typesize {
|
||||||
|
let n = bsize / typesize;
|
||||||
|
if n.is_multiple_of(8) {
|
||||||
|
let body = n * typesize;
|
||||||
|
bitunshuffle_block(&src[..body], &mut dest[..body], n, typesize);
|
||||||
|
dest[body..].copy_from_slice(&src[body..]);
|
||||||
|
} else {
|
||||||
|
dest.copy_from_slice(src);
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
dest.copy_from_slice(src);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// BloscLZ decompression (c-blosc 1.21 `blosclz_decompress`): returns the
|
||||||
|
/// number of bytes written, or 0 on malformed input — exactly as the C
|
||||||
|
/// decoder, including stopping before a match that ends the stream, so a
|
||||||
|
/// stream libblosc rejects is rejected here too.
|
||||||
|
///
|
||||||
|
/// Instructions: a control byte `ctrl`. Below 32, a literal run of
|
||||||
|
/// `ctrl + 1` bytes. Otherwise a match: length `(ctrl >> 5) + 2`, extended
|
||||||
|
/// by following bytes while they are 255 when the top three bits are all
|
||||||
|
/// set; distance `((ctrl & 31) << 8) + next byte + 1`, or — when that byte
|
||||||
|
/// is 255 and the high bits are 31 — a 16-bit big-endian distance plus 8192.
|
||||||
|
/// The first instruction is always a literal.
|
||||||
|
pub(crate) fn blosclz_decompress(input: &[u8], out: &mut [u8]) -> usize {
|
||||||
|
const MAX_DISTANCE: usize = 8191;
|
||||||
|
let limit = input.len();
|
||||||
|
if limit == 0 {
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
let mut ip = 1usize;
|
||||||
|
let mut op = 0usize;
|
||||||
|
let mut ctrl = (input[0] & 31) as usize;
|
||||||
|
loop {
|
||||||
|
if ctrl >= 32 {
|
||||||
|
let mut len = (ctrl >> 5) - 1;
|
||||||
|
let ofs = (ctrl & 31) << 8;
|
||||||
|
if len == 6 {
|
||||||
|
loop {
|
||||||
|
if ip + 1 >= limit {
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
let code = input[ip] as usize;
|
||||||
|
ip += 1;
|
||||||
|
len += code;
|
||||||
|
if code != 255 {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
} else if ip + 1 >= limit {
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
let code = input[ip] as usize;
|
||||||
|
ip += 1;
|
||||||
|
len += 3;
|
||||||
|
// The copy source is `distance` bytes back.
|
||||||
|
let mut distance = ofs + code + 1;
|
||||||
|
if code == 255 && ofs == 31 << 8 {
|
||||||
|
if ip + 1 >= limit {
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
let far = ((input[ip] as usize) << 8) + input[ip + 1] as usize;
|
||||||
|
ip += 2;
|
||||||
|
distance = far + MAX_DISTANCE + 1;
|
||||||
|
}
|
||||||
|
if op + len > out.len() {
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
if distance > op {
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
if ip >= limit {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
ctrl = input[ip] as usize;
|
||||||
|
ip += 1;
|
||||||
|
let start = op - distance;
|
||||||
|
if distance >= len {
|
||||||
|
out.copy_within(start..start + len, op);
|
||||||
|
} else {
|
||||||
|
for k in 0..len {
|
||||||
|
out[op + k] = out[start + k];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
op += len;
|
||||||
|
} else {
|
||||||
|
let run = ctrl + 1;
|
||||||
|
if op + run > out.len() || ip + run > limit {
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
out[op..op + run].copy_from_slice(&input[ip..ip + run]);
|
||||||
|
op += run;
|
||||||
|
ip += run;
|
||||||
|
if ip >= limit {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
ctrl = input[ip] as usize;
|
||||||
|
ip += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
op
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The codec our encoder puts inside the frame.
|
||||||
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||||
|
pub(crate) enum EncodeCodec {
|
||||||
|
Lz4,
|
||||||
|
Snappy,
|
||||||
|
Zlib,
|
||||||
|
Zstd,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl EncodeCodec {
|
||||||
|
/// From the filter's `cd_values[6]` compressor code.
|
||||||
|
fn from_cd(code: u32) -> Result<EncodeCodec, FormatError> {
|
||||||
|
match code {
|
||||||
|
1 | 2 => Ok(EncodeCodec::Lz4),
|
||||||
|
3 => Ok(EncodeCodec::Snappy),
|
||||||
|
4 => Ok(EncodeCodec::Zlib),
|
||||||
|
5 => Ok(EncodeCodec::Zstd),
|
||||||
|
0 => Err(FormatError::CompressionError(
|
||||||
|
"blosc: clawhdf5 cannot write BloscLZ; choose lz4, snappy, zlib or zstd".into(),
|
||||||
|
)),
|
||||||
|
other => Err(FormatError::CompressionError(format!(
|
||||||
|
"blosc: unknown compressor {other}"
|
||||||
|
))),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn flags(self) -> u8 {
|
||||||
|
(match self {
|
||||||
|
EncodeCodec::Lz4 => 1,
|
||||||
|
EncodeCodec::Snappy => 2,
|
||||||
|
EncodeCodec::Zlib => 3,
|
||||||
|
EncodeCodec::Zstd => 4,
|
||||||
|
}) << 5
|
||||||
|
}
|
||||||
|
|
||||||
|
fn encode(self, data: &[u8], level: u32) -> Result<Vec<u8>, FormatError> {
|
||||||
|
match self {
|
||||||
|
EncodeCodec::Lz4 => Ok(lz4_flex::block::compress(data)),
|
||||||
|
EncodeCodec::Snappy => snap::raw::Encoder::new()
|
||||||
|
.compress_vec(data)
|
||||||
|
.map_err(|e| FormatError::CompressionError(format!("blosc: snappy: {e}"))),
|
||||||
|
EncodeCodec::Zlib => crate::filters::deflate_bounded(data, level.min(9))
|
||||||
|
.map_err(|e| FormatError::CompressionError(format!("blosc: zlib: {e}"))),
|
||||||
|
EncodeCodec::Zstd => Ok(crate::filters_bitshuffle::zstd_encode(data)),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Block size our encoder uses: at most 256 KiB, a whole number of
|
||||||
|
/// elements (and, for bit shuffle, of 8-element groups).
|
||||||
|
fn encode_block_size(nbytes: usize, typesize: usize, bitshuffle: bool) -> usize {
|
||||||
|
let unit = if bitshuffle { 8 * typesize } else { typesize };
|
||||||
|
let target = (256 * 1024).min(nbytes);
|
||||||
|
if target < unit {
|
||||||
|
return nbytes.max(1);
|
||||||
|
}
|
||||||
|
target / unit * unit
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Encode a chunk as one Blosc 1 frame. `cd_values` as hdf5-blosc:
|
||||||
|
/// `[2]` type size, `[4]` level (0 = store), `[5]` shuffle, `[6]` codec.
|
||||||
|
pub(crate) fn blosc_encode(input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||||
|
let cd = ctx.client_data();
|
||||||
|
let cerr = |m: &str| FormatError::CompressionError(format!("blosc: {m}"));
|
||||||
|
let typesize = match cd.get(2) {
|
||||||
|
Some(&t) if t != 0 => t as usize,
|
||||||
|
_ => ctx.element_size.max(1),
|
||||||
|
};
|
||||||
|
// Blosc records the type size in one byte; c-blosc treats larger types
|
||||||
|
// as bytes.
|
||||||
|
let typesize = if typesize > 255 { 1 } else { typesize };
|
||||||
|
let level = cd.get(4).copied().unwrap_or(5);
|
||||||
|
let shuffle = cd.get(5).copied().unwrap_or(1);
|
||||||
|
let codec = EncodeCodec::from_cd(cd.get(6).copied().unwrap_or(1))?;
|
||||||
|
let nbytes = input.len();
|
||||||
|
if nbytes > i32::MAX as usize - HEADER {
|
||||||
|
return Err(cerr("chunk too large for a Blosc frame"));
|
||||||
|
}
|
||||||
|
let mut flags = codec.flags();
|
||||||
|
match shuffle {
|
||||||
|
0 => {}
|
||||||
|
1 => flags |= FLAG_SHUFFLE,
|
||||||
|
2 => flags |= FLAG_BITSHUFFLE,
|
||||||
|
other => return Err(cerr(&format!("unknown shuffle mode {other}"))),
|
||||||
|
}
|
||||||
|
let blocksize = encode_block_size(nbytes, typesize, shuffle == 2);
|
||||||
|
let header = |flags: u8, blocksize: usize, cbytes: usize| {
|
||||||
|
let mut h = Vec::with_capacity(HEADER);
|
||||||
|
h.extend_from_slice(&[2, 1, flags, typesize as u8]);
|
||||||
|
h.extend_from_slice(&(nbytes as u32).to_le_bytes());
|
||||||
|
h.extend_from_slice(&(blocksize as u32).to_le_bytes());
|
||||||
|
h.extend_from_slice(&(cbytes as u32).to_le_bytes());
|
||||||
|
h
|
||||||
|
};
|
||||||
|
let stored = || {
|
||||||
|
let mut out = header(
|
||||||
|
FLAG_MEMCPYED | (flags & !(FLAG_SHUFFLE | FLAG_BITSHUFFLE)),
|
||||||
|
blocksize,
|
||||||
|
nbytes + HEADER,
|
||||||
|
);
|
||||||
|
out.extend_from_slice(input);
|
||||||
|
out
|
||||||
|
};
|
||||||
|
if level == 0 || nbytes == 0 {
|
||||||
|
return Ok(stored());
|
||||||
|
}
|
||||||
|
|
||||||
|
let nblocks = nbytes.div_ceil(blocksize);
|
||||||
|
let leftover = nbytes % blocksize;
|
||||||
|
let mut body = Vec::with_capacity(nbytes / 2);
|
||||||
|
let mut starts = Vec::with_capacity(nblocks);
|
||||||
|
let table_end = HEADER + 4 * nblocks;
|
||||||
|
let mut shuffled = vec![0u8; blocksize];
|
||||||
|
for j in 0..nblocks {
|
||||||
|
let is_leftover = j == nblocks - 1 && leftover > 0;
|
||||||
|
let bsize = if is_leftover { leftover } else { blocksize };
|
||||||
|
let block = &input[j * blocksize..j * blocksize + bsize];
|
||||||
|
let sh = &mut shuffled[..bsize];
|
||||||
|
shuffle_block(flags, typesize, block, sh);
|
||||||
|
starts.push(table_end + body.len());
|
||||||
|
let nsplits = if typesize <= MAX_SPLITS
|
||||||
|
&& bsize / typesize >= MIN_BUFFERSIZE
|
||||||
|
&& !is_leftover
|
||||||
|
&& bsize.is_multiple_of(typesize)
|
||||||
|
{
|
||||||
|
typesize
|
||||||
|
} else {
|
||||||
|
1
|
||||||
|
};
|
||||||
|
let neblock = bsize / nsplits;
|
||||||
|
for s in 0..nsplits {
|
||||||
|
let part = &sh[s * neblock..(s + 1) * neblock];
|
||||||
|
let comp = codec.encode(part, level)?;
|
||||||
|
if comp.len() < neblock {
|
||||||
|
body.extend_from_slice(&(comp.len() as u32).to_le_bytes());
|
||||||
|
body.extend_from_slice(&comp);
|
||||||
|
} else {
|
||||||
|
body.extend_from_slice(&(neblock as u32).to_le_bytes());
|
||||||
|
body.extend_from_slice(part);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if table_end + body.len() >= nbytes + HEADER {
|
||||||
|
// Incompressible: store instead, as c-blosc does.
|
||||||
|
return Ok(stored());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// A split block must decode as split: the decoder infers splitting from
|
||||||
|
// the same rule, which requires a whole number of elements per block.
|
||||||
|
let cbytes = table_end + body.len();
|
||||||
|
let mut out = header(flags, blocksize, cbytes);
|
||||||
|
for s in starts {
|
||||||
|
out.extend_from_slice(&(s as u32).to_le_bytes());
|
||||||
|
}
|
||||||
|
out.extend_from_slice(&body);
|
||||||
|
Ok(out)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Apply the frame's shuffle to one block (the inverse of
|
||||||
|
/// [`unshuffle_block`]).
|
||||||
|
fn shuffle_block(flags: u8, typesize: usize, src: &[u8], dest: &mut [u8]) {
|
||||||
|
let bsize = src.len();
|
||||||
|
if flags & FLAG_SHUFFLE != 0 && typesize > 1 {
|
||||||
|
let n = bsize / typesize;
|
||||||
|
for i in 0..n {
|
||||||
|
for b in 0..typesize {
|
||||||
|
dest[b * n + i] = src[i * typesize + b];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
dest[n * typesize..].copy_from_slice(&src[n * typesize..]);
|
||||||
|
} else if flags & FLAG_BITSHUFFLE != 0 && bsize >= typesize {
|
||||||
|
let n = bsize / typesize;
|
||||||
|
if n.is_multiple_of(8) {
|
||||||
|
let body = n * typesize;
|
||||||
|
bitshuffle_block(&src[..body], &mut dest[..body], n, typesize);
|
||||||
|
dest[body..].copy_from_slice(&src[body..]);
|
||||||
|
} else {
|
||||||
|
dest.copy_from_slice(src);
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
dest.copy_from_slice(src);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
use crate::filter_pipeline::{FILTER_BLOSC, FilterDescription};
|
||||||
|
|
||||||
|
/// A blosclz stream: literal "abc", then a 9-byte match 3 back (a run
|
||||||
|
/// of "abc"), then literal "Z".
|
||||||
|
#[test]
|
||||||
|
fn blosclz_decodes_literals_and_overlapping_matches() {
|
||||||
|
// Match: length (ctrl >> 5) + 2 = 8, distance ofs + code + 1 = 3.
|
||||||
|
let stream = [2, b'a', b'b', b'c', (6 << 5), 2, 0, b'Z'];
|
||||||
|
let mut out = [0u8; 12];
|
||||||
|
assert_eq!(blosclz_decompress(&stream, &mut out), 12);
|
||||||
|
assert_eq!(&out, b"abcabcabcabZ");
|
||||||
|
// A stream cut inside a match is malformed.
|
||||||
|
let mut out = [0u8; 11];
|
||||||
|
assert_eq!(blosclz_decompress(&stream[..6], &mut out), 0);
|
||||||
|
// A match before the start of the output is malformed.
|
||||||
|
assert_eq!(blosclz_decompress(&[0, b'a', 32, 5, 0, b'x'], &mut out), 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn desc(cd: Vec<u32>) -> FilterDescription {
|
||||||
|
FilterDescription {
|
||||||
|
filter_id: FILTER_BLOSC,
|
||||||
|
name: None,
|
||||||
|
flags: 0,
|
||||||
|
client_data: cd,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn frame_round_trips_every_codec_and_shuffle() {
|
||||||
|
for ts in [1usize, 2, 4, 8, 3, 32] {
|
||||||
|
for n in [0usize, 5, 100, 1000, 70_000, 300_001] {
|
||||||
|
if n * ts > 1 << 20 && ts > 1 {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let data: Vec<u8> = (0..n * ts)
|
||||||
|
.map(|i| ((i / ts) % 200) as u8 ^ (i % ts) as u8)
|
||||||
|
.collect();
|
||||||
|
for codec in [1u32, 3, 4, 5] {
|
||||||
|
for shuffle in [0u32, 1, 2] {
|
||||||
|
for level in [0u32, 5] {
|
||||||
|
let f = desc(vec![2, 2, ts as u32, 0, level, shuffle, codec]);
|
||||||
|
let ctx = FilterContext {
|
||||||
|
filter: &f,
|
||||||
|
element_size: ts,
|
||||||
|
max_output: data.len(),
|
||||||
|
};
|
||||||
|
let enc = blosc_encode(&data, &ctx).unwrap();
|
||||||
|
let dec = blosc_decode(&enc, &ctx).unwrap_or_else(|e| {
|
||||||
|
panic!("ts={ts} n={n} codec={codec} shuffle={shuffle}: {e}")
|
||||||
|
});
|
||||||
|
assert!(
|
||||||
|
dec == data,
|
||||||
|
"ts={ts} n={n} codec={codec} shuffle={shuffle} level={level}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn rejects_bad_frames() {
|
||||||
|
let data = vec![9u8; 50_000];
|
||||||
|
let f = desc(vec![2, 2, 4, 0, 5, 1, 1]);
|
||||||
|
let ctx = FilterContext {
|
||||||
|
filter: &f,
|
||||||
|
element_size: 4,
|
||||||
|
max_output: data.len(),
|
||||||
|
};
|
||||||
|
let enc = blosc_encode(&data, &ctx).unwrap();
|
||||||
|
assert!(blosc_decode(&enc[..enc.len() - 3], &ctx).is_err());
|
||||||
|
let small = FilterContext {
|
||||||
|
max_output: 49_999,
|
||||||
|
..ctx
|
||||||
|
};
|
||||||
|
assert!(blosc_decode(&enc, &small).is_err());
|
||||||
|
let mut v3 = enc.clone();
|
||||||
|
v3[0] = 3;
|
||||||
|
assert!(blosc_decode(&v3, &ctx).is_err());
|
||||||
|
let f0 = desc(vec![2, 2, 4, 0, 5, 1, 0]);
|
||||||
|
let ctx0 = FilterContext { filter: &f0, ..ctx };
|
||||||
|
assert!(blosc_encode(&data, &ctx0).is_err());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A frame that declares no data, for a chunk that has some.
|
||||||
|
#[test]
|
||||||
|
fn empty_frame_for_a_non_empty_chunk_is_an_error() {
|
||||||
|
let mut frame = vec![2u8, 1, 0x20, 4];
|
||||||
|
for v in [0u32, 64, 16] {
|
||||||
|
frame.extend_from_slice(&v.to_le_bytes());
|
||||||
|
}
|
||||||
|
assert_eq!(blosc_decompress(&frame, 64).unwrap(), b"");
|
||||||
|
let f = desc(vec![2, 2, 4, 64, 5, 1, 1]);
|
||||||
|
let ctx = FilterContext {
|
||||||
|
filter: &f,
|
||||||
|
element_size: 4,
|
||||||
|
max_output: 64,
|
||||||
|
};
|
||||||
|
assert!(blosc_decode(&frame, &ctx).is_err());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A frame whose header claims a compressed size smaller than the
|
||||||
|
/// header itself, not stored raw: an error, not an arithmetic overflow
|
||||||
|
/// (it panicked in debug builds).
|
||||||
|
#[test]
|
||||||
|
fn frame_size_below_the_header_is_an_error() {
|
||||||
|
let mut frame = vec![2u8, 1, 1 << 5, 4];
|
||||||
|
for v in [64u32, 64, 8] {
|
||||||
|
frame.extend_from_slice(&v.to_le_bytes());
|
||||||
|
}
|
||||||
|
frame.extend_from_slice(&[0; 40]);
|
||||||
|
assert!(blosc_decompress(&frame, 1000).is_err());
|
||||||
|
for cbytes in 0..16u32 {
|
||||||
|
frame[12..16].copy_from_slice(&cbytes.to_le_bytes());
|
||||||
|
assert!(blosc_decompress(&frame, 1000).is_err(), "cbytes={cbytes}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A BloscLZ frame (our encoder cannot write one): a single block,
|
||||||
|
/// one stream, no shuffle.
|
||||||
|
fn blosclz_frame() -> Vec<u8> {
|
||||||
|
let stream = [2, b'a', b'b', b'c', (6 << 5), 2, 0, b'Z'];
|
||||||
|
let mut f = vec![2u8, 1, 0, 1];
|
||||||
|
for v in [12u32, 12, (HEADER + 4 + 4 + stream.len()) as u32] {
|
||||||
|
f.extend_from_slice(&v.to_le_bytes());
|
||||||
|
}
|
||||||
|
f.extend_from_slice(&((HEADER + 4) as u32).to_le_bytes());
|
||||||
|
f.extend_from_slice(&(stream.len() as u32).to_le_bytes());
|
||||||
|
f.extend_from_slice(&stream);
|
||||||
|
f
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Random and mutated frames, every codec and shuffle: errors are fine,
|
||||||
|
/// panics are not.
|
||||||
|
#[test]
|
||||||
|
fn fuzzed_frames_never_panic() {
|
||||||
|
let limit = 6000;
|
||||||
|
let data: Vec<u8> = (0..1500u32).flat_map(|i| (i / 5).to_le_bytes()).collect();
|
||||||
|
let mut seeds = vec![blosclz_frame()];
|
||||||
|
for codec in [1u32, 3, 4, 5] {
|
||||||
|
for shuffle in [0u32, 1, 2] {
|
||||||
|
for (ts, n) in [(4usize, data.len()), (4, 520), (1, 300), (2, 4)] {
|
||||||
|
let f = desc(vec![2, 2, ts as u32, 0, 5, shuffle, codec]);
|
||||||
|
let ctx = FilterContext {
|
||||||
|
filter: &f,
|
||||||
|
element_size: ts,
|
||||||
|
max_output: n,
|
||||||
|
};
|
||||||
|
seeds.push(blosc_encode(&data[..n], &ctx).unwrap());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Stored raw.
|
||||||
|
let f = desc(vec![2, 2, 4, 0, 0, 1, 1]);
|
||||||
|
let ctx = FilterContext {
|
||||||
|
filter: &f,
|
||||||
|
element_size: 4,
|
||||||
|
max_output: 64,
|
||||||
|
};
|
||||||
|
seeds.push(blosc_encode(&data[..64], &ctx).unwrap());
|
||||||
|
crate::test_fuzz::fuzz_decoder(0xb10, &seeds, 30_000, limit, |s| {
|
||||||
|
blosc_decompress(s, limit)
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// BloscLZ streams on their own, random and mutated.
|
||||||
|
#[test]
|
||||||
|
fn fuzzed_blosclz_streams_never_panic() {
|
||||||
|
let seed = blosclz_frame()[HEADER + 8..].to_vec();
|
||||||
|
let mut out = [0u8; 64];
|
||||||
|
crate::test_fuzz::fuzz_decoder(0xb11, &[seed], 30_000, 64, |s| {
|
||||||
|
let n = blosclz_decompress(s, &mut out);
|
||||||
|
if n == 0 {
|
||||||
|
Err(err("malformed"))
|
||||||
|
} else {
|
||||||
|
Ok(out[..n].to_vec())
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,133 @@
|
|||||||
|
//! bzip2 (HDF5 filter 307, PyTables' `H5Zbzip2.c`, hdf5plugin's `BZip2`).
|
||||||
|
//!
|
||||||
|
//! The chunk is one bzip2 stream; `cd_values[0]` is the block size (1-9,
|
||||||
|
//! the compression level). Decoded with the `bzip2` crate's default backend,
|
||||||
|
//! `libbz2-rs-sys`, a pure-Rust port of libbzip2.
|
||||||
|
|
||||||
|
use crate::addr::saturating_usize;
|
||||||
|
use crate::error::FormatError;
|
||||||
|
use crate::filter_registry::FilterContext;
|
||||||
|
|
||||||
|
fn err(msg: &str) -> FormatError {
|
||||||
|
FormatError::DecompressionError(format!("bzip2: {msg}"))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Decode a bzip2-filtered chunk, refusing output beyond the chunk size.
|
||||||
|
pub(crate) fn bzip2_decode(input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||||
|
use bzip2::{Decompress, Status};
|
||||||
|
let limit = ctx.output_limit();
|
||||||
|
let max_capacity = limit.saturating_add(1);
|
||||||
|
let hint = if ctx.max_output != 0 {
|
||||||
|
ctx.max_output
|
||||||
|
} else {
|
||||||
|
input.len().saturating_mul(4)
|
||||||
|
};
|
||||||
|
let mut out = Vec::new();
|
||||||
|
out.try_reserve_exact(hint.clamp(1, max_capacity))
|
||||||
|
.map_err(|_| err("cannot allocate the output buffer"))?;
|
||||||
|
let mut dec = Decompress::new(false);
|
||||||
|
loop {
|
||||||
|
let (in_before, out_before) = (dec.total_in(), dec.total_out());
|
||||||
|
let status = dec
|
||||||
|
.decompress_vec(&input[saturating_usize(in_before)..], &mut out)
|
||||||
|
.map_err(|e| err(&e.to_string()))?;
|
||||||
|
if out.len() > limit {
|
||||||
|
return Err(err("output exceeds the chunk size"));
|
||||||
|
}
|
||||||
|
if status == Status::StreamEnd {
|
||||||
|
return Ok(out);
|
||||||
|
}
|
||||||
|
if out.len() == out.capacity() {
|
||||||
|
let grow = out
|
||||||
|
.capacity()
|
||||||
|
.min(max_capacity.saturating_sub(out.capacity()))
|
||||||
|
.max(1);
|
||||||
|
out.try_reserve_exact(grow)
|
||||||
|
.map_err(|_| err("cannot allocate the output buffer"))?;
|
||||||
|
} else if saturating_usize(dec.total_in()) >= input.len()
|
||||||
|
|| (dec.total_in(), dec.total_out()) == (in_before, out_before)
|
||||||
|
{
|
||||||
|
return Err(err("truncated stream"));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Encode a chunk as one bzip2 stream at block size `cd_values[0]`
|
||||||
|
/// (default 9, as hdf5plugin).
|
||||||
|
pub(crate) fn bzip2_encode(input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||||
|
use bzip2::{Action, Compress, Compression, Status};
|
||||||
|
let level = ctx.client_data().first().copied().unwrap_or(9).clamp(1, 9);
|
||||||
|
let cerr = |m: String| FormatError::CompressionError(format!("bzip2: {m}"));
|
||||||
|
let mut enc = Compress::new(Compression::new(level), 0);
|
||||||
|
// bzip2's worst case is about 1% + 600 bytes over the input.
|
||||||
|
let mut out = Vec::with_capacity(input.len() + input.len() / 100 + 600);
|
||||||
|
loop {
|
||||||
|
let consumed = saturating_usize(enc.total_in());
|
||||||
|
let status = enc
|
||||||
|
.compress_vec(&input[consumed..], &mut out, Action::Finish)
|
||||||
|
.map_err(|e| cerr(e.to_string()))?;
|
||||||
|
if status == Status::StreamEnd {
|
||||||
|
return Ok(out);
|
||||||
|
}
|
||||||
|
if out.len() == out.capacity() {
|
||||||
|
out.reserve(out.capacity().max(4096));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
use crate::filter_pipeline::{FILTER_BZIP2, FilterDescription};
|
||||||
|
|
||||||
|
fn desc(level: u32) -> FilterDescription {
|
||||||
|
FilterDescription {
|
||||||
|
filter_id: FILTER_BZIP2,
|
||||||
|
name: None,
|
||||||
|
flags: 0,
|
||||||
|
client_data: vec![level],
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn round_trips_and_bounds() {
|
||||||
|
let data: Vec<u8> = (0..100_000u32)
|
||||||
|
.flat_map(|i| (i % 777).to_le_bytes())
|
||||||
|
.collect();
|
||||||
|
for level in [1, 5, 9] {
|
||||||
|
let f = desc(level);
|
||||||
|
let ctx = FilterContext {
|
||||||
|
filter: &f,
|
||||||
|
element_size: 4,
|
||||||
|
max_output: data.len(),
|
||||||
|
};
|
||||||
|
let enc = bzip2_encode(&data, &ctx).unwrap();
|
||||||
|
assert!(enc.len() < data.len() / 4);
|
||||||
|
assert_eq!(bzip2_decode(&enc, &ctx).unwrap(), data);
|
||||||
|
// Truncated, and larger than the chunk: errors, not data.
|
||||||
|
assert!(bzip2_decode(&enc[..enc.len() / 2], &ctx).is_err());
|
||||||
|
let small = FilterContext {
|
||||||
|
max_output: data.len() - 1,
|
||||||
|
..ctx
|
||||||
|
};
|
||||||
|
assert!(bzip2_decode(&enc, &small).is_err());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Random and mutated streams: errors are fine, panics are not.
|
||||||
|
#[test]
|
||||||
|
fn fuzzed_streams_never_panic() {
|
||||||
|
let f = desc(9);
|
||||||
|
let data: Vec<u8> = (0..4000u32).flat_map(|i| (i % 91).to_le_bytes()).collect();
|
||||||
|
let ctx = FilterContext {
|
||||||
|
filter: &f,
|
||||||
|
element_size: 4,
|
||||||
|
max_output: data.len(),
|
||||||
|
};
|
||||||
|
let seeds = vec![
|
||||||
|
bzip2_encode(&data, &ctx).unwrap(),
|
||||||
|
bzip2_encode(&data[..40], &ctx).unwrap(),
|
||||||
|
];
|
||||||
|
crate::test_fuzz::fuzz_decoder(0xb2, &seeds, 3_000, data.len(), |s| bzip2_decode(s, &ctx));
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,260 @@
|
|||||||
|
//! LZF (HDF5 filter 32000) — h5py's built-in compression filter
|
||||||
|
//! (`compression="lzf"`), in pure Rust.
|
||||||
|
//!
|
||||||
|
//! The chunk is one raw LZF stream (liblzf 3.x format, no header). The
|
||||||
|
//! stream is a sequence of instructions, each starting with a control byte:
|
||||||
|
//!
|
||||||
|
//! * `000LLLLL` — a literal run: the next `L + 1` bytes (1..=32) are copied.
|
||||||
|
//! * `LLLOOOOO [E] OOOOOOOO` — a back reference: copy `len + 2` bytes from
|
||||||
|
//! `distance` bytes back, where `len` is the top three bits (1..=6), or
|
||||||
|
//! `7 + E` when they are all ones, and `distance` is the 13-bit offset
|
||||||
|
//! (high five bits in the control byte, low eight in the last byte) plus 1.
|
||||||
|
//!
|
||||||
|
//! h5py's filter (`lzf_filter.c`) records the chunk's size in bytes in
|
||||||
|
//! `cd_values[2]` (slots 0 and 1 hold the filter and liblzf versions) and
|
||||||
|
//! sizes its output buffer from it.
|
||||||
|
|
||||||
|
#[cfg(not(feature = "std"))]
|
||||||
|
extern crate alloc;
|
||||||
|
#[cfg(not(feature = "std"))]
|
||||||
|
use alloc::{format, vec, vec::Vec};
|
||||||
|
|
||||||
|
use crate::error::FormatError;
|
||||||
|
use crate::filter_registry::FilterContext;
|
||||||
|
|
||||||
|
/// `H5PY_FILTER_LZF_VERSION`, written to `cd_values[0]`.
|
||||||
|
pub const LZF_FILTER_VERSION: u32 = 4;
|
||||||
|
/// `LZF_VERSION` (liblzf 1.5), written to `cd_values[1]`.
|
||||||
|
pub const LZF_API_VERSION: u32 = 0x0105;
|
||||||
|
|
||||||
|
const MAX_LITERAL: usize = 32;
|
||||||
|
const MAX_OFFSET: usize = 1 << 13;
|
||||||
|
const MAX_REF: usize = (1 << 8) + (1 << 3);
|
||||||
|
const HASH_LOG: u32 = 14;
|
||||||
|
|
||||||
|
fn err(msg: &str) -> FormatError {
|
||||||
|
FormatError::DecompressionError(format!("lzf: {msg}"))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Decode an LZF-filtered chunk.
|
||||||
|
pub(crate) fn lzf_decode(input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||||
|
let limit = ctx.output_limit();
|
||||||
|
let hint = match ctx.client_data().get(2) {
|
||||||
|
Some(&n) if n != 0 => n as usize,
|
||||||
|
_ => input.len().saturating_mul(2),
|
||||||
|
};
|
||||||
|
lzf_decompress(input, hint.min(limit), limit)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Decompress a raw LZF stream, refusing to produce more than `limit` bytes.
|
||||||
|
pub fn lzf_decompress(
|
||||||
|
input: &[u8],
|
||||||
|
size_hint: usize,
|
||||||
|
limit: usize,
|
||||||
|
) -> Result<Vec<u8>, FormatError> {
|
||||||
|
let mut out: Vec<u8> = Vec::new();
|
||||||
|
out.try_reserve(size_hint)
|
||||||
|
.map_err(|_| err("cannot allocate the output buffer"))?;
|
||||||
|
let mut ip = 0usize;
|
||||||
|
while ip < input.len() {
|
||||||
|
let ctrl = input[ip] as usize;
|
||||||
|
ip += 1;
|
||||||
|
if ctrl < 32 {
|
||||||
|
let run = ctrl + 1;
|
||||||
|
let lit = input
|
||||||
|
.get(ip..ip + run)
|
||||||
|
.ok_or_else(|| err("literal run past the end of the input"))?;
|
||||||
|
if out.len() + run > limit {
|
||||||
|
return Err(err("output exceeds the chunk size"));
|
||||||
|
}
|
||||||
|
out.extend_from_slice(lit);
|
||||||
|
ip += run;
|
||||||
|
} else {
|
||||||
|
let mut len = ctrl >> 5;
|
||||||
|
if len == 7 {
|
||||||
|
len += *input
|
||||||
|
.get(ip)
|
||||||
|
.ok_or_else(|| err("truncated back reference"))?
|
||||||
|
as usize;
|
||||||
|
ip += 1;
|
||||||
|
}
|
||||||
|
let low = *input
|
||||||
|
.get(ip)
|
||||||
|
.ok_or_else(|| err("truncated back reference"))? as usize;
|
||||||
|
ip += 1;
|
||||||
|
let distance = ((ctrl & 0x1f) << 8) + low + 1;
|
||||||
|
let len = len + 2;
|
||||||
|
if distance > out.len() {
|
||||||
|
return Err(err("back reference before the start of the output"));
|
||||||
|
}
|
||||||
|
if out.len() + len > limit {
|
||||||
|
return Err(err("output exceeds the chunk size"));
|
||||||
|
}
|
||||||
|
let start = out.len() - distance;
|
||||||
|
if distance >= len {
|
||||||
|
out.extend_from_within(start..start + len);
|
||||||
|
} else {
|
||||||
|
// Overlapping copy: repeats the last `distance` bytes.
|
||||||
|
for k in 0..len {
|
||||||
|
let b = out[start + k];
|
||||||
|
out.push(b);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(out)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Encode a chunk with the LZF filter.
|
||||||
|
pub(crate) fn lzf_encode(input: &[u8], _ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||||
|
Ok(lzf_compress(input))
|
||||||
|
}
|
||||||
|
|
||||||
|
fn hash3(b: &[u8]) -> usize {
|
||||||
|
let v = (u32::from(b[0]) << 16) | (u32::from(b[1]) << 8) | u32::from(b[2]);
|
||||||
|
(v.wrapping_mul(2_654_435_761) >> (32 - HASH_LOG)) as usize
|
||||||
|
}
|
||||||
|
|
||||||
|
fn flush_literals(out: &mut Vec<u8>, lit: &[u8]) {
|
||||||
|
for run in lit.chunks(MAX_LITERAL) {
|
||||||
|
out.push((run.len() - 1) as u8);
|
||||||
|
out.extend_from_slice(run);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Compress `input` into a raw LZF stream any liblzf decoder reads.
|
||||||
|
///
|
||||||
|
/// Incompressible input grows by one byte per 32. (h5py's own filter gives
|
||||||
|
/// up on such a chunk and stores it unfiltered; storing the slightly larger
|
||||||
|
/// stream is equally readable.)
|
||||||
|
pub fn lzf_compress(input: &[u8]) -> Vec<u8> {
|
||||||
|
let n = input.len();
|
||||||
|
let mut out = Vec::with_capacity(n + n / MAX_LITERAL + 1);
|
||||||
|
let mut table = vec![0u32; 1 << HASH_LOG];
|
||||||
|
let mut lit_start = 0usize;
|
||||||
|
let mut i = 0usize;
|
||||||
|
while i + 2 < n {
|
||||||
|
let h = hash3(&input[i..]);
|
||||||
|
let cand = table[h] as usize;
|
||||||
|
table[h] = (i + 1) as u32;
|
||||||
|
if cand != 0 {
|
||||||
|
let r = cand - 1;
|
||||||
|
let distance = i - r;
|
||||||
|
if distance <= MAX_OFFSET && input[r..r + 3] == input[i..i + 3] {
|
||||||
|
let max_len = (n - i).min(MAX_REF);
|
||||||
|
let mut len = 3;
|
||||||
|
while len < max_len && input[r + len] == input[i + len] {
|
||||||
|
len += 1;
|
||||||
|
}
|
||||||
|
flush_literals(&mut out, &input[lit_start..i]);
|
||||||
|
let code = len - 2;
|
||||||
|
let off = distance - 1;
|
||||||
|
if code < 7 {
|
||||||
|
out.push(((code << 5) | (off >> 8)) as u8);
|
||||||
|
} else {
|
||||||
|
out.push(((7 << 5) | (off >> 8)) as u8);
|
||||||
|
out.push((code - 7) as u8);
|
||||||
|
}
|
||||||
|
out.push((off & 0xff) as u8);
|
||||||
|
// Index the positions the match covered so later data can
|
||||||
|
// refer back into it.
|
||||||
|
let end = i + len;
|
||||||
|
let mut j = i + 1;
|
||||||
|
while j < end && j + 2 < n {
|
||||||
|
table[hash3(&input[j..])] = (j + 1) as u32;
|
||||||
|
j += 1;
|
||||||
|
}
|
||||||
|
i = end;
|
||||||
|
lit_start = i;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
i += 1;
|
||||||
|
}
|
||||||
|
flush_literals(&mut out, &input[lit_start..]);
|
||||||
|
out
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
fn round_trip(data: &[u8]) {
|
||||||
|
let c = lzf_compress(data);
|
||||||
|
assert_eq!(lzf_decompress(&c, data.len(), data.len()).unwrap(), data);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn round_trips() {
|
||||||
|
round_trip(b"");
|
||||||
|
round_trip(b"a");
|
||||||
|
round_trip(b"abcabcabcabcabcabcabcabcabcabcabcabc");
|
||||||
|
round_trip(&[7u8; 10_000]);
|
||||||
|
let noise: Vec<u8> = (0..70_000u32)
|
||||||
|
.map(|i| (i.wrapping_mul(2_654_435_761) >> 13) as u8)
|
||||||
|
.collect();
|
||||||
|
round_trip(&noise);
|
||||||
|
let ramp: Vec<u8> = (0..100_000u32)
|
||||||
|
.flat_map(|i| (i % 1000).to_le_bytes())
|
||||||
|
.collect();
|
||||||
|
round_trip(&ramp);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn compresses_repetitive_data() {
|
||||||
|
let data = [42u8; 4096];
|
||||||
|
assert!(lzf_compress(&data).len() < 100);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The chunk h5py 3.16's bundled liblzf writes for
|
||||||
|
/// `b"hello hello hello hello"` (read back with `read_direct_chunk`): a
|
||||||
|
/// 7-byte literal, a 14-byte back reference 6 bytes back (extended
|
||||||
|
/// length), and a 2-byte literal.
|
||||||
|
#[test]
|
||||||
|
fn decodes_liblzf_output() {
|
||||||
|
let stream = b"\x06hello h\xe0\x05\x05\x01lo";
|
||||||
|
assert_eq!(
|
||||||
|
lzf_decompress(stream, 23, 23).unwrap(),
|
||||||
|
b"hello hello hello hello"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn rejects_corrupt_streams() {
|
||||||
|
// Back reference before the start.
|
||||||
|
assert!(lzf_decompress(&[0x20, 0x00], 10, 10).is_err());
|
||||||
|
// Literal run past the end.
|
||||||
|
assert!(lzf_decompress(&[0x05, 1, 2], 10, 10).is_err());
|
||||||
|
// Output over the limit.
|
||||||
|
let c = lzf_compress(&[1u8; 100]);
|
||||||
|
assert!(lzf_decompress(&c, 10, 99).is_err());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Random and mutated streams: errors are fine, panics are not.
|
||||||
|
#[test]
|
||||||
|
fn fuzzed_streams_never_panic() {
|
||||||
|
let seeds: Vec<Vec<u8>> = [
|
||||||
|
b"hello hello hello hello".to_vec(),
|
||||||
|
vec![7u8; 3000],
|
||||||
|
(0..2000u32).flat_map(|i| (i % 37).to_le_bytes()).collect(),
|
||||||
|
(0..500u32)
|
||||||
|
.map(|i| (i.wrapping_mul(2_654_435_761) >> 13) as u8)
|
||||||
|
.collect(),
|
||||||
|
]
|
||||||
|
.iter()
|
||||||
|
.map(|d| lzf_compress(d))
|
||||||
|
.collect();
|
||||||
|
for limit in [0usize, 23, 4096, 8000] {
|
||||||
|
crate::test_fuzz::fuzz_decoder(
|
||||||
|
0x1f2 + limit as u64,
|
||||||
|
&seeds[..1],
|
||||||
|
5_000,
|
||||||
|
limit.max(23),
|
||||||
|
|s| lzf_decompress(s, limit, limit.max(23)),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
crate::test_fuzz::fuzz_decoder(0x1f3, &seeds, 20_000, 8000, |s| {
|
||||||
|
lzf_decompress(s, 8000, 8000)
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -1,19 +1,40 @@
|
|||||||
//! SZIP (libaec Adaptive Entropy Coding) decompression.
|
//! SZIP (libaec Adaptive Entropy Coding) decompression.
|
||||||
//!
|
//!
|
||||||
//! Gated by the `szip` feature which links against the system libaec library.
|
//! Gated by the `szip` feature which links against the system libaec library.
|
||||||
|
//!
|
||||||
|
//! libhdf5's SZIP filter (`H5Zszip.c`) prefixes each chunk with its
|
||||||
|
//! uncompressed size and hands the rest to szlib's `SZ_BufftoBuffDecompress`.
|
||||||
|
//! libaec implements that call (`sz_compat.c`) on top of `aec_buffer_decode`
|
||||||
|
//! with some reshaping — 32/64-bit samples are coded as byte planes of 8-bit
|
||||||
|
//! samples, and scanlines that are not a whole number of blocks are padded —
|
||||||
|
//! which [`szip_decompress`] reproduces so its output matches libhdf5's.
|
||||||
|
|
||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
use alloc::vec::Vec;
|
use alloc::vec::Vec;
|
||||||
|
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
|
|
||||||
/// Decompress SZIP-compressed data using libaec.
|
/// `SZ_MSB_OPTION_MASK`: samples are big-endian.
|
||||||
|
#[cfg(feature = "szip")]
|
||||||
|
const SZ_MSB_OPTION_MASK: u32 = 16;
|
||||||
|
/// `SZ_NN_OPTION_MASK`: nearest-neighbour preprocessing.
|
||||||
|
#[cfg(feature = "szip")]
|
||||||
|
const SZ_NN_OPTION_MASK: u32 = 32;
|
||||||
|
|
||||||
|
/// Decompress one SZIP-filtered chunk.
|
||||||
///
|
///
|
||||||
/// `cd` is the HDF5 SZIP filter client data (matches `H5Z_SZIP_PARM_*` indices):
|
/// `cd` is the HDF5 SZIP filter client data (`H5Z_SZIP_PARM_*` indices):
|
||||||
/// cd[0] = options mask (`H5_SZIP_NN_OPTION_MASK = 0x20` enables NN preprocessing)
|
/// cd[0] = options mask (`SZ_*_OPTION_MASK`: 16 = MSB byte order,
|
||||||
/// cd[1] = pixels per block (H5Z_SZIP_PARM_PPB; 8, 10, 16, or 32)
|
/// 32 = nearest-neighbour preprocessing; K13/EC/LSB/RAW bits carry
|
||||||
/// cd[2] = bits per sample (H5Z_SZIP_PARM_BPP; element bit width)
|
/// no decoding information for libaec)
|
||||||
/// cd[3] = pixels per scan line (H5Z_SZIP_PARM_PPS; informational only)
|
/// cd[1] = pixels per block
|
||||||
|
/// cd[2] = bits per pixel (sample precision, rounded up to 32 or 64 above
|
||||||
|
/// 24 by libhdf5)
|
||||||
|
/// cd[3] = pixels per scanline
|
||||||
|
///
|
||||||
|
/// The chunk is a 4-byte little-endian uncompressed size followed by the
|
||||||
|
/// szlib stream.
|
||||||
|
#[cfg_attr(not(feature = "szip"), allow(dead_code))]
|
||||||
pub(crate) fn szip_decompress(
|
pub(crate) fn szip_decompress(
|
||||||
_data: &[u8],
|
_data: &[u8],
|
||||||
_cd: &[u32],
|
_cd: &[u32],
|
||||||
@@ -33,62 +54,174 @@ pub(crate) fn szip_decompress(
|
|||||||
|
|
||||||
#[cfg(feature = "szip")]
|
#[cfg(feature = "szip")]
|
||||||
fn szip_decode_impl(data: &[u8], cd: &[u32], chunk_size: usize) -> Result<Vec<u8>, FormatError> {
|
fn szip_decode_impl(data: &[u8], cd: &[u32], chunk_size: usize) -> Result<Vec<u8>, FormatError> {
|
||||||
if cd.len() < 3 {
|
let err = |m: &str| FormatError::ChunkedReadError(format!("szip: {m}"));
|
||||||
return Err(FormatError::ChunkedReadError(
|
if cd.len() < 4 {
|
||||||
"szip: missing client data".into(),
|
return Err(err("missing client data"));
|
||||||
));
|
|
||||||
}
|
}
|
||||||
let options = cd[0];
|
let options = cd[0];
|
||||||
let pixels_per_block = cd[1];
|
let pixels_per_block = cd[1] as usize;
|
||||||
let bits_per_sample = cd[2]; // H5Z_SZIP_PARM_BPP
|
let bits_per_pixel = cd[2];
|
||||||
if bits_per_sample == 0 || bits_per_sample > 32 {
|
let pixels_per_scanline = cd[3] as usize;
|
||||||
return Err(FormatError::ChunkedReadError(
|
if !(1..=32).contains(&bits_per_pixel) && bits_per_pixel != 64 {
|
||||||
"szip: invalid bits per sample".into(),
|
return Err(err("invalid bits per sample"));
|
||||||
));
|
|
||||||
}
|
}
|
||||||
if chunk_size == 0 {
|
if pixels_per_block == 0 || pixels_per_scanline == 0 {
|
||||||
return Err(FormatError::ChunkedReadError(
|
return Err(err("invalid block or scanline size"));
|
||||||
"szip: unknown output size".into(),
|
|
||||||
));
|
|
||||||
}
|
}
|
||||||
if data.is_empty() {
|
if data.len() < 4 {
|
||||||
return Err(FormatError::ChunkedReadError("szip: empty input".into()));
|
return Err(err("chunk too short"));
|
||||||
}
|
}
|
||||||
|
// H5Zszip.c: UINT32DECODE of the uncompressed size, then the stream.
|
||||||
|
let dest_len = u32::from_le_bytes([data[0], data[1], data[2], data[3]]) as usize;
|
||||||
|
let limit = if chunk_size != 0 {
|
||||||
|
chunk_size
|
||||||
|
} else {
|
||||||
|
crate::filters::MAX_DECOMPRESS_SIZE
|
||||||
|
};
|
||||||
|
if dest_len > limit {
|
||||||
|
return Err(err("declared size exceeds chunk size"));
|
||||||
|
}
|
||||||
|
let stream = &data[4..];
|
||||||
|
|
||||||
// Map HDF5 option mask to libaec flags.
|
// --- libaec sz_compat.c: SZ_BufftoBuffDecompress ---
|
||||||
// HDF5 always stores SZIP data in MSB order, so AEC_DATA_MSB is unconditional.
|
let rsi = pixels_per_scanline.div_ceil(pixels_per_block);
|
||||||
// H5_SZIP_NN_OPTION_MASK (0x20): NN differential preprocessing.
|
let mut flags = 0;
|
||||||
let mut flags: u32 = libaec_sys::AEC_DATA_MSB;
|
if options & SZ_MSB_OPTION_MASK != 0 {
|
||||||
if options & 0x20 != 0 {
|
flags |= libaec_sys::AEC_DATA_MSB;
|
||||||
|
}
|
||||||
|
if options & SZ_NN_OPTION_MASK != 0 {
|
||||||
flags |= libaec_sys::AEC_DATA_PREPROCESS;
|
flags |= libaec_sys::AEC_DATA_PREPROCESS;
|
||||||
}
|
}
|
||||||
|
let pad_scanline = !pixels_per_scanline.is_multiple_of(pixels_per_block);
|
||||||
|
let deinterleave = bits_per_pixel == 32 || bits_per_pixel == 64;
|
||||||
|
let bits_per_sample = if deinterleave { 8 } else { bits_per_pixel };
|
||||||
|
let pixel_size = match bits_per_sample {
|
||||||
|
17.. => 4,
|
||||||
|
9.. => 2,
|
||||||
|
_ => 1,
|
||||||
|
};
|
||||||
|
let scanlines = (dest_len / pixel_size).div_ceil(pixels_per_scanline);
|
||||||
|
let buf_size = if pad_scanline {
|
||||||
|
rsi.checked_mul(pixels_per_block)
|
||||||
|
.and_then(|n| n.checked_mul(pixel_size))
|
||||||
|
.and_then(|n| n.checked_mul(scanlines))
|
||||||
|
.filter(|&n| n <= crate::filters::MAX_DECOMPRESS_SIZE.max(limit))
|
||||||
|
.ok_or_else(|| err("scanline padding too large"))?
|
||||||
|
} else {
|
||||||
|
dest_len
|
||||||
|
};
|
||||||
|
|
||||||
let mut out = vec![0u8; chunk_size];
|
let mut buf = vec![0u8; buf_size];
|
||||||
let mut strm = libaec_sys::AecStream::zeroed();
|
let mut strm = libaec_sys::AecStream::zeroed();
|
||||||
strm.next_in = data.as_ptr();
|
strm.next_in = stream.as_ptr();
|
||||||
strm.avail_in = data.len();
|
strm.avail_in = stream.len();
|
||||||
strm.next_out = out.as_mut_ptr();
|
strm.next_out = buf.as_mut_ptr();
|
||||||
strm.avail_out = chunk_size;
|
strm.avail_out = buf_size;
|
||||||
strm.bits_per_sample = bits_per_sample;
|
strm.bits_per_sample = bits_per_sample;
|
||||||
strm.block_size = pixels_per_block;
|
strm.block_size = pixels_per_block as u32;
|
||||||
strm.rsi = 128; // HDF5 default: 128 blocks per reference sample interval
|
strm.rsi = rsi as u32;
|
||||||
strm.flags = flags;
|
strm.flags = flags;
|
||||||
|
// SAFETY: next_in/avail_in and next_out/avail_out describe live buffers
|
||||||
|
// (`stream` and `buf`) that outlive the call.
|
||||||
let result = unsafe { libaec_sys::aec_buffer_decode(&mut strm) };
|
let result = unsafe { libaec_sys::aec_buffer_decode(&mut strm) };
|
||||||
if result != 0 {
|
if result != 0 {
|
||||||
return Err(FormatError::DecompressionError(format!(
|
return Err(FormatError::DecompressionError(format!(
|
||||||
"szip: libaec error {result}"
|
"szip: libaec error {result}"
|
||||||
)));
|
)));
|
||||||
}
|
}
|
||||||
let decoded_len = chunk_size - strm.avail_out;
|
let mut total_out = strm.total_out;
|
||||||
out.truncate(decoded_len);
|
if pad_scanline {
|
||||||
Ok(out)
|
let line = pixels_per_scanline * pixel_size;
|
||||||
|
let padded_line = rsi * pixels_per_block * pixel_size;
|
||||||
|
// remove_padding: compact each padded line down to `line` bytes.
|
||||||
|
let mut i = line;
|
||||||
|
let mut j = padded_line;
|
||||||
|
while j < total_out {
|
||||||
|
let end = (j + line).min(buf.len());
|
||||||
|
buf.copy_within(j..end, i);
|
||||||
|
i += line;
|
||||||
|
j += padded_line;
|
||||||
|
}
|
||||||
|
total_out = scanlines * line;
|
||||||
|
}
|
||||||
|
if total_out < dest_len {
|
||||||
|
return Err(err("stream decoded to fewer bytes than declared"));
|
||||||
|
}
|
||||||
|
buf.truncate(dest_len);
|
||||||
|
if deinterleave {
|
||||||
|
// deinterleave_buffer: byte planes back into words.
|
||||||
|
let w = (bits_per_pixel / 8) as usize;
|
||||||
|
let n = dest_len / w;
|
||||||
|
let mut out = vec![0u8; dest_len];
|
||||||
|
for i in 0..n {
|
||||||
|
for j in 0..w {
|
||||||
|
out[i * w + j] = buf[j * n + i];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(out)
|
||||||
|
} else {
|
||||||
|
Ok(buf)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
mod tests {
|
mod tests {
|
||||||
use super::*;
|
use super::*;
|
||||||
|
|
||||||
|
#[cfg(feature = "szip")]
|
||||||
|
fn unhex(s: &str) -> Vec<u8> {
|
||||||
|
(0..s.len())
|
||||||
|
.step_by(2)
|
||||||
|
.map(|i| u8::from_str_radix(&s[i..i + 2], 16).unwrap())
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// SZIP chunks written by libhdf5, decoded exactly as libhdf5 decodes
|
||||||
|
/// them. Each case: fixture, chunk byte offset and size (from h5py's
|
||||||
|
/// `get_chunk_info`), the filter's cd_values, and the chunk's values as
|
||||||
|
/// h5py reads them (file byte order, hex). Before the fix every one of
|
||||||
|
/// these came back as garbage or zeros (or "invalid bits per sample" for
|
||||||
|
/// 64-bit): the 4-byte size prefix was fed to libaec, 32/64-bit samples
|
||||||
|
/// were not de-interleaved from byte planes, the reference sample
|
||||||
|
/// interval was fixed at 128 instead of derived from the scanline, padded
|
||||||
|
/// scanlines were not unpadded, and LE data was decoded as MSB.
|
||||||
|
#[cfg(feature = "szip")]
|
||||||
|
#[test]
|
||||||
|
fn szip_decodes_libhdf5_chunks_exactly() {
|
||||||
|
/// (name, file, chunk offset, chunk size, cd_values, decoded hex)
|
||||||
|
type Case<'a> = (&'a str, &'a [u8], usize, usize, [u32; 4], &'a str);
|
||||||
|
let noencoder: &[u8] = include_bytes!("../tests/fixtures/filters/noencoder.h5");
|
||||||
|
let le_data: &[u8] = include_bytes!("../tests/fixtures/filters/le_data.h5");
|
||||||
|
let h5py: &[u8] = include_bytes!("../tests/fixtures/filters/szip_h5py.h5");
|
||||||
|
#[rustfmt::skip]
|
||||||
|
let cases: &[Case] = &[
|
||||||
|
// <i4, 10 px/scanline over 4 px/block: padded scanlines + byte planes.
|
||||||
|
("noencoder /noencoder_szip_dset.h5", noencoder, 6040, 16, [168, 4, 32, 10],
|
||||||
|
"00000000010000000200000003000000040000000500000006000000070000000800000009000000"),
|
||||||
|
// <f4, LSB + NN.
|
||||||
|
("le_data /Szip_float_data_le", le_data, 55224, 48, [169, 4, 32, 12],
|
||||||
|
"abaaaa3eabaa2a3f0000803fabaa2a3f0000803fabaaaa3f0000803fabaaaa3f5555d53fabaaaa3f5555d53f00000040"),
|
||||||
|
// >f4, MSB + NN.
|
||||||
|
("le_data /Szip_float_data_be", le_data, 55396, 48, [177, 4, 32, 12],
|
||||||
|
"3eaaaaab3f2aaaab3f8000003f2aaaab3f8000003faaaaab3f8000003faaaaab3fd555553faaaaab3fd5555540000000"),
|
||||||
|
// <f8 (64-bit), NN.
|
||||||
|
("szip_h5py /f8", h5py, 4016, 100, [169, 8, 64, 10],
|
||||||
|
"00000000000008c000000000000008c000000000000008c000000000000008c000000000000004c000000000000004c000000000000004c000000000000004c000000000000000c000000000000000c000000000000000c000000000000000c0000000000000f8bf000000000000f8bf000000000000f8bf000000000000f8bf000000000000f0bf000000000000f0bf000000000000f0bf000000000000f0bf000000000000e0bf000000000000e0bf000000000000e0bf000000000000e0bf0000000000000000000000000000000000000000000000000000000000000000000000000000e03f000000000000e03f000000000000e03f000000000000e03f000000000000f03f000000000000f03f000000000000f03f000000000000f03f000000000000f83f000000000000f83f000000000000f83f000000000000f83f"),
|
||||||
|
// <i8 (64-bit), entropy coding without NN.
|
||||||
|
("szip_h5py /i8", h5py, 4188, 53, [141, 4, 64, 10],
|
||||||
|
"000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000300000000000000030000000000000003000000000000000300000000000000030000000000000003000000000000000300000000000000030000000000000006000000000000000600000000000000060000000000000006000000000000000600000000000000060000000000000006000000000000000600000000000000090000000000000009000000000000000900000000000000090000000000000009000000000000000900000000000000090000000000000009000000000000000c000000000000000c000000000000000c000000000000000c000000000000000c000000000000000c000000000000000c000000000000000c00000000000000"),
|
||||||
|
// <u2, 35 px/scanline over 8 px/block: padded scanlines, 16-bit samples.
|
||||||
|
("szip_h5py /u2", h5py, 4308, 43, [169, 8, 16, 35],
|
||||||
|
"00000000000000006100610061006100c200c200c200c20023012301230123018401840184018401e501e501e501e5014602460246024602a702a702a702a702080308030803"),
|
||||||
|
];
|
||||||
|
for (name, file, off, len, cd, want) in cases {
|
||||||
|
let want = unhex(want);
|
||||||
|
let got = szip_decompress(&file[*off..off + len], cd, want.len())
|
||||||
|
.unwrap_or_else(|e| panic!("{name}: {e:?}"));
|
||||||
|
assert_eq!(got, want, "{name}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn szip_disabled_returns_unsupported() {
|
fn szip_disabled_returns_unsupported() {
|
||||||
#[cfg(not(feature = "szip"))]
|
#[cfg(not(feature = "szip"))]
|
||||||
@@ -132,6 +265,8 @@ mod tests {
|
|||||||
assert_eq!(rc, 0, "aec_buffer_encode failed: {rc}");
|
assert_eq!(rc, 0, "aec_buffer_encode failed: {rc}");
|
||||||
let enc_len = encoded.len() - enc.avail_out;
|
let enc_len = encoded.len() - enc.avail_out;
|
||||||
encoded.truncate(enc_len);
|
encoded.truncate(enc_len);
|
||||||
|
// H5Zszip.c prefixes the stream with the uncompressed size.
|
||||||
|
encoded.splice(0..0, (original.len() as u32).to_le_bytes());
|
||||||
|
|
||||||
// Decode through our public interface.
|
// Decode through our public interface.
|
||||||
// cd[0]=0 (no NN bit 0x20), cd[1]=8 (ppb), cd[2]=8 (bpp), cd[3]=1024 (pps).
|
// cd[0]=0 (no NN bit 0x20), cd[1]=8 (ppb), cd[2]=8 (bpp), cd[3]=1024 (pps).
|
||||||
@@ -163,6 +298,8 @@ mod tests {
|
|||||||
assert_eq!(rc, 0, "aec_buffer_encode with NN failed: {rc}");
|
assert_eq!(rc, 0, "aec_buffer_encode with NN failed: {rc}");
|
||||||
let enc_len = encoded.len() - enc.avail_out;
|
let enc_len = encoded.len() - enc.avail_out;
|
||||||
encoded.truncate(enc_len);
|
encoded.truncate(enc_len);
|
||||||
|
// H5Zszip.c prefixes the stream with the uncompressed size.
|
||||||
|
encoded.splice(0..0, (original.len() as u32).to_le_bytes());
|
||||||
|
|
||||||
// cd[0] = 0x20 (H5_SZIP_NN_OPTION_MASK) → decoder must set AEC_DATA_PREPROCESS.
|
// cd[0] = 0x20 (H5_SZIP_NN_OPTION_MASK) → decoder must set AEC_DATA_PREPROCESS.
|
||||||
let cd = [0x20u32, 8, 8, 1024];
|
let cd = [0x20u32, 8, 8, 1024];
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -6,18 +6,23 @@ extern crate alloc;
|
|||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
use alloc::{format, vec, vec::Vec};
|
use alloc::{format, vec, vec::Vec};
|
||||||
|
|
||||||
|
use crate::addr::to_usize;
|
||||||
|
use crate::chunk_grid::ChunkGrid;
|
||||||
use crate::chunked_read::ChunkInfo;
|
use crate::chunked_read::ChunkInfo;
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
|
use crate::storage::{PAGED_BLOCK_ONE_READ_MAX, Storage, Window, len_usize, read_exact_at};
|
||||||
|
|
||||||
/// Verify the Jenkins lookup3 checksum stored immediately after
|
/// Verify the Jenkins lookup3 checksum stored immediately after
|
||||||
/// `data[start..end]`, as every Fixed Array structure carries one.
|
/// `data[start..end]`, as every Fixed Array structure carries one. `w` is
|
||||||
|
/// a window of the file and `start`/`end` are relative to it.
|
||||||
///
|
///
|
||||||
/// A corrupt chunk index silently yields addresses pointing at the wrong
|
/// A corrupt chunk index silently yields addresses pointing at the wrong
|
||||||
/// bytes, so a mismatch has to be an error rather than a shrug: without this
|
/// bytes, so a mismatch has to be an error rather than a shrug: without this
|
||||||
/// the damage surfaces as plausible-looking data from the wrong chunk.
|
/// the damage surfaces as plausible-looking data from the wrong chunk.
|
||||||
#[cfg(feature = "checksum")]
|
#[cfg(feature = "checksum")]
|
||||||
fn verify_checksum(data: &[u8], start: usize, end: usize) -> Result<(), FormatError> {
|
fn verify_checksum(w: &Window<'_>, start: usize, end: usize) -> Result<(), FormatError> {
|
||||||
ensure_len(data, end, 4)?;
|
w.ensure(end, 4)?;
|
||||||
|
let data = &w.bytes;
|
||||||
let stored = u32::from_le_bytes([data[end], data[end + 1], data[end + 2], data[end + 3]]);
|
let stored = u32::from_le_bytes([data[end], data[end + 1], data[end + 2], data[end + 3]]);
|
||||||
let computed = crate::checksum::jenkins_lookup3(&data[start..end]);
|
let computed = crate::checksum::jenkins_lookup3(&data[start..end]);
|
||||||
if computed != stored {
|
if computed != stored {
|
||||||
@@ -30,7 +35,7 @@ fn verify_checksum(data: &[u8], start: usize, end: usize) -> Result<(), FormatEr
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[cfg(not(feature = "checksum"))]
|
#[cfg(not(feature = "checksum"))]
|
||||||
fn verify_checksum(_data: &[u8], _start: usize, _end: usize) -> Result<(), FormatError> {
|
fn verify_checksum(_w: &Window<'_>, _start: usize, _end: usize) -> Result<(), FormatError> {
|
||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -72,19 +77,6 @@ fn read_length(data: &[u8], pos: usize, size: u8) -> Result<u64, FormatError> {
|
|||||||
read_offset(data, pos, size)
|
read_offset(data, pos, size)
|
||||||
}
|
}
|
||||||
|
|
||||||
fn ensure_len(data: &[u8], offset: usize, needed: usize) -> Result<(), FormatError> {
|
|
||||||
if offset
|
|
||||||
.checked_add(needed)
|
|
||||||
.is_none_or(|end| end > data.len())
|
|
||||||
{
|
|
||||||
return Err(FormatError::UnexpectedEof {
|
|
||||||
expected: offset.saturating_add(needed),
|
|
||||||
available: data.len(),
|
|
||||||
});
|
|
||||||
}
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
fn is_undefined(data: &[u8], pos: usize, size: u8) -> bool {
|
fn is_undefined(data: &[u8], pos: usize, size: u8) -> bool {
|
||||||
let s = size as usize;
|
let s = size as usize;
|
||||||
if pos + s > data.len() {
|
if pos + s > data.len() {
|
||||||
@@ -100,13 +92,24 @@ impl FixedArrayHeader {
|
|||||||
offset: usize,
|
offset: usize,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
|
) -> Result<Self, FormatError> {
|
||||||
|
Self::parse_in(file_data, offset as u64, offset_size, length_size)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`Self::parse`] over any [`Storage`]: one read of the header.
|
||||||
|
pub fn parse_in<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
offset: u64,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
) -> Result<Self, FormatError> {
|
) -> Result<Self, FormatError> {
|
||||||
// FAHD signature(4) + version(1) + client_id(1) + element_size(1) +
|
// FAHD signature(4) + version(1) + client_id(1) + element_size(1) +
|
||||||
// max_nelmts_bits(1) + num_elements(length_size) + data_block_addr(offset_size) + checksum(4)
|
// max_nelmts_bits(1) + num_elements(length_size) + data_block_addr(offset_size) + checksum(4)
|
||||||
let min_size = 4 + 1 + 1 + 1 + 1 + length_size as usize + offset_size as usize + 4;
|
let min_size = 4 + 1 + 1 + 1 + 1 + length_size as usize + offset_size as usize + 4;
|
||||||
ensure_len(file_data, offset, min_size)?;
|
let w = Window::read(file, offset, min_size)?;
|
||||||
|
w.ensure(0, min_size)?;
|
||||||
|
|
||||||
let d = &file_data[offset..];
|
let d: &[u8] = &w.bytes;
|
||||||
if &d[0..4] != b"FAHD" {
|
if &d[0..4] != b"FAHD" {
|
||||||
return Err(FormatError::ChunkedReadError(
|
return Err(FormatError::ChunkedReadError(
|
||||||
"invalid Fixed Array header signature".into(),
|
"invalid Fixed Array header signature".into(),
|
||||||
@@ -129,7 +132,7 @@ impl FixedArrayHeader {
|
|||||||
pos += length_size as usize;
|
pos += length_size as usize;
|
||||||
let data_block_address = read_offset(d, pos, offset_size)?;
|
let data_block_address = read_offset(d, pos, offset_size)?;
|
||||||
pos += offset_size as usize;
|
pos += offset_size as usize;
|
||||||
verify_checksum(file_data, offset, offset + pos)?;
|
verify_checksum(&w, 0, pos)?;
|
||||||
|
|
||||||
Ok(FixedArrayHeader {
|
Ok(FixedArrayHeader {
|
||||||
client_id,
|
client_id,
|
||||||
@@ -151,19 +154,43 @@ pub fn read_fixed_array_chunks(
|
|||||||
file_data: &[u8],
|
file_data: &[u8],
|
||||||
header: &FixedArrayHeader,
|
header: &FixedArrayHeader,
|
||||||
dataset_dims: &[u64],
|
dataset_dims: &[u64],
|
||||||
|
max_dims: Option<&[u64]>,
|
||||||
|
chunk_dimensions: &[u32],
|
||||||
|
element_size: u32,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<Vec<ChunkInfo>, FormatError> {
|
||||||
|
read_fixed_array_chunks_in(
|
||||||
|
&file_data,
|
||||||
|
header,
|
||||||
|
dataset_dims,
|
||||||
|
max_dims,
|
||||||
|
chunk_dimensions,
|
||||||
|
element_size,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`read_fixed_array_chunks`] over any [`Storage`]: one read of the data
|
||||||
|
/// block's prefix, one of the whole data block (pages included).
|
||||||
|
#[allow(clippy::too_many_arguments)]
|
||||||
|
pub fn read_fixed_array_chunks_in<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
header: &FixedArrayHeader,
|
||||||
|
dataset_dims: &[u64],
|
||||||
|
max_dims: Option<&[u64]>,
|
||||||
chunk_dimensions: &[u32],
|
chunk_dimensions: &[u32],
|
||||||
element_size: u32,
|
element_size: u32,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
_length_size: u8,
|
_length_size: u8,
|
||||||
) -> Result<Vec<ChunkInfo>, FormatError> {
|
) -> Result<Vec<ChunkInfo>, FormatError> {
|
||||||
let db_offset = header.data_block_address as usize;
|
let file_len = len_usize(file);
|
||||||
let rank = chunk_dimensions.len();
|
let db_offset = to_usize(header.data_block_address)?;
|
||||||
|
|
||||||
// Parse data block header: FADB(4) + version(1) + client_id(1) + header_address(offset_size)
|
// Parse data block header: FADB(4) + version(1) + client_id(1) + header_address(offset_size)
|
||||||
let db_header_size = 4 + 1 + 1 + offset_size as usize;
|
let db_header_size = 4 + 1 + 1 + offset_size as usize;
|
||||||
ensure_len(file_data, db_offset, db_header_size)?;
|
let d = read_exact_at(file, db_offset as u64, db_header_size)?;
|
||||||
|
|
||||||
let d = &file_data[db_offset..];
|
|
||||||
if &d[0..4] != b"FADB" {
|
if &d[0..4] != b"FADB" {
|
||||||
return Err(FormatError::ChunkedReadError(
|
return Err(FormatError::ChunkedReadError(
|
||||||
"invalid Fixed Array data block signature".into(),
|
"invalid Fixed Array data block signature".into(),
|
||||||
@@ -173,11 +200,11 @@ pub fn read_fixed_array_chunks(
|
|||||||
// Elements start immediately after the data block prefix.
|
// Elements start immediately after the data block prefix.
|
||||||
let elements_start = db_offset + db_header_size;
|
let elements_start = db_offset + db_header_size;
|
||||||
|
|
||||||
let num_elements = header.num_elements as usize;
|
let num_elements = to_usize(header.num_elements)?;
|
||||||
// A chunk index cannot describe more elements than the file has bytes (each
|
// A chunk index cannot describe more elements than the file has bytes (each
|
||||||
// element occupies at least `offset_size` bytes). Reject a corrupt count
|
// element occupies at least `offset_size` bytes). Reject a corrupt count
|
||||||
// before it can drive a huge loop or overflow an offset computation.
|
// before it can drive a huge loop or overflow an offset computation.
|
||||||
if num_elements > file_data.len() {
|
if num_elements > file_len {
|
||||||
return Err(FormatError::ChunkedReadError(
|
return Err(FormatError::ChunkedReadError(
|
||||||
"Fixed Array element count exceeds file size".into(),
|
"Fixed Array element count exceeds file size".into(),
|
||||||
));
|
));
|
||||||
@@ -198,44 +225,43 @@ pub fn read_fixed_array_chunks(
|
|||||||
))
|
))
|
||||||
};
|
};
|
||||||
|
|
||||||
// Compute chunk offsets based on index.
|
// The index is laid out over the chunk grid of the *maximum* dimensions
|
||||||
// Chunks are stored in row-major order within the dataset space.
|
// (row-major), so a dataset smaller than its maxshape has gaps.
|
||||||
let mut num_chunks_per_dim = Vec::with_capacity(rank);
|
let dims_u64: Vec<u64> = chunk_dimensions.iter().map(|&d| d as u64).collect();
|
||||||
for d_idx in 0..rank {
|
let grid = ChunkGrid::fixed_array(dataset_dims, max_dims, &dims_u64)?;
|
||||||
let ch_dim = chunk_dimensions[d_idx] as u64;
|
|
||||||
if ch_dim == 0 {
|
|
||||||
return Err(FormatError::ChunkedReadError(
|
|
||||||
"chunk dimension is zero".into(),
|
|
||||||
));
|
|
||||||
}
|
|
||||||
let ds_dim = dataset_dims[d_idx];
|
|
||||||
num_chunks_per_dim.push(ds_dim.div_ceil(ch_dim));
|
|
||||||
}
|
|
||||||
|
|
||||||
let chunk_byte_size: u64 =
|
let chunk_byte_size: u64 =
|
||||||
chunk_dimensions.iter().map(|&d| d as u64).product::<u64>() * element_size as u64;
|
chunk_dimensions.iter().map(|&d| d as u64).product::<u64>() * element_size as u64;
|
||||||
|
|
||||||
let mut chunks = Vec::new();
|
let mut chunks = Vec::new();
|
||||||
let push_element =
|
// `rel` is relative to the data block, whose bytes are in `w`.
|
||||||
|i: usize, abs: usize, chunks: &mut Vec<ChunkInfo>| -> Result<(), FormatError> {
|
let push_element = |w: &Window<'_>,
|
||||||
if let Some((address, chunk_size, filter_mask)) = parse_fa_element(
|
i: usize,
|
||||||
file_data,
|
rel: usize,
|
||||||
abs,
|
chunks: &mut Vec<ChunkInfo>|
|
||||||
header.client_id,
|
-> Result<(), FormatError> {
|
||||||
offset_size,
|
if let Some((address, chunk_size, filter_mask)) = parse_fa_element(
|
||||||
header.element_size,
|
w,
|
||||||
chunk_byte_size,
|
rel,
|
||||||
)? {
|
header.client_id,
|
||||||
let offsets = index_to_chunk_offsets(i, &num_chunks_per_dim, chunk_dimensions);
|
offset_size,
|
||||||
chunks.push(ChunkInfo {
|
header.element_size,
|
||||||
chunk_size,
|
chunk_byte_size,
|
||||||
filter_mask,
|
)? {
|
||||||
offsets,
|
// A slot beyond the current extent is ignored, as the
|
||||||
address,
|
// library does.
|
||||||
});
|
let Some(offsets) = grid.offsets(i as u64) else {
|
||||||
}
|
return Ok(());
|
||||||
Ok(())
|
};
|
||||||
};
|
chunks.push(ChunkInfo {
|
||||||
|
chunk_size,
|
||||||
|
filter_mask,
|
||||||
|
offsets,
|
||||||
|
address,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
};
|
||||||
|
|
||||||
// A data block is paged when it holds more elements than fit in one page.
|
// A data block is paged when it holds more elements than fit in one page.
|
||||||
// `max_nelmts_bits` is an untrusted u8; a shift >= the pointer width would
|
// `max_nelmts_bits` is an untrusted u8; a shift >= the pointer width would
|
||||||
@@ -250,10 +276,16 @@ pub fn read_fixed_array_chunks(
|
|||||||
|
|
||||||
if !is_paged {
|
if !is_paged {
|
||||||
// Non-paged: prefix, then `num_elements` elements packed directly,
|
// Non-paged: prefix, then `num_elements` elements packed directly,
|
||||||
// then a checksum over both.
|
// then a checksum over both. One window holds all of it (or ends at
|
||||||
verify_checksum(file_data, db_offset, elem_at(elements_start, num_elements)?)?;
|
// the end of the file), so its bounds checks are the whole-file ones.
|
||||||
|
let end = elem_at(elements_start, num_elements)?;
|
||||||
|
// The checksum's bounds check comes first: make it before reading.
|
||||||
|
#[cfg(feature = "checksum")]
|
||||||
|
Window::check_extent(file, db_offset as u64, end - db_offset, 4)?;
|
||||||
|
let w = Window::read(file, db_offset as u64, end.saturating_add(4) - db_offset)?;
|
||||||
|
verify_checksum(&w, 0, end - db_offset)?;
|
||||||
for i in 0..num_elements {
|
for i in 0..num_elements {
|
||||||
push_element(i, elem_at(elements_start, i)?, &mut chunks)?;
|
push_element(&w, i, elem_at(elements_start, i)? - db_offset, &mut chunks)?;
|
||||||
}
|
}
|
||||||
return Ok(chunks);
|
return Ok(chunks);
|
||||||
}
|
}
|
||||||
@@ -276,22 +308,40 @@ pub fn read_fixed_array_chunks(
|
|||||||
.and_then(|x| x.checked_add(4))
|
.and_then(|x| x.checked_add(4))
|
||||||
.ok_or_else(stride_overflow)?;
|
.ok_or_else(stride_overflow)?;
|
||||||
|
|
||||||
if bitmap_start + bitmap_size > file_data.len() {
|
if bitmap_start + bitmap_size > file_len {
|
||||||
return Err(FormatError::UnexpectedEof {
|
return Err(FormatError::UnexpectedEof {
|
||||||
expected: bitmap_start + bitmap_size,
|
expected: bitmap_start + bitmap_size,
|
||||||
available: file_data.len(),
|
available: file_len,
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
// The whole data block in one window when it is small: every page slot
|
||||||
|
// is at most `page_stride` bytes, so every position checked below lies
|
||||||
|
// inside it (or past the end of the file). A larger block is read as its
|
||||||
|
// prefix and bitmap, then each page in use on its own.
|
||||||
|
let block_len = (pages_start - db_offset).saturating_add(npages.saturating_mul(page_stride));
|
||||||
|
let whole = if block_len <= PAGED_BLOCK_ONE_READ_MAX {
|
||||||
|
Some(Window::read(file, db_offset as u64, block_len)?)
|
||||||
|
} else {
|
||||||
|
None
|
||||||
|
};
|
||||||
|
let head_w;
|
||||||
|
let head = match &whole {
|
||||||
|
Some(w) => w,
|
||||||
|
None => {
|
||||||
|
head_w = Window::read(file, db_offset as u64, pages_start - db_offset)?;
|
||||||
|
&head_w
|
||||||
|
}
|
||||||
|
};
|
||||||
// The prefix and page bitmap are covered by their own checksum, and each
|
// The prefix and page bitmap are covered by their own checksum, and each
|
||||||
// initialised page by one of its own.
|
// initialised page by one of its own.
|
||||||
verify_checksum(file_data, db_offset, bitmap_start + bitmap_size)?;
|
verify_checksum(head, 0, bitmap_start + bitmap_size - db_offset)?;
|
||||||
|
|
||||||
for p in 0..npages {
|
for p in 0..npages {
|
||||||
let page_first = p * page_nelmts; // < num_elements, cannot overflow
|
let page_first = p * page_nelmts; // < num_elements, cannot overflow
|
||||||
let page_count = core::cmp::min(page_nelmts, num_elements - page_first);
|
let page_count = core::cmp::min(page_nelmts, num_elements - page_first);
|
||||||
|
|
||||||
// Check the page-init bit (MSB-first within each byte).
|
// Check the page-init bit (MSB-first within each byte).
|
||||||
let bit_byte = file_data[bitmap_start + p / 8];
|
let bit_byte = head.bytes[bitmap_start + p / 8 - db_offset];
|
||||||
let bit_mask = 1u8 << (7 - (p % 8));
|
let bit_mask = 1u8 << (7 - (p % 8));
|
||||||
if bit_byte & bit_mask == 0 {
|
if bit_byte & bit_mask == 0 {
|
||||||
continue; // entire page unallocated
|
continue; // entire page unallocated
|
||||||
@@ -301,21 +351,33 @@ pub fn read_fixed_array_chunks(
|
|||||||
.checked_mul(page_stride)
|
.checked_mul(page_stride)
|
||||||
.and_then(|o| pages_start.checked_add(o))
|
.and_then(|o| pages_start.checked_add(o))
|
||||||
.ok_or_else(stride_overflow)?;
|
.ok_or_else(stride_overflow)?;
|
||||||
verify_checksum(file_data, page_off, elem_at(page_off, page_count)?)?;
|
let page_end = elem_at(page_off, page_count)?;
|
||||||
|
// `w` holds the page from `base` on (positions below are relative
|
||||||
|
// to it).
|
||||||
|
let page_w;
|
||||||
|
let (w, base) = match &whole {
|
||||||
|
Some(w) => (w, db_offset),
|
||||||
|
None => {
|
||||||
|
page_w =
|
||||||
|
Window::read(file, page_off as u64, page_end.saturating_add(4) - page_off)?;
|
||||||
|
(&page_w, page_off)
|
||||||
|
}
|
||||||
|
};
|
||||||
|
verify_checksum(w, page_off - base, page_end - base)?;
|
||||||
for e in 0..page_count {
|
for e in 0..page_count {
|
||||||
push_element(page_first + e, elem_at(page_off, e)?, &mut chunks)?;
|
push_element(w, page_first + e, elem_at(page_off, e)? - base, &mut chunks)?;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
Ok(chunks)
|
Ok(chunks)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Parse a single Fixed Array element at absolute file offset `abs`.
|
/// Parse a single Fixed Array element at offset `abs` of the window `w`.
|
||||||
///
|
///
|
||||||
/// Returns `Some((address, chunk_size, filter_mask))` for an allocated chunk, or
|
/// Returns `Some((address, chunk_size, filter_mask))` for an allocated chunk, or
|
||||||
/// `None` if the element is undefined (an unallocated chunk, address all-`0xFF`).
|
/// `None` if the element is undefined (an unallocated chunk, address all-`0xFF`).
|
||||||
fn parse_fa_element(
|
fn parse_fa_element(
|
||||||
file_data: &[u8],
|
w: &Window<'_>,
|
||||||
abs: usize,
|
abs: usize,
|
||||||
client_id: u8,
|
client_id: u8,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
@@ -325,12 +387,8 @@ fn parse_fa_element(
|
|||||||
let os = offset_size as usize;
|
let os = offset_size as usize;
|
||||||
if client_id == 0 {
|
if client_id == 0 {
|
||||||
// Non-filtered: element is just the chunk address.
|
// Non-filtered: element is just the chunk address.
|
||||||
if abs + os > file_data.len() {
|
w.ensure(abs, os)?;
|
||||||
return Err(FormatError::UnexpectedEof {
|
let file_data: &[u8] = &w.bytes;
|
||||||
expected: abs + os,
|
|
||||||
available: file_data.len(),
|
|
||||||
});
|
|
||||||
}
|
|
||||||
if is_undefined(file_data, abs, offset_size) {
|
if is_undefined(file_data, abs, offset_size) {
|
||||||
return Ok(None);
|
return Ok(None);
|
||||||
}
|
}
|
||||||
@@ -345,17 +403,14 @@ fn parse_fa_element(
|
|||||||
));
|
));
|
||||||
}
|
}
|
||||||
let chunk_size_bytes = es - os - 4;
|
let chunk_size_bytes = es - os - 4;
|
||||||
if abs + es > file_data.len() {
|
w.ensure(abs, es)?;
|
||||||
return Err(FormatError::UnexpectedEof {
|
let file_data: &[u8] = &w.bytes;
|
||||||
expected: abs + es,
|
|
||||||
available: file_data.len(),
|
|
||||||
});
|
|
||||||
}
|
|
||||||
if is_undefined(file_data, abs, offset_size) {
|
if is_undefined(file_data, abs, offset_size) {
|
||||||
return Ok(None);
|
return Ok(None);
|
||||||
}
|
}
|
||||||
let address = read_offset(file_data, abs, offset_size)?;
|
let address = read_offset(file_data, abs, offset_size)?;
|
||||||
let chunk_size = read_variable_length(&file_data[abs + os..], chunk_size_bytes)?;
|
let chunk_size =
|
||||||
|
read_variable_length(&file_data[abs + os..abs + es - 4], chunk_size_bytes)?;
|
||||||
let fm_off = abs + os + chunk_size_bytes;
|
let fm_off = abs + os + chunk_size_bytes;
|
||||||
let filter_mask = u32::from_le_bytes([
|
let filter_mask = u32::from_le_bytes([
|
||||||
file_data[fm_off],
|
file_data[fm_off],
|
||||||
@@ -367,27 +422,6 @@ fn parse_fa_element(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Convert a linear chunk index to N-dimensional chunk offsets in dataset space.
|
|
||||||
fn index_to_chunk_offsets(
|
|
||||||
index: usize,
|
|
||||||
num_chunks_per_dim: &[u64],
|
|
||||||
chunk_dimensions: &[u32],
|
|
||||||
) -> Vec<u64> {
|
|
||||||
let rank = num_chunks_per_dim.len();
|
|
||||||
let mut offsets = vec![0u64; rank];
|
|
||||||
let mut remaining = index as u64;
|
|
||||||
for d in (0..rank).rev() {
|
|
||||||
let nchunks = num_chunks_per_dim[d];
|
|
||||||
if nchunks == 0 {
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
let chunk_idx = remaining % nchunks;
|
|
||||||
remaining /= nchunks;
|
|
||||||
offsets[d] = chunk_idx * chunk_dimensions[d] as u64;
|
|
||||||
}
|
|
||||||
offsets
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Read a variable-length little-endian unsigned integer.
|
/// Read a variable-length little-endian unsigned integer.
|
||||||
fn read_variable_length(data: &[u8], size: usize) -> Result<u64, FormatError> {
|
fn read_variable_length(data: &[u8], size: usize) -> Result<u64, FormatError> {
|
||||||
if size > 8 || data.len() < size {
|
if size > 8 || data.len() < size {
|
||||||
@@ -416,44 +450,21 @@ mod tests {
|
|||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn index_to_offsets_1d() {
|
fn index_to_offsets_1d() {
|
||||||
let num_chunks = vec![5u64];
|
let g = ChunkGrid::fixed_array(&[100], None, &[20]).unwrap();
|
||||||
let chunk_dims = vec![20u32];
|
assert_eq!(g.offsets(0).unwrap(), vec![0]);
|
||||||
assert_eq!(index_to_chunk_offsets(0, &num_chunks, &chunk_dims), vec![0]);
|
assert_eq!(g.offsets(1).unwrap(), vec![20]);
|
||||||
assert_eq!(
|
assert_eq!(g.offsets(4).unwrap(), vec![80]);
|
||||||
index_to_chunk_offsets(1, &num_chunks, &chunk_dims),
|
|
||||||
vec![20]
|
|
||||||
);
|
|
||||||
assert_eq!(
|
|
||||||
index_to_chunk_offsets(4, &num_chunks, &chunk_dims),
|
|
||||||
vec![80]
|
|
||||||
);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn index_to_offsets_2d() {
|
fn index_to_offsets_2d() {
|
||||||
// 10x6 dataset with 4x3 chunks => ceil(10/4)=3, ceil(6/3)=2 => 6 chunks
|
// 10x6 dataset with 4x3 chunks => ceil(10/4)=3, ceil(6/3)=2 => 6 chunks
|
||||||
let num_chunks = vec![3u64, 2];
|
let g = ChunkGrid::fixed_array(&[10, 6], None, &[4, 3]).unwrap();
|
||||||
let chunk_dims = vec![4u32, 3];
|
assert_eq!(g.offsets(0).unwrap(), vec![0, 0]);
|
||||||
assert_eq!(
|
assert_eq!(g.offsets(1).unwrap(), vec![0, 3]);
|
||||||
index_to_chunk_offsets(0, &num_chunks, &chunk_dims),
|
assert_eq!(g.offsets(2).unwrap(), vec![4, 0]);
|
||||||
vec![0, 0]
|
assert_eq!(g.offsets(3).unwrap(), vec![4, 3]);
|
||||||
);
|
assert_eq!(g.offsets(5).unwrap(), vec![8, 3]);
|
||||||
assert_eq!(
|
|
||||||
index_to_chunk_offsets(1, &num_chunks, &chunk_dims),
|
|
||||||
vec![0, 3]
|
|
||||||
);
|
|
||||||
assert_eq!(
|
|
||||||
index_to_chunk_offsets(2, &num_chunks, &chunk_dims),
|
|
||||||
vec![4, 0]
|
|
||||||
);
|
|
||||||
assert_eq!(
|
|
||||||
index_to_chunk_offsets(3, &num_chunks, &chunk_dims),
|
|
||||||
vec![4, 3]
|
|
||||||
);
|
|
||||||
assert_eq!(
|
|
||||||
index_to_chunk_offsets(5, &num_chunks, &chunk_dims),
|
|
||||||
vec![8, 3]
|
|
||||||
);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
@@ -517,7 +528,7 @@ mod tests {
|
|||||||
|
|
||||||
let read = |f: &[u8], fahd: usize| -> Result<Vec<ChunkInfo>, FormatError> {
|
let read = |f: &[u8], fahd: usize| -> Result<Vec<ChunkInfo>, FormatError> {
|
||||||
let h = FixedArrayHeader::parse(f, fahd, 8, 8)?;
|
let h = FixedArrayHeader::parse(f, fahd, 8, 8)?;
|
||||||
read_fixed_array_chunks(f, &h, &[60], &[20], 8, 8, 8)
|
read_fixed_array_chunks(f, &h, &[60], None, &[20], 8, 8, 8)
|
||||||
};
|
};
|
||||||
|
|
||||||
let (clean, fahd) = build();
|
let (clean, fahd) = build();
|
||||||
@@ -562,7 +573,7 @@ mod tests {
|
|||||||
let db = 0x100usize;
|
let db = 0x100usize;
|
||||||
buf[db..db + 4].copy_from_slice(b"FADB");
|
buf[db..db + 4].copy_from_slice(b"FADB");
|
||||||
let header = FixedArrayHeader::parse(&buf, fahd, 8, 8).unwrap();
|
let header = FixedArrayHeader::parse(&buf, fahd, 8, 8).unwrap();
|
||||||
let r = read_fixed_array_chunks(&buf, &header, &[100], &[20], 8, 8, 8);
|
let r = read_fixed_array_chunks(&buf, &header, &[100], None, &[20], 8, 8, 8);
|
||||||
assert!(r.is_err());
|
assert!(r.is_err());
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -579,7 +590,7 @@ mod tests {
|
|||||||
stamp_checksum(&mut buf, fahd, fahd + 24);
|
stamp_checksum(&mut buf, fahd, fahd + 24);
|
||||||
buf[0x80..0x84].copy_from_slice(b"FADB");
|
buf[0x80..0x84].copy_from_slice(b"FADB");
|
||||||
let header = FixedArrayHeader::parse(&buf, fahd, 8, 8).unwrap();
|
let header = FixedArrayHeader::parse(&buf, fahd, 8, 8).unwrap();
|
||||||
let r = read_fixed_array_chunks(&buf, &header, &[100], &[20], 8, 8, 8);
|
let r = read_fixed_array_chunks(&buf, &header, &[100], None, &[20], 8, 8, 8);
|
||||||
assert!(r.is_err());
|
assert!(r.is_err());
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -602,7 +613,7 @@ mod tests {
|
|||||||
data_block_address: (usize::MAX - 4) as u64,
|
data_block_address: (usize::MAX - 4) as u64,
|
||||||
};
|
};
|
||||||
let buf = vec![0u8; 64];
|
let buf = vec![0u8; 64];
|
||||||
let r = read_fixed_array_chunks(&buf, &header, &[100], &[20], 8, 8, 8);
|
let r = read_fixed_array_chunks(&buf, &header, &[100], None, &[20], 8, 8, 8);
|
||||||
assert!(r.is_err());
|
assert!(r.is_err());
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -664,6 +675,7 @@ mod tests {
|
|||||||
&file_data,
|
&file_data,
|
||||||
&header,
|
&header,
|
||||||
&ds_dims,
|
&ds_dims,
|
||||||
|
None,
|
||||||
&chunk_dims,
|
&chunk_dims,
|
||||||
8,
|
8,
|
||||||
offset_size,
|
offset_size,
|
||||||
@@ -740,6 +752,7 @@ mod tests {
|
|||||||
&file_data,
|
&file_data,
|
||||||
&header,
|
&header,
|
||||||
&ds_dims,
|
&ds_dims,
|
||||||
|
None,
|
||||||
&chunk_dims,
|
&chunk_dims,
|
||||||
8,
|
8,
|
||||||
offset_size,
|
offset_size,
|
||||||
@@ -840,6 +853,7 @@ mod tests {
|
|||||||
&file_data,
|
&file_data,
|
||||||
&header,
|
&header,
|
||||||
&ds_dims,
|
&ds_dims,
|
||||||
|
None,
|
||||||
&chunk_dims,
|
&chunk_dims,
|
||||||
8,
|
8,
|
||||||
offset_size,
|
offset_size,
|
||||||
@@ -858,4 +872,126 @@ mod tests {
|
|||||||
.collect();
|
.collect();
|
||||||
assert_eq!(got, expect);
|
assert_eq!(got, expect);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// A fixed array (header at 0x100, data block at 0x200) of `n` chunks,
|
||||||
|
/// filtered or not, paged when `n` exceeds `1 << page_bits`; every
|
||||||
|
/// page initialised except page 1.
|
||||||
|
fn build_fixed_array(n: usize, filtered: bool, page_bits: u8) -> Vec<u8> {
|
||||||
|
let os = 8usize;
|
||||||
|
let es = if filtered { os + 4 + 4 } else { os };
|
||||||
|
let (fahd, db) = (0x100usize, 0x200usize);
|
||||||
|
let mut f = vec![0u8; 0x2000];
|
||||||
|
f[fahd..fahd + 4].copy_from_slice(b"FAHD");
|
||||||
|
f[fahd + 5] = u8::from(filtered);
|
||||||
|
f[fahd + 6] = es as u8;
|
||||||
|
f[fahd + 7] = page_bits;
|
||||||
|
f[fahd + 8..fahd + 16].copy_from_slice(&(n as u64).to_le_bytes());
|
||||||
|
f[fahd + 16..fahd + 24].copy_from_slice(&(db as u64).to_le_bytes());
|
||||||
|
stamp_checksum(&mut f, fahd, fahd + 24);
|
||||||
|
f[db..db + 4].copy_from_slice(b"FADB");
|
||||||
|
f[db + 5] = u8::from(filtered);
|
||||||
|
f[db + 6..db + 14].copy_from_slice(&(fahd as u64).to_le_bytes());
|
||||||
|
let elems = db + 6 + os;
|
||||||
|
let write = |f: &mut Vec<u8>, at: usize, i: usize| {
|
||||||
|
let addr = if i == 2 {
|
||||||
|
u64::MAX
|
||||||
|
} else {
|
||||||
|
0x1000 + i as u64 * 0x100
|
||||||
|
};
|
||||||
|
f[at..at + os].copy_from_slice(&addr.to_le_bytes());
|
||||||
|
if filtered {
|
||||||
|
f[at + os..at + os + 4].copy_from_slice(&(100 + i as u32).to_le_bytes());
|
||||||
|
f[at + os + 4..at + os + 8].copy_from_slice(&(i as u32 & 1).to_le_bytes());
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let page = 1usize << page_bits;
|
||||||
|
if n <= page {
|
||||||
|
for i in 0..n {
|
||||||
|
write(&mut f, elems + i * es, i);
|
||||||
|
}
|
||||||
|
stamp_checksum(&mut f, db, elems + n * es);
|
||||||
|
} else {
|
||||||
|
let npages = n.div_ceil(page);
|
||||||
|
let bitmap = npages.div_ceil(8);
|
||||||
|
for p in 0..npages {
|
||||||
|
if p != 1 {
|
||||||
|
f[elems + p / 8] |= 0x80 >> (p % 8);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
stamp_checksum(&mut f, db, elems + bitmap);
|
||||||
|
let pages_start = elems + bitmap + 4;
|
||||||
|
for p in (0..npages).filter(|&p| p != 1) {
|
||||||
|
let at = pages_start + p * (page * es + 4);
|
||||||
|
let count = page.min(n - p * page);
|
||||||
|
for e in 0..count {
|
||||||
|
write(&mut f, at + e * es, p * page + e);
|
||||||
|
}
|
||||||
|
stamp_checksum(&mut f, at, at + count * es);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
f
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Non-paged and paged, filtered and unfiltered arrays, cut at every
|
||||||
|
/// length through the data block and with a damaged byte, read
|
||||||
|
/// identically through a `read_at`-only storage.
|
||||||
|
#[test]
|
||||||
|
fn storage_reads_match_slice_reads() {
|
||||||
|
use crate::storage::CountingStorage;
|
||||||
|
for (n, filtered, bits) in [(3, false, 10), (3, true, 10), (11, false, 2), (11, true, 2)] {
|
||||||
|
let full = build_fixed_array(n, filtered, bits);
|
||||||
|
let es = if filtered { 16 } else { 8 };
|
||||||
|
let dims = [n as u64 * 20];
|
||||||
|
let h = FixedArrayHeader::parse(&full, 0x100, 8, 8).unwrap();
|
||||||
|
let chunks = read_fixed_array_chunks(&full, &h, &dims, None, &[20], 8, 8, 8).unwrap();
|
||||||
|
// Chunk 2 is unallocated, and so is page 1 of a paged array.
|
||||||
|
let expect = if n > 4 { n - 1 - 4 } else { n - 1 };
|
||||||
|
assert_eq!(chunks.len(), expect);
|
||||||
|
let mut files = Vec::new();
|
||||||
|
for cut in (0x100..0x200 + 40 + n * (es + 4) + 16).step_by(3) {
|
||||||
|
files.push(full[..cut].to_vec());
|
||||||
|
}
|
||||||
|
for at in [0x104, 0x210, 0x21a, 0x230] {
|
||||||
|
let mut damaged = full.clone();
|
||||||
|
damaged[at] ^= 1;
|
||||||
|
files.push(damaged);
|
||||||
|
}
|
||||||
|
files.push(full);
|
||||||
|
for f in files {
|
||||||
|
let storage = CountingStorage::new(f.clone());
|
||||||
|
let want = FixedArrayHeader::parse(&f, 0x100, 8, 8);
|
||||||
|
let got = FixedArrayHeader::parse_in(&storage, 0x100, 8, 8);
|
||||||
|
assert_eq!(format!("{got:?}"), format!("{want:?}"));
|
||||||
|
let Ok(h) = want else { continue };
|
||||||
|
let want = read_fixed_array_chunks(&f, &h, &dims, None, &[20], 8, 8, 8);
|
||||||
|
let got = read_fixed_array_chunks_in(&storage, &h, &dims, None, &[20], 8, 8, 8);
|
||||||
|
assert_eq!(format!("{got:?}"), format!("{want:?}"), "{} bytes", f.len());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A header whose element count stretches its data block (one checksum
|
||||||
|
/// over the whole block) far past the end of a 16 MiB file: the
|
||||||
|
/// checksum's bounds check fails before the block is read, with the
|
||||||
|
/// slice read's error.
|
||||||
|
#[cfg(feature = "checksum")]
|
||||||
|
#[test]
|
||||||
|
fn oversized_block_fails_before_reading() {
|
||||||
|
use crate::storage::CountingStorage;
|
||||||
|
let mut f = build_fixed_array(3, false, 10);
|
||||||
|
f.resize(16 << 20, 0);
|
||||||
|
let mut h = FixedArrayHeader::parse(&f, 0x100, 8, 8).unwrap();
|
||||||
|
h.max_nelmts_bits = 30;
|
||||||
|
h.num_elements = 4 << 20;
|
||||||
|
let dims = [h.num_elements * 20];
|
||||||
|
let want = read_fixed_array_chunks(&f, &h, &dims, None, &[20], 8, 8, 8);
|
||||||
|
assert!(
|
||||||
|
matches!(want, Err(FormatError::UnexpectedEof { .. })),
|
||||||
|
"{want:?}"
|
||||||
|
);
|
||||||
|
let storage = CountingStorage::new(f);
|
||||||
|
let got = read_fixed_array_chunks_in(&storage, &h, &dims, None, &[20], 8, 8, 8);
|
||||||
|
assert_eq!(format!("{got:?}"), format!("{want:?}"));
|
||||||
|
assert!(storage.bytes_read() < 64, "{} bytes", storage.bytes_read());
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user